diff --git a/README.rst b/README.rst index 0c10463a..c55e8d2f 100644 --- a/README.rst +++ b/README.rst @@ -201,6 +201,7 @@ Features fr French he Hebrew it Italian + ja Japanese ko Korean no Norwegian pt Portuguese @@ -265,6 +266,25 @@ NOTE: If you find problem installing ``libpng12-dev``, try installing ``libpng-d $ curl https://raw.githubusercontent.com/codelucas/newspaper/master/download_corpora.py | python3 +**If you are on CentOS** and use Japanese language support, install using the following: + +- Install ``mecab`` - Japanese morphological analyzer:: + + $ rpm -ivh http://packages.groonga.org/centos/groonga-release-1.1.0-1.noarch.rpm + + $ yum install mecab mecab-devel mecab-ipadic + + $ pip install mecab-python3 + +- Install ``mecab-ipadic-NEologd`` - Neologism dictionary for MeCab (Optional):: + + $ mkdir ~/src && cd ~/src && git clone --depth 1 https://github.com/neologd/mecab-ipadic-neologd.git + + $ cd ~/src/mecab-ipadic-neologd && ./bin/install-mecab-ipadic-neologd -n + + $ curl https://raw.githubusercontent.com/codelucas/newspaper/master/download_corpora.py | python3 + + **Otherwise**, install with the following: NOTE: You will still most likely need to install the following libraries via your package manager diff --git a/docs/index.rst b/docs/index.rst index aaf60fc6..198fe112 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -32,6 +32,7 @@ Inspired by `requests`_ for its simplicity and powered by `lxml`_ for its speed. fr French he Hebrew it Italian + ja Japanese ko Korean no Norwegian pt Portuguese diff --git a/docs/user_guide/install.rst b/docs/user_guide/install.rst index 7d7f645a..a2236fcc 100644 --- a/docs/user_guide/install.rst +++ b/docs/user_guide/install.rst @@ -33,6 +33,23 @@ However, you will run into fixable issues if you are trying to install on ubuntu $ curl https://raw.githubusercontent.com/codelucas/newspaper/master/download_corpora.py | python2.7 +**If you are on CentOS** and use Japanese language support, install using the following: + +- Install ``mecab`` - Japanese morphological analyzer:: + + $ rpm -ivh http://packages.groonga.org/centos/groonga-release-1.1.0-1.noarch.rpm + + $ yum install mecab mecab-devel mecab-ipadic + + $ pip install mecab-python3 + +- Install ``mecab-ipadic-NEologd`` - Neologism dictionary for MeCab (Optional):: + + $ mkdir ~/src && cd ~/src && git clone --depth 1 https://github.com/neologd/mecab-ipadic-neologd.git + + $ cd ~/src/mecab-ipadic-neologd && ./bin/install-mecab-ipadic-neologd -n + + $ curl https://raw.githubusercontent.com/codelucas/newspaper/master/download_corpora.py | python3 **If you are on OSX**, install using the following, you may use both homebrew or macports: diff --git a/docs/user_guide/quickstart.rst b/docs/user_guide/quickstart.rst index a0fdaab6..07b045eb 100644 --- a/docs/user_guide/quickstart.rst +++ b/docs/user_guide/quickstart.rst @@ -254,6 +254,7 @@ of popular news source urls.. In case you need help choosing a news source! fr French he Hebrew it Italian + ja Japanese ko Korean no Norwegian pt Portuguese diff --git a/newspaper/configuration.py b/newspaper/configuration.py index f93a532c..e506a180 100644 --- a/newspaper/configuration.py +++ b/newspaper/configuration.py @@ -14,7 +14,7 @@ from .parsers import Parser from .text import (StopWords, StopWordsArabic, StopWordsChinese, - StopWordsKorean) + StopWordsKorean, StopWordsJapanese) from .version import __version__ log = logging.getLogger(__name__) @@ -108,6 +108,8 @@ def get_stopwords_class(self, language): return StopWordsChinese elif language == 'ar': return StopWordsArabic + elif language == 'ja': + return StopWordsJapanese return StopWords def get_parser(self): diff --git a/newspaper/resources/text/stopwords-ja.txt b/newspaper/resources/text/stopwords-ja.txt new file mode 100644 index 00000000..9f2b3e42 --- /dev/null +++ b/newspaper/resources/text/stopwords-ja.txt @@ -0,0 +1,416 @@ +あか +あそこ +あたり +あたりまえ +あちら +あっち +あと +あな +あなた +あなたに +あの +あのかた +あの人 +あります +ありません +あれ +あん +いいね +いかが +いが +いくつ +いくら +いずれ +いっぱい +いつ +いつか +いづみ +いま +います +いも +いや +いろいろ +うえ +うそ +うた +うち +うな +え +おおまか +おかげ +おなじみ +おば +おまえ +および +おります +おれ +おろそか +お互い +お気に入り +お知らせ +かく +かずい +かた +かたち +かなり +かやの +から +かわな +が +がい +がち +がら +きた +きっかけ +きゅう +くじ +くせ +ここ +こちら +こっち +こと +この +このほど +こまめ +これ +これら +ころ +ごっちゃ +ごと +ごろ +さまざま +さらい +さん +し +しかし +しかた +しながら +しなやか +しよう +しん +じき +すか +すで +すね +すべて +ずつ +ぜんぶ +そう +そこ +そこそこ +そちら +そっち +そで +その +その他 +その後 +その間 +それ +それぞれ +それで +それなり +たくさん +たち +たび +ため +だめ +だれ +ちゃ +ちゃん +っぷり +つば +つぶ +てん +で +である +できないよ +です +ですよ +では +と +とおり +とき +ときどき +とこ +ところ +どこ +どこか +どちら +どっか +どっち +どの +どれ +なか +なかば +など +なに +ならでは +なん +なんだろう +に +にあ +にも +の +のち +のり +は +はじまり +はじめ +はず +はるか +ひと +ひとつ +ひとつひとつ +ひととき +ひとみ +ひろ +ふう +ふく +ふたり +ぶり +ぷらす +へん +べつ +ぺん +ほう +ほか +ぼくら +まさ +まし +ます +またがり +まで +まとめ +まとも +まま +みす +みたい +みつ +みなさん +みんな +めど +も +もと +もの +もん +やつ +やりとり +よう +よそ +より +わけ +わたし +われわれ +を +カ所 +カ月 +ハイ +レ +ヵ所 +ヵ月 +ヶ所 +ヶ月 +一 +一つ +七 +万 +三 +上 +上記 +下 +下記 +中 +九 +事 +二 +五 +人 +今 +今回 +他 +以上 +以下 +以前 +以後 +以降 +会 +伸 +体 +何 +何人 +作 +例 +係 +俺 +個 +億 +元 +兆 +先 +全部 +八 +六 +内 +円 +冬 +分 +列 +別 +前 +前回 +力 +化 +匹 +区 +十 +千 +半ば +口 +台 +右 +各 +同じ +名 +名前 +向こう +哀 +品 +員 +喜 +器 +四 +回 +国 +土 +地 +場 +場合 +境 +士 +夏 +外 +多く +女 +奴 +婦 +子 +字 +室 +家 +屋 +左 +市 +席 +年 +年生 +幾つ +店 +府 +度 +式 +形 +彼 +彼女 +後 +怒 +性 +情 +感 +感じ +我々 +所 +手 +手段 +扱い +数 +文 +新た +方 +方法 +日 +春 +時 +時点 +時間 +書 +月 +木 +未満 +本当 +村 +束 +枚 +校 +楽 +様 +様々 +次 +歳 +歴 +段 +毎 +毎日 +氏 +気 +水 +法 +火 +点 +玉 +用 +男 +町 +界 +略 +百 +的 +目 +県 +確か +私 +私達 +秋 +秒 +第 +等 +箇所 +箇月 +簿 +系 +紀 +結局 +続きを読む +線 +者 +自体 +自分 +行 +見 +観 +話 +誌 +誰 +課 +論 +貴方 +貴方方 +輪 +近く +通 +連 +週 +道 +達 +違い +部 +都 +金 +間 +関係 +際 +集 +面 +頃 +類 +首 +高 \ No newline at end of file diff --git a/newspaper/text.py b/newspaper/text.py index 0c8a1f7f..641e47e3 100644 --- a/newspaper/text.py +++ b/newspaper/text.py @@ -156,3 +156,65 @@ def get_stopword_count(self, content): ws.set_stopword_count(len(overlapping_stopwords)) ws.set_stop_words(overlapping_stopwords) return ws + + +class StopWordsJapanese(StopWords): + """Japanese segmentation + """ + def __init__(self, language='ja'): + super(StopWordsJapanese, self).__init__(language='ja') + + def candidate_words(self, stripped_input): + words = [] + try: + import MeCab + import os, subprocess + # Check if MecCab is installed + mecab_path = subprocess.Popen( + ['which', 'mecab'], + stdout=subprocess.PIPE + ).communicate()[0].decode().replace('\n', '') + if os.path.exists(mecab_path): + # Check if MecCab dictionary exists + dic_basedir = subprocess.Popen( + ['mecab-config', '--dicdir'], + stdout=subprocess.PIPE + ).communicate()[0].decode().replace('\n', '') + if os.path.exists(dic_basedir): + # Setup MecCab dictionary + dic_dir = os.path.join(dic_basedir, 'mecab-ipadic-neologd') + if os.path.exists(dic_dir): + mecab = MeCab.Tagger('-d {}'.format(dic_dir)) + else: + mecab = MeCab.Tagger('mecabrc') + if mecab: + mecab.parse('') + node = mecab.parseToNode(stripped_input) + while node: + word = node.surface.lower() + pos = node.feature.split(',')[0] + if len(word) > 0 and not pos == '記号': + words.append(word) + node = node.next + except Exception as e: + pass + + return words + + def get_stopword_count(self, content): + if not content: + return WordStats() + ws = WordStats() + candidate_words = self.candidate_words(content) + overlapping_stopwords = [] + regexp = u'^({})$'.format(u'|'.join(self.STOP_WORDS)) + c = 0 + for w in candidate_words: + c += 1 + if re.match(regexp, w): + overlapping_stopwords.append(w) + + ws.set_word_count(c) + ws.set_stopword_count(len(overlapping_stopwords)) + ws.set_stop_words(overlapping_stopwords) + return ws diff --git a/newspaper/utils.py b/newspaper/utils.py index 93c67a9f..9bb787ff 100644 --- a/newspaper/utils.py +++ b/newspaper/utils.py @@ -363,6 +363,7 @@ def print_available_languages(): 'fr': 'French', 'he': 'Hebrew', 'it': 'Italian', + 'ja': 'Japanese', 'ko': 'Korean', 'no': 'Norwegian', 'nb': 'Norwegian (Bokmål)', diff --git a/tests/data/html/japanese_article.html b/tests/data/html/japanese_article.html new file mode 100644 index 00000000..b6e10d63 --- /dev/null +++ b/tests/data/html/japanese_article.html @@ -0,0 +1,967 @@ + + + + +CNN.co.jp : トランプ氏へのハッキング、「計画ない」 ウィキリークス + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
+
+ + + +
+
+ + +
+ +
+
+
+ +
+ +
+ + + + +
+
+

トランプ氏へのハッキング、「計画ない」 ウィキリークス

+

2016.08.07 Sun posted at 11:30 JST

+
+ +
+

[PR]

+
+
+ + + + +
+

ワシントン(CNN) 内部告発サイト「ウィキリークス」が米民主党全国委員会(DNC)関連の電子メールなどを暴露した問題をめぐり、同サイトの創始者ジュリアン・アサンジュ氏は米大統領選の共和党候補ドナルド・トランプ氏に対するハッキングも企てていることを示唆したが、同サイトは6日、そのような計画はないと明確に否定した。

+ +

ウィキリークスは先月、DNCが大統領選に向けた党指名レースでヒラリー・クリントン氏に肩入れしていたことを示すメールなどを公開して波紋を呼んだ。アサンジュ氏は、クリントン陣営についてこのほかにも情報を握っていると主張し、公開する構えを示している。

+ +

アサンジュ氏は5日、滞在先の在英エクアドル大使館から米HBOテレビのインタビュー番組に出演。クリントン氏を支持する司会者から「トランプ氏の納税申告書はどうして入手しないのか」と問われ、「今取り組んでいるところだ」と答えた。

+ +

この発言に対し、ウィキリークスのメンバーが6日午前、「ウィキリークスはトランプ氏の納税申告に対するハッキングに取り組んではいない」とツイート。アサンジュ氏の発言は冗談にすぎなかったと説明した。

+ +

大統領選の候補者は納税申告書を公表するのが慣例となっている。しかしトランプ氏は監査中だとして公表を拒否し続けてきたため、「ロシアとの取引を隠しているのでは」などの憶測や批判を招いている。

+ +

DNCからの情報入手には、ロシアのハッカーが関与した可能性が指摘されている。インタビューの司会者は、公開された情報を取得したのがロシアであることは「明白」だと追及したが、アサンジュ氏は「だれもが知っている通り、情報源は民主党だ」とかわしていた。

+
+ +
    +
  • + + + +
  • +
  • +
    + + +
    +
    +
  • +
+ + + +
+ +
+ + + + +
+ + +

CNN.co.jpの最新情報をメールでチェック

+
+ + +
+ + +

CNN.co.jpの最新情報をメールでチェック

+
+ +
+ + +

世界の今を知る。CNN.co.jpメルマガを無料購読

+
+ +
+ + + + + + + +
+
+ +
+
+
+
+

[PR]

+
+
+
+
+ + + + + + + + +
+
+
+ + +
+
+
+
+ + + +
+
+
+
+
+
+
+
+
+
+
+ +
+ + + 図で見る「イラク・シリア・イスラム国(ISIS)」 + +
+ + +
+
+ +
+ +
+
+ +
+
+ +
+
+ + +
+
+

CNN からのご案内

+ +
+
    +
  • +
    + 携帯端末サービス +
  • +
  • +
    + CNN.co.jp App for iPhone/iPad +
  • +
  • +
    + CNNテレビ視聴 +
  • +
  • +
    + アメリカ国内版CNN/USテレビをそのまま日本で +
  • +
  • +
    + ちょっと手ごわい、でも効果絶大! +
  • +
+
+ + + + +
+
+
+ +
+ + +
+
+ + + + + + + + + + + + + + +
+
+ + \ No newline at end of file diff --git a/tests/data/text/japanese.txt b/tests/data/text/japanese.txt new file mode 100644 index 00000000..b3552de1 --- /dev/null +++ b/tests/data/text/japanese.txt @@ -0,0 +1,11 @@ +ワシントン(CNN) 内部告発サイト「ウィキリークス」が米民主党全国委員会(DNC)関連の電子メールなどを暴露した問題をめぐり、同サイトの創始者ジュリアン・アサンジュ氏は米大統領選の共和党候補ドナルド・トランプ氏に対するハッキングも企てていることを示唆したが、同サイトは6日、そのような計画はないと明確に否定した。 + +ウィキリークスは先月、DNCが大統領選に向けた党指名レースでヒラリー・クリントン氏に肩入れしていたことを示すメールなどを公開して波紋を呼んだ。アサンジュ氏は、クリントン陣営についてこのほかにも情報を握っていると主張し、公開する構えを示している。 + +アサンジュ氏は5日、滞在先の在英エクアドル大使館から米HBOテレビのインタビュー番組に出演。クリントン氏を支持する司会者から「トランプ氏の納税申告書はどうして入手しないのか」と問われ、「今取り組んでいるところだ」と答えた。 + +この発言に対し、ウィキリークスのメンバーが6日午前、「ウィキリークスはトランプ氏の納税申告に対するハッキングに取り組んではいない」とツイート。アサンジュ氏の発言は冗談にすぎなかったと説明した。 + +大統領選の候補者は納税申告書を公表するのが慣例となっている。しかしトランプ氏は監査中だとして公表を拒否し続けてきたため、「ロシアとの取引を隠しているのでは」などの憶測や批判を招いている。 + +DNCからの情報入手には、ロシアのハッカーが関与した可能性が指摘されている。インタビューの司会者は、公開された情報を取得したのがロシアであることは「明白」だと追及したが、アサンジュ氏は「だれもが知っている通り、情報源は民主党だ」とかわしていた。 \ No newline at end of file diff --git a/tests/unit_tests.py b/tests/unit_tests.py index eb5bf6db..44d59e84 100644 --- a/tests/unit_tests.py +++ b/tests/unit_tests.py @@ -602,6 +602,20 @@ def test_spanish_fulltext_extract(self): self.assertEqual(text, article.text) self.assertEqual(text, fulltext(article.html, 'es')) + @print_test + def test_japanese_fulltext_extract(self): + try: + url = 'http://www.cnn.co.jp/tech/35087106.html' + article = Article(url=url, language='ja') + html = mock_resource_with('japanese_article', 'html') + article.download(html) + article.parse() + text = mock_resource_with('japanese', 'txt') + self.assertEqual(text, article.text) + self.assertEqual(text, fulltext(article.html, 'ja')) + except Exception as e: + print('ERR', str(e)) + if __name__ == '__main__': argv = list(sys.argv)