jieba/test/demo.py

#encoding=utf-8
from __future__ import unicode_literals
import sys
sys.path.append("../")

import jieba
import jieba.posseg
import jieba.analyse

print('='*40)
print('1. 分词')
print('-'*40)

seg_list = jieba.cut("我来到北京清华大学", cut_all=True)
print("Full Mode: " + "/ ".join(seg_list))  # 全模式

seg_list = jieba.cut("我来到北京清华大学", cut_all=False)
print("Default Mode: " + "/ ".join(seg_list))  # 默认模式

seg_list = jieba.cut("他来到了网易杭研大厦")
print(", ".join(seg_list))

seg_list = jieba.cut_for_search("小明硕士毕业于中国科学院计算所，后在日本京都大学深造")  # 搜索引擎模式
print(", ".join(seg_list))

print('='*40)
print('2. 添加自定义词典/调整词典')
print('-'*40)

print('/'.join(jieba.cut('如果放到post中将出错。', HMM=False)))
#如果/放到/post/中将/出错/。
print(jieba.suggest_freq(('中', '将'), True))
#494
print('/'.join(jieba.cut('如果放到post中将出错。', HMM=False)))
#如果/放到/post/中/将/出错/。
print('/'.join(jieba.cut('「台中」正确应该不会被切开', HMM=False)))
#「/台/中/」/正确/应该/不会/被/切开
print(jieba.suggest_freq('台中', True))
#69
print('/'.join(jieba.cut('「台中」正确应该不会被切开', HMM=False)))
#「/台中/」/正确/应该/不会/被/切开

print('='*40)
print('3. 关键词提取')
print('-'*40)
print(' TF-IDF')
print('-'*40)

s = "此外，公司拟对全资子公司吉林欧亚置业有限公司增资4.3亿元，增资后，吉林欧亚置业注册资本由7000万元增加到5亿元。吉林欧亚置业主要经营范围为房地产开发及百货零售等业务。目前在建吉林欧亚城市商业综合体项目。2013年，实现营业收入0万元，实现净利润-139.13万元。"
for x, w in jieba.analyse.extract_tags(s, withWeight=True):
    print('%s %s' % (x, w))

print('-'*40)
print(' TextRank')
print('-'*40)

for x, w in jieba.analyse.textrank(s, withWeight=True):
    print('%s %s' % (x, w))

print('='*40)
print('4. 词性标注')
print('-'*40)

words = jieba.posseg.cut("我爱北京天安门")
for word, flag in words:
    print('%s %s' % (word, flag))

print('='*40)
print('6. Tokenize: 返回词语在原文的起止位置')
print('-'*40)
print(' 默认模式')
print('-'*40)

result = jieba.tokenize('永和服装饰品有限公司')
for tk in result:
    print("word %s\t\t start: %d \t\t end:%d" % (tk[0],tk[1],tk[2]))

print('-'*40)
print(' 搜索模式')
print('-'*40)

result = jieba.tokenize('永和服装饰品有限公司', mode='search')
for tk in result:
    print("word %s\t\t start: %d \t\t end:%d" % (tk[0],tk[1],tk[2]))
-												add a sample script about tags extraction

											
										
										
											13 years ago
+								#encoding=utf-8
-												Merge master and jieba3k, make the code Python 2/3 compatible

											
										
										
											10 years ago
+								from __future__ import unicode_literals
-												version chage; doc update

											
										
										
											12 years ago
+								import sys
 								sys.path.append("../")
-												add a sample script about tags extraction

											
										
										
											13 years ago
+								import jieba
-												wraps most globals in classes

API changes:
* class jieba.Tokenizer, jieba.posseg.POSTokenizer
* class jieba.analyse.TFIDF, jieba.analyse.TextRank
* global functions are mapped to jieba.(posseg.)dt, the default (POS)Tokenizer
* multiprocessing only works with jieba.(posseg.)dt
* new lcut, lcut_for_search functions that returns a list
* jieba.analyse.textrank now returns 20 items by default

Tests:
* added test_lock.py to test multithread locking
* demo.py now contains most of the examples in README

											
										
										
											10 years ago
+								import jieba.posseg
 								import jieba.analyse
 								print('='*40)
 								print('1. 分词')
 								print('-'*40)
-												add a sample script about tags extraction

											
										
										
											13 years ago
-												update to v0.33

											
										
										
											11 years ago
+								seg_list = jieba.cut("我来到北京清华大学", cut_all=True)
-												Merge master and jieba3k, make the code Python 2/3 compatible

											
										
										
											10 years ago
+								print("Full Mode: " + "/ ".join(seg_list))  # 全模式
-												add a sample script about tags extraction

											
										
										
											13 years ago
-												update to v0.33

											
										
										
											11 years ago
+								seg_list = jieba.cut("我来到北京清华大学", cut_all=False)
-												Merge master and jieba3k, make the code Python 2/3 compatible

											
										
										
											10 years ago
+								print("Default Mode: " + "/ ".join(seg_list))  # 默认模式
-												add a sample script about tags extraction

											
										
										
											13 years ago
 								seg_list = jieba.cut("他来到了网易杭研大厦")
-												first py3k version of jieba

											
										
										
											12 years ago
+								print(", ".join(seg_list))
-												version chage; doc update

											
										
										
											12 years ago
-												update to v0.33

											
										
										
											11 years ago
+								seg_list = jieba.cut_for_search("小明硕士毕业于中国科学院计算所，后在日本京都大学深造")  # 搜索引擎模式
-												first py3k version of jieba

											
										
										
											12 years ago
+								print(", ".join(seg_list))
-												wraps most globals in classes

API changes:
* class jieba.Tokenizer, jieba.posseg.POSTokenizer
* class jieba.analyse.TFIDF, jieba.analyse.TextRank
* global functions are mapped to jieba.(posseg.)dt, the default (POS)Tokenizer
* multiprocessing only works with jieba.(posseg.)dt
* new lcut, lcut_for_search functions that returns a list
* jieba.analyse.textrank now returns 20 items by default

Tests:
* added test_lock.py to test multithread locking
* demo.py now contains most of the examples in README

											
										
										
											10 years ago
 								print('='*40)
 								print('2. 添加自定义词典/调整词典')
 								print('-'*40)
 								print('/'.join(jieba.cut('如果放到post中将出错。', HMM=False)))
 								#如果/放到/post/中将/出错/。
 								print(jieba.suggest_freq(('中', '将'), True))
 								#494
 								print('/'.join(jieba.cut('如果放到post中将出错。', HMM=False)))
 								#如果/放到/post/中/将/出错/。
 								print('/'.join(jieba.cut('「台中」正确应该不会被切开', HMM=False)))
 								#「/台/中/」/正确/应该/不会/被/切开
 								print(jieba.suggest_freq('台中', True))
 								#69
 								print('/'.join(jieba.cut('「台中」正确应该不会被切开', HMM=False)))
 								#「/台中/」/正确/应该/不会/被/切开
 								print('='*40)
 								print('3. 关键词提取')
 								print('-'*40)
 								print(' TF-IDF')
 								print('-'*40)
 								s = "此外，公司拟对全资子公司吉林欧亚置业有限公司增资4.3亿元，增资后，吉林欧亚置业注册资本由7000万元增加到5亿元。吉林欧亚置业主要经营范围为房地产开发及百货零售等业务。目前在建吉林欧亚城市商业综合体项目。2013年，实现营业收入0万元，实现净利润-139.13万元。"
 								for x, w in jieba.analyse.extract_tags(s, withWeight=True):
 								    print('%s %s' % (x, w))
 								print('-'*40)
 								print(' TextRank')
 								print('-'*40)
 								for x, w in jieba.analyse.textrank(s, withWeight=True):
 								    print('%s %s' % (x, w))
 								print('='*40)
 								print('4. 词性标注')
 								print('-'*40)
 								words = jieba.posseg.cut("我爱北京天安门")
-												fix self.FREQ in cut_for_search; make pair object iterable

											
										
										
											10 years ago
+								for word, flag in words:
 								    print('%s %s' % (word, flag))
-												wraps most globals in classes

API changes:
* class jieba.Tokenizer, jieba.posseg.POSTokenizer
* class jieba.analyse.TFIDF, jieba.analyse.TextRank
* global functions are mapped to jieba.(posseg.)dt, the default (POS)Tokenizer
* multiprocessing only works with jieba.(posseg.)dt
* new lcut, lcut_for_search functions that returns a list
* jieba.analyse.textrank now returns 20 items by default

Tests:
* added test_lock.py to test multithread locking
* demo.py now contains most of the examples in README

											
										
										
											10 years ago
 								print('='*40)
 								print('6. Tokenize: 返回词语在原文的起止位置')
 								print('-'*40)
 								print(' 默认模式')
 								print('-'*40)
 								result = jieba.tokenize('永和服装饰品有限公司')
 								for tk in result:
 								    print("word %s\t\t start: %d \t\t end:%d" % (tk[0],tk[1],tk[2]))
 								print('-'*40)
 								print(' 搜索模式')
 								print('-'*40)
 								result = jieba.tokenize('永和服装饰品有限公司', mode='search')
 								for tk in result:
 								    print("word %s\t\t start: %d \t\t end:%d" % (tk[0],tk[1],tk[2]))