完整代碼
開源代碼
統計三國演義人物高頻次數
#!/usr/bin/env python
# coding=utf-8
#e10.4CalThreeKingdoms.py
import jieba
excludes = {"來到","人馬","領兵","將軍","卻說","荊州","二人","不可","不能","如此"}
txt = open("threekingdom.txt", "rb").read()
words = jieba.lcut(txt)
counts = {}
for word in words:if len(word) == 1:continueelif word == "諸葛亮" or word == "孔明曰":rword = "孔明"elif word == "關公" or word == "云長":rword = "關羽"elif word == "玄德" or word == "玄德曰":rword = "劉備"elif word == "孟德" or word == "丞相":rword = "曹操"else:rword = wordcounts[rword] = counts.get(rword,0) + 1
for word in excludes:del(counts[word])
items = list(counts.items())
items.sort(key=lambda x:x[1], reverse=True)
for i in range(55):word, count = items[i]print ("{0:<10}{1:>5}".format(word, count))
代碼運行:人物頻率統計
threekingdom.txt kingdom.py
kou@ubuntu:~/python/file_文本處理$ python3 kingdom.py
Building prefix dict from the default dictionary ...
Dumping model to file cache /tmp/jieba.cache
Loading model cost 2.446 seconds.
Prefix dict has been built succesfully.
曹操 1348
劉備 1144
孔明 865
關羽 557
呂布 322
張飛 300
詞云圖片
#!/usr/bin/env python
# coding=utf-8import jieba
import wordcloudf = open("threekingdom.txt","rb")
t = f.read()
f.close()
ls = jieba.lcut(t)
txt = " ".join(ls)
w = wordcloud.WordCloud( font_path = "NotoSerifCJK-Bold.ttc",\width = 1000,height = 700,background_color = "white",\)w.generate(txt)
w.to_file("gr.png")