笔记-哈姆雷特词频统计
def getText():
txt=open(r"Hamlet.txt","r").read()#打开文件
txt=txt.lower()#将文本替换为小写
for ch in '!@#$%^&*():">?<{}~+_=-;,./':#替换标点符号为空格
txt=txt.replace(ch," ")
return txt
hamletTxt=getText()
words=hamletTxt.split()
counts={}
for word in words:
counts[word]=counts.get(word,0)+1
items=list(counts.items())
items.sort(key=lambda x:x[1],reverse=True)
for i in range(10):
word,count=items[i]
print("{0:<10}{1:>5}".format(word,count))