Python 文本词频统计
excludes={"the","and","of","you","my"}
def getText():
txt=open(r"C:\Users\DELL\Desktop\英语短文.txt","r",encoding="UTF-8").read()
txt=txt.lower()
for ch in '"!@#$%&()+-,.:;?/{}[]~`':
txt=txt.replace(ch," ")
return txt
英语短文Txt=getText()
words=英语短文Txt.split()
counts={}
for word in words:
counts[word]=counts.get(word,0)+1
for word in excludes:
del(counts[word])
items=list(counts.items())
items.sort(key=lambda x:x[1],reverse=True)
for i in range(10):
word,count=items[i]
print("{0:<10}{1:>5}".format(word,count))