7-2 词频统计（30 分）

最新推荐文章于 2023-11-13 21:57:54 发布

周环

最新推荐文章于 2023-11-13 21:57:54 发布

阅读量3.4k

点赞数 1

本文链接：https://blog.csdn.net/qq_41478705/article/details/80389327

版权

请编写程序，对一段英文文本，统计其中所有不同单词的个数，以及词频最大的前10%的单词。

所谓“单词”，是指由不超过80个单词字符组成的连续字符串，但长度超过15的单词将只截取保留前15个单词字符。而合法的“单词字符”为大小写字母、数字和下划线，其它字符均认为是单词分隔符。

输入格式:

输入给出一段非空文本，最后以符号#结尾。输入保证存在至少10个不同的单词。

输出格式:

在第一行中输出文本中所有不同单词的个数。注意“单词”不区分英文大小写，例如“PAT”和“pat”被认为是同一个单词。

随后按照词频递减的顺序，按照词频:单词的格式输出词频最大的前10%的单词。若有并列，则按递增字典序输出。

输入样例：

This is a test.

The word "this" is the word with the highest frequency.

Longlonglonglongword should be cut off, so is considered as the same as longlonglonglonee.  But this_8 is different than this, and this, and this...#
this line should be ignored.

输出样例：（注意：虽然单词`the`也出现了4次，但因为我们只要输出前10%（即23个单词中的前2个）单词，而按照字母序，`the`排第3位，所以不输出。）

23
5:this
4:is

#include<iostream>
#include<vector>
#include<map>
#include<string>
#include<algorithm>

using namespace std;

map<string, int> mp;

struct Words {//定义了一个word结构体
	string str;
	int count;
};

bool cmp(const Words &w1, const Words &w2) {//排序用到的比较函数
	if (w1.count > w2.count) {
		return true;
	}
	if (w1.count == w2.count) {
		return w1.str < w2.str;
	}
	return false;
}

int main() {
	string word;
	char c;
	while (true) {
		scanf("%c",&c);//一个一个字母往里输入
		if (c >= 'A'&&c <= 'Z' || c >= 'a'&&c <= 'z' || c >= '0'&&c <= '9' || c == '_') {
			if (c >= 'A'&&c <= 'Z') {
				c = (c - 'A' + 'a');//将大写字母转化为小写字母，统计词频不分大小写
			}
			if (word.length() < 15) {
				word += c;//如果这个单词长度小于15，就把这个字母加到这个单词的末尾
			}
		}
		else if (c == '#' || word.length() > 0) {
			if (!mp[word]) {//如果map里面没有这个单词，就插入这个单词，并把key（单词出现的次数）设为1
				mp[word] = 1;
			}
			else {
				mp[word]++;//如果已经有了，就把频率+1
			}
			word.clear();//把word清空，方便下一个单词输入
			if (c == '#') {
				break;//如果输入等于#，则输入结束
			}
		}
	}
	map<string, int>::iterator iter;//迭代器
	vector<Words> vec;//定义一个Words类型的vector，方便排序
	Words w;
	for (iter = mp.begin(); iter != mp.end(); iter++) {//通过迭代器遍历map
		//cout << iter->first << " " << iter->second << endl;
		if (iter->first.length()>0)
		{
			w.str = iter->first;
			w.count = iter->second;
			vec.push_back(w);
		}
		
	}
	sort(vec.begin(), vec.end(), cmp);//对vector进行排序
	int count = vec.size()*0.1;
	cout << vec.size() << endl;
	for (int i = 0; i < count; i++) {
		cout << vec[i].count << ":" << vec[i].str << endl;
	}
	return 0;
}