关于字符集的专题知识 UTF-8 GB2312 UNICODE

最新推荐文章于 2022-05-12 17:14:55 发布

Timmy_zhou

最新推荐文章于 2022-05-12 17:14:55 发布

阅读量661

点赞数

文章标签： string delete null class

此文介绍了UTF8和GB2312间的互换并提供代码，但是代码有误，现修改如下：

class CChineseCodeLib
{
public:
static void UTF_8ToGB2312(string& pOut,char *pText, int pLen);
static void GB2312ToUTF_8(string& pOut,char *pText, int pLen);
// Unicode 转换成UTF-8
static void UnicodeToUTF_8(char* pOut,WCHAR* pText);
// GB2312 转换成　Unicode
static void Gb2312ToUnicode(WCHAR* pOut,char *gbBuffer);
// 把Unicode 转换成 GB2312
static void UnicodeToGB2312(char* pOut,WCHAR uData);
// 把UTF-8转换成Unicode
static void UTF_8ToUnicode(WCHAR* pOut,char* pText);

CChineseCodeLib();
virtual ~CChineseCodeLib();
};

void CChineseCodeLib::UTF_8ToUnicode(WCHAR* pOut,char *pText)
{
char* uchar = (char *)pOut;

uchar[1] = ((pText[0] & 0x0F) << 4) + ((pText[1] >> 2) & 0x0F);
uchar[0] = ((pText[1] & 0x03) << 6) + (pText[2] & 0x3F);

return;
}

void CChineseCodeLib::UnicodeToGB2312(char* pOut,WCHAR uData)
{
WideCharToMultiByte(CP_ACP,NULL,&uData,1,pOut,sizeof(WCHAR),NULL,NULL);
return;
}

void CChineseCodeLib::Gb2312ToUnicode(WCHAR* pOut,char *gbBuffer)
{
::MultiByteToWideChar(CP_ACP,MB_PRECOMPOSED,gbBuffer,2,pOut,1);
return;
}

void CChineseCodeLib::UnicodeToUTF_8(char* pOut,WCHAR* pText)
{
// 注意 WCHAR高低字的顺序,低字节在前，高字节在后
char* pchar = (char *)pText;

pOut[0] = (0xE0 | ((pchar[1] & 0xF0) >> 4));
pOut[1] = (0x80 | ((pchar[1] & 0x0F) << 2)) + ((pchar[0] & 0xC0) >> 6); // 此处也有错误，该文作者已提出！
pOut[2] = (0x80 | (pchar[0] & 0x3F));

return;
}

void CChineseCodeLib::GB2312ToUTF_8(string& pOut,char *pText, int pLen)
{
char buf[4];
char* rst = new char[(pLen/2)*3+1];    // 原代码此处分配的空间不够，导致越界！

memset(buf,0,4);
memset(rst,0,sizeof(rst));

int i = 0;
int j = 0;
while(i < pLen)
{
//如果是英文直接复制就可以
if( *(pText + i) >= 0)
{
   rst[j++] = pText[i++];
}
else
{
   WCHAR pbuffer;
   Gb2312ToUnicode(&pbuffer,pText+i);

   UnicodeToUTF_8(buf,&pbuffer);

   unsigned short int tmp = 0;
   tmp = rst[j] = buf[0];
   tmp = rst[j+1] = buf[1];
   tmp = rst[j+2] = buf[2];


   j += 3;
   i += 2;
}
}
rst[j] = '\0';

//返回结果
pOut = rst;
delete []rst;

return;
}

void CChineseCodeLib::UTF_8ToGB2312(string &pOut, char *pText, int pLen)
{
char * newBuf = new char[pLen+1]; // 原代码此处分配空间不够，当pText全部为英文时会出错。pLen为pText字 // 符串的长度，不包含终止符
char Ctemp[4];
memset(Ctemp,0,4);

int i =0;
int j = 0;

while(i < pLen)
{
   if(pText[i] > 0)
{
   newBuf[j++] = pText[i++];
}
else
{
   WCHAR Wtemp;
   UTF_8ToUnicode(&Wtemp,pText + i);

   UnicodeToGB2312(Ctemp,Wtemp);

   newBuf[j] = Ctemp[0];
   newBuf[j + 1] = Ctemp[1];

i += 3;
j += 2;
}
}

newBuf[j] = '\0';

pOut = newBuf;
delete []newBuf;

return;
}

Karlson,2009-07-25 13:42:35

具体实践
-         data << gItem.m_gMessage;                           // text for gossip item
+            //Karlson 解决VS2005中文问题 >>>
+         std::string str = gItem.m_gMessage;
+        if(str[0] == ' '){ // 根据首字符是否为空格来判断是否需要转换
+            char *chr = (char *)str.c_str();
+            CChineseCode::GB2312ToUTF_8(str, chr, strlen(chr));
+        }
+        data << str;
+         // <<< Karlson
+        //data << gItem.m_gMessage; 关于字符集的专题知识 UTF-8 GB2312 UNICODE Karlson,2009-07-25 13:39:39

Timmy_zhou

关注

0
点赞
踩
0

收藏

觉得还不错? 一键收藏
0
评论
关于字符集的专题知识 UTF-8 GB2312 UNICODE

此文介绍了UTF8和GB2312间的互换并提供代码，但是代码有误，现修改如下：class CChineseCodeLib {public:static void UTF_8ToGB2312(string& pOut,char *pText, int pLen);static void GB2312ToUTF_8(string& pOut,char *pText, int pL
复制链接

扫一扫