当前位置:   article > 正文

c++ utf8转为gbk_c++标准库中unicode ,utf8 ,gbk字符转化的函数叫什么

c++ unicode、utf8、gbk编码之间转换

展开全部

没有现成的,e69da5e887aa3231313335323631343130323136353331333361313262需要自己实现函数,给你贴几段

GBK转UTF-8

#include

#include

//GBK编码转换到UTF8编码

int GBKToUTF8(unsigned char * lpGBKStr,unsigned char * lpUTF8Str,int nUTF8StrLen)

{

wchar_t * lpUnicodeStr = NULL;

int nRetLen = 0;

if(!lpGBKStr) //如果GBK字符串为NULL则出错退出

return 0;

nRetLen = ::MultiByteToWideChar(CP_ACP,0,(char *)lpGBKStr,-1,NULL,NULL); //获取转换到Unicode编码后所需要的字符空间长度

lpUnicodeStr = new WCHAR[nRetLen + 1]; //为Unicode字符串空间

nRetLen = ::MultiByteToWideChar(CP_ACP,0,(char *)lpGBKStr,-1,lpUnicodeStr,nRetLen); //转换到Unicode编码

if(!nRetLen) //转换失败则出错退出

return 0;

nRetLen = ::WideCharToMultiByte(CP_UTF8,0,lpUnicodeStr,-1,NULL,0,NULL,NULL); //获取转换到UTF8编码后所需要的字符空间长度

if(!lpUTF8Str) //输出缓冲区为空则返回转换后需要的空间大小

{

if(lpUnicodeStr)

delete []lpUnicodeStr;

return nRetLen;

}

if(nUTF8StrLen < nRetLen) //如果输出缓冲区长度不够则退出

{

if(lpUnicodeStr)

delete []lpUnicodeStr;

return 0;

}

nRetLen = ::WideCharToMultiByte(CP_UTF8,0,lpUnicodeStr,-1,(char *)lpUTF8Str,nUTF8StrLen,NULL,NULL); //转换到UTF8编码

if(lpUnicodeStr)

delete []lpUnicodeStr;

return nRetLen;

}

//使用这两个函数的例子

int main()

{

char cGBKStr[] = "我是中国人!";

char * lpGBKStr = NULL;

char * lpUTF8Str = NULL;

FILE * fp = NULL;

int nRetLen = 0;

nRetLen = GBKToUTF8((unsigned char *) cGBKStr,NULL,NULL);

printf("转换后的字符串需要的空间长度为:%d ",nRetLen);

lpUTF8Str = new char[nRetLen + 1];

nRetLen = GBKToUTF8((unsigned char *)cGBKStr,(unsigned char *)lpUTF8Str,nRetLen);

if(nRetLen)

{

printf("GBKToUTF8转换成功!");

}

else

{

printf("GBKToUTF8转换失败!");

goto Ret0;

}

fp = fopen("C:\\GBKtoUTF8.txt","wb"); //保存到文本文件

fwrite(lpUTF8Str,nRetLen,1,fp);

fclose(fp);

getchar(); //先去打开那个文本文件看看,单击记事本的“文件”-“另存为”菜单,在对话框中看到编码框变为了“UTF-8”说明转换成功了

Ret0:

{

if(lpGBKStr)

delete []lpGBKStr;

if(lpUTF8Str)

delete []lpUTF8Str;

}

return 0;

}

Unicode UTF_8互转

class CChineseCode

{

public:

static void UTF_8ToUnicode(wchar_t* pOut,char *pText); // 把UTF-8转换成Unicode

static void UnicodeToUTF_8(char* pOut,wchar_t* pText); //Unicode 转换成UTF-8

static void UnicodeToGB2312(char* pOut,wchar_t uData); // 把Unicode 转换成 GB2312

static void Gb2312ToUnicode(wchar_t* pOut,char *gbBuffer);// GB2312 转换成 Unicode

static void GB2312ToUTF_8(string& pOut,char *pText, int pLen);//GB2312 转为 UTF-8

static void UTF_8ToGB2312(string &pOut, char *pText, int pLen);//UTF-8 转为 GB2312

};

类实现

void CChineseCode::UTF_8ToUnicode(wchar_t* pOut,char *pText)

{

char* uchar = (char *)pOut;

uchar[1] = ((pText[0] & 0x0F) << 4) + ((pText[1] >> 2) & 0x0F);

uchar[0] = ((pText[1] & 0x03) << 6) + (pText[2] & 0x3F);

return;

}

void CChineseCode::UnicodeToUTF_8(char* pOut,wchar_t* pText)

{

// 注意 WCHAR高低字的顺序,低字节在前,高字节在后

char* pchar = (char *)pText;

pOut[0] = (0xE0 | ((pchar[1] & 0xF0) >> 4));

pOut[1] = (0x80 | ((pchar[1] & 0x0F) << 2)) + ((pchar[0] & 0xC0) >> 6);

pOut[2] = (0x80 | (pchar[0] & 0x3F));

return;

}

void CChineseCode::UnicodeToGB2312(char* pOut,wchar_t uData)

{

WideCharToMultiByte(CP_ACP,NULL,&uData,1,pOut,sizeof(wchar_t),NULL,NULL);

return;

}

void CChineseCode::Gb2312ToUnicode(wchar_t* pOut,char *gbBuffer)

{

::MultiByteToWideChar(CP_ACP,MB_PRECOMPOSED,gbBuffer,2,pOut,1);

return ;

}

void CChineseCode::GB2312ToUTF_8(string& pOut,char *pText, int pLen)

{

char buf[4];

int nLength = pLen* 3;

char* rst = new char[nLength];

memset(buf,0,4);

memset(rst,0,nLength);

int i = 0;

int j = 0;

while(i < pLen)

{

//如果是英文直接复制就可以

if( *(pText + i) >= 0)

{

rst[j++] = pText[i++];

}

else

{

wchar_t pbuffer;

Gb2312ToUnicode(&pbuffer,pText+i);

UnicodeToUTF_8(buf,&pbuffer);

unsigned short int tmp = 0;

tmp = rst[j] = buf[0];

tmp = rst[j+1] = buf[1];

tmp = rst[j+2] = buf[2];

j += 3;

i += 2;

}

}

rst[j] = '';

//返回结果

pOut = rst;

delete []rst;

return;

}

void CChineseCode::UTF_8ToGB2312(string &pOut, char *pText, int pLen)

{

char * newBuf = new char[pLen];

char Ctemp[4];

memset(Ctemp,0,4);

int i =0;

int j = 0;

while(i < pLen)

{

if(pText > 0)

{

newBuf[j++] = pText[i++];

}

else

{

WCHAR Wtemp;

UTF_8ToUnicode(&Wtemp,pText + i);

UnicodeToGB2312(Ctemp,Wtemp);

newBuf[j] = Ctemp[0];

newBuf[j + 1] = Ctemp[1];

i += 3;

j += 2;

}

}

newBuf[j] = '';

pOut = newBuf;

delete []newBuf;

return;

}

将GBK转换成UTF8

string GBKToUTF8(const std::string& strGBK)

{ string strOutUTF8 = "";

WCHAR * str1;

int n = MultiByteToWideChar(CP_ACP, 0, strGBK.c_str(), -1, NULL, 0);

str1 = new WCHAR[n];

MultiByteToWideChar(CP_ACP, 0, strGBK.c_str(), -1, str1, n); n = WideCharToMultiByte(CP_UTF8, 0, str1, -1, NULL, 0, NULL, NULL);

char * str2 = new char[n];

WideCharToMultiByte(CP_UTF8, 0, str1, -1, str2, n, NULL, NULL);

strOutUTF8 = str2;

delete[]str1;

str1 = NULL;

delete[]str2;

str2 = NULL;

return strOutUTF8;

}

将UTF8转换成GBK

string UTF8ToGBK(const std::string& strUTF8)

{

int len = MultiByteToWideChar(CP_UTF8, 0, strUTF8.c_str(), -1, NULL, 0);

unsigned short * wszGBK = new unsigned short[len + 1]; memset(wszGBK, 0, len * 2 + 2);

MultiByteToWideChar(CP_UTF8, 0, (LPCTSTR)strUTF8.c_str(), -1, wszGBK, len);

len = WideCharToMultiByte(CP_ACP, 0, wszGBK, -1, NULL, 0, NULL, NULL);

char *szGBK = new char[len + 1];

memset(szGBK, 0, len + 1);

WideCharToMultiByte(CP_ACP,0, wszGBK, -1, szGBK, len, NULL, NULL); //strUTF8 = szGBK;

std::string strTemp(szGBK);

delete[]szGBK;

delete[]wszGBK;

return strTemp;

}

已赞过

已踩过<

你对这个回答的评价是?

评论

收起

声明:本文内容由网友自发贡献,不代表【wpsshop博客】立场,版权归原作者所有,本站不承担相应法律责任。如您发现有侵权的内容,请联系我们。转载请注明出处:https://www.wpsshop.cn/w/我家小花儿/article/detail/123154
推荐阅读
相关标签
  

闽ICP备14008679号