UTF8 UTF16 之间的互相转换

http://www.oschina.net/code/snippet_179574_15065
按照如下的编码方式,对UTF8和UTF16之间进行转换 

从UCS-2到UTF-8的编码方式如下:

UCS-2编码(16进制) UTF-8 字节流(二进制)
0000 - 007F 0xxxxxxx
0080 - 07FF 110xxxxx 10xxxxxx
0800 - FFFF 1110xxxx 10xxxxxx 10xxxxxx
 
typedef unsigned long   UTF32;  /* at least 32 bits */
typedef unsigned short  UTF16;  /* at least 16 bits */
typedef unsigned char   UTF8;   /* typically 8 bits */
typedef unsigned int    INT ;
 
/*
UCS-2编码    UTF-8 字节流(二进制)
0000 - 007F  0xxxxxxx
0080 - 07FF 110xxxxx 10xxxxxx
0800 - FFFF 1110xxxx 10xxxxxx 10xxxxxx
*/
 
#define UTF8_ONE_START      (0xOOO1)
#define UTF8_ONE_END        (0x007F)
#define UTF8_TWO_START      (0x0080)
#define UTF8_TWO_END        (0x07FF)
#define UTF8_THREE_START    (0x0800)
#define UTF8_THREE_END      (0xFFFF)
 
 
void UTF16ToUTF8(UTF16* pUTF16Start, UTF16* pUTF16End, UTF8* pUTF8Start, UTF8* pUTF8End)
{
     UTF16* pTempUTF16 = pUTF16Start;
     UTF8* pTempUTF8 = pUTF8Start;
 
     while (pTempUTF16 < pUTF16End)
     {
         if (*pTempUTF16 <= UTF8_ONE_END
             && pTempUTF8 + 1 < pUTF8End)
         {
             //0000 - 007F  0xxxxxxx
             *pTempUTF8++ = (UTF8)*pTempUTF16;
         }
         else if (*pTempUTF16 >= UTF8_TWO_START && *pTempUTF16 <= UTF8_TWO_END
             && pTempUTF8 + 2 < pUTF8End)
         {
             //0080 - 07FF 110xxxxx 10xxxxxx
             *pTempUTF8++ = (*pTempUTF16 >> 6) | 0xC0;
             *pTempUTF8++ = (*pTempUTF16 & 0x3F) | 0x80;
         }
         else if (*pTempUTF16 >= UTF8_THREE_START && *pTempUTF16 <= UTF8_THREE_END
             && pTempUTF8 + 3 < pUTF8End)
         {
             //0800 - FFFF 1110xxxx 10xxxxxx 10xxxxxx
             *pTempUTF8++ = (*pTempUTF16 >> 12) | 0xE0;
             *pTempUTF8++ = ((*pTempUTF16 >> 6) & 0x3F) | 0x80;
             *pTempUTF8++ = (*pTempUTF16 & 0x3F) | 0x80;
         }
         else
         {
             break ;
         }
         pTempUTF16++;
     }
     *pTempUTF8 = 0;
}
 
void UTF8ToUTF16(UTF8* pUTF8Start, UTF8* pUTF8End, UTF16* pUTF16Start, UTF16* pUTF16End)
{
     UTF16* pTempUTF16 = pUTF16Start;
     UTF8* pTempUTF8 = pUTF8Start;
 
     while (pTempUTF8 < pUTF8End && pTempUTF16+1 < pUTF16End)
     {
         if (*pTempUTF8 >= 0xE0 && *pTempUTF8 <= 0xEF) //是3个字节的格式
         {
             //0800 - FFFF 1110xxxx 10xxxxxx 10xxxxxx
             *pTempUTF16 |= ((*pTempUTF8++ & 0xEF) << 12);
             *pTempUTF16 |= ((*pTempUTF8++ & 0x3F) << 6);
             *pTempUTF16 |= (*pTempUTF8++ & 0x3F);
 
         }
         else if (*pTempUTF8 >= 0xC0 && *pTempUTF8 <= 0xDF) //是2个字节的格式
         {
             //0080 - 07FF 110xxxxx 10xxxxxx
             *pTempUTF16 |= ((*pTempUTF8++ & 0x1F) << 6);
             *pTempUTF16 |= (*pTempUTF8++ & 0x3F);
         }
         else if (*pTempUTF8 >= 0 && *pTempUTF8 <= 0x7F) //是1个字节的格式
         {
             //0000 - 007F  0xxxxxxx
             *pTempUTF16 = *pTempUTF8++;
         }
         else
         {
             break ;
         }
         pTempUTF16++;
     }
     *pTempUTF16 = 0;
}
 
 
int main()
{
     UTF16 utf16[256] = {L "你a好b吗234中国~!" };
     UTF8 utf8[256];
     
     UTF16ToUTF8(utf16, utf16+wcslen(utf16), utf8, utf8+256);
 
     memset (utf16, 0, sizeof (utf16));
 
     UTF8ToUTF16(utf8, utf8 + strlen (utf8), utf16, utf16+256);
 
 
     return 0;
}

猜你喜欢

转载自www.cnblogs.com/develon/p/9164566.html
今日推荐