C_ConvertString.h 4.6 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195
  1. #pragma once
  2. #include <string>
  3. struct ConvertString
  4. {
  5. static void ConvertOtherToUnicode(const char *strUtf8 , std::wstring &strWRet , UINT uiOtherCode)
  6. {
  7. #if _MSC_VER > 1500
  8. strWRet.clear();
  9. #endif
  10. if( strUtf8 == NULL || strlen(strUtf8) <= 0 ) return ;
  11. int len=MultiByteToWideChar(uiOtherCode, 0, strUtf8, -1, NULL,0);
  12. unsigned short* wszGBK = new unsigned short[len];
  13. memset(wszGBK, 0, len*sizeof(unsigned short));
  14. MultiByteToWideChar(uiOtherCode, 0, strUtf8, -1, (LPWSTR)wszGBK, len);
  15. strWRet = (WCHAR*)wszGBK;
  16. delete[] wszGBK;
  17. }
  18. static void ConvertUnicodeToOther(const wchar_t *strUnicode, std::string &strUtf8 , UINT uiOtherCode)
  19. {
  20. #if _MSC_VER > 1500
  21. strUtf8.clear();
  22. #endif
  23. if( strUnicode == NULL || wcslen(strUnicode) <= 0 ) return;
  24. int len = WideCharToMultiByte(uiOtherCode, 0, strUnicode, -1, NULL, 0, NULL, NULL);
  25. char *szUtf8=new char[len];
  26. memset(szUtf8, 0, len*sizeof(char));
  27. WideCharToMultiByte (uiOtherCode, 0, strUnicode, -1, szUtf8, len, NULL,NULL);
  28. strUtf8 = szUtf8;
  29. delete[] szUtf8;
  30. }
  31. static void ConvertUtf8ToUnicode(const char *strUtf8 , std::wstring &strWRet)
  32. {
  33. ConvertOtherToUnicode(strUtf8 , strWRet , CP_UTF8);
  34. }
  35. static void ConvertUnicodeToUtf8(const wchar_t *strUnicode , std::string &strUtf8)
  36. {
  37. ConvertUnicodeToOther(strUnicode , strUtf8 , CP_UTF8);
  38. }
  39. static void ConvertUnicodeToGBK(const wchar_t *strUnicode , std::string &strGBK)
  40. {
  41. ConvertUnicodeToOther(strUnicode,strGBK,CP_ACP);
  42. }
  43. static void ConvertGBKToUnicode(const char *strUtf8 , std::wstring &strWRet)
  44. {
  45. ConvertOtherToUnicode(strUtf8,strWRet,CP_ACP);
  46. }
  47. static void ConvertGBKToUtf8(const char *strGBK, std::string &strUtf8)
  48. {
  49. std::wstring wstrUnicode;
  50. ConvertString::ConvertGBKToUnicode(strGBK, wstrUnicode);
  51. ConvertString::ConvertUnicodeToUtf8(wstrUnicode.c_str(), strUtf8);
  52. }
  53. static void ConvertUtf8ToGBK(const char *strUtf8, std::string &strGBK)
  54. {
  55. std::wstring wstrUnicode;
  56. ConvertString::ConvertUtf8ToUnicode(strUtf8, wstrUnicode);
  57. ConvertString::ConvertUnicodeToGBK(wstrUnicode.c_str(), strGBK);
  58. }
  59. #if _MSC_VER > 1500
  60. static void stringFormat(std::string &StrDis , const char* pString , ...)
  61. {
  62. if(!pString) return;
  63. va_list argList;
  64. va_start( argList, pString );
  65. int Length = _vscprintf( pString, argList )+1;
  66. char *pBuffer = new char[Length];
  67. vsprintf_s( pBuffer, Length, pString, argList );
  68. StrDis = std::string(pBuffer);
  69. delete []pBuffer;
  70. va_end( argList );
  71. }
  72. static void stringFormat(std::wstring &StrDis , const wchar_t* pString , ...)
  73. {
  74. if(pString == NULL) return;
  75. va_list argList;
  76. va_start( argList, pString );
  77. int Length = _vscwprintf( pString, argList )+1;
  78. wchar_t *pBuffer = new wchar_t[Length];
  79. vswprintf_s( pBuffer, Length, pString, argList );
  80. StrDis = std::wstring(pBuffer);
  81. delete []pBuffer;
  82. va_end( argList );
  83. }
  84. #endif
  85. static bool IsTextUTF8(const char* str)
  86. {
  87. unsigned long nBytes = 0;//UFT8可用1-6个字节编码,ASCII用一个字节
  88. unsigned char chr;
  89. bool bAllAscii = true; //如果全部都是ASCII, 说明不是UTF-8
  90. size_t length = strlen(str);
  91. for (size_t i = 0; i < length; ++i)
  92. {
  93. chr = *(str + i);
  94. if ((chr & 0x80) != 0) // 判断是否ASCII编码,如果不是,说明有可能是UTF-8,ASCII用7位编码,但用一个字节存,最高位标记为0,o0xxxxxxx
  95. bAllAscii = false;
  96. if (nBytes == 0) //如果不是ASCII码,应该是多字节符,计算字节数
  97. {
  98. if (chr >= 0x80)
  99. {
  100. if (chr >= 0xFC && chr <= 0xFD)
  101. nBytes = 6;
  102. else if (chr >= 0xF8)
  103. nBytes = 5;
  104. else if (chr >= 0xF0)
  105. nBytes = 4;
  106. else if (chr >= 0xE0)
  107. nBytes = 3;
  108. else if (chr >= 0xC0)
  109. nBytes = 2;
  110. else
  111. return false;
  112. nBytes--;
  113. }
  114. }
  115. else //多字节符的非首字节,应为 10xxxxxx
  116. {
  117. if ((chr & 0xC0) != 0x80)
  118. return false;
  119. nBytes--;
  120. }
  121. }
  122. if (nBytes > 0) //违返规则
  123. return false;
  124. if (bAllAscii) //如果全部都是ASCII, 说明不是UTF-8
  125. return false;
  126. return true;
  127. }
  128. };
  129. class C_UTF8_UNICODE
  130. {
  131. public:
  132. C_UTF8_UNICODE(const wchar_t *strUnicode)
  133. {
  134. m_strUnicode = strUnicode;
  135. ConvertString::ConvertUnicodeToUtf8(strUnicode,m_strUtf8);
  136. }
  137. C_UTF8_UNICODE(const char *strUtf8)
  138. {
  139. m_strUtf8 = strUtf8;
  140. ConvertString::ConvertUtf8ToUnicode(strUtf8,m_strUnicode);
  141. }
  142. void GetAnsi(std::string &m_Ansi)
  143. {
  144. ConvertString::ConvertUnicodeToGBK(m_strUnicode.c_str() , m_Ansi);
  145. }
  146. operator const std::string*() const
  147. {
  148. return &m_strUtf8;
  149. }
  150. operator const std::wstring*() const
  151. {
  152. return &m_strUnicode;
  153. }
  154. operator const wchar_t *() const
  155. {
  156. return m_strUnicode.c_str();
  157. }
  158. operator const char *() const
  159. {
  160. return m_strUtf8.c_str();
  161. }
  162. private:
  163. std::string m_strUtf8;
  164. std::wstring m_strUnicode;
  165. };