| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195 |
- #pragma once
- #include <string>
- struct ConvertString
- {
- static void ConvertOtherToUnicode(const char *strUtf8 , std::wstring &strWRet , UINT uiOtherCode)
- {
- #if _MSC_VER > 1500
- strWRet.clear();
- #endif
-
- if( strUtf8 == NULL || strlen(strUtf8) <= 0 ) return ;
- int len=MultiByteToWideChar(uiOtherCode, 0, strUtf8, -1, NULL,0);
- unsigned short* wszGBK = new unsigned short[len];
- memset(wszGBK, 0, len*sizeof(unsigned short));
- MultiByteToWideChar(uiOtherCode, 0, strUtf8, -1, (LPWSTR)wszGBK, len);
- strWRet = (WCHAR*)wszGBK;
- delete[] wszGBK;
- }
- static void ConvertUnicodeToOther(const wchar_t *strUnicode, std::string &strUtf8 , UINT uiOtherCode)
- {
- #if _MSC_VER > 1500
- strUtf8.clear();
- #endif
- if( strUnicode == NULL || wcslen(strUnicode) <= 0 ) return;
- int len = WideCharToMultiByte(uiOtherCode, 0, strUnicode, -1, NULL, 0, NULL, NULL);
- char *szUtf8=new char[len];
- memset(szUtf8, 0, len*sizeof(char));
- WideCharToMultiByte (uiOtherCode, 0, strUnicode, -1, szUtf8, len, NULL,NULL);
- strUtf8 = szUtf8;
- delete[] szUtf8;
- }
- static void ConvertUtf8ToUnicode(const char *strUtf8 , std::wstring &strWRet)
- {
- ConvertOtherToUnicode(strUtf8 , strWRet , CP_UTF8);
- }
- static void ConvertUnicodeToUtf8(const wchar_t *strUnicode , std::string &strUtf8)
- {
- ConvertUnicodeToOther(strUnicode , strUtf8 , CP_UTF8);
- }
- static void ConvertUnicodeToGBK(const wchar_t *strUnicode , std::string &strGBK)
- {
- ConvertUnicodeToOther(strUnicode,strGBK,CP_ACP);
- }
- static void ConvertGBKToUnicode(const char *strUtf8 , std::wstring &strWRet)
- {
- ConvertOtherToUnicode(strUtf8,strWRet,CP_ACP);
- }
-
- static void ConvertGBKToUtf8(const char *strGBK, std::string &strUtf8)
- {
- std::wstring wstrUnicode;
- ConvertString::ConvertGBKToUnicode(strGBK, wstrUnicode);
- ConvertString::ConvertUnicodeToUtf8(wstrUnicode.c_str(), strUtf8);
- }
- static void ConvertUtf8ToGBK(const char *strUtf8, std::string &strGBK)
- {
- std::wstring wstrUnicode;
- ConvertString::ConvertUtf8ToUnicode(strUtf8, wstrUnicode);
- ConvertString::ConvertUnicodeToGBK(wstrUnicode.c_str(), strGBK);
- }
- #if _MSC_VER > 1500
- static void stringFormat(std::string &StrDis , const char* pString , ...)
- {
- if(!pString) return;
- va_list argList;
- va_start( argList, pString );
- int Length = _vscprintf( pString, argList )+1;
- char *pBuffer = new char[Length];
- vsprintf_s( pBuffer, Length, pString, argList );
- StrDis = std::string(pBuffer);
- delete []pBuffer;
- va_end( argList );
- }
- static void stringFormat(std::wstring &StrDis , const wchar_t* pString , ...)
- {
- if(pString == NULL) return;
- va_list argList;
- va_start( argList, pString );
- int Length = _vscwprintf( pString, argList )+1;
- wchar_t *pBuffer = new wchar_t[Length];
- vswprintf_s( pBuffer, Length, pString, argList );
- StrDis = std::wstring(pBuffer);
- delete []pBuffer;
- va_end( argList );
- }
- #endif
- static bool IsTextUTF8(const char* str)
- {
- unsigned long nBytes = 0;//UFT8可用1-6个字节编码,ASCII用一个字节
- unsigned char chr;
- bool bAllAscii = true; //如果全部都是ASCII, 说明不是UTF-8
- size_t length = strlen(str);
- for (size_t i = 0; i < length; ++i)
- {
- chr = *(str + i);
- if ((chr & 0x80) != 0) // 判断是否ASCII编码,如果不是,说明有可能是UTF-8,ASCII用7位编码,但用一个字节存,最高位标记为0,o0xxxxxxx
- bAllAscii = false;
- if (nBytes == 0) //如果不是ASCII码,应该是多字节符,计算字节数
- {
- if (chr >= 0x80)
- {
- if (chr >= 0xFC && chr <= 0xFD)
- nBytes = 6;
- else if (chr >= 0xF8)
- nBytes = 5;
- else if (chr >= 0xF0)
- nBytes = 4;
- else if (chr >= 0xE0)
- nBytes = 3;
- else if (chr >= 0xC0)
- nBytes = 2;
- else
- return false;
- nBytes--;
- }
- }
- else //多字节符的非首字节,应为 10xxxxxx
- {
- if ((chr & 0xC0) != 0x80)
- return false;
- nBytes--;
- }
- }
- if (nBytes > 0) //违返规则
- return false;
- if (bAllAscii) //如果全部都是ASCII, 说明不是UTF-8
- return false;
- return true;
- }
- };
- class C_UTF8_UNICODE
- {
- public:
- C_UTF8_UNICODE(const wchar_t *strUnicode)
- {
- m_strUnicode = strUnicode;
- ConvertString::ConvertUnicodeToUtf8(strUnicode,m_strUtf8);
- }
- C_UTF8_UNICODE(const char *strUtf8)
- {
- m_strUtf8 = strUtf8;
- ConvertString::ConvertUtf8ToUnicode(strUtf8,m_strUnicode);
- }
- void GetAnsi(std::string &m_Ansi)
- {
- ConvertString::ConvertUnicodeToGBK(m_strUnicode.c_str() , m_Ansi);
- }
- operator const std::string*() const
- {
- return &m_strUtf8;
- }
- operator const std::wstring*() const
- {
- return &m_strUnicode;
- }
- operator const wchar_t *() const
- {
- return m_strUnicode.c_str();
- }
- operator const char *() const
- {
- return m_strUtf8.c_str();
- }
- private:
- std::string m_strUtf8;
- std::wstring m_strUnicode;
- };
|