1.讀取UTF-8編碼文本原理

首先了解UTF-8的編碼方式,UTF-8采用可變長編碼的方式，一個字符可占1字節-6字節，其中每個字符所占的字節數由字符開始的1的個數確定，具體的編碼方式如下：

U-00000000 - U-0000007F: 0xxxxxxx
U-00000080 - U-000007FF: 110xxxxx 10xxxxxx
U-00000800 - U-0000FFFF: 1110xxxx 10xxxxxx 10xxxxxx
U-00010000 - U-001FFFFF: 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx
U-00200000 - U-03FFFFFF: 111110xx 10xxxxxx 10xxxxxx 10xxxxxx 10xxxxxx
U-04000000 - U-7FFFFFFF: 1111110x 10xxxxxx 10xxxxxx 10xxxxxx 10xxxxxx 10xxxxxx

因此，對於每個字節如果起始位為“0”則說明，該字符占有1字節。

如果起始位為“10”則說明該字節不是字符的起始字節。

如果起始為為$n$個“1”+1個“0”，則說明改字符占有$n$個字節。其中$1 \leq n \leq 6$。

因此對於UTF-8的編碼，我們只需要每次計算每個字符開始字節的1的個數，就可以確定這個字符的長度。

2.讀取GBK系列文本原理

對於ASCII、GB2312、GBK到GB18030編碼方法是向下兼容的，即同一個字符在這些方案中總是有相同的編碼，后面的標准支持更多的字符。

在這些編碼中，英文和中文可以統一地處理。區分中文編碼的方法是高字節的最高位不為0。

因此我們只需處理好GB18130，就可以處理與他兼容的所有編碼，對於GB18130使用雙字節變長編碼。

單字節部分從 0x0~0x7F 與 ASCII 編碼兼容。雙字節部分，首字節從 0x81~0xFE，尾字節從 0x40~0x7E以及 0x80~0xFE，與GBK標准基本兼容。

因此只需檢測首字節是否小於0x81即可確定其為單字節編碼還是雙字節編碼。

3.C++代碼實現

對於一個語言處理系統，讀取不同編碼的文本應該是最基礎的需求，文本的編碼方式應該對系統其他調用者透明，只需每次獲取一個字符即可，而不需要關注這個文本的編碼方式。從而我們定義了抽象類Text，及其接口ReadOneChar，並使兩個文本類GbkText和UtfText繼承這個抽象類，當系統需要讀取更多種編碼的文件時，只需要定義新的類然后繼承該抽象類即可，並不需要更改調用該類的代碼。從而獲得更好的擴展性。

更好的方式是使用簡單工廠模式，使不同的文本編碼格式對於調用類完全透明，簡單工廠模式詳解請參看：C++實現設計模式之 — 簡單工廠模式

其中Text抽象類的定義如下：

 1 #ifndef TEXT_H
 2 #define TEXT_H
 3 #include <iostream>
 4 #include <fstream>
 5 using namespace std;
 6 class Text
 7 {
 8     protected:
 9         char * m_binaryStr;
10         size_t m_length;
11         size_t m_index;
12     public:
13         Text(string path);
14         void SetIndex(size_t index);
15         virtual bool ReadOneChar(string &oneChar) = 0;
16         size_t Size();
17         virtual ~Text();
18 };
19 #endif

View Code

Text抽象類的實現如下：

 1 #include "Text.h"
 2 using namespace std;
 3 Text::Text(string path):m_index(0)
 4 {
 5     filebuf *pbuf;
 6     ifstream filestr;
 7     // 采用二進制打開 
 8     filestr.open(path.c_str(), ios::binary);
 9     if(!filestr)
10     {
11         cerr<<path<<" Load text error."<<endl;
12         return;
13     }
14     // 獲取filestr對應buffer對象的指針 
15     pbuf=filestr.rdbuf();
16     // 調用buffer對象方法獲取文件大小
17     m_length=(int)pbuf->pubseekoff(0,ios::end,ios::in);
18     pbuf->pubseekpos(0,ios::in);
19     // 分配內存空間
20     m_binaryStr = new char[m_length+1];
21     // 獲取文件內容
22     pbuf->sgetn(m_binaryStr,m_length);
23     //關閉文件
24     filestr.close();
25 }
26 
27 void Text::SetIndex(size_t index)
28 {
29     m_index = index;
30 }
31 
32 size_t Text::Size()
33 {
34     return m_length;
35 }
36 
37 Text::~Text()
38 {
39     delete [] m_binaryStr;
40 }

View Code

GBKText類的定義如下：

#ifndef GBKTEXT_H
#define GBKTEXT_H
#include <iostream>
#include <string>
#include "Text.h"
using namespace std;
class GbkText:public Text
{
public:
    GbkText(string path);
    ~GbkText(void);
    bool ReadOneChar(string & oneChar);
};
#endif

View Code

GBKText類的實現如下：

 1 #include "GbkText.h"
 2 GbkText::GbkText(string path):Text(path){}
 3 GbkText::~GbkText(void) {}
 4 bool GbkText::ReadOneChar(string & oneChar)
 5 {
 6     // return true 表示讀取成功，
 7     // return false 表示已經讀取到流末尾
 8     if(m_length == m_index)
 9         return false;
10         if((unsigned char)m_binaryStr[m_index] < 0x81)
11     {
12         oneChar = m_binaryStr[m_index];
13         m_index++;
14     }
15     else
16     {
17         oneChar = string(m_binaryStr, 2);
18         m_index += 2;
19     }
20     return true;
21 }

View Code

UtfText類的定義如下：

 1 #ifndef UTFTEXT_H
 2 #define UTFTEXT_H
 3 #include <iostream>
 4 #include <string>
 5 #include "Text.h"
 6 using namespace std;
 7 class UtfText:public Text
 8 {
 9 public:
10     UtfText(string path);
11     ~UtfText(void);
12     bool ReadOneChar(string & oneChar);
13 private:
14     size_t get_utf8_char_len(const char & byte);
15 };
16 #endif

View Code

UtfText類的實現如下：

 1 #include "UtfText.h"
 2 UtfText::UtfText(string path):Text(path){}
 3 UtfText::~UtfText(void) {}
 4 bool UtfText::ReadOneChar(string & oneChar)
 5 {
 6     // return true 表示讀取成功，
 7     // return false 表示已經讀取到流末尾
 8     if(m_length == m_index)
 9         return false;
10     size_t utf8_char_len = get_utf8_char_len(m_binaryStr[m_index]);
11     if( 0 == utf8_char_len )
12     {
13             oneChar = "";
14             m_index++;
15         return true;
16     }
17     size_t next_idx = m_index + utf8_char_len;
18     if( m_length < next_idx )
19     {
20         //cerr << "Get utf8 first byte out of input src string." << endl;
21         next_idx = m_length;
22     }
23     //輸出UTF-8的一個字符
24     oneChar = string(m_binaryStr + m_index, next_idx - m_index);
25     //重置偏移量
26     m_index = next_idx;
27     return true;
28 }
29 
30 
31 size_t UtfText::get_utf8_char_len(const char & byte)
32 {
33     // return 0 表示錯誤
34     // return 1-6 表示正確值
35     // 不會 return 其他值 
36 
37     //UTF8 編碼格式：
38     //     U-00000000 - U-0000007F: 0xxxxxxx  
39     //     U-00000080 - U-000007FF: 110xxxxx 10xxxxxx  
40     //     U-00000800 - U-0000FFFF: 1110xxxx 10xxxxxx 10xxxxxx  
41     //     U-00010000 - U-001FFFFF: 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx  
42     //     U-00200000 - U-03FFFFFF: 111110xx 10xxxxxx 10xxxxxx 10xxxxxx 10xxxxxx  
43     //     U-04000000 - U-7FFFFFFF: 1111110x 10xxxxxx 10xxxxxx 10xxxxxx 10xxxxxx 10xxxxxx  
44 
45     size_t len = 0;
46     unsigned char mask = 0x80;
47     while( byte & mask )
48     {
49         len++;
50         if( len > 6 )
51         {
52             //cerr << "The mask get len is over 6." << endl;
53             return 0;
54         }
55         mask >>= 1;
56     }
57     if( 0 == len)
58     {
59         return 1;
60     }
61     return len;
62 }

View Code

工廠類TextFactory的類定義如下：

 1 #ifndef TEXTFACTORY_H
 2 #define TEXTFACTORY_H
 3 #include <iostream>
 4 #include "Text.h"
 5 #include "UtfText.h"
 6 #include "GbkText.h"
 7 using namespace std;
 8 class TextFactory
 9 {
10     public:
11         static Text * CreateText(string textCode, string path);
12 };
13 #endif

View Code

工廠類的實現如下：

 1 #include "TextFactory.h"
 2 #include "Text.h"
 3 Text * TextFactory::CreateText(string textCode, string path)
 4 {
 5     if( (textCode == "utf-8") 
 6                 || (textCode == "UTF-8") 
 7                 || (textCode == "ISO-8859-2")
 8                 || (textCode == "ascii") 
 9                 || (textCode == "ASCII")
10                 || (textCode == "TIS-620")
11                 || (textCode == "ISO-8859-5") 
12                 || (textCode == "ISO-8859-7") ) 
13     {
14         return new UtfText(path);
15     }
16     else if((textCode == "windows-1252") 
17                 || (textCode == "Big5")
18                 || (textCode == "EUC-KR") 
19                 || (textCode == "GB2312") 
20                 || (textCode == "ISO-2022-CN") 
21                 || (textCode == "HZ-GB-2312") 
22                 || (textCode == "gb18030"))
23     {
24         return new GbkText(path);
25     }
26     return NULL;
27 }

View Code

測試的Main函數如下：

 1 #include <stdio.h>
 2 #include <string.h>
 3 #include <iostream>
 4 #include "Text.h"
 5 #include "TextFactory.h"
 6 #include "CodeDetector.h"
 7 using namespace std;
 8 int main(int argc, char *argv[])
 9 {
10     string path ="日文"; 
11     string code ="utf-8";
12     Text * t = TextFactory::CreateText(code, path);
13     string s;
14     while(t->ReadOneChar(s))
15     {
16         cout<<s;
17     }
18     delete t;
19 }

View Code

編譯運行后即可在控制台輸出正確的文本。

免責聲明！

本站轉載的文章為個人學習借鑒使用，本站對版權不負任何法律責任。如果侵犯了您的隱私權益，請聯系本站郵箱yoyou2525@163.com刪除。

猜您在找 C++讀取mysql中utf8mb4編碼表數據亂碼問題及UTF8轉GBK編碼 C/C++ GBK和UTF8之間的轉換 C++ 字符串UTF8與GBK轉化 C++實現utf8和gbk編碼字符串互相轉換 Gbk互相轉換UTF8 GBK 和 UTF8編碼 Unicode,GBK和UTF8 MyEclipse默認編碼為GBK，修改為UTF8的方法 gbk和utf8的json轉化 UTF8 & GBK之間的轉換