使用中文寫文章,當篇幅超過一定程度,必然會使用到諸如:“的”、“你”、“我”這樣的常用字。本類思想便是提取中文最常用的一百個字,使用中文世界常用編碼(主要有GBK、GB2312、GB18030、UTF-8、UTF-32、Unicode、BigEndianUnicode及UTF-7等)獲得其編碼位元組, ...
使用中文寫文章,當篇幅超過一定程度,必然會使用到諸如:“的”、“你”、“我”這樣的常用字。本類思想便是提取中文最常用的一百個字,使用中文世界常用編碼(主要有GBK、GB2312、GB18030、UTF-8、UTF-32、Unicode、BigEndianUnicode及UTF-7等)獲得其編碼位元組,以其為搜索詞到目標流進行查找,如果查找得到則表示該流使用此種編碼。顯而易見此類不適用於小篇幅。
using System.Collections.Generic; using System.IO; using System.Text; namespace YunShenBuZhiChu.MiMaBenJiaMiFa { /// <summary> /// 文字編碼檢測。 /// 用於檢測一篇文章使用什麼編碼方式進行編碼。 /// </summary> public class StreamBianMaJianCe { /// <summary> /// BigEndianUnicode編碼高頻漢字編碼 /// </summary> private List<byte[]> _BigEndianUnicodeGaoPinZiFuBianMaLsit = new List<byte[]>() { new byte[2]{118,132},new byte[2]{78,0},new byte[2]{86,253},new byte[2]{87,40},new byte[2]{78,186},new byte[2]{78,134},new byte[2]{103,9},new byte[2]{78,45}, new byte[2]{102,47},new byte[2]{94,116},new byte[2]{84,140},new byte[2]{89,39},new byte[2]{78,26},new byte[2]{78,13},new byte[2]{78,58},new byte[2]{83,209}, new byte[2]{79,26},new byte[2]{93,229},new byte[2]{126,207},new byte[2]{78,10},new byte[2]{87,48},new byte[2]{94,2},new byte[2]{137,129},new byte[2]{78,42}, new byte[2]{78,167},new byte[2]{143,217},new byte[2]{81,250},new byte[2]{136,76},new byte[2]{79,92},new byte[2]{117,31},new byte[2]{91,182},new byte[2]{78,229}, new byte[2]{98,16},new byte[2]{82,48},new byte[2]{101,229},new byte[2]{108,17},new byte[2]{103,101},new byte[2]{98,17},new byte[2]{144,232},new byte[2]{91,249}, new byte[2]{143,219},new byte[2]{89,26},new byte[2]{81,104},new byte[2]{94,250},new byte[2]{78,214},new byte[2]{81,108},new byte[2]{95,0},new byte[2]{78,236}, new byte[2]{87,58},new byte[2]{92,85},new byte[2]{101,246},new byte[2]{116,6},new byte[2]{101,176},new byte[2]{101,185},new byte[2]{78,59},new byte[2]{79,1}, new byte[2]{141,68},new byte[2]{91,158},new byte[2]{91,102},new byte[2]{98,165},new byte[2]{82,54},new byte[2]{101,63},new byte[2]{109,78},new byte[2]{117,40}, new byte[2]{84,12},new byte[2]{78,142},new byte[2]{108,213},new byte[2]{154,216},new byte[2]{149,127},new byte[2]{115,176},new byte[2]{103,44},new byte[2]{103,8}, new byte[2]{91,154},new byte[2]{83,22},new byte[2]{82,160},new byte[2]{82,168},new byte[2]{84,8},new byte[2]{84,193},new byte[2]{145,205},new byte[2]{81,115}, new byte[2]{103,58},new byte[2]{82,6},new byte[2]{82,155},new byte[2]{129,234},new byte[2]{89,22},new byte[2]{128,5},new byte[2]{83,58},new byte[2]{128,253}, new byte[2]{139,190},new byte[2]{84,14},new byte[2]{92,49},new byte[2]{123,73},new byte[2]{79,83},new byte[2]{78,11},new byte[2]{78,7},new byte[2]{81,67}, new byte[2]{121,62},new byte[2]{143,199},new byte[2]{82,77},new byte[2]{151,98},new byte[2]{48,2},new byte[2]{255,12},new byte[2]{255,31},new byte[2]{255,1} }; /// <summary> /// UTF8編碼高頻漢字編碼 /// </summary> private List<byte[]> _UTF8GaoPinZiFuBianMaLsit = new List<byte[]>() { new byte[3]{231,154,132},new byte[3]{228,184,128},new byte[3]{229,155,189},new byte[3]{229,156,168},new byte[3]{228,186,186},new byte[3]{228,186,134}, new byte[3]{230,156,137},new byte[3]{228,184,173},new byte[3]{230,152,175},new byte[3]{229,185,180},new byte[3]{229,146,140},new byte[3]{229,164,167}, new byte[3]{228,184,154},new byte[3]{228,184,141},new byte[3]{228,184,186},new byte[3]{229,143,145},new byte[3]{228,188,154},new byte[3]{229,183,165}, new byte[3]{231,187,143},new byte[3]{228,184,138},new byte[3]{229,156,176},new byte[3]{229,184,130},new byte[3]{232,166,129},new byte[3]{228,184,170}, new byte[3]{228,186,167},new byte[3]{232,191,153},new byte[3]{229,135,186},new byte[3]{232,161,140},new byte[3]{228,189,156},new byte[3]{231,148,159}, new byte[3]{229,174,182},new byte[3]{228,187,165},new byte[3]{230,136,144},new byte[3]{229,136,176},new byte[3]{230,151,165},new byte[3]{230,176,145}, new byte[3]{230,157,165},new byte[3]{230,136,145},new byte[3]{233,131,168},new byte[3]{229,175,185},new byte[3]{232,191,155},new byte[3]{229,164,154}, new byte[3]{229,133,168},new byte[3]{229,187,186},new byte[3]{228,187,150},new byte[3]{229,133,172},new byte[3]{229,188,128},new byte[3]{228,187,172}, new byte[3]{229,156,186},new byte[3]{229,177,149},new byte[3]{230,151,182},new byte[3]{231,144,134},new byte[3]{230,150,176},new byte[3]{230,150,185}, new byte[3]{228,184,187},new byte[3]{228,188,129},new byte[3]{232,181,132},new byte[3]{229,174,158},new byte[3]{229,173,166},new byte[3]{230,138,165}, new byte[3]{229,136,182},new byte[3]{230,148,191},new byte[3]{230,181,142},new byte[3]{231,148,168},new byte[3]{229,144,140},new byte[3]{228,186,142}, new byte[3]{230,179,149},new byte[3]{233,171,152},new byte[3]{233,149,191},new byte[3]{231,142,176},new byte[3]{230,156,172},new byte[3]{230,156,136}, new byte[3]{229,174,154},new byte[3]{229,140,150},new byte[3]{229,138,160},new byte[3]{229,138,168},new byte[3]{229,144,136},new byte[3]{229,147,129}, new byte[3]{233,135,141},new byte[3]{229,133,179},new byte[3]{230,156,186},new byte[3]{229,136,134},new byte[3]{229,138,155},new byte[3]{232,135,170}, new byte[3]{229,164,150},new byte[3]{232,128,133},new byte[3]{229,140,186},new byte[3]{232,131,189},new byte[3]{232,174,190},new byte[3]{229,144,142}, new byte[3]{229,176,177},new byte[3]{231,173,137},new byte[3]{228,189,147},new byte[3]{228,184,139},new byte[3]{228,184,135},new byte[3]{229,133,131}, new byte[3]{231,164,190},new byte[3]{232,191,135},new byte[3]{229,137,141},new byte[3]{233,157,162},new byte[3]{227,128,130},new byte[3]{239,188,140}, new byte[3]{239,188,159},new byte[3]{239,188,129}, }; /// <summary> /// Unicode編碼高頻漢字編碼 /// </summary> private List<byte[]> _UnicodeGaoPinZiFuBianMaLsit = new List<byte[]>() { new byte[2]{132,118},new byte[2]{0,78},new byte[2]{253,86},new byte[2]{40,87},new byte[2]{186,78},new byte[2]{134,78},new byte[2]{9,103},new byte[2]{45,78}, new byte[2]{47,102},new byte[2]{116,94},new byte[2]{140,84},new byte[2]{39,89},new byte[2]{26,78},new byte[2]{13,78},new byte[2]{58,78},new byte[2]{209,83}, new byte[2]{26,79},new byte[2]{229,93},new byte[2]{207,126},new byte[2]{10,78},new byte[2]{48,87},new byte[2]{2,94},new byte[2]{129,137},new byte[2]{42,78}, new byte[2]{167,78},new byte[2]{217,143},new byte[2]{250,81},new byte[2]{76,136},new byte[2]{92,79},new byte[2]{31,117},new byte[2]{182,91},new byte[2]{229,78}, new byte[2]{16,98},new byte[2]{48,82},new byte[2]{229,101},new byte[2]{17,108},new byte[2]{101,103},new byte[2]{17,98},new byte[2]{232,144},new byte[2]{249,91}, new byte[2]{219,143},new byte[2]{26,89},new byte[2]{104,81},new byte[2]{250,94},new byte[2]{214,78},new byte[2]{108,81},new byte[2]{0,95},new byte[2]{236,78}, new byte[2]{58,87},new byte[2]{85,92},new byte[2]{246,101},new byte[2]{6,116},new byte[2]{176,101},new byte[2]{185,101},new byte[2]{59,78},new byte[2]{1,79}, new byte[2]{68,141},new byte[2]{158,91},new byte[2]{102,91},new byte[2]{165,98},new byte[2]{54,82},new byte[2]{63,101},new byte[2]{78,109},new byte[2]{40,117}, new byte[2]{12,84},new byte[2]{142,78},new byte[2]{213,108},new byte[2]{216,154},new byte[2]{127,