1/** 2 * 默认GB18030 3 */ 4 public static final String detectCharset(byte[] byteArray){ 5 // 建立InputStream 6 ByteArrayInputStream bais = new ByteArrayInputStream(byteArray); 7 8 // 默认编码 9 String utf8 = "UTF-8"; 10 String charset = "GB18030"; 11 12 // 开始检测是否为UTF-8 13 try { 14 // 标记初始位置 15 bais.mark(0); 16 17 // 读取前3字节 18 byte[] first3Bytes = new byte[3]; 19 bais.read(first3Bytes); 20 21 // 如果前三字节为 0xEFBBBF ,则为带签名的UTF-8 22 if (first3Bytes[0] == (byte) 0xEF && first3Bytes[1] == (byte) 0xBB && first3Bytes[2] == (byte) 0xBF) { 23 return utf8; 24 } 25 26 // 前三字节判定失败,开始检测是否是不带签名的UTF-8 27 // 重置读取位置 28 bais.reset(); 29 30 // 逐字节判定,直到遇到一个UTF-8编码字符 31 byte[] oneByte = new byte[1]; 32 boolean isUtf8 = false; 33 while (-1 != bais.read(oneByte)) { 34 // 如果是ASCII码,跳过 35 if (CharUtils.isAscii((char) oneByte[0])) { 36 continue; 37 } 38 39 // 双字节格式 40 // 110yyyyy(C0-DF) 10xxxxxx 41 if ((oneByte[0] & 0xE0) == 0xC0) { 42 bais.mark(0); 43 byte[] nextOneByte = new byte[1]; 44 if (bais.available() >= 1 && -1 != bais.read(nextOneByte)) { 45 if ((nextOneByte[0] & 0xC0) == 0x80) { 46 47 // 是GBK双字节重叠部分?暂时当GBK处理,中文系统下,GBK默认编码,UTF-8不常见 48 // 双字节,第一个字节的值从0x81到0xFE,第二个字节的值从0x40到0xFE(不包括0x7F) 49 int oneByteInt = oneByte[0] & 0xff; 50 int nextOneByteInt = nextOneByte[0] & 0xff; 51 if ( 52 ((0x81 & 0xff) < = oneByteInt && oneByteInt <= (0xFE & 0xff)) 53 && ((0x40 & 0xff) <= nextOneByteInt && nextOneByteInt <= (0xfe & 0xff)) 54 && (nextOneByte[0] != 0x7F) 55 ) { 56 continue; 57 } 58 59 // 非GBK重叠部分,归于UTF-8 60 isUtf8 = true; 61 break; 62 } 63 bais.reset(); 64 } 65 } 66 67 // 三字节格式 68 // 1110xxxx(E0-EF) 10xxxxxx 10xxxxxx 69 if ((oneByte[0] & 0xF0) == 0xE0) { 70 byte[] twoByte = new byte[2]; 71 bais.mark(0); 72 if (bais.available() >= 2 && -1 != bais.read(twoByte)) { 73 if (((twoByte[0] & 0xC0) == 0x80) && ((twoByte[1] & 0xC0) == 0x80)) { 74 isUtf8 = true; 75 break; 76 } 77 bais.reset(); 78 } 79 } 80 81 // 四字节格式 82 // 11110www(F0-F7) 10xxxxxx 10xxxxxx 10xxxxxx 83 if ((oneByte[0] & 0xF8) == 0xF0) { 84 byte[] threeByte = new byte[3]; 85 bais.mark(0); 86 if (bais.available() >= 3 && -1 != bais.read(threeByte)) { 87 if (((threeByte[0] & 0xC0) == 0x80) && ((threeByte[1] & 0xC0) == 0x80) 88 && ((threeByte[2] & 0xC0) == 0x80)) { 89 isUtf8 = true; 90 break; 91 } 92 bais.reset(); 93 } 94 } 95 96 // 五字节格式 97 // 111110xx(F8-FB) 10xxxxxx 10xxxxxx 10xxxxxx 10xxxxxx 98 if ((oneByte[0] & 0xFC) == 0xF8) { 99 byte[] fourByte = new byte[4]; 100 bais.mark(0); 101 if (bais.available() >= 4 && -1 != bais.read(fourByte)) { 102 if (((fourByte[0] & 0xC0) == 0x80) && ((fourByte[1] & 0xC0) == 0x80) 103 && ((fourByte[2] & 0xC0) == 0x80) 104 && ((fourByte[3] & 0xC0) == 0x80)) { 105 isUtf8 = true; 106 break; 107 } 108 bais.reset(); 109 } 110 } 111 112 // 六字节格式 113 // 1111110x(FC-FD) 10xxxxxx 10xxxxxx 10xxxxxx 10xxxxxx 10xxxxxx 114 if ((oneByte[0] & 0xFE) == 0xFC) { 115 byte[] fiveByte = new byte[5]; 116 bais.mark(0); 117 if (bais.available() >= 5 && -1 != bais.read(fiveByte)) { 118 if (((fiveByte[0] & 0xC0) == 0x80) && ((fiveByte[1] & 0xC0) == 0x80) 119 && ((fiveByte[2] & 0xC0) == 0x80) 120 && ((fiveByte[3] & 0xC0) == 0x80) 121 && ((fiveByte[4] & 0xC0) == 0x80)) { 122 isUtf8 = true; 123 break; 124 } 125 bais.reset(); 126 } 127 } 128 } 129 130 // 依据标志位设定返回值 131 if (isUtf8) { 132 return utf8; 133 } 134 } catch (IOException e) { 135 } 136 137 // 返回字符编码格式 138 return charset; 139 }
java 检测文本、文件编码
Wesley13
2021-10-11
1122 1 0
点赞
收藏
评论区
加载中...