java读取UTF-8的txt文件，第一行有?

最新推荐文章于 2024-08-20 04:38:40 发布

不带刺仙人球

最新推荐文章于 2024-08-20 04:38:40 发布

阅读量1.4k

点赞数

分类专栏： Java 文章标签： txtUTF-8编码

本文链接：https://blog.csdn.net/zyzn1425077119/article/details/79859168

版权

Java 专栏收录该内容

101 篇文章 1 订阅

订阅专栏

参考jackpk的博客，网址：https://blog.csdn.net/jackpk/article/details/5702964/

test.txt文件采用写字板保存为UTF-8格式

保存并关闭后使用写字板再次打开该UTF-8文档，中文、字母正常显示

import java.io.BufferedReader;
import java.io.File;
import java.io.FileInputStream;
import java.io.InputStreamReader;
public class ReadTxtFile {
public static void main(String[] args) {
try {
String charsetName = "UTF-8";
String path = "D:/to_delete/test.txt";
File file = new File(path);
if (file.isFile() && file.exists())
{
InputStreamReader insReader = new InputStreamReader(
new FileInputStream(file), charsetName);
BufferedReader bufReader = new BufferedReader(insReader);
String line = new String();
while ((line = bufReader.readLine()) != null) {
System.out.println(line);
}
bufReader.close();
insReader.close();
}
} catch (Exception e) {
System.out.println("读取文件内容操作出错");
e.printStackTrace();
}
}
}

第一行? ，

Java读取BOM（Byte Order　Mark）的问题，在使用UTF-8时，可以在文件的开始使用3个字节的"EF BB BF"来标识文件使用了UTF-8的编码，当然也可以不用这个3个字节。

上面的问题应该就是因为对开头3个字节的读取导致的。

/**
version: 1.1 / 2007-01-25
- changed BOM recognition ordering (longer boms first)

Original pseudocode   : Thomas Weidenfeller
Implementation tweaked: Aki Nieminen

http://www.unicode.org/unicode/faq/utf_bom.html
BOMs:
   00 00 FE FF    = UTF-32, big-endian
   FF FE 00 00    = UTF-32, little-endian
   EF BB BF       = UTF-8,
   FE FF          = UTF-16, big-endian
   FF FE          = UTF-16, little-endian

Win2k Notepad:
   Unicode format = UTF-16LE
***/

import java.io.*;

/**
* Generic unicode textreader, which will use BOM mark to identify the encoding
* to be used. If BOM is not found then use a given default or system encoding.
*/
public class UnicodeReader extends Reader {
   PushbackInputStream internalIn;
   InputStreamReader internalIn2 = null;
   String defaultEnc;

   private static final int BOM_SIZE = 4;

   /**
   *
   * @param in
   *            inputstream to be read
   * @param defaultEnc
   *            default encoding if stream does not have BOM marker. Give NULL
   *            to use system-level default.
   */
   UnicodeReader(InputStream in, String defaultEnc) {
       internalIn = new PushbackInputStream(in, BOM_SIZE);
       this.defaultEnc = defaultEnc;
   }

   public String getDefaultEncoding() {
       return defaultEnc;
   }

   /**
   * Get stream encoding or NULL if stream is uninitialized. Call init() or
   * read() method to initialize it.
   */
   public String getEncoding() {
       if (internalIn2 == null)
           return null;
       return internalIn2.getEncoding();
   }

   /**
   * Read-ahead four bytes and check for BOM marks. Extra bytes are unread
   * back to the stream, only BOM bytes are skipped.
   */
   protected void init() throws IOException {
       if (internalIn2 != null)
           return;

       String encoding;
       byte bom[] = new byte[BOM_SIZE];
       int n, unread;
       n = internalIn.read(bom, 0, bom.length);

       if ((bom[0] == (byte) 0x00) && (bom[1] == (byte) 0x00)
               && (bom[2] == (byte) 0xFE) && (bom[3] == (byte) 0xFF)) {
           encoding = "UTF-32BE";
           unread = n - 4;
       } else if ((bom[0] == (byte) 0xFF) && (bom[1] == (byte) 0xFE)
               && (bom[2] == (byte) 0x00) && (bom[3] == (byte) 0x00)) {
           encoding = "UTF-32LE";
           unread = n - 4;
       } else if ((bom[0] == (byte) 0xEF) && (bom[1] == (byte) 0xBB)
               && (bom[2] == (byte) 0xBF)) {
           encoding = "UTF-8";
           unread = n - 3;
       } else if ((bom[0] == (byte) 0xFE) && (bom[1] == (byte) 0xFF)) {
           encoding = "UTF-16BE";
           unread = n - 2;
       } else if ((bom[0] == (byte) 0xFF) && (bom[1] == (byte) 0xFE)) {
           encoding = "UTF-16LE";
           unread = n - 2;
       } else {
           // Unicode BOM mark not found, unread all bytes
           encoding = defaultEnc;
           unread = n;
       }
       // System.out.println("read=" + n + ", unread=" + unread);

       if (unread > 0)
           internalIn.unread(bom, (n - unread), unread);

       // Use given encoding
       if (encoding == null) {
           internalIn2 = new InputStreamReader(internalIn);
       } else {
           internalIn2 = new InputStreamReader(internalIn, encoding);
       }
   }

   public void close() throws IOException {
       init();
       internalIn2.close();
   }

   public int read(char[] cbuf, int off, int len) throws IOException {
       init();
       return internalIn2.read(cbuf, off, len);
   }

}

import java.io.BufferedReader;
import java.io.File;
import java.io.FileInputStream;
import java.io.IOException;
import java.io.InputStreamReader;
import java.nio.charset.Charset;

public class UTF8Test {
   public static void main(String[] args) throws IOException {
       File f = new File("./utf.txt");
       FileInputStream in = new FileInputStream(f);
       // 指定读取文件时以UTF-8的格式读取
//       BufferedReader br = new BufferedReader(new InputStreamReader(in, "UTF-8"));
       BufferedReader br = new BufferedReader(new UnicodeReader(in, Charset.defaultCharset().name()));

       String line = br.readLine();
       while(line != null)
       {
           System.out.println(line);
           line = br.readLine();
       }
   }
}