Fit a lot of encodings for text file. (#458)

### What problem does this PR solve? #384 ### Type of change - [x] Performance Improvement
2026-01-31 23:55:06 +08:00 · 2024-04-19 18:02:53 +08:00
parent cda7b607cb
commit ed6081845a
19 changed files with 118 additions and 55 deletions
--- a/rag/app/book.py
+++ b/rag/app/book.py
@ -15,7 +15,8 @@ import re
 from io import BytesIO

 from rag.nlp import bullets_category, is_english, tokenize, remove_contents_table, \
-    hierarchical_merge, make_colon_as_title, naive_merge, random_choices, tokenize_table, add_positions, tokenize_chunks
+    hierarchical_merge, make_colon_as_title, naive_merge, random_choices, tokenize_table, add_positions, \
+    tokenize_chunks, find_codec
 from rag.nlp import huqie
 from deepdoc.parser import PdfParser, DocxParser, PlainParser

@ -87,7 +88,8 @@ def chunk(filename, binary=None, from_page=0, to_page=100000,
        callback(0.1, "Start to parse.")
        txt = ""
        if binary:
-            txt = binary.decode("utf-8")
+            encoding = find_codec(binary)
+            txt = binary.decode(encoding)
        else:
            with open(filename, "r") as f:
                while True:
--- a/rag/app/laws.py
+++ b/rag/app/laws.py
@ -17,7 +17,7 @@ from docx import Document

 from api.db import ParserType
 from rag.nlp import bullets_category, is_english, tokenize, remove_contents_table, hierarchical_merge, \
-    make_colon_as_title, add_positions, tokenize_chunks
+    make_colon_as_title, add_positions, tokenize_chunks, find_codec
 from rag.nlp import huqie
 from deepdoc.parser import PdfParser, DocxParser, PlainParser
 from rag.settings import cron_logger
@ -111,7 +111,8 @@ def chunk(filename, binary=None, from_page=0, to_page=100000,
        callback(0.1, "Start to parse.")
        txt = ""
        if binary:
-            txt = binary.decode("utf-8")
+            encoding = find_codec(binary)
+            txt = binary.decode(encoding)
        else:
            with open(filename, "r") as f:
                while True:
--- a/rag/app/naive.py
+++ b/rag/app/naive.py
@ -14,7 +14,7 @@ from io import BytesIO
 from docx import Document
 import re
 from deepdoc.parser.pdf_parser import PlainParser
-from rag.nlp import huqie, naive_merge, tokenize_table, tokenize_chunks
+from rag.nlp import huqie, naive_merge, tokenize_table, tokenize_chunks, find_codec
 from deepdoc.parser import PdfParser, ExcelParser, DocxParser
 from rag.settings import cron_logger

@ -139,10 +139,8 @@ def chunk(filename, binary=None, from_page=0, to_page=100000,
        callback(0.1, "Start to parse.")
        txt = ""
        if binary:
-            try:
-                txt = binary.decode("utf-8")
-            except Exception as e:
-                txt = binary.decode("gb2312")
+            encoding = find_codec(binary)
+            txt = binary.decode(encoding)
        else:
            with open(filename, "r") as f:
                while True:
--- a/rag/app/one.py
+++ b/rag/app/one.py
@ -12,7 +12,7 @@
 #
 import re
 from rag.app import laws
-from rag.nlp import huqie, tokenize
+from rag.nlp import huqie, tokenize, find_codec
 from deepdoc.parser import PdfParser, ExcelParser, PlainParser


@ -82,7 +82,8 @@ def chunk(filename, binary=None, from_page=0, to_page=100000,
        callback(0.1, "Start to parse.")
        txt = ""
        if binary:
-            txt = binary.decode("utf-8")
+            encoding = find_codec(binary)
+            txt = binary.decode(encoding)
        else:
            with open(filename, "r") as f:
                while True:
--- a/rag/app/qa.py
+++ b/rag/app/qa.py
@ -15,7 +15,7 @@ from copy import deepcopy
 from io import BytesIO
 from nltk import word_tokenize
 from openpyxl import load_workbook
-from rag.nlp import is_english, random_choices
+from rag.nlp import is_english, random_choices, find_codec
 from rag.nlp import huqie
 from deepdoc.parser import ExcelParser

@ -106,7 +106,8 @@ def chunk(filename, binary=None, lang="Chinese", callback=None, **kwargs):
        callback(0.1, "Start to parse.")
        txt = ""
        if binary:
-            txt = binary.decode("utf-8")
+            encoding = find_codec(binary)
+            txt = binary.decode(encoding)
        else:
            with open(filename, "r") as f:
                while True:
--- a/rag/app/table.py
+++ b/rag/app/table.py
@ -20,7 +20,7 @@ from openpyxl import load_workbook
 from dateutil.parser import parse as datetime_parse

 from api.db.services.knowledgebase_service import KnowledgebaseService
-from rag.nlp import huqie, is_english, tokenize
+from rag.nlp import huqie, is_english, tokenize, find_codec
 from deepdoc.parser import ExcelParser


@ -147,7 +147,8 @@ def chunk(filename, binary=None, from_page=0, to_page=10000000000,
        callback(0.1, "Start to parse.")
        txt = ""
        if binary:
-            txt = binary.decode("utf-8")
+            encoding = find_codec(binary)
+            txt = binary.decode(encoding)
        else:
            with open(filename, "r") as f:
                while True: