refac
This commit is contained in:
@@ -233,7 +233,33 @@ class Loader:
|
||||
|
||||
def load(self, filename: str, file_content_type: str, file_path: str) -> list[Document]:
|
||||
loader = self._get_loader(filename, file_content_type, file_path)
|
||||
docs = loader.load()
|
||||
|
||||
try:
|
||||
docs = loader.load()
|
||||
except (UnicodeDecodeError, RuntimeError, TimeoutError) as e:
|
||||
# Only retry with latin-1 for text-encoding loaders (TextLoader,
|
||||
# CSVLoader). Binary format loaders (PDF, DOCX, PPTX, …) can
|
||||
# raise RuntimeError for unrelated reasons; retrying those as
|
||||
# plain text would produce garbage.
|
||||
if not isinstance(loader, (TextLoader, CSVLoader)):
|
||||
raise
|
||||
|
||||
# TextLoader/CSVLoader with autodetect_encoding=True can fail when:
|
||||
# - chardet times out (5 s hardcoded in langchain)
|
||||
# - chardet returns a wrong/low-confidence encoding
|
||||
# - the file contains non-UTF-8 bytes (Latin-1, Windows-1252, …)
|
||||
#
|
||||
# Latin-1 is a safe last resort: every byte 0x00–0xFF is valid,
|
||||
# and ftfy.fix_text() (applied below) repairs most mojibake that
|
||||
# results from treating Windows-1252 content as Latin-1.
|
||||
log.warning(
|
||||
"Primary loader failed for %s (%s), retrying with latin-1 encoding: %s",
|
||||
filename,
|
||||
type(e).__name__,
|
||||
e,
|
||||
)
|
||||
fallback_loader = TextLoader(file_path, encoding="latin-1")
|
||||
docs = fallback_loader.load()
|
||||
|
||||
return [Document(page_content=ftfy.fix_text(doc.page_content), metadata=doc.metadata) for doc in docs]
|
||||
|
||||
|
||||
@@ -61,7 +61,11 @@ from open_webui.utils.access_control.files import has_access_to_file
|
||||
|
||||
|
||||
def _is_text_file(file_path: str, chunk_size: int = 8192) -> bool:
|
||||
"""Check if a file is likely a text file by reading a chunk and validating UTF-8.
|
||||
"""Check if a file is likely a text file by reading a chunk and decoding it.
|
||||
|
||||
Tries UTF-8 first, then falls back to Latin-1 (which accepts every byte
|
||||
in 0x00–0xFF) so that legacy-encoded files from Windows environments are
|
||||
not misclassified as binary.
|
||||
|
||||
This catches files whose extensions are mis-mapped by mimetypes/browsers
|
||||
(e.g. TypeScript .ts → video/mp2t) without maintaining an extension whitelist.
|
||||
@@ -75,9 +79,15 @@ def _is_text_file(file_path: str, chunk_size: int = 8192) -> bool:
|
||||
# Null bytes are a strong indicator of binary content
|
||||
if b'\x00' in chunk:
|
||||
return False
|
||||
chunk.decode('utf-8')
|
||||
try:
|
||||
chunk.decode('utf-8')
|
||||
except UnicodeDecodeError:
|
||||
# Latin-1 always succeeds (every byte is valid), so this
|
||||
# effectively just means "the file has no null bytes and is
|
||||
# therefore likely text, even if not valid UTF-8".
|
||||
chunk.decode('latin-1')
|
||||
return True
|
||||
except (UnicodeDecodeError, Exception):
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user