This commit is contained in:
Timothy Jaeryang Baek
2026-06-01 09:53:04 -07:00
parent 55ca719bbf
commit 6f0277db52
2 changed files with 40 additions and 4 deletions
+27 -1
View File
@@ -233,7 +233,33 @@ class Loader:
def load(self, filename: str, file_content_type: str, file_path: str) -> list[Document]:
loader = self._get_loader(filename, file_content_type, file_path)
docs = loader.load()
try:
docs = loader.load()
except (UnicodeDecodeError, RuntimeError, TimeoutError) as e:
# Only retry with latin-1 for text-encoding loaders (TextLoader,
# CSVLoader). Binary format loaders (PDF, DOCX, PPTX, …) can
# raise RuntimeError for unrelated reasons; retrying those as
# plain text would produce garbage.
if not isinstance(loader, (TextLoader, CSVLoader)):
raise
# TextLoader/CSVLoader with autodetect_encoding=True can fail when:
# - chardet times out (5 s hardcoded in langchain)
# - chardet returns a wrong/low-confidence encoding
# - the file contains non-UTF-8 bytes (Latin-1, Windows-1252, …)
#
# Latin-1 is a safe last resort: every byte 0x00–0xFF is valid,
# and ftfy.fix_text() (applied below) repairs most mojibake that
# results from treating Windows-1252 content as Latin-1.
log.warning(
"Primary loader failed for %s (%s), retrying with latin-1 encoding: %s",
filename,
type(e).__name__,
e,
)
fallback_loader = TextLoader(file_path, encoding="latin-1")
docs = fallback_loader.load()
return [Document(page_content=ftfy.fix_text(doc.page_content), metadata=doc.metadata) for doc in docs]
+13 -3
View File
@@ -61,7 +61,11 @@ from open_webui.utils.access_control.files import has_access_to_file
def _is_text_file(file_path: str, chunk_size: int = 8192) -> bool:
"""Check if a file is likely a text file by reading a chunk and validating UTF-8.
"""Check if a file is likely a text file by reading a chunk and decoding it.
Tries UTF-8 first, then falls back to Latin-1 (which accepts every byte
in 0x00–0xFF) so that legacy-encoded files from Windows environments are
not misclassified as binary.
This catches files whose extensions are mis-mapped by mimetypes/browsers
(e.g. TypeScript .ts → video/mp2t) without maintaining an extension whitelist.
@@ -75,9 +79,15 @@ def _is_text_file(file_path: str, chunk_size: int = 8192) -> bool:
# Null bytes are a strong indicator of binary content
if b'\x00' in chunk:
return False
chunk.decode('utf-8')
try:
chunk.decode('utf-8')
except UnicodeDecodeError:
# Latin-1 always succeeds (every byte is valid), so this
# effectively just means "the file has no null bytes and is
# therefore likely text, even if not valid UTF-8".
chunk.decode('latin-1')
return True
except (UnicodeDecodeError, Exception):
except Exception:
return False