From 6f0277db52d005420480abb0702d421525d6ea8b Mon Sep 17 00:00:00 2001 From: Timothy Jaeryang Baek Date: Mon, 1 Jun 2026 09:53:04 -0700 Subject: [PATCH] refac --- backend/open_webui/retrieval/loaders/main.py | 28 +++++++++++++++++++- backend/open_webui/routers/files.py | 16 ++++++++--- 2 files changed, 40 insertions(+), 4 deletions(-) diff --git a/backend/open_webui/retrieval/loaders/main.py b/backend/open_webui/retrieval/loaders/main.py index 436d9a0acd..ede890793b 100644 --- a/backend/open_webui/retrieval/loaders/main.py +++ b/backend/open_webui/retrieval/loaders/main.py @@ -233,7 +233,33 @@ class Loader: def load(self, filename: str, file_content_type: str, file_path: str) -> list[Document]: loader = self._get_loader(filename, file_content_type, file_path) - docs = loader.load() + + try: + docs = loader.load() + except (UnicodeDecodeError, RuntimeError, TimeoutError) as e: + # Only retry with latin-1 for text-encoding loaders (TextLoader, + # CSVLoader). Binary format loaders (PDF, DOCX, PPTX, …) can + # raise RuntimeError for unrelated reasons; retrying those as + # plain text would produce garbage. + if not isinstance(loader, (TextLoader, CSVLoader)): + raise + + # TextLoader/CSVLoader with autodetect_encoding=True can fail when: + # - chardet times out (5 s hardcoded in langchain) + # - chardet returns a wrong/low-confidence encoding + # - the file contains non-UTF-8 bytes (Latin-1, Windows-1252, …) + # + # Latin-1 is a safe last resort: every byte 0x00–0xFF is valid, + # and ftfy.fix_text() (applied below) repairs most mojibake that + # results from treating Windows-1252 content as Latin-1. + log.warning( + "Primary loader failed for %s (%s), retrying with latin-1 encoding: %s", + filename, + type(e).__name__, + e, + ) + fallback_loader = TextLoader(file_path, encoding="latin-1") + docs = fallback_loader.load() return [Document(page_content=ftfy.fix_text(doc.page_content), metadata=doc.metadata) for doc in docs] diff --git a/backend/open_webui/routers/files.py b/backend/open_webui/routers/files.py index 31157dd803..3e036613d4 100644 --- a/backend/open_webui/routers/files.py +++ b/backend/open_webui/routers/files.py @@ -61,7 +61,11 @@ from open_webui.utils.access_control.files import has_access_to_file def _is_text_file(file_path: str, chunk_size: int = 8192) -> bool: - """Check if a file is likely a text file by reading a chunk and validating UTF-8. + """Check if a file is likely a text file by reading a chunk and decoding it. + + Tries UTF-8 first, then falls back to Latin-1 (which accepts every byte + in 0x00–0xFF) so that legacy-encoded files from Windows environments are + not misclassified as binary. This catches files whose extensions are mis-mapped by mimetypes/browsers (e.g. TypeScript .ts → video/mp2t) without maintaining an extension whitelist. @@ -75,9 +79,15 @@ def _is_text_file(file_path: str, chunk_size: int = 8192) -> bool: # Null bytes are a strong indicator of binary content if b'\x00' in chunk: return False - chunk.decode('utf-8') + try: + chunk.decode('utf-8') + except UnicodeDecodeError: + # Latin-1 always succeeds (every byte is valid), so this + # effectively just means "the file has no null bytes and is + # therefore likely text, even if not valid UTF-8". + chunk.decode('latin-1') return True - except (UnicodeDecodeError, Exception): + except Exception: return False