From 6b438f1a79716048f81659f175891ff06ec77e0d Mon Sep 17 00:00:00 2001 From: Classic298 <27028174+Classic298@users.noreply.github.com> Date: Tue, 25 Aug 2026 01:06:13 +0200 Subject: [PATCH] refac: stream web-fetched documents to disk instead of buffering them (#28945) The binary branch of the web fetch read the entire response body into memory before writing it out. It now streams in blocks, applies the configured file size limit the same way the sibling URL endpoint already does, and removes the temporary file when a download fails partway instead of leaving it behind. --- backend/open_webui/retrieval/utils.py | 17 ++++++++++++++--- 1 file changed, 14 insertions(+), 3 deletions(-) diff --git a/backend/open_webui/retrieval/utils.py b/backend/open_webui/retrieval/utils.py index 8bf9189e12..a6e1bd6857 100644 --- a/backend/open_webui/retrieval/utils.py +++ b/backend/open_webui/retrieval/utils.py @@ -24,6 +24,7 @@ from open_webui.config import ( RAG_EMBEDDING_QUERY_PREFIX, VECTOR_DB, ) +from open_webui.constants import ERROR_MESSAGES from open_webui.env import ( AIOHTTP_CLIENT_ALLOW_REDIRECTS, AIOHTTP_CLIENT_SESSION_SSL, @@ -67,6 +68,7 @@ def is_youtube_url(url: str) -> bool: LOADER_CONFIG_KEYS = { + 'file_max_size': 'rag.file.max_size', 'youtube_language': 'rag.youtube_loader_language', 'youtube_proxy_url': 'rag.youtube_loader_proxy_url', 'web_loader_ssl_verification': 'web.loader.ssl_verification', @@ -182,11 +184,20 @@ def _extract_text_from_binary_response( suffix = '.' + filename.split('.')[-1].lower() if '.' in filename else '' - with tempfile.NamedTemporaryFile(suffix=suffix, delete=False) as tmp: - tmp.write(response.content) - tmp_path = tmp.name + max_size = loader_config.get('file_max_size') + max_bytes = int(max_size) * 1024 * 1024 if max_size else 0 + tmp_fd, tmp_path = tempfile.mkstemp(suffix=suffix) try: + downloaded = 0 + # Stream to disk; response.content buffers the whole body in memory first. + with os.fdopen(tmp_fd, 'wb') as tmp: + for chunk in response.iter_content(64 * 1024): + downloaded += len(chunk) + if max_bytes and downloaded > max_bytes: + raise ValueError(ERROR_MESSAGES.FILE_TOO_LARGE(size=f'{max_size} MB')) + tmp.write(chunk) + loader = build_loader_from_config(request, loader_config) docs = loader.load(filename, content_type, tmp_path) for doc in docs: