From acf586c0062f8e844419e605ba889cee05ff24e4 Mon Sep 17 00:00:00 2001 From: Classic298 <27028174+Classic298@users.noreply.github.com> Date: Fri, 24 Jul 2026 03:35:27 +0200 Subject: [PATCH] fix: resolve the web loader parser per URL instead of locking in the first one (#27367) SafeWebBaseLoader._unpack_fetch_results assigned the resolved parser to the parser parameter itself, so the None check only ran for the first URL. In a mixed batch every later document was parsed with whatever the first URL happened to select: an .xml feed first meant all following HTML pages went through the xml parser (broken text extraction), and an HTML page first meant .xml URLs were parsed as HTML. Web search regularly fetches mixed batches, so this silently degraded extraction quality depending on result order. The parser is now resolved per URL; an explicitly passed parser still applies to the whole batch as before. Verified with mixed xml/html batches in both orders and with an explicit parser override. --- backend/open_webui/retrieval/web/utils.py | 12 +++++------- 1 file changed, 5 insertions(+), 7 deletions(-) diff --git a/backend/open_webui/retrieval/web/utils.py b/backend/open_webui/retrieval/web/utils.py index 0cb10eb9a6..3b9a6ed065 100644 --- a/backend/open_webui/retrieval/web/utils.py +++ b/backend/open_webui/retrieval/web/utils.py @@ -785,13 +785,11 @@ class SafeWebBaseLoader(WebBaseLoader): final_results = [] for i, result in enumerate(results): url = urls[i] - if parser is None: - if url.endswith('.xml'): - parser = 'xml' - else: - parser = self.default_parser - self._check_parser(parser) - final_results.append(BeautifulSoup(result, parser, **self.bs_kwargs)) + url_parser = parser + if url_parser is None: + url_parser = 'xml' if url.endswith('.xml') else self.default_parser + self._check_parser(url_parser) + final_results.append(BeautifulSoup(result, url_parser, **self.bs_kwargs)) return final_results async def ascrape_all(self, urls: List[str], parser: Union[str, None] = None) -> List[Any]: