fix: resolve the web loader parser per URL instead of locking in the first one (#27367)
SafeWebBaseLoader._unpack_fetch_results assigned the resolved parser to the parser parameter itself, so the None check only ran for the first URL. In a mixed batch every later document was parsed with whatever the first URL happened to select: an .xml feed first meant all following HTML pages went through the xml parser (broken text extraction), and an HTML page first meant .xml URLs were parsed as HTML. Web search regularly fetches mixed batches, so this silently degraded extraction quality depending on result order. The parser is now resolved per URL; an explicitly passed parser still applies to the whole batch as before. Verified with mixed xml/html batches in both orders and with an explicit parser override.
This commit is contained in:
@@ -785,13 +785,11 @@ class SafeWebBaseLoader(WebBaseLoader):
|
||||
final_results = []
|
||||
for i, result in enumerate(results):
|
||||
url = urls[i]
|
||||
if parser is None:
|
||||
if url.endswith('.xml'):
|
||||
parser = 'xml'
|
||||
else:
|
||||
parser = self.default_parser
|
||||
self._check_parser(parser)
|
||||
final_results.append(BeautifulSoup(result, parser, **self.bs_kwargs))
|
||||
url_parser = parser
|
||||
if url_parser is None:
|
||||
url_parser = 'xml' if url.endswith('.xml') else self.default_parser
|
||||
self._check_parser(url_parser)
|
||||
final_results.append(BeautifulSoup(result, url_parser, **self.bs_kwargs))
|
||||
return final_results
|
||||
|
||||
async def ascrape_all(self, urls: List[str], parser: Union[str, None] = None) -> List[Any]:
|
||||
|
||||
Reference in New Issue
Block a user