When `RAG_DOCUMENT_LOADER_ENGINE` is set to `paddleocr_vl`, the dispatch branch in `Loader._get_loader` checked only the engine name and a non-empty token, so every uploaded file was handed to the PaddleOCR-VL loader regardless of its type. Text based uploads such as `.md`, `.txt` and `.csv` were base64 encoded and posted to the `/layout-parsing` endpoint tagged as PDFs, and the API rejected them with `422 Unprocessable Entity` ("PDFium: Data format error"), so those files never indexed at all.
The loader already knows which extensions it can handle: it tags images with `fileType: 1` and treats everything else as a PDF. That list is now a module level constant, and the dispatch branch gates on `['pdf'] + images`, the same way `mistral_ocr`, `datalab_marker`, `document_intelligence` and `mineru` already limit themselves. Deriving the gate from the loader's own list keeps the two in sync, so a file can never be admitted by the gate and then mislabelled as a PDF on the wire. Everything outside that set falls through to the default loader chain, so `.md` and `.txt` load as text, `.csv` through `CSVLoader`, `.docx` through `Docx2txtLoader`, and so on.
The branch also never checked `PADDLEOCR_VL_BASE_URL`. With the URL cleared, `PaddleOCRVLLoader` raised `ValueError` from its constructor and the upload failed outright instead of falling back. Both settings are now required for the branch to be taken, matching how the other engines guard their own configuration.
Fixes #24988
Fixes #26759
128 lines
4.3 KiB
Python
128 lines
4.3 KiB
Python
import base64
|
|
import logging
|
|
import os
|
|
import sys
|
|
from typing import List
|
|
|
|
import requests
|
|
from langchain_core.documents import Document
|
|
from open_webui.env import GLOBAL_LOG_LEVEL
|
|
|
|
logging.basicConfig(stream=sys.stdout, level=GLOBAL_LOG_LEVEL)
|
|
log = logging.getLogger(__name__)
|
|
|
|
PADDLEOCR_VL_IMAGE_EXTENSIONS = ['png', 'jpg', 'jpeg', 'bmp', 'tiff', 'webp']
|
|
PADDLEOCR_VL_SUPPORTED_EXTENSIONS = ['pdf'] + PADDLEOCR_VL_IMAGE_EXTENSIONS
|
|
|
|
|
|
class PaddleOCRVLLoader:
|
|
"""Loader that uses PaddleOCR-vl API to extract text from PDF/images."""
|
|
|
|
def __init__(
|
|
self,
|
|
api_url: str,
|
|
token: str,
|
|
file_path: str,
|
|
):
|
|
if not api_url or not token:
|
|
raise ValueError('PaddleOCR-vl API URL and Token are required.')
|
|
if not os.path.exists(file_path):
|
|
raise FileNotFoundError(f'File not found at {file_path}')
|
|
|
|
self.api_url = api_url.rstrip('/')
|
|
self.token = token
|
|
self.file_path = file_path
|
|
self.file_name = os.path.basename(file_path)
|
|
|
|
def load(self) -> List[Document]:
|
|
log.info(f'Processing with PaddleOCR-vl: {self.file_path}')
|
|
|
|
try:
|
|
with open(self.file_path, 'rb') as file:
|
|
file_bytes = file.read()
|
|
file_data = base64.b64encode(file_bytes).decode('ascii')
|
|
except Exception as e:
|
|
log.error(f'Failed to read file {self.file_path}: {e}')
|
|
raise
|
|
|
|
headers = {'Authorization': f'token {self.token}', 'Content-Type': 'application/json'}
|
|
|
|
# Detect fileType based on file extension
|
|
ext = self.file_path.lower().split('.')[-1]
|
|
file_type = 1 if ext in PADDLEOCR_VL_IMAGE_EXTENSIONS else 0
|
|
|
|
payload = {
|
|
'file': file_data,
|
|
'fileType': file_type,
|
|
'useDocOrientationClassify': False,
|
|
'useDocUnwarping': False,
|
|
'useChartRecognition': False,
|
|
}
|
|
|
|
try:
|
|
response = requests.post(f'{self.api_url}/layout-parsing', json=payload, headers=headers)
|
|
response.raise_for_status()
|
|
|
|
result = response.json().get('result', {})
|
|
layout_results = result.get('layoutParsingResults', [])
|
|
|
|
documents = []
|
|
total_pages = len(layout_results)
|
|
skipped_pages = 0
|
|
|
|
for i, res in enumerate(layout_results):
|
|
markdown_text = res.get('markdown', {}).get('text', '')
|
|
|
|
if isinstance(markdown_text, str):
|
|
cleaned_content = markdown_text.strip()
|
|
else:
|
|
cleaned_content = str(markdown_text).strip()
|
|
|
|
if not cleaned_content:
|
|
skipped_pages += 1
|
|
continue
|
|
|
|
documents.append(
|
|
Document(
|
|
page_content=cleaned_content,
|
|
metadata={
|
|
'page': i,
|
|
'page_label': i + 1,
|
|
'total_pages': total_pages,
|
|
'file_name': self.file_name,
|
|
'processing_engine': 'paddleocr-vl',
|
|
},
|
|
)
|
|
)
|
|
|
|
if skipped_pages > 0:
|
|
log.info(f'PaddleOCR-vl: Processed {len(documents)} pages, skipped {skipped_pages} empty pages.')
|
|
|
|
if not documents:
|
|
log.warning('No valid text content found by PaddleOCR-vl.')
|
|
return [
|
|
Document(
|
|
page_content='No valid text content found in document',
|
|
metadata={
|
|
'error': 'no_valid_pages',
|
|
'file_name': self.file_name,
|
|
'processing_engine': 'paddleocr-vl',
|
|
},
|
|
)
|
|
]
|
|
|
|
return documents
|
|
|
|
except Exception as e:
|
|
log.error(f'Error calling PaddleOCR-vl: {e}')
|
|
return [
|
|
Document(
|
|
page_content=f'Error during OCR processing: {e}',
|
|
metadata={
|
|
'error': 'processing_failed',
|
|
'file_name': self.file_name,
|
|
'processing_engine': 'paddleocr-vl',
|
|
},
|
|
)
|
|
]
|