chore: format
This commit is contained in:
@@ -400,10 +400,7 @@ class Loader:
|
||||
api_key=self.kwargs.get('MISTRAL_OCR_API_KEY'),
|
||||
file_path=file_path,
|
||||
)
|
||||
elif (
|
||||
self.engine == 'paddleocr_vl'
|
||||
and self.kwargs.get('PADDLEOCR_VL_TOKEN') != ''
|
||||
):
|
||||
elif self.engine == 'paddleocr_vl' and self.kwargs.get('PADDLEOCR_VL_TOKEN') != '':
|
||||
loader = PaddleOCRVLLoader(
|
||||
api_url=self.kwargs.get('PADDLEOCR_VL_BASE_URL'),
|
||||
token=self.kwargs.get('PADDLEOCR_VL_TOKEN'),
|
||||
|
||||
@@ -11,6 +11,7 @@ from open_webui.env import GLOBAL_LOG_LEVEL
|
||||
logging.basicConfig(stream=sys.stdout, level=GLOBAL_LOG_LEVEL)
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class PaddleOCRVLLoader:
|
||||
"""Loader that uses PaddleOCR-vl API to extract text from PDF/images."""
|
||||
|
||||
@@ -21,9 +22,9 @@ class PaddleOCRVLLoader:
|
||||
file_path: str,
|
||||
):
|
||||
if not api_url or not token:
|
||||
raise ValueError("PaddleOCR-vl API URL and Token are required.")
|
||||
raise ValueError('PaddleOCR-vl API URL and Token are required.')
|
||||
if not os.path.exists(file_path):
|
||||
raise FileNotFoundError(f"File not found at {file_path}")
|
||||
raise FileNotFoundError(f'File not found at {file_path}')
|
||||
|
||||
self.api_url = api_url.rstrip('/')
|
||||
self.token = token
|
||||
@@ -31,20 +32,17 @@ class PaddleOCRVLLoader:
|
||||
self.file_name = os.path.basename(file_path)
|
||||
|
||||
def load(self) -> List[Document]:
|
||||
log.info(f"Processing with PaddleOCR-vl: {self.file_path}")
|
||||
log.info(f'Processing with PaddleOCR-vl: {self.file_path}')
|
||||
|
||||
try:
|
||||
with open(self.file_path, "rb") as file:
|
||||
with open(self.file_path, 'rb') as file:
|
||||
file_bytes = file.read()
|
||||
file_data = base64.b64encode(file_bytes).decode("ascii")
|
||||
file_data = base64.b64encode(file_bytes).decode('ascii')
|
||||
except Exception as e:
|
||||
log.error(f"Failed to read file {self.file_path}: {e}")
|
||||
log.error(f'Failed to read file {self.file_path}: {e}')
|
||||
raise
|
||||
|
||||
headers = {
|
||||
"Authorization": f"token {self.token}",
|
||||
"Content-Type": "application/json"
|
||||
}
|
||||
headers = {'Authorization': f'token {self.token}', 'Content-Type': 'application/json'}
|
||||
|
||||
# Detect fileType based on file extension
|
||||
ext = self.file_path.lower().split('.')[-1]
|
||||
@@ -52,76 +50,76 @@ class PaddleOCRVLLoader:
|
||||
file_type = 1 if ext in image_extensions else 0
|
||||
|
||||
payload = {
|
||||
"file": file_data,
|
||||
"fileType": file_type,
|
||||
"useDocOrientationClassify": False,
|
||||
"useDocUnwarping": False,
|
||||
"useChartRecognition": False,
|
||||
'file': file_data,
|
||||
'fileType': file_type,
|
||||
'useDocOrientationClassify': False,
|
||||
'useDocUnwarping': False,
|
||||
'useChartRecognition': False,
|
||||
}
|
||||
|
||||
try:
|
||||
response = requests.post(f"{self.api_url}/layout-parsing", json=payload, headers=headers)
|
||||
response = requests.post(f'{self.api_url}/layout-parsing', json=payload, headers=headers)
|
||||
response.raise_for_status()
|
||||
|
||||
result = response.json().get("result", {})
|
||||
layout_results = result.get("layoutParsingResults", [])
|
||||
|
||||
|
||||
result = response.json().get('result', {})
|
||||
layout_results = result.get('layoutParsingResults', [])
|
||||
|
||||
documents = []
|
||||
total_pages = len(layout_results)
|
||||
skipped_pages = 0
|
||||
|
||||
|
||||
for i, res in enumerate(layout_results):
|
||||
markdown_text = res.get("markdown", {}).get("text", "")
|
||||
|
||||
markdown_text = res.get('markdown', {}).get('text', '')
|
||||
|
||||
if isinstance(markdown_text, str):
|
||||
cleaned_content = markdown_text.strip()
|
||||
else:
|
||||
cleaned_content = str(markdown_text).strip()
|
||||
|
||||
|
||||
if not cleaned_content:
|
||||
skipped_pages += 1
|
||||
continue
|
||||
|
||||
|
||||
documents.append(
|
||||
Document(
|
||||
page_content=cleaned_content,
|
||||
metadata={
|
||||
"page": i,
|
||||
"page_label": i + 1,
|
||||
"total_pages": total_pages,
|
||||
"file_name": self.file_name,
|
||||
"processing_engine": "paddleocr-vl"
|
||||
}
|
||||
'page': i,
|
||||
'page_label': i + 1,
|
||||
'total_pages': total_pages,
|
||||
'file_name': self.file_name,
|
||||
'processing_engine': 'paddleocr-vl',
|
||||
},
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
if skipped_pages > 0:
|
||||
log.info(f"PaddleOCR-vl: Processed {len(documents)} pages, skipped {skipped_pages} empty pages.")
|
||||
|
||||
log.info(f'PaddleOCR-vl: Processed {len(documents)} pages, skipped {skipped_pages} empty pages.')
|
||||
|
||||
if not documents:
|
||||
log.warning("No valid text content found by PaddleOCR-vl.")
|
||||
log.warning('No valid text content found by PaddleOCR-vl.')
|
||||
return [
|
||||
Document(
|
||||
page_content="No valid text content found in document",
|
||||
page_content='No valid text content found in document',
|
||||
metadata={
|
||||
"error": "no_valid_pages",
|
||||
"file_name": self.file_name,
|
||||
"processing_engine": "paddleocr-vl"
|
||||
}
|
||||
'error': 'no_valid_pages',
|
||||
'file_name': self.file_name,
|
||||
'processing_engine': 'paddleocr-vl',
|
||||
},
|
||||
)
|
||||
]
|
||||
|
||||
|
||||
return documents
|
||||
|
||||
|
||||
except Exception as e:
|
||||
log.error(f"Error calling PaddleOCR-vl: {e}")
|
||||
log.error(f'Error calling PaddleOCR-vl: {e}')
|
||||
return [
|
||||
Document(
|
||||
page_content=f"Error during OCR processing: {e}",
|
||||
page_content=f'Error during OCR processing: {e}',
|
||||
metadata={
|
||||
"error": "processing_failed",
|
||||
"file_name": self.file_name,
|
||||
"processing_engine": "paddleocr-vl"
|
||||
}
|
||||
'error': 'processing_failed',
|
||||
'file_name': self.file_name,
|
||||
'processing_engine': 'paddleocr-vl',
|
||||
},
|
||||
)
|
||||
]
|
||||
|
||||
Reference in New Issue
Block a user