refactor: make PDF parser dependencies optional and lazy-load parser

This commit is contained in:
kdeterme 2026-10-01 13:57:12 +02:00
parent de6828221b
commit 391966da07
3 changed files with 20 additions and 6 deletions

View file

@ -12,8 +12,15 @@ import logging
from io import BytesIO
from typing import Dict, List, Any, Optional, Tuple
try:
import pypdfium2 as pdfium
except ImportError:
pdfium = None
try:
import numpy as np
except ImportError:
np = None
from django.db.models import Q
from .models import Supplier, Product, WarehouseLocation
@ -91,8 +98,13 @@ class PdfQuoteParser:
self._temp_path = None
self.is_ocr = False
def _get_document(self) -> Tuple[pdfium.PdfDocument, str]:
def _get_document(self) -> Tuple[Any, str]:
"""Ouvre le document PDF via pypdfium2."""
if pdfium is None:
raise RuntimeError(
"Le module 'pypdfium2' n'est pas installé sur le serveur. "
"Veuillez exécuter 'pip install -r requirements/base.txt' pour activer l'analyse PDF locale."
)
if isinstance(self.file_or_path, str):
return pdfium.PdfDocument(self.file_or_path), self.file_or_path
elif hasattr(self.file_or_path, 'temporary_file_path'):
@ -146,7 +158,7 @@ class PdfQuoteParser:
pass
self.close()
def _parse_vector(self, doc: pdfium.PdfDocument) -> Dict[str, Any]:
def _parse_vector(self, doc: Any) -> Dict[str, Any]:
"""Analyse un PDF avec couche texte vectorielle (recherche spatiale précise)."""
full_text = ""
for page in doc:
@ -167,7 +179,7 @@ class PdfQuoteParser:
'total_items': len(enriched_items),
}
def _parse_with_ocr(self, doc: pdfium.PdfDocument) -> Dict[str, Any]:
def _parse_with_ocr(self, doc: Any) -> Dict[str, Any]:
"""Analyse un PDF scanné via RapidOCR."""
try:
from rapidocr_onnxruntime import RapidOCR

View file

@ -64,7 +64,7 @@ from .models import (
SerializedItemEvent,
notify_supervisors_pending_product,
)
from .pdf_parser import PdfQuoteParser, normalize_unit
from .pdf_parser import normalize_unit
from .permissions import (
warehouse_view_required,
warehouse_manage_required,
@ -1649,6 +1649,7 @@ def purchase_order_from_pdf(request):
orig_pdf_name = pdf_file.name
try:
from .pdf_parser import PdfQuoteParser
parser = PdfQuoteParser(temp_pdf_path)
parse_result = parser.parse()
step = 'review'

View file

@ -44,5 +44,6 @@ boto3>=1.34.0
markdown
onnxruntime>=1.19.0
rapidocr-onnxruntime>=1.2.0
pypdfium2>=5.0.0
pywebpush>=2.0.0
py-vapid>=1.9.0