refactor: make PDF parser dependencies optional and lazy-load parser

This commit is contained in:
kdeterme 2026-10-01 13:57:12 +02:00
parent de6828221b
commit 391966da07
3 changed files with 20 additions and 6 deletions

View file

@ -12,8 +12,15 @@ import logging
from io import BytesIO from io import BytesIO
from typing import Dict, List, Any, Optional, Tuple from typing import Dict, List, Any, Optional, Tuple
try:
import pypdfium2 as pdfium import pypdfium2 as pdfium
except ImportError:
pdfium = None
try:
import numpy as np import numpy as np
except ImportError:
np = None
from django.db.models import Q from django.db.models import Q
from .models import Supplier, Product, WarehouseLocation from .models import Supplier, Product, WarehouseLocation
@ -91,8 +98,13 @@ class PdfQuoteParser:
self._temp_path = None self._temp_path = None
self.is_ocr = False self.is_ocr = False
def _get_document(self) -> Tuple[pdfium.PdfDocument, str]: def _get_document(self) -> Tuple[Any, str]:
"""Ouvre le document PDF via pypdfium2.""" """Ouvre le document PDF via pypdfium2."""
if pdfium is None:
raise RuntimeError(
"Le module 'pypdfium2' n'est pas installé sur le serveur. "
"Veuillez exécuter 'pip install -r requirements/base.txt' pour activer l'analyse PDF locale."
)
if isinstance(self.file_or_path, str): if isinstance(self.file_or_path, str):
return pdfium.PdfDocument(self.file_or_path), self.file_or_path return pdfium.PdfDocument(self.file_or_path), self.file_or_path
elif hasattr(self.file_or_path, 'temporary_file_path'): elif hasattr(self.file_or_path, 'temporary_file_path'):
@ -146,7 +158,7 @@ class PdfQuoteParser:
pass pass
self.close() self.close()
def _parse_vector(self, doc: pdfium.PdfDocument) -> Dict[str, Any]: def _parse_vector(self, doc: Any) -> Dict[str, Any]:
"""Analyse un PDF avec couche texte vectorielle (recherche spatiale précise).""" """Analyse un PDF avec couche texte vectorielle (recherche spatiale précise)."""
full_text = "" full_text = ""
for page in doc: for page in doc:
@ -167,7 +179,7 @@ class PdfQuoteParser:
'total_items': len(enriched_items), 'total_items': len(enriched_items),
} }
def _parse_with_ocr(self, doc: pdfium.PdfDocument) -> Dict[str, Any]: def _parse_with_ocr(self, doc: Any) -> Dict[str, Any]:
"""Analyse un PDF scanné via RapidOCR.""" """Analyse un PDF scanné via RapidOCR."""
try: try:
from rapidocr_onnxruntime import RapidOCR from rapidocr_onnxruntime import RapidOCR

View file

@ -64,7 +64,7 @@ from .models import (
SerializedItemEvent, SerializedItemEvent,
notify_supervisors_pending_product, notify_supervisors_pending_product,
) )
from .pdf_parser import PdfQuoteParser, normalize_unit from .pdf_parser import normalize_unit
from .permissions import ( from .permissions import (
warehouse_view_required, warehouse_view_required,
warehouse_manage_required, warehouse_manage_required,
@ -1649,6 +1649,7 @@ def purchase_order_from_pdf(request):
orig_pdf_name = pdf_file.name orig_pdf_name = pdf_file.name
try: try:
from .pdf_parser import PdfQuoteParser
parser = PdfQuoteParser(temp_pdf_path) parser = PdfQuoteParser(temp_pdf_path)
parse_result = parser.parse() parse_result = parser.parse()
step = 'review' step = 'review'

View file

@ -44,5 +44,6 @@ boto3>=1.34.0
markdown markdown
onnxruntime>=1.19.0 onnxruntime>=1.19.0
rapidocr-onnxruntime>=1.2.0 rapidocr-onnxruntime>=1.2.0
pypdfium2>=5.0.0
pywebpush>=2.0.0 pywebpush>=2.0.0
py-vapid>=1.9.0 py-vapid>=1.9.0