refactor: make PDF parser dependencies optional and lazy-load parser

This commit is contained in:
kdeterme 2026-10-01 13:57:12 +02:00
parent de6828221b
commit 391966da07
3 changed files with 20 additions and 6 deletions

View file

@ -12,8 +12,15 @@ import logging
from io import BytesIO from io import BytesIO
from typing import Dict, List, Any, Optional, Tuple from typing import Dict, List, Any, Optional, Tuple
import pypdfium2 as pdfium try:
import numpy as np import pypdfium2 as pdfium
except ImportError:
pdfium = None
try:
import numpy as np
except ImportError:
np = None
from django.db.models import Q from django.db.models import Q
from .models import Supplier, Product, WarehouseLocation from .models import Supplier, Product, WarehouseLocation
@ -91,8 +98,13 @@ class PdfQuoteParser:
self._temp_path = None self._temp_path = None
self.is_ocr = False self.is_ocr = False
def _get_document(self) -> Tuple[pdfium.PdfDocument, str]: def _get_document(self) -> Tuple[Any, str]:
"""Ouvre le document PDF via pypdfium2.""" """Ouvre le document PDF via pypdfium2."""
if pdfium is None:
raise RuntimeError(
"Le module 'pypdfium2' n'est pas installé sur le serveur. "
"Veuillez exécuter 'pip install -r requirements/base.txt' pour activer l'analyse PDF locale."
)
if isinstance(self.file_or_path, str): if isinstance(self.file_or_path, str):
return pdfium.PdfDocument(self.file_or_path), self.file_or_path return pdfium.PdfDocument(self.file_or_path), self.file_or_path
elif hasattr(self.file_or_path, 'temporary_file_path'): elif hasattr(self.file_or_path, 'temporary_file_path'):
@ -146,7 +158,7 @@ class PdfQuoteParser:
pass pass
self.close() self.close()
def _parse_vector(self, doc: pdfium.PdfDocument) -> Dict[str, Any]: def _parse_vector(self, doc: Any) -> Dict[str, Any]:
"""Analyse un PDF avec couche texte vectorielle (recherche spatiale précise).""" """Analyse un PDF avec couche texte vectorielle (recherche spatiale précise)."""
full_text = "" full_text = ""
for page in doc: for page in doc:
@ -167,7 +179,7 @@ class PdfQuoteParser:
'total_items': len(enriched_items), 'total_items': len(enriched_items),
} }
def _parse_with_ocr(self, doc: pdfium.PdfDocument) -> Dict[str, Any]: def _parse_with_ocr(self, doc: Any) -> Dict[str, Any]:
"""Analyse un PDF scanné via RapidOCR.""" """Analyse un PDF scanné via RapidOCR."""
try: try:
from rapidocr_onnxruntime import RapidOCR from rapidocr_onnxruntime import RapidOCR

View file

@ -64,7 +64,7 @@ from .models import (
SerializedItemEvent, SerializedItemEvent,
notify_supervisors_pending_product, notify_supervisors_pending_product,
) )
from .pdf_parser import PdfQuoteParser, normalize_unit from .pdf_parser import normalize_unit
from .permissions import ( from .permissions import (
warehouse_view_required, warehouse_view_required,
warehouse_manage_required, warehouse_manage_required,
@ -1649,6 +1649,7 @@ def purchase_order_from_pdf(request):
orig_pdf_name = pdf_file.name orig_pdf_name = pdf_file.name
try: try:
from .pdf_parser import PdfQuoteParser
parser = PdfQuoteParser(temp_pdf_path) parser = PdfQuoteParser(temp_pdf_path)
parse_result = parser.parse() parse_result = parser.parse()
step = 'review' step = 'review'

View file

@ -44,5 +44,6 @@ boto3>=1.34.0
markdown markdown
onnxruntime>=1.19.0 onnxruntime>=1.19.0
rapidocr-onnxruntime>=1.2.0 rapidocr-onnxruntime>=1.2.0
pypdfium2>=5.0.0
pywebpush>=2.0.0 pywebpush>=2.0.0
py-vapid>=1.9.0 py-vapid>=1.9.0