refactor: make PDF parser dependencies optional and lazy-load parser
This commit is contained in:
parent
de6828221b
commit
391966da07
3 changed files with 20 additions and 6 deletions
|
|
@ -12,8 +12,15 @@ import logging
|
|||
from io import BytesIO
|
||||
from typing import Dict, List, Any, Optional, Tuple
|
||||
|
||||
import pypdfium2 as pdfium
|
||||
import numpy as np
|
||||
try:
|
||||
import pypdfium2 as pdfium
|
||||
except ImportError:
|
||||
pdfium = None
|
||||
|
||||
try:
|
||||
import numpy as np
|
||||
except ImportError:
|
||||
np = None
|
||||
|
||||
from django.db.models import Q
|
||||
from .models import Supplier, Product, WarehouseLocation
|
||||
|
|
@ -91,8 +98,13 @@ class PdfQuoteParser:
|
|||
self._temp_path = None
|
||||
self.is_ocr = False
|
||||
|
||||
def _get_document(self) -> Tuple[pdfium.PdfDocument, str]:
|
||||
def _get_document(self) -> Tuple[Any, str]:
|
||||
"""Ouvre le document PDF via pypdfium2."""
|
||||
if pdfium is None:
|
||||
raise RuntimeError(
|
||||
"Le module 'pypdfium2' n'est pas installé sur le serveur. "
|
||||
"Veuillez exécuter 'pip install -r requirements/base.txt' pour activer l'analyse PDF locale."
|
||||
)
|
||||
if isinstance(self.file_or_path, str):
|
||||
return pdfium.PdfDocument(self.file_or_path), self.file_or_path
|
||||
elif hasattr(self.file_or_path, 'temporary_file_path'):
|
||||
|
|
@ -146,7 +158,7 @@ class PdfQuoteParser:
|
|||
pass
|
||||
self.close()
|
||||
|
||||
def _parse_vector(self, doc: pdfium.PdfDocument) -> Dict[str, Any]:
|
||||
def _parse_vector(self, doc: Any) -> Dict[str, Any]:
|
||||
"""Analyse un PDF avec couche texte vectorielle (recherche spatiale précise)."""
|
||||
full_text = ""
|
||||
for page in doc:
|
||||
|
|
@ -167,7 +179,7 @@ class PdfQuoteParser:
|
|||
'total_items': len(enriched_items),
|
||||
}
|
||||
|
||||
def _parse_with_ocr(self, doc: pdfium.PdfDocument) -> Dict[str, Any]:
|
||||
def _parse_with_ocr(self, doc: Any) -> Dict[str, Any]:
|
||||
"""Analyse un PDF scanné via RapidOCR."""
|
||||
try:
|
||||
from rapidocr_onnxruntime import RapidOCR
|
||||
|
|
|
|||
|
|
@ -64,7 +64,7 @@ from .models import (
|
|||
SerializedItemEvent,
|
||||
notify_supervisors_pending_product,
|
||||
)
|
||||
from .pdf_parser import PdfQuoteParser, normalize_unit
|
||||
from .pdf_parser import normalize_unit
|
||||
from .permissions import (
|
||||
warehouse_view_required,
|
||||
warehouse_manage_required,
|
||||
|
|
@ -1649,6 +1649,7 @@ def purchase_order_from_pdf(request):
|
|||
orig_pdf_name = pdf_file.name
|
||||
|
||||
try:
|
||||
from .pdf_parser import PdfQuoteParser
|
||||
parser = PdfQuoteParser(temp_pdf_path)
|
||||
parse_result = parser.parse()
|
||||
step = 'review'
|
||||
|
|
|
|||
|
|
@ -44,5 +44,6 @@ boto3>=1.34.0
|
|||
markdown
|
||||
onnxruntime>=1.19.0
|
||||
rapidocr-onnxruntime>=1.2.0
|
||||
pypdfium2>=5.0.0
|
||||
pywebpush>=2.0.0
|
||||
py-vapid>=1.9.0
|
||||
|
|
|
|||
Loading…
Reference in a new issue