diff --git a/loko/stock/pdf_parser.py b/loko/stock/pdf_parser.py
index efad984..a53d8b1 100644
--- a/loko/stock/pdf_parser.py
+++ b/loko/stock/pdf_parser.py
@@ -191,36 +191,42 @@ class PdfQuoteParser:
IMAGE_EXTENSIONS = ('.png', '.jpg', '.jpeg', '.webp', '.bmp', '.tiff', '.tif')
- def __init__(self, file_or_path):
- self.file_or_path = file_or_path
- self._temp_path = None
+ def __init__(self, files_or_paths):
+ if isinstance(files_or_paths, (list, tuple)):
+ self.files = list(files_or_paths)
+ elif files_or_paths:
+ self.files = [files_or_paths]
+ else:
+ self.files = []
+
+ self.file_or_path = self.files[0] if self.files else None
+ self._temp_paths = []
self.is_ocr = False
self.is_image = False
- def _is_image(self) -> bool:
- """Détecte si le fichier fourni est une image ou une photo."""
- if isinstance(self.file_or_path, str):
- ext = os.path.splitext(self.file_or_path)[1].lower()
+ def _is_file_image(self, target) -> bool:
+ """Vérifie si un fichier ou chemin cible est une image."""
+ if isinstance(target, str):
+ ext = os.path.splitext(target)[1].lower()
if ext in self.IMAGE_EXTENSIONS:
return True
- elif hasattr(self.file_or_path, 'name'):
- ext = os.path.splitext(self.file_or_path.name)[1].lower()
+ elif hasattr(target, 'name'):
+ ext = os.path.splitext(target.name)[1].lower()
if ext in self.IMAGE_EXTENSIONS:
return True
- # Détection par signature d'octets si disponible
- if hasattr(self.file_or_path, 'read'):
- header = self.file_or_path.read(16)
- if hasattr(self.file_or_path, 'seek'):
- self.file_or_path.seek(0)
+ if hasattr(target, 'read'):
+ header = target.read(16)
+ if hasattr(target, 'seek'):
+ target.seek(0)
if header.startswith(b'\x89PNG') or header.startswith(b'\xff\xd8\xff') or header.startswith(b'RIFF'):
return True
if header.startswith(b'%PDF'):
return False
- if isinstance(self.file_or_path, str) and os.path.exists(self.file_or_path):
+ if isinstance(target, str) and os.path.exists(target):
try:
- with open(self.file_or_path, 'rb') as f:
+ with open(target, 'rb') as f:
header = f.read(16)
if header.startswith(b'\x89PNG') or header.startswith(b'\xff\xd8\xff') or header.startswith(b'RIFF'):
return True
@@ -229,47 +235,92 @@ class PdfQuoteParser:
return False
+ def _is_image(self) -> bool:
+ """Détecte si les fichiers fournis sont des images ou des photos."""
+ if not self.files:
+ return False
+ return all(self._is_file_image(f) for f in self.files)
+
def _get_document(self) -> Tuple[Any, str]:
- """Ouvre le document PDF via pypdfium2."""
+ """Ouvre le ou les documents PDF via pypdfium2."""
if pdfium is None:
raise RuntimeError(
"Le module 'pypdfium2' n'est pas installé sur le serveur. "
"Veuillez exécuter 'pip install -r requirements/base.txt' pour activer l'analyse PDF locale."
)
- if isinstance(self.file_or_path, str):
- return pdfium.PdfDocument(self.file_or_path), self.file_or_path
- elif hasattr(self.file_or_path, 'temporary_file_path'):
- path = self.file_or_path.temporary_file_path()
- return pdfium.PdfDocument(path), path
- elif hasattr(self.file_or_path, 'read'):
- content = self.file_or_path.read()
- if hasattr(self.file_or_path, 'seek'):
- self.file_or_path.seek(0)
- fd, tmp_path = tempfile.mkstemp(suffix='.pdf')
- with os.fdopen(fd, 'wb') as f:
- f.write(content)
- self._temp_path = tmp_path
- return pdfium.PdfDocument(tmp_path), tmp_path
- else:
- raise ValueError("Type de fichier non supporté pour l'analyse PDF.")
+ if not self.files:
+ raise ValueError("Aucun fichier fourni pour l'analyse PDF.")
+
+ if len(self.files) == 1:
+ f = self.files[0]
+ if isinstance(f, str):
+ return pdfium.PdfDocument(f), f
+ elif hasattr(f, 'temporary_file_path'):
+ path = f.temporary_file_path()
+ return pdfium.PdfDocument(path), path
+ elif hasattr(f, 'read'):
+ content = f.read()
+ if hasattr(f, 'seek'):
+ f.seek(0)
+ fd, tmp_path = tempfile.mkstemp(suffix='.pdf')
+ with os.fdopen(fd, 'wb') as out_f:
+ out_f.write(content)
+ self._temp_paths.append(tmp_path)
+ return pdfium.PdfDocument(tmp_path), tmp_path
+ else:
+ raise ValueError("Type de fichier non supporté pour l'analyse PDF.")
+
+ # Plusieurs fichiers PDF : on fusionne leurs pages dans un document unique
+ merged_doc = pdfium.PdfDocument.new()
+ first_path = ""
+ for f in self.files:
+ sub_doc = None
+ if isinstance(f, str):
+ sub_doc = pdfium.PdfDocument(f)
+ if not first_path:
+ first_path = f
+ elif hasattr(f, 'temporary_file_path'):
+ path = f.temporary_file_path()
+ sub_doc = pdfium.PdfDocument(path)
+ if not first_path:
+ first_path = path
+ elif hasattr(f, 'read'):
+ content = f.read()
+ if hasattr(f, 'seek'):
+ f.seek(0)
+ fd, tmp_path = tempfile.mkstemp(suffix='.pdf')
+ with os.fdopen(fd, 'wb') as out_f:
+ out_f.write(content)
+ self._temp_paths.append(tmp_path)
+ sub_doc = pdfium.PdfDocument(tmp_path)
+ if not first_path:
+ first_path = tmp_path
+
+ if sub_doc:
+ merged_doc.import_pages(sub_doc)
+ sub_doc.close()
+
+ return merged_doc, first_path
def close(self):
"""Nettoie les fichiers temporaires si besoin."""
- if self._temp_path and os.path.exists(self._temp_path):
- try:
- os.remove(self._temp_path)
- except OSError:
- pass
+ for p in self._temp_paths:
+ if p and os.path.exists(p):
+ try:
+ os.remove(p)
+ except OSError:
+ pass
+ self._temp_paths = []
def parse(self) -> Dict[str, Any]:
"""
- Extrait les métadonnées et la liste des articles d'un devis (PDF ou Image).
+ Extrait les métadonnées et la liste des articles d'un devis (PDF ou Image(s)).
Effectue le rapprochement automatique avec le stock existant (fournisseur et produits).
"""
if self._is_image():
self.is_ocr = True
self.is_image = True
- return self._parse_image()
+ return self._parse_images()
doc = None
try:
@@ -294,8 +345,8 @@ class PdfQuoteParser:
pass
self.close()
- def _parse_image(self) -> Dict[str, Any]:
- """Analyse directe d'une photo ou d'un scan d'offre via RapidOCR local."""
+ def _parse_images(self) -> Dict[str, Any]:
+ """Analyse directe d'une ou plusieurs photos ou scans d'offre via RapidOCR local."""
if Image is None:
raise RuntimeError("Le module PIL/Pillow n'est pas disponible pour l'analyse d'images.")
@@ -306,53 +357,62 @@ class PdfQuoteParser:
logger.error("RapidOCR n'est pas disponible pour l'OCR local.")
raise RuntimeError("Le module OCR local n'est pas installé dans l'environnement.")
- # Charger l'image avec gestion de l'orientation EXIF des smartphones
- if isinstance(self.file_or_path, str):
- pil_img = Image.open(self.file_or_path)
- elif hasattr(self.file_or_path, 'temporary_file_path'):
- pil_img = Image.open(self.file_or_path.temporary_file_path())
- elif hasattr(self.file_or_path, 'read'):
- pil_img = Image.open(self.file_or_path)
- if hasattr(self.file_or_path, 'seek'):
- self.file_or_path.seek(0)
- else:
- raise ValueError("Type d'image non supporté.")
+ ocr_pages_data = []
+ full_text_lines = []
- if ImageOps is not None:
- pil_img = ImageOps.exif_transpose(pil_img)
- pil_img = pil_img.convert('RGB')
+ for p_idx, f_item in enumerate(self.files, start=1):
+ if isinstance(f_item, str):
+ pil_img = Image.open(f_item)
+ elif hasattr(f_item, 'temporary_file_path'):
+ pil_img = Image.open(f_item.temporary_file_path())
+ elif hasattr(f_item, 'read'):
+ pil_img = Image.open(f_item)
+ if hasattr(f_item, 'seek'):
+ f_item.seek(0)
+ else:
+ raise ValueError(f"Type d'image non supporté : {type(f_item)}")
- # Redimensionnement maîtrisé si image très volumineuse (ex: photo smartphone 48MP)
- max_dim = max(pil_img.size)
- if max_dim > 2500:
- scale = 2500.0 / max_dim
- new_size = (int(pil_img.width * scale), int(pil_img.height * scale))
- pil_img = pil_img.resize(new_size, Image.Resampling.LANCZOS)
+ # Charger l'image avec gestion de l'orientation EXIF des smartphones
+ if ImageOps is not None:
+ pil_img = ImageOps.exif_transpose(pil_img)
+ pil_img = pil_img.convert('RGB')
- pil_img, ocr_res = self._detect_and_deskew_image(pil_img, ocr)
- rects = []
- h = pil_img.height
+ # Redimensionnement maîtrisé si image très volumineuse (ex: photo smartphone 48MP)
+ max_dim = max(pil_img.size)
+ if max_dim > 2500:
+ scale = 2500.0 / max_dim
+ new_size = (int(pil_img.width * scale), int(pil_img.height * scale))
+ pil_img = pil_img.resize(new_size, Image.Resampling.LANCZOS)
- if ocr_res:
- for item in ocr_res:
- box, txt, conf = item
- if not txt.strip():
- continue
- left = min(pt[0] for pt in box)
- right = max(pt[0] for pt in box)
- top = min(pt[1] for pt in box)
- bottom = max(pt[1] for pt in box)
- rects.append((left, h - bottom, right, h - top, txt.strip()))
+ pil_img, ocr_res = self._detect_and_deskew_image(pil_img, ocr)
+ rects = []
+ h = pil_img.height
- rects.sort(key=lambda x: -x[3])
- avg_h = float(np.mean([r[3] - r[1] for r in rects])) if rects else 15.0
- y_tol = max(6.0, avg_h * 0.40)
+ if ocr_res:
+ for item in ocr_res:
+ box, txt, conf = item
+ if not txt.strip():
+ continue
+ left = min(pt[0] for pt in box)
+ right = max(pt[0] for pt in box)
+ top = min(pt[1] for pt in box)
+ bottom = max(pt[1] for pt in box)
+ rects.append((left, h - bottom, right, h - top, txt.strip()))
- page_lines = self._cluster_lines(rects, y_tol=y_tol)
- full_text = "\n".join(" ".join(r[4] for r in l) for l in page_lines)
+ rects.sort(key=lambda x: -x[3])
+ avg_h = float(np.mean([r[3] - r[1] for r in rects])) if rects else 15.0
+ y_tol = max(6.0, avg_h * 0.40)
+
+ page_lines = self._cluster_lines(rects, y_tol=y_tol)
+ ocr_pages_data.append(page_lines)
+ for l in page_lines:
+ full_text_lines.append(' '.join(item[4] for item in l))
+ full_text_lines.append(f"--- Page {p_idx} ---")
+
+ full_text = "\n".join(full_text_lines)
metadata = self._extract_metadata(full_text)
- items = self._extract_items_from_clustered_lines([page_lines])
+ items = self._extract_items_from_clustered_lines(ocr_pages_data)
matched_supplier = self._match_supplier(metadata)
enriched_items = self._match_products(items)
@@ -360,12 +420,19 @@ class PdfQuoteParser:
return {
'is_ocr': True,
'is_image': True,
+ 'is_multi_image': len(self.files) > 1,
+ 'image_count': len(self.files),
+ 'page_count': len(self.files),
'metadata': metadata,
'items': enriched_items,
'matched_supplier': matched_supplier,
'total_items': len(enriched_items),
}
+ def _parse_image(self) -> Dict[str, Any]:
+ """Méthode de compatibilité pour analyse d'image unique."""
+ return self._parse_images()
+
@staticmethod
def _detect_and_deskew_image(pil_img: Any, ocr: Any) -> Tuple[Any, Any]:
"""
diff --git a/loko/stock/templates/stock/purchase_order_from_pdf.html b/loko/stock/templates/stock/purchase_order_from_pdf.html
index 3d3337b..213c3a3 100644
--- a/loko/stock/templates/stock/purchase_order_from_pdf.html
+++ b/loko/stock/templates/stock/purchase_order_from_pdf.html
@@ -41,7 +41,7 @@
{% translate "Importer une offre ou un devis fournisseur" %}
- {% translate "Déposez un devis au format PDF, un scan ou une photo prise avec un smartphone. Le système analyse automatiquement le document avec reconnaissance OCR locale, extrait les articles, compare avec le stock existant et prépare le bon de commande." %}
+ {% translate "Déposez un devis au format PDF, un scan ou des photos prises avec un smartphone (sélectionnez toutes les pages en une fois). Le système combine et analyse automatiquement les pages avec reconnaissance OCR locale, extrait les articles et prépare le bon de commande." %}
@@ -64,22 +64,22 @@
-
+
- {% translate "Cliquez ou glissez-déposez le document ici" %}
- {% translate "Formats acceptés : PDF (numérique ou scanné), photos & scans (JPG, PNG, WEBP, TIFF)" %}
+ {% translate "Cliquez ou glissez-déposez vos documents ou photos ici" %}
+ {% translate "Formats acceptés : PDF, photos & scans (JPG, PNG, WEBP, TIFF) — Sélection multiple de photos autorisée pour plusieurs pages" %}
{% translate "Reconnaissance OCR 100% locale" %}
-
+