diff --git a/loko/stock/pdf_parser.py b/loko/stock/pdf_parser.py
index 9167dba..2e9c3e2 100644
--- a/loko/stock/pdf_parser.py
+++ b/loko/stock/pdf_parser.py
@@ -29,6 +29,12 @@ try:
except ImportError:
np = None
+try:
+ from PIL import Image, ImageOps
+except ImportError:
+ Image = None
+ ImageOps = None
+
from django.db.models import Q
from .models import Supplier, Product, WarehouseLocation
@@ -159,13 +165,50 @@ def compute_name_similarity(name_a: str, name_b: str) -> Tuple[float, set]:
class PdfQuoteParser:
"""
- Parseur local de devis / offres PDF pour la création de bons de commande.
+ Parseur local de devis / offres au format PDF ou image (scan, photo).
+ - Utilise pypdfium2 pour l'extraction vectorielle directe sur PDF numérique (rapide et exact).
+ - Bascule automatiquement sur RapidOCR (moteur 100% local CPU) pour les PDF scannés et les photos/images.
"""
+ IMAGE_EXTENSIONS = ('.png', '.jpg', '.jpeg', '.webp', '.bmp', '.tiff', '.tif')
+
def __init__(self, file_or_path):
self.file_or_path = file_or_path
self._temp_path = None
self.is_ocr = False
+ self.is_image = False
+
+ def _is_image(self) -> bool:
+ """Détecte si le fichier fourni est une image ou une photo."""
+ if isinstance(self.file_or_path, str):
+ ext = os.path.splitext(self.file_or_path)[1].lower()
+ if ext in self.IMAGE_EXTENSIONS:
+ return True
+ elif hasattr(self.file_or_path, 'name'):
+ ext = os.path.splitext(self.file_or_path.name)[1].lower()
+ if ext in self.IMAGE_EXTENSIONS:
+ return True
+
+ # Détection par signature d'octets si disponible
+ if hasattr(self.file_or_path, 'read'):
+ header = self.file_or_path.read(16)
+ if hasattr(self.file_or_path, 'seek'):
+ self.file_or_path.seek(0)
+ if header.startswith(b'\x89PNG') or header.startswith(b'\xff\xd8\xff') or header.startswith(b'RIFF'):
+ return True
+ if header.startswith(b'%PDF'):
+ return False
+
+ if isinstance(self.file_or_path, str) and os.path.exists(self.file_or_path):
+ try:
+ with open(self.file_or_path, 'rb') as f:
+ header = f.read(16)
+ if header.startswith(b'\x89PNG') or header.startswith(b'\xff\xd8\xff') or header.startswith(b'RIFF'):
+ return True
+ except OSError:
+ pass
+
+ return False
def _get_document(self) -> Tuple[Any, str]:
"""Ouvre le document PDF via pypdfium2."""
@@ -201,9 +244,14 @@ class PdfQuoteParser:
def parse(self) -> Dict[str, Any]:
"""
- Extrait les métadonnées et la liste des articles d'un devis.
+ Extrait les métadonnées et la liste des articles d'un devis (PDF ou Image).
Effectue le rapprochement automatique avec le stock existant (fournisseur et produits).
"""
+ if self._is_image():
+ self.is_ocr = True
+ self.is_image = True
+ return self._parse_image()
+
doc = None
try:
doc, path = self._get_document()
@@ -214,7 +262,7 @@ class PdfQuoteParser:
# Vérifier si le document possède du texte vectoriel ou nécessite un OCR
total_char_count = sum(len(page.get_textpage().get_text_range()) for page in doc)
if total_char_count < 50:
- logger.info(f"PDF sans couche texte suffisante ({total_char_count} chars). Utilisation du moteur OCR.")
+ logger.info(f"PDF sans couche texte suffisante ({total_char_count} chars). Utilisation du moteur OCR local.")
self.is_ocr = True
return self._parse_with_ocr(doc)
else:
@@ -227,6 +275,78 @@ class PdfQuoteParser:
pass
self.close()
+ def _parse_image(self) -> Dict[str, Any]:
+ """Analyse directe d'une photo ou d'un scan d'offre via RapidOCR local."""
+ if Image is None:
+ raise RuntimeError("Le module PIL/Pillow n'est pas disponible pour l'analyse d'images.")
+
+ try:
+ from rapidocr_onnxruntime import RapidOCR
+ ocr = RapidOCR()
+ except ImportError:
+ logger.error("RapidOCR n'est pas disponible pour l'OCR local.")
+ raise RuntimeError("Le module OCR local n'est pas installé dans l'environnement.")
+
+ # Charger l'image avec gestion de l'orientation EXIF des smartphones
+ if isinstance(self.file_or_path, str):
+ pil_img = Image.open(self.file_or_path)
+ elif hasattr(self.file_or_path, 'temporary_file_path'):
+ pil_img = Image.open(self.file_or_path.temporary_file_path())
+ elif hasattr(self.file_or_path, 'read'):
+ pil_img = Image.open(self.file_or_path)
+ if hasattr(self.file_or_path, 'seek'):
+ self.file_or_path.seek(0)
+ else:
+ raise ValueError("Type d'image non supporté.")
+
+ if ImageOps is not None:
+ pil_img = ImageOps.exif_transpose(pil_img)
+ pil_img = pil_img.convert('RGB')
+
+ # Redimensionnement maîtrisé si image très volumineuse (ex: photo smartphone 48MP)
+ max_dim = max(pil_img.size)
+ if max_dim > 2500:
+ scale = 2500.0 / max_dim
+ new_size = (int(pil_img.width * scale), int(pil_img.height * scale))
+ pil_img = pil_img.resize(new_size, Image.Resampling.LANCZOS)
+
+ ocr_res, _ = ocr(np.array(pil_img))
+ rects = []
+ h = pil_img.height
+
+ if ocr_res:
+ for item in ocr_res:
+ box, txt, conf = item
+ if not txt.strip():
+ continue
+ left = min(pt[0] for pt in box)
+ right = max(pt[0] for pt in box)
+ top = min(pt[1] for pt in box)
+ bottom = max(pt[1] for pt in box)
+ rects.append((left, h - bottom, right, h - top, txt.strip()))
+
+ rects.sort(key=lambda x: -x[3])
+ avg_h = float(np.mean([r[3] - r[1] for r in rects])) if rects else 15.0
+ y_tol = max(8.0, avg_h * 0.45)
+
+ page_lines = self._cluster_lines(rects, y_tol=y_tol)
+ full_text = "\n".join(" ".join(r[4] for r in l) for l in page_lines)
+
+ metadata = self._extract_metadata(full_text)
+ items = self._extract_items_from_clustered_lines([page_lines])
+
+ matched_supplier = self._match_supplier(metadata)
+ enriched_items = self._match_products(items)
+
+ return {
+ 'is_ocr': True,
+ 'is_image': True,
+ 'metadata': metadata,
+ 'items': enriched_items,
+ 'matched_supplier': matched_supplier,
+ 'total_items': len(enriched_items),
+ }
+
def _parse_vector(self, doc: Any) -> Dict[str, Any]:
"""Analyse un PDF avec couche texte vectorielle (recherche spatiale précise)."""
full_text = ""
@@ -236,12 +356,12 @@ class PdfQuoteParser:
metadata = self._extract_metadata(full_text, doc)
items = self._extract_items_vector(doc)
- # Rapprochement avec le catalogue
matched_supplier = self._match_supplier(metadata)
enriched_items = self._match_products(items)
return {
'is_ocr': False,
+ 'is_image': False,
'metadata': metadata,
'items': enriched_items,
'matched_supplier': matched_supplier,
@@ -267,6 +387,7 @@ class PdfQuoteParser:
continue
rects = []
+ h = pil_img.height
for item in ocr_res:
box, txt, conf = item
if not txt.strip():
@@ -275,12 +396,13 @@ class PdfQuoteParser:
right = max(pt[0] for pt in box)
top = min(pt[1] for pt in box)
bottom = max(pt[1] for pt in box)
- # Invert y for bottom-up coordinate alignment
- h = pil_img.height
rects.append((left, h - bottom, right, h - top, txt.strip()))
rects.sort(key=lambda x: -x[3])
- page_lines = self._cluster_lines(rects)
+ avg_h = float(np.mean([r[3] - r[1] for r in rects])) if rects else 15.0
+ y_tol = max(8.0, avg_h * 0.45)
+
+ page_lines = self._cluster_lines(rects, y_tol=y_tol)
ocr_pages_data.append(page_lines)
for l in page_lines:
full_text_lines.append(' '.join(item[4] for item in l))
@@ -294,6 +416,7 @@ class PdfQuoteParser:
return {
'is_ocr': True,
+ 'is_image': False,
'metadata': metadata,
'items': enriched_items,
'matched_supplier': matched_supplier,
@@ -308,9 +431,11 @@ class PdfQuoteParser:
for r in rects:
y = (r[1] + r[3]) / 2.0
- if current_y is None or abs(current_y - y) <= y_tol:
+ h_box = r[3] - r[1]
+ tol = max(y_tol, h_box * 0.45)
+ if current_y is None or abs(current_y - y) <= tol:
current_line.append(r)
- current_y = y
+ current_y = (current_y + y) / 2.0 if current_y else y
else:
current_line.sort(key=lambda x: x[0])
lines.append(current_line)
@@ -321,7 +446,7 @@ class PdfQuoteParser:
lines.append(current_line)
return lines
- def _extract_metadata(self, full_text: str, doc: pdfium.PdfDocument) -> Dict[str, Any]:
+ def _extract_metadata(self, full_text: str, doc: Optional[Any] = None) -> Dict[str, Any]:
"""Extrait les métadonnées de l'en-tête (Fournisseur, Devis n°, Date, Totaux)."""
meta = {
'supplier_name': '',
@@ -338,7 +463,9 @@ class PdfQuoteParser:
}
# 1. Numéro de TVA / Entreprise
- vat_m = re.search(r'(?:TVA|BTW|VAT)\s*:\s*(BE\s*0?\d{3}[\.\s]?\d{3}[\.\s]?\d{3})', full_text, re.I)
+ vat_m = re.search(r'(?:TVA|BTW|VAT|N°\s*d\'entreprise)[\s.:-]*(BE\s*0?\d{3}[\.\s]?\d{3}[\.\s]?\d{3})', full_text, re.I)
+ if not vat_m:
+ vat_m = re.search(r'\b(BE\s*0?\d{3}[\.\s]?\d{3}[\.\s]?\d{3})\b', full_text, re.I)
if vat_m:
meta['supplier_vat'] = vat_m.group(1).strip()
@@ -348,12 +475,12 @@ class PdfQuoteParser:
meta['supplier_email'] = email_m.group(0).strip()
# 3. Téléphone
- phone_m = re.search(r'(?:Tel|Tél|Phone)[\s:]*([+\d\s\.\(\)\/-]{8,25})', full_text, re.I)
+ phone_m = re.search(r'(?:Tel|Tél|Phone|Gsm)[\s.:-]*([+\d\s\.\(\)\/-]{8,25})', full_text, re.I)
if phone_m:
meta['supplier_phone'] = phone_m.group(1).strip().rstrip(' -')
- # 4. Numéro de devis / offre (ex: "OF 2026000723/", "DEV-12345", etc.)
- quote_m = re.search(r'\b(OF\s*[-#]?\s*\d+[\w\/-]*|DEV(?:IS)?\s*[-#]?\s*\d+[\w\/-]*|QUO(?:TE)?\s*[-#]?\s*\d+[\w\/-]*)\b', full_text, re.I)
+ # 4. Numéro de devis / offre (ex: "OF 2026000723", "DEV-12345", etc.)
+ quote_m = re.search(r'\b(OF\s*[-#]?\s*\d+[\w\/-]*|DEV(?:IS)?\s*[-#]?\s*\d+[\w\/-]*|QUO(?:TE)?\s*[-#]?\s*\d+[\w\/-]*|OFFRE\s*[-#]?\s*\d+[\w\/-]*)\b', full_text, re.I)
if quote_m:
meta['quote_number'] = quote_m.group(0).strip().rstrip('/')
else:
@@ -367,15 +494,35 @@ class PdfQuoteParser:
meta['date'] = date_m.group(1).replace('-', '/').replace('.', '/')
# 6. Nom du fournisseur
- name_m = re.search(r'(?:sa|nv|srl|sprl|bvba|bv|s\.a\.|n\.v\.)\s+([A-Za-z\s0-9\.-]+?)(?:\r|\n|Rue|Avenue|Chaussée|Boulevard|Lentestraat)', full_text, re.I)
- if name_m:
- meta['supplier_name'] = name_m.group(0).strip()
- else:
- domain_m = re.search(r'([\w-]+\.(?:com|be|fr|eu))', full_text, re.I)
- if domain_m:
- meta['supplier_name'] = domain_m.group(1)
+ LEGAL = r'(?:s\.?a\.?|n\.?v\.?|s\.?r\.?l\.?|s\.?p\.?r\.?l\.?|b\.?v\.?b\.?a\.?|b\.?v\.?|s\.?a\.?s\.?|s\.?a\.?r\.?l\.?)'
+ for line in full_text.split('\n')[:30]:
+ line_s = line.strip()
+ # Chercher une ligne contenant une forme légale avec nom substantiel
+ m1 = re.search(rf'^\b{LEGAL}\b\s+([A-Za-z0-9\ \.-]{{3,60}})', line_s, re.I)
+ if m1:
+ cand = m1.group(1).strip()
+ cand_clean = re.sub(rf'\b{LEGAL}\b', '', cand, flags=re.I).strip(' .-')
+ if len(cand_clean) >= 4:
+ meta['supplier_name'] = line_s
+ break
+ m2 = re.search(rf'^([A-Za-z0-9\ \.-]{{3,60}})\s+\b{LEGAL}\b', line_s, re.I)
+ if m2:
+ cand = m2.group(1).strip()
+ cand_clean = re.sub(rf'\b{LEGAL}\b', '', cand, flags=re.I).strip(' .-')
+ if len(cand_clean) >= 4:
+ meta['supplier_name'] = line_s
+ break
+
+ if not meta['supplier_name']:
+ name_m = re.search(r'(?:sa|nv|srl|sprl|bvba|bv|s\.a\.|n\.v\.)\s+([A-Za-z\s0-9\.-]+?)(?:\r|\n|Rue|Avenue|Chaussée|Boulevard|Lentestraat)', full_text, re.I)
+ if name_m:
+ meta['supplier_name'] = name_m.group(0).strip()
else:
- meta['supplier_name'] = "Fournisseur inconnu"
+ domain_m = re.search(r'([\w-]+\.(?:com|be|fr|eu))', full_text, re.I)
+ if domain_m:
+ meta['supplier_name'] = domain_m.group(1)
+ else:
+ meta['supplier_name'] = ""
# Adresse
addr_m = re.search(r'((?:Rue|Avenue|Boulevard|Chaussée|Straat|Lentestraat)[^\n\r]+[\d]{4}\s+[A-Za-zÀ-ÿ]+)', full_text, re.I)
@@ -414,7 +561,6 @@ class PdfQuoteParser:
rects.sort(key=lambda x: -x[3])
- # Délimiter verticalement la zone de tableau sur la page
table_top = None
table_bottom = 50.0
@@ -443,11 +589,18 @@ class PdfQuoteParser:
items = []
current_item = None
- # Modèle de ligne produit :
+ # Modèle 1 (vectoriel classique) :
# description | qté | unité | prix unitaire | [remise%] | total | tva%
row_pattern = re.compile(
r'^(.*?)\s*\|\s*(\d+[.,]\d+)\s*\|\s*([A-Za-zÀ-ÿ]+)\s*\|\s*(\d+[.,]\d+)\s*(?:\|\s*(\d+[.,]?\d*%)|\s*)\s*\|\s*(\d+[.,]\d+)\s*\|\s*(\d+%)'
)
+ # Modèle 2 (OCR / Scan souple où qté+unité ou total+tva peuvent être regroupés sans pipe) :
+ ocr_row_base = re.compile(
+ r'^(.*?)\s*\|\s*(\d+[.,]\d+)\s*(?:\|\s*|\s+)([A-Za-zÀ-ÿ]+)\s*\|\s*(\d+[.,]\d+)(.*?)$'
+ )
+ ocr_row_tail = re.compile(
+ r'(?:\|\s*(\d+[.,]?\d*%)\s*)?\|\s*(\d+[.,]\d+)(?:[\|\s]+(\d+%))?'
+ )
stop_keywords = [
'Total hors', 'Total EUR', 'Total TTC', 'Total HT',
@@ -473,7 +626,23 @@ class PdfQuoteParser:
discount_str = m.group(5) if m.group(5) else '0%'
total_val = float(m.group(6).replace(',', '.'))
vat_val = float(m.group(7).replace('%', ''))
+ else:
+ m_ocr = ocr_row_base.match(line_str)
+ if m_ocr:
+ desc_part = m_ocr.group(1).strip()
+ qty_val = float(m_ocr.group(2).replace(',', '.'))
+ raw_unit = m_ocr.group(3).strip()
+ gross_price = float(m_ocr.group(4).replace(',', '.'))
+ tail = m_ocr.group(5)
+ tail_m = ocr_row_tail.search(tail) if tail else None
+ discount_str = tail_m.group(1) if tail_m and tail_m.group(1) else '0%'
+ total_val = float(tail_m.group(2).replace(',', '.')) if tail_m and tail_m.group(2) else gross_price * qty_val
+ vat_val = float(tail_m.group(3).replace('%', '')) if tail_m and tail_m.group(3) else 21.0
+ m = m_ocr
+ else:
+ m = None
+ if m:
# Séparer la référence fournisseur de la description
parts = desc_part.split(' | ')
if len(parts) >= 2:
@@ -511,10 +680,9 @@ class PdfQuoteParser:
if continuation and not any(k in continuation for k in ['RECUPEL', 'ING - BE95', 'BNP Paribas', '1/2', '2/2']):
current_item['name'] += ' ' + continuation
- # Nettoyage final des libellés (suppression des indicateurs de pagination orphelins)
+ # Nettoyage final des libellés
for it in items:
it['name'] = re.sub(r'\s*\b[1-9]/[1-9]\b\s*$', '', it['name']).strip()
- # Nettoyer les paramètres URL résiduels éventuels
it['name'] = re.sub(r'categoryId=\d+.*', '', it['name']).strip()
return items
@@ -638,3 +806,7 @@ class PdfQuoteParser:
item['is_new'] = not bool(matched_id)
return items
+
+
+# Alias pour usage universel (PDF, scans, photos)
+QuoteDocumentParser = PdfQuoteParser
diff --git a/loko/stock/templates/stock/purchase_order_from_pdf.html b/loko/stock/templates/stock/purchase_order_from_pdf.html
index 0b94c1b..3d3337b 100644
--- a/loko/stock/templates/stock/purchase_order_from_pdf.html
+++ b/loko/stock/templates/stock/purchase_order_from_pdf.html
@@ -1,7 +1,7 @@
{% extends "base.html" %}
{% load i18n static %}
-{% block title %}{% translate "Créer un bon de commande depuis une offre PDF" %} — Loko{% endblock %}
+{% block title %}{% translate "Créer un bon de commande depuis une offre (PDF, Scan, Photo)" %} — Loko{% endblock %}
{% block content %}
@@ -11,12 +11,13 @@
- {% translate "Créer un bon de commande depuis une offre PDF" %}
+
+ {% translate "Créer un bon de commande depuis une offre (PDF, Scan, Photo)" %}
@@ -27,7 +28,7 @@
{% if step == 'upload' %}
-
+
@@ -40,7 +41,7 @@
{% translate "Importer une offre ou un devis fournisseur" %}
- {% translate "Déposez le fichier PDF reçu du fournisseur. Le système analyse automatiquement le document, extrait les articles, compare avec le stock existant et prépare le bon de commande." %}
+ {% translate "Déposez un devis au format PDF, un scan ou une photo prise avec un smartphone. Le système analyse automatiquement le document avec reconnaissance OCR locale, extrait les articles, compare avec le stock existant et prépare le bon de commande." %}
@@ -63,16 +64,22 @@
-
+
-
- {% translate "Cliquez ou glissez-déposez le devis PDF ici" %}
- {% translate "Formats acceptés : PDF numérique ou scanné (OCR automatique)" %}
-
+
+
+
+
+ {% translate "Cliquez ou glissez-déposez le document ici" %}
+ {% translate "Formats acceptés : PDF (numérique ou scanné), photos & scans (JPG, PNG, WEBP, TIFF)" %}
+
+ {% translate "Reconnaissance OCR 100% locale" %}
+
+
@@ -83,8 +90,8 @@
- {% translate "Règle de gestion des articles :" %}
- {% translate "Les articles trouvés dans le stock Loko seront directement rattachés. Pour tout article non répertorié, vous pourrez confirmer sa création en tant que nouveau produit avec un statut spécial 'En attente de validation superviseur'." %}
+ {% translate "Règle de gestion des articles & du fournisseur :" %}
+ {% translate "Les articles du devis sont rapprochés avec les produits existants de votre catalogue Loko. Si le fournisseur n'est pas encore enregistré, vous pourrez confirmer sa création directement lors de la revue." %}
+ {% translate "Fournisseur non trouvé dans le répertoire Loko" %}
+ {% translate "Le fournisseur détecté n'a pas été retrouvé. Souhaitez-vous le créer dans Loko ou rattacher un fournisseur existant ?" %}