loko/loko/stock/pdf_parser.py

545 lines
21 KiB
Python

"""
Service d'analyse locale de devis / offres au format PDF.
Utilise pypdfium2 pour l'extraction vectorielle directe (100% exacte, rapide)
et bascule sur RapidOCR (local CPU) en cas de PDF scanné / image.
"""
import os
import re
import tempfile
import difflib
import logging
from io import BytesIO
from typing import Dict, List, Any, Optional, Tuple
try:
import pypdfium2 as pdfium
except ImportError:
pdfium = None
try:
import numpy as np
except ImportError:
np = None
from django.db.models import Q
from .models import Supplier, Product, WarehouseLocation
logger = logging.getLogger(__name__)
UNIT_MAPPING = {
'pièce': Product.UNIT_PC,
'piece': Product.UNIT_PC,
'pièces': Product.UNIT_PC,
'pieces': Product.UNIT_PC,
'pcs': Product.UNIT_PC,
'pc': Product.UNIT_PC,
'stk': Product.UNIT_PC,
'un': Product.UNIT_PC,
'unité': Product.UNIT_PC,
'unite': Product.UNIT_PC,
'boîte': Product.UNIT_PC,
'boite': Product.UNIT_PC,
'bte': Product.UNIT_PC,
'sac': Product.UNIT_PC,
'sacs': Product.UNIT_PC,
'colis': Product.UNIT_PC,
'paquet': Product.UNIT_PC,
'kg': Product.UNIT_KG,
'kilogramme': Product.UNIT_KG,
'kilogrammes': Product.UNIT_KG,
'kilo': Product.UNIT_KG,
'g': Product.UNIT_G,
'gramme': Product.UNIT_G,
'grammes': Product.UNIT_G,
't': Product.UNIT_T,
'tonne': Product.UNIT_T,
'tonnes': Product.UNIT_T,
'l': Product.UNIT_L,
'litre': Product.UNIT_L,
'litres': Product.UNIT_L,
'ltr': Product.UNIT_L,
'm': Product.UNIT_M,
'mètre': Product.UNIT_M,
'metre': Product.UNIT_M,
'mètres': Product.UNIT_M,
'metres': Product.UNIT_M,
'ml': Product.UNIT_M,
'm2': Product.UNIT_M2,
'm²': Product.UNIT_M2,
'm3': Product.UNIT_M3,
'm³': Product.UNIT_M3,
}
def normalize_unit(raw_unit: str) -> str:
"""Normalise l'unité textuelle vers les choix reconnus par Loko Product."""
if not raw_unit:
return Product.UNIT_PC
clean = raw_unit.strip().lower()
return UNIT_MAPPING.get(clean, Product.UNIT_PC)
def clean_vat_number(raw_vat: str) -> str:
"""Nettoie un numéro de TVA / entreprise pour comparaison (ex: BE 0447.966.784 -> BE0447966784)."""
if not raw_vat:
return ""
return re.sub(r'[\s\.]', '', raw_vat).upper()
class PdfQuoteParser:
"""
Parseur local de devis / offres PDF pour la création de bons de commande.
"""
def __init__(self, file_or_path):
self.file_or_path = file_or_path
self._temp_path = None
self.is_ocr = False
def _get_document(self) -> Tuple[Any, str]:
"""Ouvre le document PDF via pypdfium2."""
if pdfium is None:
raise RuntimeError(
"Le module 'pypdfium2' n'est pas installé sur le serveur. "
"Veuillez exécuter 'pip install -r requirements/base.txt' pour activer l'analyse PDF locale."
)
if isinstance(self.file_or_path, str):
return pdfium.PdfDocument(self.file_or_path), self.file_or_path
elif hasattr(self.file_or_path, 'temporary_file_path'):
path = self.file_or_path.temporary_file_path()
return pdfium.PdfDocument(path), path
elif hasattr(self.file_or_path, 'read'):
content = self.file_or_path.read()
if hasattr(self.file_or_path, 'seek'):
self.file_or_path.seek(0)
fd, tmp_path = tempfile.mkstemp(suffix='.pdf')
with os.fdopen(fd, 'wb') as f:
f.write(content)
self._temp_path = tmp_path
return pdfium.PdfDocument(tmp_path), tmp_path
else:
raise ValueError("Type de fichier non supporté pour l'analyse PDF.")
def close(self):
"""Nettoie les fichiers temporaires si besoin."""
if self._temp_path and os.path.exists(self._temp_path):
try:
os.remove(self._temp_path)
except OSError:
pass
def parse(self) -> Dict[str, Any]:
"""
Extrait les métadonnées et la liste des articles d'un devis.
Effectue le rapprochement automatique avec le stock existant (fournisseur et produits).
"""
doc = None
try:
doc, path = self._get_document()
total_pages = len(doc)
if total_pages == 0:
raise ValueError("Le document PDF ne contient aucune page.")
# Vérifier si le document possède du texte vectoriel ou nécessite un OCR
total_char_count = sum(len(page.get_textpage().get_text_range()) for page in doc)
if total_char_count < 50:
logger.info(f"PDF sans couche texte suffisante ({total_char_count} chars). Utilisation du moteur OCR.")
self.is_ocr = True
return self._parse_with_ocr(doc)
else:
return self._parse_vector(doc)
finally:
if doc:
try:
doc.close()
except Exception:
pass
self.close()
def _parse_vector(self, doc: Any) -> Dict[str, Any]:
"""Analyse un PDF avec couche texte vectorielle (recherche spatiale précise)."""
full_text = ""
for page in doc:
full_text += page.get_textpage().get_text_range() + "\n"
metadata = self._extract_metadata(full_text, doc)
items = self._extract_items_vector(doc)
# Rapprochement avec le catalogue
matched_supplier = self._match_supplier(metadata)
enriched_items = self._match_products(items)
return {
'is_ocr': False,
'metadata': metadata,
'items': enriched_items,
'matched_supplier': matched_supplier,
'total_items': len(enriched_items),
}
def _parse_with_ocr(self, doc: Any) -> Dict[str, Any]:
"""Analyse un PDF scanné via RapidOCR."""
try:
from rapidocr_onnxruntime import RapidOCR
ocr = RapidOCR()
except ImportError:
logger.error("RapidOCR n'est pas disponible pour l'OCR local.")
raise RuntimeError("Le module OCR local n'est pas installé dans l'environnement.")
full_text_lines = []
ocr_pages_data = []
for p_idx, page in enumerate(doc):
pil_img = page.render(scale=2.0).to_pil()
ocr_res, _ = ocr(np.array(pil_img))
if not ocr_res:
continue
rects = []
for item in ocr_res:
box, txt, conf = item
if not txt.strip():
continue
left = min(pt[0] for pt in box)
right = max(pt[0] for pt in box)
top = min(pt[1] for pt in box)
bottom = max(pt[1] for pt in box)
# Invert y for bottom-up coordinate alignment
h = pil_img.height
rects.append((left, h - bottom, right, h - top, txt.strip()))
rects.sort(key=lambda x: -x[3])
page_lines = self._cluster_lines(rects)
ocr_pages_data.append(page_lines)
for l in page_lines:
full_text_lines.append(' '.join(item[4] for item in l))
full_text = "\n".join(full_text_lines)
metadata = self._extract_metadata(full_text, doc)
items = self._extract_items_from_clustered_lines(ocr_pages_data)
matched_supplier = self._match_supplier(metadata)
enriched_items = self._match_products(items)
return {
'is_ocr': True,
'metadata': metadata,
'items': enriched_items,
'matched_supplier': matched_supplier,
'total_items': len(enriched_items),
}
def _cluster_lines(self, rects: List[Tuple[float, float, float, float, str]], y_tol: float = 4.5) -> List[List[Tuple]]:
"""Regroupe les fragments de texte spatialement alignés sur la même ligne horizontale."""
lines = []
current_line = []
current_y = None
for r in rects:
y = (r[1] + r[3]) / 2.0
if current_y is None or abs(current_y - y) <= y_tol:
current_line.append(r)
current_y = y
else:
current_line.sort(key=lambda x: x[0])
lines.append(current_line)
current_line = [r]
current_y = y
if current_line:
current_line.sort(key=lambda x: x[0])
lines.append(current_line)
return lines
def _extract_metadata(self, full_text: str, doc: pdfium.PdfDocument) -> Dict[str, Any]:
"""Extrait les métadonnées de l'en-tête (Fournisseur, Devis n°, Date, Totaux)."""
meta = {
'supplier_name': '',
'supplier_vat': '',
'supplier_email': '',
'supplier_phone': '',
'supplier_address': '',
'quote_number': '',
'date': '',
'total_net': None,
'total_gross': None,
'currency': 'EUR',
'customer_name': '',
}
# 1. Numéro de TVA / Entreprise
vat_m = re.search(r'(?:TVA|BTW|VAT)\s*:\s*(BE\s*0?\d{3}[\.\s]?\d{3}[\.\s]?\d{3})', full_text, re.I)
if vat_m:
meta['supplier_vat'] = vat_m.group(1).strip()
# 2. Email
email_m = re.search(r'\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Z|a-z]{2,}\b', full_text)
if email_m:
meta['supplier_email'] = email_m.group(0).strip()
# 3. Téléphone
phone_m = re.search(r'(?:Tel|Tél|Phone)[\s:]*([+\d\s\.\(\)\/-]{8,25})', full_text, re.I)
if phone_m:
meta['supplier_phone'] = phone_m.group(1).strip().rstrip(' -')
# 4. Numéro de devis / offre (ex: "OF 2026000723/", "DEV-12345", etc.)
quote_m = re.search(r'\b(OF\s*[-#]?\s*\d+[\w\/-]*|DEV(?:IS)?\s*[-#]?\s*\d+[\w\/-]*|QUO(?:TE)?\s*[-#]?\s*\d+[\w\/-]*)\b', full_text, re.I)
if quote_m:
meta['quote_number'] = quote_m.group(0).strip().rstrip('/')
else:
bon_m = re.search(r'Bon\s+n[°o]\s*[\r\n\s]*([A-Z0-9\s\/-]{4,25})', full_text, re.I)
if bon_m:
meta['quote_number'] = bon_m.group(1).strip().rstrip('/')
# 5. Date
date_m = re.search(r'\b(\d{2}[/\.-]\d{2}[/\.-]\d{4})\b', full_text)
if date_m:
meta['date'] = date_m.group(1).replace('-', '/').replace('.', '/')
# 6. Nom du fournisseur
name_m = re.search(r'(?:sa|nv|srl|sprl|bvba|bv|s\.a\.|n\.v\.)\s+([A-Za-z\s0-9\.-]+?)(?:\r|\n|Rue|Avenue|Chaussée|Boulevard|Lentestraat)', full_text, re.I)
if name_m:
meta['supplier_name'] = name_m.group(0).strip()
else:
domain_m = re.search(r'([\w-]+\.(?:com|be|fr|eu))', full_text, re.I)
if domain_m:
meta['supplier_name'] = domain_m.group(1)
else:
meta['supplier_name'] = "Fournisseur inconnu"
# Adresse
addr_m = re.search(r'((?:Rue|Avenue|Boulevard|Chaussée|Straat|Lentestraat)[^\n\r]+[\d]{4}\s+[A-Za-zÀ-ÿ]+)', full_text, re.I)
if addr_m:
meta['supplier_address'] = addr_m.group(1).strip()
# 7. Totaux
tot_ht = re.search(r'Total\s+hors\s*(?:tva)?[\r\n\s]*([0-9\.,]+)', full_text, re.I)
if tot_ht:
try:
meta['total_net'] = float(tot_ht.group(1).replace('.', '').replace(',', '.'))
except ValueError:
pass
tot_ttc = re.search(r'Total\s+(?:EUR|TTC)[\r\n\s]*([0-9\.,]+)', full_text, re.I)
if tot_ttc:
try:
meta['total_gross'] = float(tot_ttc.group(1).replace('.', '').replace(',', '.'))
except ValueError:
pass
return meta
def _extract_items_vector(self, doc: pdfium.PdfDocument) -> List[Dict[str, Any]]:
"""Extrait les lignes d'articles via l'analyse géométrique vectorielle de chaque page."""
pages_data = []
for p_idx, page in enumerate(doc):
tp = page.get_textpage()
rects = []
for i in range(tp.count_rects()):
r = tp.get_rect(i)
txt = tp.get_text_bounded(*r).strip()
if txt:
rects.append((r[0], r[1], r[2], r[3], txt))
rects.sort(key=lambda x: -x[3])
# Délimiter verticalement la zone de tableau sur la page
table_top = None
table_bottom = 50.0
for r in rects:
if any(w in r[4] for w in ['Article', 'Quantité', 'Description']):
table_top = r[1] - 4.0
break
for r in rects:
if any(k in r[4] for k in ['Total hors', 'Total HT', 'Meilleures salutations', "D'avance je vous remercie"]):
if r[3] > table_bottom and (not table_top or r[3] < table_top):
table_bottom = max(table_bottom, r[3] + 4.0)
table_rects = [
r for r in rects
if (table_top is None or r[3] < table_top) and r[1] > table_bottom
]
lines = self._cluster_lines(table_rects)
pages_data.append(lines)
return self._extract_items_from_clustered_lines(pages_data)
def _extract_items_from_clustered_lines(self, pages_lines: List[List[List[Tuple]]]) -> List[Dict[str, Any]]:
"""Extrait les dictionnaires d'articles depuis les lignes regroupées."""
items = []
current_item = None
# Modèle de ligne produit :
# description | qté | unité | prix unitaire | [remise%] | total | tva%
row_pattern = re.compile(
r'^(.*?)\s*\|\s*(\d+[.,]\d+)\s*\|\s*([A-Za-zÀ-ÿ]+)\s*\|\s*(\d+[.,]\d+)\s*(?:\|\s*(\d+[.,]?\d*%)|\s*)\s*\|\s*(\d+[.,]\d+)\s*\|\s*(\d+%)'
)
stop_keywords = [
'Total hors', 'Total EUR', 'Total TTC', 'Total HT',
'Meilleures salutations', 'Dominique Vandevelde',
"D'avance je vous remercie", 'TVA - BTW', 'SERVICE PUBLIC',
'ITS - tools', 'N°. Article', 'Concerne', 'A l\'attention'
]
for page_lines in pages_lines:
for line in page_lines:
line_str = ' | '.join(item[4] for item in line)
if any(kw in line_str for kw in stop_keywords):
current_item = None
continue
m = row_pattern.search(line_str)
if m:
desc_part = m.group(1).strip()
qty_val = float(m.group(2).replace(',', '.'))
raw_unit = m.group(3).strip()
gross_price = float(m.group(4).replace(',', '.'))
discount_str = m.group(5) if m.group(5) else '0%'
total_val = float(m.group(6).replace(',', '.'))
vat_val = float(m.group(7).replace('%', ''))
# Séparer la référence fournisseur de la description
parts = desc_part.split(' | ')
if len(parts) >= 2:
ref = parts[0].strip()
name = ' '.join(parts[1:]).strip()
else:
tokens = desc_part.split(None, 1)
if len(tokens) == 2 and len(tokens[0]) <= 25 and any(c.isdigit() for c in tokens[0]):
ref = tokens[0]
name = tokens[1]
else:
ref = ''
name = desc_part
unit_price_net = round(total_val / qty_val, 2) if qty_val > 0 else gross_price
unit_code = normalize_unit(raw_unit)
current_item = {
'reference': ref,
'name': name,
'quantity': int(qty_val) if qty_val.is_integer() else qty_val,
'unit': unit_code,
'raw_unit': raw_unit,
'gross_unit_price': gross_price,
'discount': discount_str,
'unit_price': unit_price_net,
'total_amount': total_val,
'vat_rate': vat_val,
}
items.append(current_item)
elif current_item:
# Ligne de continuation de la description
if not line_str.startswith('http') and not line_str.startswith('https') and not '---' in line_str:
continuation = line_str.replace(' | ', ' ').strip()
if continuation and not any(k in continuation for k in ['RECUPEL', 'ING - BE95', 'BNP Paribas', '1/2', '2/2']):
current_item['name'] += ' ' + continuation
# Nettoyage final des libellés (suppression des indicateurs de pagination orphelins)
for it in items:
it['name'] = re.sub(r'\s*\b[1-9]/[1-9]\b\s*$', '', it['name']).strip()
# Nettoyer les paramètres URL résiduels éventuels
it['name'] = re.sub(r'categoryId=\d+.*', '', it['name']).strip()
return items
def _match_supplier(self, metadata: Dict[str, Any]) -> Optional[Supplier]:
"""Tente de retrouver un fournisseur existant dans la base de données."""
vat = metadata.get('supplier_vat')
if vat:
clean_vat = clean_vat_number(vat)
# Recherche par numéro d'entreprise / TVA
for s in Supplier.objects.filter(is_active=True):
if s.enterprise_number and clean_vat_number(s.enterprise_number) == clean_vat:
return s
email = metadata.get('supplier_email')
if email:
s = Supplier.objects.filter(is_active=True, contact_email__iexact=email).first()
if s:
return s
name = metadata.get('supplier_name')
if name and name != "Fournisseur inconnu":
s = Supplier.objects.filter(is_active=True, name__icontains=name[:20]).first()
if s:
return s
return None
def _match_products(self, items: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
"""
Rapproche chaque article extrait avec les produits du catalogue Loko.
Cherche par référence fournisseur, code interne, SKU, ou nom similaire.
"""
active_products = list(
Product.objects.filter(is_active=True).values('id', 'code', 'name', 'sku', 'price', 'unit', 'supplier_reference')
)
for item in items:
ref = item.get('reference', '').strip()
name = item.get('name', '').strip()
matched_id = None
matched_name = None
matched_code = None
match_type = None
# 1. Correspondance exacte sur référence fournisseur
if ref:
for p in active_products:
if p.get('supplier_reference') and p['supplier_reference'].strip().lower() == ref.lower():
matched_id = p['id']
matched_name = p['name']
matched_code = p['code']
match_type = 'exact_supplier_ref'
break
# 2. Correspondance exacte sur SKU ou Code interne
if not matched_id and ref:
for p in active_products:
if p['sku'].strip().lower() == ref.lower() or p['code'].strip().lower() == ref.lower():
matched_id = p['id']
matched_name = p['name']
matched_code = p['code']
match_type = 'exact_sku_or_code'
break
# 3. Correspondance exacte sur le nom
if not matched_id and name:
for p in active_products:
if p['name'].strip().lower() == name.lower():
matched_id = p['id']
matched_name = p['name']
matched_code = p['code']
match_type = 'exact_name'
break
# 4. Correspondance floue (similarité textuelle >= 80%)
if not matched_id and name:
best_ratio = 0.0
best_prod = None
for p in active_products:
ratio = difflib.SequenceMatcher(None, name.lower(), p['name'].lower()).ratio()
if ratio > best_ratio and ratio >= 0.80:
best_ratio = ratio
best_prod = p
if best_prod:
matched_id = best_prod['id']
matched_name = best_prod['name']
matched_code = best_prod['code']
match_type = 'fuzzy_name'
item['matched_product_id'] = matched_id
item['matched_product_name'] = matched_name
item['matched_product_code'] = matched_code
item['match_type'] = match_type
item['is_new'] = (matched_id is None)
return items