545 lines
21 KiB
Python
545 lines
21 KiB
Python
"""
|
|
Service d'analyse locale de devis / offres au format PDF.
|
|
Utilise pypdfium2 pour l'extraction vectorielle directe (100% exacte, rapide)
|
|
et bascule sur RapidOCR (local CPU) en cas de PDF scanné / image.
|
|
"""
|
|
|
|
import os
|
|
import re
|
|
import tempfile
|
|
import difflib
|
|
import logging
|
|
from io import BytesIO
|
|
from typing import Dict, List, Any, Optional, Tuple
|
|
|
|
try:
|
|
import pypdfium2 as pdfium
|
|
except ImportError:
|
|
pdfium = None
|
|
|
|
try:
|
|
import numpy as np
|
|
except ImportError:
|
|
np = None
|
|
|
|
from django.db.models import Q
|
|
from .models import Supplier, Product, WarehouseLocation
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
UNIT_MAPPING = {
|
|
'pièce': Product.UNIT_PC,
|
|
'piece': Product.UNIT_PC,
|
|
'pièces': Product.UNIT_PC,
|
|
'pieces': Product.UNIT_PC,
|
|
'pcs': Product.UNIT_PC,
|
|
'pc': Product.UNIT_PC,
|
|
'stk': Product.UNIT_PC,
|
|
'un': Product.UNIT_PC,
|
|
'unité': Product.UNIT_PC,
|
|
'unite': Product.UNIT_PC,
|
|
'boîte': Product.UNIT_PC,
|
|
'boite': Product.UNIT_PC,
|
|
'bte': Product.UNIT_PC,
|
|
'sac': Product.UNIT_PC,
|
|
'sacs': Product.UNIT_PC,
|
|
'colis': Product.UNIT_PC,
|
|
'paquet': Product.UNIT_PC,
|
|
'kg': Product.UNIT_KG,
|
|
'kilogramme': Product.UNIT_KG,
|
|
'kilogrammes': Product.UNIT_KG,
|
|
'kilo': Product.UNIT_KG,
|
|
'g': Product.UNIT_G,
|
|
'gramme': Product.UNIT_G,
|
|
'grammes': Product.UNIT_G,
|
|
't': Product.UNIT_T,
|
|
'tonne': Product.UNIT_T,
|
|
'tonnes': Product.UNIT_T,
|
|
'l': Product.UNIT_L,
|
|
'litre': Product.UNIT_L,
|
|
'litres': Product.UNIT_L,
|
|
'ltr': Product.UNIT_L,
|
|
'm': Product.UNIT_M,
|
|
'mètre': Product.UNIT_M,
|
|
'metre': Product.UNIT_M,
|
|
'mètres': Product.UNIT_M,
|
|
'metres': Product.UNIT_M,
|
|
'ml': Product.UNIT_M,
|
|
'm2': Product.UNIT_M2,
|
|
'm²': Product.UNIT_M2,
|
|
'm3': Product.UNIT_M3,
|
|
'm³': Product.UNIT_M3,
|
|
}
|
|
|
|
|
|
def normalize_unit(raw_unit: str) -> str:
|
|
"""Normalise l'unité textuelle vers les choix reconnus par Loko Product."""
|
|
if not raw_unit:
|
|
return Product.UNIT_PC
|
|
clean = raw_unit.strip().lower()
|
|
return UNIT_MAPPING.get(clean, Product.UNIT_PC)
|
|
|
|
|
|
def clean_vat_number(raw_vat: str) -> str:
|
|
"""Nettoie un numéro de TVA / entreprise pour comparaison (ex: BE 0447.966.784 -> BE0447966784)."""
|
|
if not raw_vat:
|
|
return ""
|
|
return re.sub(r'[\s\.]', '', raw_vat).upper()
|
|
|
|
|
|
class PdfQuoteParser:
|
|
"""
|
|
Parseur local de devis / offres PDF pour la création de bons de commande.
|
|
"""
|
|
|
|
def __init__(self, file_or_path):
|
|
self.file_or_path = file_or_path
|
|
self._temp_path = None
|
|
self.is_ocr = False
|
|
|
|
def _get_document(self) -> Tuple[Any, str]:
|
|
"""Ouvre le document PDF via pypdfium2."""
|
|
if pdfium is None:
|
|
raise RuntimeError(
|
|
"Le module 'pypdfium2' n'est pas installé sur le serveur. "
|
|
"Veuillez exécuter 'pip install -r requirements/base.txt' pour activer l'analyse PDF locale."
|
|
)
|
|
if isinstance(self.file_or_path, str):
|
|
return pdfium.PdfDocument(self.file_or_path), self.file_or_path
|
|
elif hasattr(self.file_or_path, 'temporary_file_path'):
|
|
path = self.file_or_path.temporary_file_path()
|
|
return pdfium.PdfDocument(path), path
|
|
elif hasattr(self.file_or_path, 'read'):
|
|
content = self.file_or_path.read()
|
|
if hasattr(self.file_or_path, 'seek'):
|
|
self.file_or_path.seek(0)
|
|
fd, tmp_path = tempfile.mkstemp(suffix='.pdf')
|
|
with os.fdopen(fd, 'wb') as f:
|
|
f.write(content)
|
|
self._temp_path = tmp_path
|
|
return pdfium.PdfDocument(tmp_path), tmp_path
|
|
else:
|
|
raise ValueError("Type de fichier non supporté pour l'analyse PDF.")
|
|
|
|
def close(self):
|
|
"""Nettoie les fichiers temporaires si besoin."""
|
|
if self._temp_path and os.path.exists(self._temp_path):
|
|
try:
|
|
os.remove(self._temp_path)
|
|
except OSError:
|
|
pass
|
|
|
|
def parse(self) -> Dict[str, Any]:
|
|
"""
|
|
Extrait les métadonnées et la liste des articles d'un devis.
|
|
Effectue le rapprochement automatique avec le stock existant (fournisseur et produits).
|
|
"""
|
|
doc = None
|
|
try:
|
|
doc, path = self._get_document()
|
|
total_pages = len(doc)
|
|
if total_pages == 0:
|
|
raise ValueError("Le document PDF ne contient aucune page.")
|
|
|
|
# Vérifier si le document possède du texte vectoriel ou nécessite un OCR
|
|
total_char_count = sum(len(page.get_textpage().get_text_range()) for page in doc)
|
|
if total_char_count < 50:
|
|
logger.info(f"PDF sans couche texte suffisante ({total_char_count} chars). Utilisation du moteur OCR.")
|
|
self.is_ocr = True
|
|
return self._parse_with_ocr(doc)
|
|
else:
|
|
return self._parse_vector(doc)
|
|
finally:
|
|
if doc:
|
|
try:
|
|
doc.close()
|
|
except Exception:
|
|
pass
|
|
self.close()
|
|
|
|
def _parse_vector(self, doc: Any) -> Dict[str, Any]:
|
|
"""Analyse un PDF avec couche texte vectorielle (recherche spatiale précise)."""
|
|
full_text = ""
|
|
for page in doc:
|
|
full_text += page.get_textpage().get_text_range() + "\n"
|
|
|
|
metadata = self._extract_metadata(full_text, doc)
|
|
items = self._extract_items_vector(doc)
|
|
|
|
# Rapprochement avec le catalogue
|
|
matched_supplier = self._match_supplier(metadata)
|
|
enriched_items = self._match_products(items)
|
|
|
|
return {
|
|
'is_ocr': False,
|
|
'metadata': metadata,
|
|
'items': enriched_items,
|
|
'matched_supplier': matched_supplier,
|
|
'total_items': len(enriched_items),
|
|
}
|
|
|
|
def _parse_with_ocr(self, doc: Any) -> Dict[str, Any]:
|
|
"""Analyse un PDF scanné via RapidOCR."""
|
|
try:
|
|
from rapidocr_onnxruntime import RapidOCR
|
|
ocr = RapidOCR()
|
|
except ImportError:
|
|
logger.error("RapidOCR n'est pas disponible pour l'OCR local.")
|
|
raise RuntimeError("Le module OCR local n'est pas installé dans l'environnement.")
|
|
|
|
full_text_lines = []
|
|
ocr_pages_data = []
|
|
|
|
for p_idx, page in enumerate(doc):
|
|
pil_img = page.render(scale=2.0).to_pil()
|
|
ocr_res, _ = ocr(np.array(pil_img))
|
|
if not ocr_res:
|
|
continue
|
|
|
|
rects = []
|
|
for item in ocr_res:
|
|
box, txt, conf = item
|
|
if not txt.strip():
|
|
continue
|
|
left = min(pt[0] for pt in box)
|
|
right = max(pt[0] for pt in box)
|
|
top = min(pt[1] for pt in box)
|
|
bottom = max(pt[1] for pt in box)
|
|
# Invert y for bottom-up coordinate alignment
|
|
h = pil_img.height
|
|
rects.append((left, h - bottom, right, h - top, txt.strip()))
|
|
|
|
rects.sort(key=lambda x: -x[3])
|
|
page_lines = self._cluster_lines(rects)
|
|
ocr_pages_data.append(page_lines)
|
|
for l in page_lines:
|
|
full_text_lines.append(' '.join(item[4] for item in l))
|
|
|
|
full_text = "\n".join(full_text_lines)
|
|
metadata = self._extract_metadata(full_text, doc)
|
|
items = self._extract_items_from_clustered_lines(ocr_pages_data)
|
|
|
|
matched_supplier = self._match_supplier(metadata)
|
|
enriched_items = self._match_products(items)
|
|
|
|
return {
|
|
'is_ocr': True,
|
|
'metadata': metadata,
|
|
'items': enriched_items,
|
|
'matched_supplier': matched_supplier,
|
|
'total_items': len(enriched_items),
|
|
}
|
|
|
|
def _cluster_lines(self, rects: List[Tuple[float, float, float, float, str]], y_tol: float = 4.5) -> List[List[Tuple]]:
|
|
"""Regroupe les fragments de texte spatialement alignés sur la même ligne horizontale."""
|
|
lines = []
|
|
current_line = []
|
|
current_y = None
|
|
|
|
for r in rects:
|
|
y = (r[1] + r[3]) / 2.0
|
|
if current_y is None or abs(current_y - y) <= y_tol:
|
|
current_line.append(r)
|
|
current_y = y
|
|
else:
|
|
current_line.sort(key=lambda x: x[0])
|
|
lines.append(current_line)
|
|
current_line = [r]
|
|
current_y = y
|
|
if current_line:
|
|
current_line.sort(key=lambda x: x[0])
|
|
lines.append(current_line)
|
|
return lines
|
|
|
|
def _extract_metadata(self, full_text: str, doc: pdfium.PdfDocument) -> Dict[str, Any]:
|
|
"""Extrait les métadonnées de l'en-tête (Fournisseur, Devis n°, Date, Totaux)."""
|
|
meta = {
|
|
'supplier_name': '',
|
|
'supplier_vat': '',
|
|
'supplier_email': '',
|
|
'supplier_phone': '',
|
|
'supplier_address': '',
|
|
'quote_number': '',
|
|
'date': '',
|
|
'total_net': None,
|
|
'total_gross': None,
|
|
'currency': 'EUR',
|
|
'customer_name': '',
|
|
}
|
|
|
|
# 1. Numéro de TVA / Entreprise
|
|
vat_m = re.search(r'(?:TVA|BTW|VAT)\s*:\s*(BE\s*0?\d{3}[\.\s]?\d{3}[\.\s]?\d{3})', full_text, re.I)
|
|
if vat_m:
|
|
meta['supplier_vat'] = vat_m.group(1).strip()
|
|
|
|
# 2. Email
|
|
email_m = re.search(r'\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Z|a-z]{2,}\b', full_text)
|
|
if email_m:
|
|
meta['supplier_email'] = email_m.group(0).strip()
|
|
|
|
# 3. Téléphone
|
|
phone_m = re.search(r'(?:Tel|Tél|Phone)[\s:]*([+\d\s\.\(\)\/-]{8,25})', full_text, re.I)
|
|
if phone_m:
|
|
meta['supplier_phone'] = phone_m.group(1).strip().rstrip(' -')
|
|
|
|
# 4. Numéro de devis / offre (ex: "OF 2026000723/", "DEV-12345", etc.)
|
|
quote_m = re.search(r'\b(OF\s*[-#]?\s*\d+[\w\/-]*|DEV(?:IS)?\s*[-#]?\s*\d+[\w\/-]*|QUO(?:TE)?\s*[-#]?\s*\d+[\w\/-]*)\b', full_text, re.I)
|
|
if quote_m:
|
|
meta['quote_number'] = quote_m.group(0).strip().rstrip('/')
|
|
else:
|
|
bon_m = re.search(r'Bon\s+n[°o]\s*[\r\n\s]*([A-Z0-9\s\/-]{4,25})', full_text, re.I)
|
|
if bon_m:
|
|
meta['quote_number'] = bon_m.group(1).strip().rstrip('/')
|
|
|
|
# 5. Date
|
|
date_m = re.search(r'\b(\d{2}[/\.-]\d{2}[/\.-]\d{4})\b', full_text)
|
|
if date_m:
|
|
meta['date'] = date_m.group(1).replace('-', '/').replace('.', '/')
|
|
|
|
# 6. Nom du fournisseur
|
|
name_m = re.search(r'(?:sa|nv|srl|sprl|bvba|bv|s\.a\.|n\.v\.)\s+([A-Za-z\s0-9\.-]+?)(?:\r|\n|Rue|Avenue|Chaussée|Boulevard|Lentestraat)', full_text, re.I)
|
|
if name_m:
|
|
meta['supplier_name'] = name_m.group(0).strip()
|
|
else:
|
|
domain_m = re.search(r'([\w-]+\.(?:com|be|fr|eu))', full_text, re.I)
|
|
if domain_m:
|
|
meta['supplier_name'] = domain_m.group(1)
|
|
else:
|
|
meta['supplier_name'] = "Fournisseur inconnu"
|
|
|
|
# Adresse
|
|
addr_m = re.search(r'((?:Rue|Avenue|Boulevard|Chaussée|Straat|Lentestraat)[^\n\r]+[\d]{4}\s+[A-Za-zÀ-ÿ]+)', full_text, re.I)
|
|
if addr_m:
|
|
meta['supplier_address'] = addr_m.group(1).strip()
|
|
|
|
# 7. Totaux
|
|
tot_ht = re.search(r'Total\s+hors\s*(?:tva)?[\r\n\s]*([0-9\.,]+)', full_text, re.I)
|
|
if tot_ht:
|
|
try:
|
|
meta['total_net'] = float(tot_ht.group(1).replace('.', '').replace(',', '.'))
|
|
except ValueError:
|
|
pass
|
|
|
|
tot_ttc = re.search(r'Total\s+(?:EUR|TTC)[\r\n\s]*([0-9\.,]+)', full_text, re.I)
|
|
if tot_ttc:
|
|
try:
|
|
meta['total_gross'] = float(tot_ttc.group(1).replace('.', '').replace(',', '.'))
|
|
except ValueError:
|
|
pass
|
|
|
|
return meta
|
|
|
|
def _extract_items_vector(self, doc: pdfium.PdfDocument) -> List[Dict[str, Any]]:
|
|
"""Extrait les lignes d'articles via l'analyse géométrique vectorielle de chaque page."""
|
|
pages_data = []
|
|
|
|
for p_idx, page in enumerate(doc):
|
|
tp = page.get_textpage()
|
|
rects = []
|
|
for i in range(tp.count_rects()):
|
|
r = tp.get_rect(i)
|
|
txt = tp.get_text_bounded(*r).strip()
|
|
if txt:
|
|
rects.append((r[0], r[1], r[2], r[3], txt))
|
|
|
|
rects.sort(key=lambda x: -x[3])
|
|
|
|
# Délimiter verticalement la zone de tableau sur la page
|
|
table_top = None
|
|
table_bottom = 50.0
|
|
|
|
for r in rects:
|
|
if any(w in r[4] for w in ['Article', 'Quantité', 'Description']):
|
|
table_top = r[1] - 4.0
|
|
break
|
|
|
|
for r in rects:
|
|
if any(k in r[4] for k in ['Total hors', 'Total HT', 'Meilleures salutations', "D'avance je vous remercie"]):
|
|
if r[3] > table_bottom and (not table_top or r[3] < table_top):
|
|
table_bottom = max(table_bottom, r[3] + 4.0)
|
|
|
|
table_rects = [
|
|
r for r in rects
|
|
if (table_top is None or r[3] < table_top) and r[1] > table_bottom
|
|
]
|
|
|
|
lines = self._cluster_lines(table_rects)
|
|
pages_data.append(lines)
|
|
|
|
return self._extract_items_from_clustered_lines(pages_data)
|
|
|
|
def _extract_items_from_clustered_lines(self, pages_lines: List[List[List[Tuple]]]) -> List[Dict[str, Any]]:
|
|
"""Extrait les dictionnaires d'articles depuis les lignes regroupées."""
|
|
items = []
|
|
current_item = None
|
|
|
|
# Modèle de ligne produit :
|
|
# description | qté | unité | prix unitaire | [remise%] | total | tva%
|
|
row_pattern = re.compile(
|
|
r'^(.*?)\s*\|\s*(\d+[.,]\d+)\s*\|\s*([A-Za-zÀ-ÿ]+)\s*\|\s*(\d+[.,]\d+)\s*(?:\|\s*(\d+[.,]?\d*%)|\s*)\s*\|\s*(\d+[.,]\d+)\s*\|\s*(\d+%)'
|
|
)
|
|
|
|
stop_keywords = [
|
|
'Total hors', 'Total EUR', 'Total TTC', 'Total HT',
|
|
'Meilleures salutations', 'Dominique Vandevelde',
|
|
"D'avance je vous remercie", 'TVA - BTW', 'SERVICE PUBLIC',
|
|
'ITS - tools', 'N°. Article', 'Concerne', 'A l\'attention'
|
|
]
|
|
|
|
for page_lines in pages_lines:
|
|
for line in page_lines:
|
|
line_str = ' | '.join(item[4] for item in line)
|
|
|
|
if any(kw in line_str for kw in stop_keywords):
|
|
current_item = None
|
|
continue
|
|
|
|
m = row_pattern.search(line_str)
|
|
if m:
|
|
desc_part = m.group(1).strip()
|
|
qty_val = float(m.group(2).replace(',', '.'))
|
|
raw_unit = m.group(3).strip()
|
|
gross_price = float(m.group(4).replace(',', '.'))
|
|
discount_str = m.group(5) if m.group(5) else '0%'
|
|
total_val = float(m.group(6).replace(',', '.'))
|
|
vat_val = float(m.group(7).replace('%', ''))
|
|
|
|
# Séparer la référence fournisseur de la description
|
|
parts = desc_part.split(' | ')
|
|
if len(parts) >= 2:
|
|
ref = parts[0].strip()
|
|
name = ' '.join(parts[1:]).strip()
|
|
else:
|
|
tokens = desc_part.split(None, 1)
|
|
if len(tokens) == 2 and len(tokens[0]) <= 25 and any(c.isdigit() for c in tokens[0]):
|
|
ref = tokens[0]
|
|
name = tokens[1]
|
|
else:
|
|
ref = ''
|
|
name = desc_part
|
|
|
|
unit_price_net = round(total_val / qty_val, 2) if qty_val > 0 else gross_price
|
|
unit_code = normalize_unit(raw_unit)
|
|
|
|
current_item = {
|
|
'reference': ref,
|
|
'name': name,
|
|
'quantity': int(qty_val) if qty_val.is_integer() else qty_val,
|
|
'unit': unit_code,
|
|
'raw_unit': raw_unit,
|
|
'gross_unit_price': gross_price,
|
|
'discount': discount_str,
|
|
'unit_price': unit_price_net,
|
|
'total_amount': total_val,
|
|
'vat_rate': vat_val,
|
|
}
|
|
items.append(current_item)
|
|
elif current_item:
|
|
# Ligne de continuation de la description
|
|
if not line_str.startswith('http') and not line_str.startswith('https') and not '---' in line_str:
|
|
continuation = line_str.replace(' | ', ' ').strip()
|
|
if continuation and not any(k in continuation for k in ['RECUPEL', 'ING - BE95', 'BNP Paribas', '1/2', '2/2']):
|
|
current_item['name'] += ' ' + continuation
|
|
|
|
# Nettoyage final des libellés (suppression des indicateurs de pagination orphelins)
|
|
for it in items:
|
|
it['name'] = re.sub(r'\s*\b[1-9]/[1-9]\b\s*$', '', it['name']).strip()
|
|
# Nettoyer les paramètres URL résiduels éventuels
|
|
it['name'] = re.sub(r'categoryId=\d+.*', '', it['name']).strip()
|
|
|
|
return items
|
|
|
|
def _match_supplier(self, metadata: Dict[str, Any]) -> Optional[Supplier]:
|
|
"""Tente de retrouver un fournisseur existant dans la base de données."""
|
|
vat = metadata.get('supplier_vat')
|
|
if vat:
|
|
clean_vat = clean_vat_number(vat)
|
|
# Recherche par numéro d'entreprise / TVA
|
|
for s in Supplier.objects.filter(is_active=True):
|
|
if s.enterprise_number and clean_vat_number(s.enterprise_number) == clean_vat:
|
|
return s
|
|
|
|
email = metadata.get('supplier_email')
|
|
if email:
|
|
s = Supplier.objects.filter(is_active=True, contact_email__iexact=email).first()
|
|
if s:
|
|
return s
|
|
|
|
name = metadata.get('supplier_name')
|
|
if name and name != "Fournisseur inconnu":
|
|
s = Supplier.objects.filter(is_active=True, name__icontains=name[:20]).first()
|
|
if s:
|
|
return s
|
|
|
|
return None
|
|
|
|
def _match_products(self, items: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
|
|
"""
|
|
Rapproche chaque article extrait avec les produits du catalogue Loko.
|
|
Cherche par référence fournisseur, code interne, SKU, ou nom similaire.
|
|
"""
|
|
active_products = list(
|
|
Product.objects.filter(is_active=True).values('id', 'code', 'name', 'sku', 'price', 'unit', 'supplier_reference')
|
|
)
|
|
|
|
for item in items:
|
|
ref = item.get('reference', '').strip()
|
|
name = item.get('name', '').strip()
|
|
matched_id = None
|
|
matched_name = None
|
|
matched_code = None
|
|
match_type = None
|
|
|
|
# 1. Correspondance exacte sur référence fournisseur
|
|
if ref:
|
|
for p in active_products:
|
|
if p.get('supplier_reference') and p['supplier_reference'].strip().lower() == ref.lower():
|
|
matched_id = p['id']
|
|
matched_name = p['name']
|
|
matched_code = p['code']
|
|
match_type = 'exact_supplier_ref'
|
|
break
|
|
|
|
# 2. Correspondance exacte sur SKU ou Code interne
|
|
if not matched_id and ref:
|
|
for p in active_products:
|
|
if p['sku'].strip().lower() == ref.lower() or p['code'].strip().lower() == ref.lower():
|
|
matched_id = p['id']
|
|
matched_name = p['name']
|
|
matched_code = p['code']
|
|
match_type = 'exact_sku_or_code'
|
|
break
|
|
|
|
# 3. Correspondance exacte sur le nom
|
|
if not matched_id and name:
|
|
for p in active_products:
|
|
if p['name'].strip().lower() == name.lower():
|
|
matched_id = p['id']
|
|
matched_name = p['name']
|
|
matched_code = p['code']
|
|
match_type = 'exact_name'
|
|
break
|
|
|
|
# 4. Correspondance floue (similarité textuelle >= 80%)
|
|
if not matched_id and name:
|
|
best_ratio = 0.0
|
|
best_prod = None
|
|
for p in active_products:
|
|
ratio = difflib.SequenceMatcher(None, name.lower(), p['name'].lower()).ratio()
|
|
if ratio > best_ratio and ratio >= 0.80:
|
|
best_ratio = ratio
|
|
best_prod = p
|
|
if best_prod:
|
|
matched_id = best_prod['id']
|
|
matched_name = best_prod['name']
|
|
matched_code = best_prod['code']
|
|
match_type = 'fuzzy_name'
|
|
|
|
item['matched_product_id'] = matched_id
|
|
item['matched_product_name'] = matched_name
|
|
item['matched_product_code'] = matched_code
|
|
item['match_type'] = match_type
|
|
item['is_new'] = (matched_id is None)
|
|
|
|
return items
|