diff --git a/loko/stock/pdf_parser.py b/loko/stock/pdf_parser.py index bce6d35..9167dba 100644 --- a/loko/stock/pdf_parser.py +++ b/loko/stock/pdf_parser.py @@ -12,6 +12,13 @@ import logging from io import BytesIO from typing import Dict, List, Any, Optional, Tuple +import unicodedata + +try: + from rapidfuzz import fuzz +except ImportError: + fuzz = None + try: import pypdfium2 as pdfium except ImportError: @@ -88,6 +95,68 @@ def clean_vat_number(raw_vat: str) -> str: return re.sub(r'[\s\.]', '', raw_vat).upper() +STOP_WORDS = { + 'de', 'du', 'la', 'le', 'les', 'des', 'en', 'et', 'au', 'aux', 'd', 'l', + 'un', 'une', 'pour', 'par', 'sur', 'avec', 'sans', 'sous', 'dans', 'clb' +} + + +def clean_tokens_for_matching(text: str) -> Tuple[set, str]: + """ + Normalise le texte pour la comparaison sémantique de produits : + - Minuscules et décomposition d'accents + - Normalisation des dimensions et unités collées/séparées (5 m -> 5m, 25 mm -> 25mm, 1000 gr -> 1000g) + - Élimination de la ponctuation et caractères spéciaux + - Retrait des stop-words + Retourne (set_de_tokens, chaine_nettoyee). + """ + if not text: + return set(), "" + t = text.lower() + # Rapprocher les dimensions: '5 m' -> '5m', '25 mm' -> '25mm', '1000 gr' -> '1000g' + t = re.sub(r'(\d+)\s*(mm|cm|m|g|gr|kg|l|ml|v|w|ah|a)\b', r'\1\2', t) + t = re.sub(r'(\d+)gr\b', r'\1g', t) + t = unicodedata.normalize('NFKD', t) + t = ''.join(c for c in t if not unicodedata.combining(c)) + t = re.sub(r'[^a-z0-9]', ' ', t) + raw_tokens = t.split() + meaningful = [w for w in raw_tokens if w not in STOP_WORDS and len(w) > 1] + return set(meaningful), ' '.join(meaningful) + + +def compute_name_similarity(name_a: str, name_b: str) -> Tuple[float, set]: + """ + Calcule un score de similarité (0-100) basé sur les mots communs et la distance floue. + Permet d'associer des descriptions comme : + 'Mètre Ruban CLB Magnétique 5m x 25mm' et 'Mètre ruban, ABS, 5 m x 25 mm, jaune/noir' + """ + tok_a, str_a = clean_tokens_for_matching(name_a) + tok_b, str_b = clean_tokens_for_matching(name_b) + + if not tok_a or not tok_b: + return 0.0, set() + + common = tok_a.intersection(tok_b) + if not common: + return 0.0, set() + + overlap_a = len(common) / len(tok_a) + overlap_b = len(common) / len(tok_b) + max_overlap = max(overlap_a, overlap_b) + min_overlap = min(overlap_a, overlap_b) + + if fuzz is not None: + tsr = fuzz.token_set_ratio(str_a, str_b) + else: + tsr = difflib.SequenceMatcher(None, str_a, str_b).ratio() * 100 + + score = (max_overlap * 40.0) + (min_overlap * 20.0) + (tsr * 0.4) + if len(common) >= 3: + score = min(100.0, score + 10.0) + + return score, common + + class PdfQuoteParser: """ Parseur local de devis / offres PDF pour la création de bons de commande. @@ -477,7 +546,7 @@ class PdfQuoteParser: def _match_products(self, items: List[Dict[str, Any]]) -> List[Dict[str, Any]]: """ Rapproche chaque article extrait avec les produits du catalogue Loko. - Cherche par référence fournisseur, code interne, SKU, ou nom similaire. + Cherche par référence fournisseur, code interne, SKU, ou mots/similarité souple. """ active_products = list( Product.objects.filter(is_active=True).values('id', 'code', 'name', 'sku', 'price', 'unit', 'supplier_reference') @@ -490,6 +559,8 @@ class PdfQuoteParser: matched_name = None matched_code = None match_type = None + match_score = 0.0 + top_matches = [] # 1. Correspondance exacte sur référence fournisseur if ref: @@ -499,6 +570,7 @@ class PdfQuoteParser: matched_name = p['name'] matched_code = p['code'] match_type = 'exact_supplier_ref' + match_score = 100.0 break # 2. Correspondance exacte sur SKU ou Code interne @@ -509,37 +581,60 @@ class PdfQuoteParser: matched_name = p['name'] matched_code = p['code'] match_type = 'exact_sku_or_code' + match_score = 100.0 break - # 3. Correspondance exacte sur le nom + # 3. Correspondance exacte sur le nom (insensible casse et accents) if not matched_id and name: + _, clean_name = clean_tokens_for_matching(name) for p in active_products: - if p['name'].strip().lower() == name.lower(): + _, clean_pname = clean_tokens_for_matching(p['name']) + if clean_name and clean_name == clean_pname: matched_id = p['id'] matched_name = p['name'] matched_code = p['code'] match_type = 'exact_name' + match_score = 100.0 break - # 4. Correspondance floue (similarité textuelle >= 80%) - if not matched_id and name: - best_ratio = 0.0 - best_prod = None + # 4. Correspondance souple basée sur les mots présents et la similarité + candidates = [] + if name: for p in active_products: - ratio = difflib.SequenceMatcher(None, name.lower(), p['name'].lower()).ratio() - if ratio > best_ratio and ratio >= 0.80: - best_ratio = ratio - best_prod = p - if best_prod: - matched_id = best_prod['id'] - matched_name = best_prod['name'] - matched_code = best_prod['code'] - match_type = 'fuzzy_name' + if matched_id and p['id'] == matched_id: + continue + sim_score, common_words = compute_name_similarity(name, p['name']) + if sim_score >= 35.0: + candidates.append({ + 'id': p['id'], + 'code': p['code'], + 'name': p['name'], + 'score': round(sim_score, 1), + 'common_words': list(common_words), + }) + + # Trier les candidats par score décroissant + candidates.sort(key=lambda x: x['score'], reverse=True) + top_matches = candidates[:5] + + # Si pas de match exact, retenir le meilleur candidat au-dessus du seuil souple + if not matched_id and candidates: + best = candidates[0] + # Seuil assoupli : score >= 45% et au moins 1 mot significatif en commun + if best['score'] >= 45.0 and len(best['common_words']) >= 1: + matched_id = best['id'] + matched_name = best['name'] + matched_code = best['code'] + match_type = 'fuzzy_name' + match_score = best['score'] item['matched_product_id'] = matched_id item['matched_product_name'] = matched_name item['matched_product_code'] = matched_code item['match_type'] = match_type - item['is_new'] = (matched_id is None) + item['match_score'] = round(match_score) + item['top_matches'] = top_matches + item['is_matched'] = bool(matched_id) + item['is_new'] = not bool(matched_id) return items diff --git a/loko/stock/templates/stock/purchase_order_from_pdf.html b/loko/stock/templates/stock/purchase_order_from_pdf.html index d5a9c21..0b94c1b 100644 --- a/loko/stock/templates/stock/purchase_order_from_pdf.html +++ b/loko/stock/templates/stock/purchase_order_from_pdf.html @@ -102,7 +102,7 @@ {% elif step == 'review' %} -