From 4010dfb13844ef1a20af8ed4ff2cc3bbbb2980b5 Mon Sep 17 00:00:00 2001 From: kdeterme Date: Thu, 1 Oct 2026 15:12:56 +0200 Subject: [PATCH] fix(stock): prevent merged quote items via auto-deskew, fixed baseline clustering and item splitting - Add auto-deskewing for tilted smartphone photos and document scans - Anchor line clustering to initial baseline to avoid vertical centroid drift - Normalize OCR numerical artifacts (semicolon decimals, glued VAT percentages) - Purge catalog URLs from product names to improve catalog matching - Implement fused item splitter to rescue embedded articles from descriptions - Add regression tests for item splitting and OCR normalization --- loko/stock/pdf_parser.py | 250 ++++++++++++++++++++++++++++++++++++--- loko/stock/tests.py | 46 +++++++ 2 files changed, 278 insertions(+), 18 deletions(-) diff --git a/loko/stock/pdf_parser.py b/loko/stock/pdf_parser.py index 2e9c3e2..efad984 100644 --- a/loko/stock/pdf_parser.py +++ b/loko/stock/pdf_parser.py @@ -101,6 +101,25 @@ def clean_vat_number(raw_vat: str) -> str: return re.sub(r'[\s\.]', '', raw_vat).upper() +ARTICLE_CODE_PATTERN = re.compile( + r'\b([A-Z]{2,4}\s*[-.]?\s*\d{4,8}|[A-Z]{1,3}\d{3,8}[A-Z0-9\.]*|[A-Z][0-9]\d{3,8}[A-Z0-9\.]*|\b\d{4,8}\b)\b' +) +CATALOG_URL_PATTERN = re.compile( + r'(?:https?://|//|tps://|www\.)[^\s]+(?:\.pdf|\.pdl|\.po|\.html|[a-z0-9/_-]*)', re.I +) + + +def normalize_ocr_text(text: str) -> str: + """Normalise les imperfections OCR courantes dans une ligne de tableau.""" + # Corriger les points-virgules pris pour des virgules décimales (ex: '1;00' -> '1,00') + t = re.sub(r'(\d+);(\d{2})\b', r'\1,\2', text) + # Décoller un montant et une TVA (ex: '64,3021%' -> '64,30 21%' ou '406,5621%' -> '406,56 21%') + t = re.sub(r'(\d+[.,]\d{2})(\d{1,2}%)', r'\1 \2', t) + # Décoller quantité et unité (ex: '12,00Piece' -> '12,00 Piece') + t = re.sub(r'(\d+[.,]\d+)([A-Za-zÀ-ÿ]{3,})', r'\1 \2', t) + return t + + STOP_WORDS = { 'de', 'du', 'la', 'le', 'les', 'des', 'en', 'et', 'au', 'aux', 'd', 'l', 'un', 'une', 'pour', 'par', 'sur', 'avec', 'sans', 'sous', 'dans', 'clb' @@ -310,7 +329,7 @@ class PdfQuoteParser: new_size = (int(pil_img.width * scale), int(pil_img.height * scale)) pil_img = pil_img.resize(new_size, Image.Resampling.LANCZOS) - ocr_res, _ = ocr(np.array(pil_img)) + pil_img, ocr_res = self._detect_and_deskew_image(pil_img, ocr) rects = [] h = pil_img.height @@ -327,7 +346,7 @@ class PdfQuoteParser: rects.sort(key=lambda x: -x[3]) avg_h = float(np.mean([r[3] - r[1] for r in rects])) if rects else 15.0 - y_tol = max(8.0, avg_h * 0.45) + y_tol = max(6.0, avg_h * 0.40) page_lines = self._cluster_lines(rects, y_tol=y_tol) full_text = "\n".join(" ".join(r[4] for r in l) for l in page_lines) @@ -347,6 +366,35 @@ class PdfQuoteParser: 'total_items': len(enriched_items), } + @staticmethod + def _detect_and_deskew_image(pil_img: Any, ocr: Any) -> Tuple[Any, Any]: + """ + Détecte l'inclinaison des lignes de texte (photos smartphone ou scans de travers) + et redresse automatiquement l'image si l'angle dépasse 0.35° pour assurer un alignement horizontal parfait. + """ + ocr_res, _ = ocr(np.array(pil_img)) + if not ocr_res or np is None: + return pil_img, ocr_res + + angles = [] + for box, txt, conf in ocr_res: + dx = box[1][0] - box[0][0] + dy = box[1][1] - box[0][1] + bw = np.hypot(dx, dy) + bh = np.hypot(box[3][0] - box[0][0], box[3][1] - box[0][1]) + if bw > bh * 2.5: + ang = np.degrees(np.arctan2(dy, dx)) + if abs(ang) < 45.0: + angles.append(ang) + + median_angle = float(np.median(angles)) if angles else 0.0 + if abs(median_angle) >= 0.35: + logger.info(f"Redressement de l'image de {median_angle:.2f}° pour aligner les lignes OCR.") + pil_img = pil_img.rotate(median_angle, expand=True, fillcolor='white') + ocr_res, _ = ocr(np.array(pil_img)) + + return pil_img, ocr_res + def _parse_vector(self, doc: Any) -> Dict[str, Any]: """Analyse un PDF avec couche texte vectorielle (recherche spatiale précise).""" full_text = "" @@ -382,7 +430,7 @@ class PdfQuoteParser: for p_idx, page in enumerate(doc): pil_img = page.render(scale=2.0).to_pil() - ocr_res, _ = ocr(np.array(pil_img)) + pil_img, ocr_res = self._detect_and_deskew_image(pil_img, ocr) if not ocr_res: continue @@ -400,7 +448,7 @@ class PdfQuoteParser: rects.sort(key=lambda x: -x[3]) avg_h = float(np.mean([r[3] - r[1] for r in rects])) if rects else 15.0 - y_tol = max(8.0, avg_h * 0.45) + y_tol = max(6.0, avg_h * 0.40) page_lines = self._cluster_lines(rects, y_tol=y_tol) ocr_pages_data.append(page_lines) @@ -424,23 +472,24 @@ class PdfQuoteParser: } def _cluster_lines(self, rects: List[Tuple[float, float, float, float, str]], y_tol: float = 4.5) -> List[List[Tuple]]: - """Regroupe les fragments de texte spatialement alignés sur la même ligne horizontale.""" + """Regroupe les fragments de texte spatialement alignés sur la même ligne horizontale sans dérive de centroïde.""" lines = [] current_line = [] - current_y = None + initial_y = None for r in rects: y = (r[1] + r[3]) / 2.0 h_box = r[3] - r[1] - tol = max(y_tol, h_box * 0.45) - if current_y is None or abs(current_y - y) <= tol: + tol = max(y_tol, h_box * 0.40) + if initial_y is None or abs(initial_y - y) <= tol: current_line.append(r) - current_y = (current_y + y) / 2.0 if current_y else y + if initial_y is None: + initial_y = y else: current_line.sort(key=lambda x: x[0]) lines.append(current_line) current_line = [r] - current_y = y + initial_y = y if current_line: current_line.sort(key=lambda x: x[0]) lines.append(current_line) @@ -605,18 +654,22 @@ class PdfQuoteParser: stop_keywords = [ 'Total hors', 'Total EUR', 'Total TTC', 'Total HT', 'Meilleures salutations', 'Dominique Vandevelde', - "D'avance je vous remercie", 'TVA - BTW', 'SERVICE PUBLIC', - 'ITS - tools', 'N°. Article', 'Concerne', 'A l\'attention' + "D'avance je vous remercie", 'TVA - BTW', 'TVA-BTW', 'TVA/BTW', + 'SERVICE PUBLIC', 'ITS - tools', 'N°. Article', 'Concerne', + 'A l\'attention', 'BNP Paribas', 'GEBABEBB', 'BBRUBEBB', + 'ING - BE', 'ING-BE' ] for page_lines in pages_lines: for line in page_lines: - line_str = ' | '.join(item[4] for item in line) + raw_line_str = ' | '.join(item[4] for item in line) - if any(kw in line_str for kw in stop_keywords): + if any(kw.lower() in raw_line_str.lower() for kw in stop_keywords): current_item = None continue + line_str = normalize_ocr_text(raw_line_str) + m = row_pattern.search(line_str) if m: desc_part = m.group(1).strip() @@ -660,6 +713,9 @@ class PdfQuoteParser: unit_price_net = round(total_val / qty_val, 2) if qty_val > 0 else gross_price unit_code = normalize_unit(raw_unit) + # Supprimer les liens de catalogue de la désignation + name = CATALOG_URL_PATTERN.sub('', name).strip(' /.-') + current_item = { 'reference': ref, 'name': name, @@ -674,18 +730,176 @@ class PdfQuoteParser: } items.append(current_item) elif current_item: - # Ligne de continuation de la description - if not line_str.startswith('http') and not line_str.startswith('https') and not '---' in line_str: + # Ligne de continuation potentielle de la description + has_url = bool(CATALOG_URL_PATTERN.search(line_str)) + has_prices = bool(re.search(r'\d+[.,]\d{2}', line_str)) + # Si la ligne est uniquement un lien de catalogue, ne pas polluer le libellé + if has_url and not has_prices: + continue + + # Si la ligne contient un code article ET des valeurs chiffrées (prix, montants), + # ce n'est PAS une continuation mais un nouvel article non tabulé ou fusionné ! + has_article_code = bool(ARTICLE_CODE_PATTERN.search(line_str)) + if has_article_code and has_prices: + current_item = None continuation = line_str.replace(' | ', ' ').strip() + continuation = CATALOG_URL_PATTERN.sub('', continuation).strip(' /.-') + items.append({ + 'reference': '', + 'name': continuation, + 'quantity': 1, + 'unit': 'piece', + 'raw_unit': 'Piece', + 'gross_unit_price': 0.0, + 'discount': '0%', + 'unit_price': 0.0, + 'total_amount': 0.0, + 'vat_rate': 21.0, + }) + continue + + if not line_str.startswith('---'): + continuation = line_str.replace(' | ', ' ').strip() + continuation = CATALOG_URL_PATTERN.sub('', continuation).strip(' /.-') if continuation and not any(k in continuation for k in ['RECUPEL', 'ING - BE95', 'BNP Paribas', '1/2', '2/2']): current_item['name'] += ' ' + continuation - # Nettoyage final des libellés + # Détecter et scinder les articles potentiellement fusionnés dans la description + items = self._split_fused_items(items) + + # Nettoyage final des libellés et filtrage des faux articles (pieds de page) + FOOTER_PATTERNS = [ + 'TVA-BTW', 'TVA - BTW', 'BNP Paribas', 'GEBABEBB', 'BBRUBEBB', + 'ING-BE', 'ING - BE', 'BIC:', 'IBAN' + ] + clean_items = [] for it in items: + it['name'] = CATALOG_URL_PATTERN.sub('', it['name']) it['name'] = re.sub(r'\s*\b[1-9]/[1-9]\b\s*$', '', it['name']).strip() it['name'] = re.sub(r'categoryId=\d+.*', '', it['name']).strip() + it['name'] = re.sub(r'\s+', ' ', it['name']).strip(' /.-') - return items + name_u = it.get('name', '').upper() + ref_u = it.get('reference', '').upper() + if any(fp.upper() in name_u or fp.upper() in ref_u for fp in FOOTER_PATTERNS): + continue + if it.get('total_amount', 0.0) == 0.0 and it.get('gross_unit_price', 0.0) == 0.0 and it.get('unit_price', 0.0) == 0.0: + continue + if not it.get('name') and not it.get('reference'): + continue + clean_items.append(it) + + return clean_items + + def _split_fused_items(self, items: List[Dict[str, Any]]) -> List[Dict[str, Any]]: + """ + Détecte et scinde les articles fusionnés par mégarde dans la description d'une ligne précédente. + Identifie les patterns : [CODE ARTICLE] [DESCRIPTION] [PRIX / MONTANTS / TVA]. + """ + out = [] + for it in items: + name = it.get('name', '') + name_clean = CATALOG_URL_PATTERN.sub('', name) + name_clean = re.sub(r'(\d+);(\d{2})\b', r'\1,\2', name_clean) + name_clean = re.sub(r'(\d+[.,]\d{2})(\d{1,2}%)', r'\1 \2', name_clean) + name_clean = re.sub(r'\s+', ' ', name_clean).strip(' /.-') + + matches = list(ARTICLE_CODE_PATTERN.finditer(name_clean)) + + # Filtrer les matches situés à l'intérieur de parenthèses (ex: '(EX CF E01-41-0180)') + valid_matches = [] + for m in matches: + before = name_clean[:m.start()] + if before.count('(') > before.count(')'): + continue + valid_matches.append(m) + matches = valid_matches + + # Si l'article n'a pas de référence et qu'un code article est présent au début + if not it.get('reference') and matches: + first_m = matches[0] + if first_m.start() < 5: + it['reference'] = first_m.group(0).strip() + name_clean = name_clean[first_m.end():].strip(' /.-') + matches = [m for m in matches[1:]] + + # Vérifier si un des matches contient des montants après le code article + has_embedded = False + for m in matches: + if m.start() < 5 and it.get('reference') and m.group(0).replace(' ', '') == it['reference'].replace(' ', ''): + continue + after_match = name_clean[m.start():] + if re.search(r'\d+[.,]\d{2}', after_match): + has_embedded = True + break + + if not has_embedded: + it['name'] = name_clean + out.append(it) + continue + + # Découpage en segments sur les codes articles identifiés + segments = [] + prev_idx = 0 + for m in matches: + if m.start() < 5 and it.get('reference') and m.group(0).replace(' ', '') == it['reference'].replace(' ', ''): + continue + if m.start() > prev_idx: + segments.append(name_clean[prev_idx:m.start()].strip(' /.-')) + prev_idx = m.start() + if prev_idx < len(name_clean): + segments.append(name_clean[prev_idx:].strip(' /.-')) + + if segments: + it['name'] = segments[0] + out.append(it) + + for seg in segments[1:]: + if not seg: + continue + m_ref = ARTICLE_CODE_PATTERN.search(seg) + ref_val = m_ref.group(0) if m_ref else '' + rest = seg[m_ref.end():].strip(' /.-') if m_ref else seg + + # Recherche des composantes chiffrées : [qté] [unité] [prix] [remise] [total] [tva] + m_num = re.search( + r'(?:(\d+[.,]\d+)\s*([A-Za-zÀ-ÿ]+)\s+)?(\d+[.,]\d+)\s*(?:(\d+[.,]?\d*%)\s+)?(\d+[.,]\d+)\s*(\d+%)?', + rest + ) + if m_num: + clean_desc = (rest[:m_num.start()] + ' ' + rest[m_num.end():]).strip(' /.-') + clean_desc = re.sub(r'\s+', ' ', clean_desc) + raw_qty = m_num.group(1) + unit_str = m_num.group(2) or 'Piece' + gross_p = float(m_num.group(3).replace(',', '.')) + disc = m_num.group(4) or '0%' + tot = float(m_num.group(5).replace(',', '.')) + vat = float(m_num.group(6).replace('%', '')) if m_num.group(6) else 21.0 + + disc_pct = float(disc.replace('%', '').replace(',', '.')) / 100.0 if '%' in disc else 0.0 + net_unit_p = round(gross_p * (1.0 - disc_pct), 2) + + if raw_qty: + qty = float(raw_qty.replace(',', '.')) + else: + qty = round(tot / net_unit_p) if net_unit_p > 0 else 1.0 + + out.append({ + 'reference': ref_val, + 'name': clean_desc, + 'quantity': int(qty) if isinstance(qty, float) and qty.is_integer() else qty, + 'unit': normalize_unit(unit_str), + 'raw_unit': unit_str, + 'gross_unit_price': gross_p, + 'discount': disc, + 'unit_price': net_unit_p, + 'total_amount': tot, + 'vat_rate': vat, + }) + else: + if out: + out[-1]['name'] = (out[-1]['name'] + ' ' + seg).strip() + return out def _match_supplier(self, metadata: Dict[str, Any]) -> Optional[Supplier]: """Tente de retrouver un fournisseur existant dans la base de données.""" diff --git a/loko/stock/tests.py b/loko/stock/tests.py index aac34cf..ef6fe8b 100644 --- a/loko/stock/tests.py +++ b/loko/stock/tests.py @@ -2771,3 +2771,49 @@ class PurchaseOrderFromDocumentTests(TestCase): self.assertIsNotNone(po) self.assertEqual(po.items.count(), 1) self.assertEqual(po.items.first().quantity, 2) + + def test_normalize_ocr_text(self): + from .pdf_parser import normalize_ocr_text + raw = "1;00 Piece 89,80 20,00% 64,3021% 12,00Piece" + normalized = normalize_ocr_text(raw) + self.assertIn("1,00", normalized) + self.assertIn("64,30 21%", normalized) + self.assertIn("12,00 Piece", normalized) + + def test_split_fused_items_recovers_merged_articles(self): + from .pdf_parser import PdfQuoteParser + parser = PdfQuoteParser("test_doc.pdf") + fused_items = [ + { + 'reference': 'SR 10361018', + 'name': ( + 'Pince universelle-180 mm -DUOTECH-High Leverage(EX CF E01-41-0180) //www.prof-praxis.be/cat2026/page/245.pdf PM 503055 ' + 'Burin de macon avec poignee demolition 16,08 20,00% 64,3021% XTREME-300x22mm www.prof-praxis.be/page/340.pdf ' + 'JD101824 Crayon de menuisier PRO 101-laque rouge 24 1;00 Piece 89,80 20,00% 71,8421% cm-prix par 100 pcs' + ), + 'quantity': 5, + 'unit': 'piece', + 'raw_unit': 'Piece', + 'gross_unit_price': 17.90, + 'discount': '20%', + 'unit_price': 14.32, + 'total_amount': 71.60, + 'vat_rate': 21.0, + } + ] + split = parser._split_fused_items(fused_items) + self.assertEqual(len(split), 3) + self.assertEqual(split[0]['reference'], 'SR 10361018') + self.assertIn('Pince universelle', split[0]['name']) + self.assertNotIn('PM 503055', split[0]['name']) + + self.assertEqual(split[1]['reference'], 'PM 503055') + self.assertIn('Burin de macon', split[1]['name']) + self.assertEqual(split[1]['quantity'], 5) + self.assertEqual(split[1]['total_amount'], 64.30) + + self.assertEqual(split[2]['reference'], 'JD101824') + self.assertIn('Crayon de menuisier', split[2]['name']) + self.assertEqual(split[2]['quantity'], 1) + self.assertEqual(split[2]['total_amount'], 71.84) +