fix(stock): prevent merged quote items via auto-deskew, fixed baseline clustering and item splitting

- Add auto-deskewing for tilted smartphone photos and document scans
- Anchor line clustering to initial baseline to avoid vertical centroid drift
- Normalize OCR numerical artifacts (semicolon decimals, glued VAT percentages)
- Purge catalog URLs from product names to improve catalog matching
- Implement fused item splitter to rescue embedded articles from descriptions
- Add regression tests for item splitting and OCR normalization
This commit is contained in:
kdeterme 2026-10-01 15:12:56 +02:00
parent 32cb154815
commit 4010dfb138
2 changed files with 278 additions and 18 deletions

View file

@ -101,6 +101,25 @@ def clean_vat_number(raw_vat: str) -> str:
return re.sub(r'[\s\.]', '', raw_vat).upper() return re.sub(r'[\s\.]', '', raw_vat).upper()
ARTICLE_CODE_PATTERN = re.compile(
r'\b([A-Z]{2,4}\s*[-.]?\s*\d{4,8}|[A-Z]{1,3}\d{3,8}[A-Z0-9\.]*|[A-Z][0-9]\d{3,8}[A-Z0-9\.]*|\b\d{4,8}\b)\b'
)
CATALOG_URL_PATTERN = re.compile(
r'(?:https?://|//|tps://|www\.)[^\s]+(?:\.pdf|\.pdl|\.po|\.html|[a-z0-9/_-]*)', re.I
)
def normalize_ocr_text(text: str) -> str:
"""Normalise les imperfections OCR courantes dans une ligne de tableau."""
# Corriger les points-virgules pris pour des virgules décimales (ex: '1;00' -> '1,00')
t = re.sub(r'(\d+);(\d{2})\b', r'\1,\2', text)
# Décoller un montant et une TVA (ex: '64,3021%' -> '64,30 21%' ou '406,5621%' -> '406,56 21%')
t = re.sub(r'(\d+[.,]\d{2})(\d{1,2}%)', r'\1 \2', t)
# Décoller quantité et unité (ex: '12,00Piece' -> '12,00 Piece')
t = re.sub(r'(\d+[.,]\d+)([A-Za-zÀ-ÿ]{3,})', r'\1 \2', t)
return t
STOP_WORDS = { STOP_WORDS = {
'de', 'du', 'la', 'le', 'les', 'des', 'en', 'et', 'au', 'aux', 'd', 'l', 'de', 'du', 'la', 'le', 'les', 'des', 'en', 'et', 'au', 'aux', 'd', 'l',
'un', 'une', 'pour', 'par', 'sur', 'avec', 'sans', 'sous', 'dans', 'clb' 'un', 'une', 'pour', 'par', 'sur', 'avec', 'sans', 'sous', 'dans', 'clb'
@ -310,7 +329,7 @@ class PdfQuoteParser:
new_size = (int(pil_img.width * scale), int(pil_img.height * scale)) new_size = (int(pil_img.width * scale), int(pil_img.height * scale))
pil_img = pil_img.resize(new_size, Image.Resampling.LANCZOS) pil_img = pil_img.resize(new_size, Image.Resampling.LANCZOS)
ocr_res, _ = ocr(np.array(pil_img)) pil_img, ocr_res = self._detect_and_deskew_image(pil_img, ocr)
rects = [] rects = []
h = pil_img.height h = pil_img.height
@ -327,7 +346,7 @@ class PdfQuoteParser:
rects.sort(key=lambda x: -x[3]) rects.sort(key=lambda x: -x[3])
avg_h = float(np.mean([r[3] - r[1] for r in rects])) if rects else 15.0 avg_h = float(np.mean([r[3] - r[1] for r in rects])) if rects else 15.0
y_tol = max(8.0, avg_h * 0.45) y_tol = max(6.0, avg_h * 0.40)
page_lines = self._cluster_lines(rects, y_tol=y_tol) page_lines = self._cluster_lines(rects, y_tol=y_tol)
full_text = "\n".join(" ".join(r[4] for r in l) for l in page_lines) full_text = "\n".join(" ".join(r[4] for r in l) for l in page_lines)
@ -347,6 +366,35 @@ class PdfQuoteParser:
'total_items': len(enriched_items), 'total_items': len(enriched_items),
} }
@staticmethod
def _detect_and_deskew_image(pil_img: Any, ocr: Any) -> Tuple[Any, Any]:
"""
Détecte l'inclinaison des lignes de texte (photos smartphone ou scans de travers)
et redresse automatiquement l'image si l'angle dépasse 0.35° pour assurer un alignement horizontal parfait.
"""
ocr_res, _ = ocr(np.array(pil_img))
if not ocr_res or np is None:
return pil_img, ocr_res
angles = []
for box, txt, conf in ocr_res:
dx = box[1][0] - box[0][0]
dy = box[1][1] - box[0][1]
bw = np.hypot(dx, dy)
bh = np.hypot(box[3][0] - box[0][0], box[3][1] - box[0][1])
if bw > bh * 2.5:
ang = np.degrees(np.arctan2(dy, dx))
if abs(ang) < 45.0:
angles.append(ang)
median_angle = float(np.median(angles)) if angles else 0.0
if abs(median_angle) >= 0.35:
logger.info(f"Redressement de l'image de {median_angle:.2f}° pour aligner les lignes OCR.")
pil_img = pil_img.rotate(median_angle, expand=True, fillcolor='white')
ocr_res, _ = ocr(np.array(pil_img))
return pil_img, ocr_res
def _parse_vector(self, doc: Any) -> Dict[str, Any]: def _parse_vector(self, doc: Any) -> Dict[str, Any]:
"""Analyse un PDF avec couche texte vectorielle (recherche spatiale précise).""" """Analyse un PDF avec couche texte vectorielle (recherche spatiale précise)."""
full_text = "" full_text = ""
@ -382,7 +430,7 @@ class PdfQuoteParser:
for p_idx, page in enumerate(doc): for p_idx, page in enumerate(doc):
pil_img = page.render(scale=2.0).to_pil() pil_img = page.render(scale=2.0).to_pil()
ocr_res, _ = ocr(np.array(pil_img)) pil_img, ocr_res = self._detect_and_deskew_image(pil_img, ocr)
if not ocr_res: if not ocr_res:
continue continue
@ -400,7 +448,7 @@ class PdfQuoteParser:
rects.sort(key=lambda x: -x[3]) rects.sort(key=lambda x: -x[3])
avg_h = float(np.mean([r[3] - r[1] for r in rects])) if rects else 15.0 avg_h = float(np.mean([r[3] - r[1] for r in rects])) if rects else 15.0
y_tol = max(8.0, avg_h * 0.45) y_tol = max(6.0, avg_h * 0.40)
page_lines = self._cluster_lines(rects, y_tol=y_tol) page_lines = self._cluster_lines(rects, y_tol=y_tol)
ocr_pages_data.append(page_lines) ocr_pages_data.append(page_lines)
@ -424,23 +472,24 @@ class PdfQuoteParser:
} }
def _cluster_lines(self, rects: List[Tuple[float, float, float, float, str]], y_tol: float = 4.5) -> List[List[Tuple]]: def _cluster_lines(self, rects: List[Tuple[float, float, float, float, str]], y_tol: float = 4.5) -> List[List[Tuple]]:
"""Regroupe les fragments de texte spatialement alignés sur la même ligne horizontale.""" """Regroupe les fragments de texte spatialement alignés sur la même ligne horizontale sans dérive de centroïde."""
lines = [] lines = []
current_line = [] current_line = []
current_y = None initial_y = None
for r in rects: for r in rects:
y = (r[1] + r[3]) / 2.0 y = (r[1] + r[3]) / 2.0
h_box = r[3] - r[1] h_box = r[3] - r[1]
tol = max(y_tol, h_box * 0.45) tol = max(y_tol, h_box * 0.40)
if current_y is None or abs(current_y - y) <= tol: if initial_y is None or abs(initial_y - y) <= tol:
current_line.append(r) current_line.append(r)
current_y = (current_y + y) / 2.0 if current_y else y if initial_y is None:
initial_y = y
else: else:
current_line.sort(key=lambda x: x[0]) current_line.sort(key=lambda x: x[0])
lines.append(current_line) lines.append(current_line)
current_line = [r] current_line = [r]
current_y = y initial_y = y
if current_line: if current_line:
current_line.sort(key=lambda x: x[0]) current_line.sort(key=lambda x: x[0])
lines.append(current_line) lines.append(current_line)
@ -605,18 +654,22 @@ class PdfQuoteParser:
stop_keywords = [ stop_keywords = [
'Total hors', 'Total EUR', 'Total TTC', 'Total HT', 'Total hors', 'Total EUR', 'Total TTC', 'Total HT',
'Meilleures salutations', 'Dominique Vandevelde', 'Meilleures salutations', 'Dominique Vandevelde',
"D'avance je vous remercie", 'TVA - BTW', 'SERVICE PUBLIC', "D'avance je vous remercie", 'TVA - BTW', 'TVA-BTW', 'TVA/BTW',
'ITS - tools', 'N°. Article', 'Concerne', 'A l\'attention' 'SERVICE PUBLIC', 'ITS - tools', 'N°. Article', 'Concerne',
'A l\'attention', 'BNP Paribas', 'GEBABEBB', 'BBRUBEBB',
'ING - BE', 'ING-BE'
] ]
for page_lines in pages_lines: for page_lines in pages_lines:
for line in page_lines: for line in page_lines:
line_str = ' | '.join(item[4] for item in line) raw_line_str = ' | '.join(item[4] for item in line)
if any(kw in line_str for kw in stop_keywords): if any(kw.lower() in raw_line_str.lower() for kw in stop_keywords):
current_item = None current_item = None
continue continue
line_str = normalize_ocr_text(raw_line_str)
m = row_pattern.search(line_str) m = row_pattern.search(line_str)
if m: if m:
desc_part = m.group(1).strip() desc_part = m.group(1).strip()
@ -660,6 +713,9 @@ class PdfQuoteParser:
unit_price_net = round(total_val / qty_val, 2) if qty_val > 0 else gross_price unit_price_net = round(total_val / qty_val, 2) if qty_val > 0 else gross_price
unit_code = normalize_unit(raw_unit) unit_code = normalize_unit(raw_unit)
# Supprimer les liens de catalogue de la désignation
name = CATALOG_URL_PATTERN.sub('', name).strip(' /.-')
current_item = { current_item = {
'reference': ref, 'reference': ref,
'name': name, 'name': name,
@ -674,18 +730,176 @@ class PdfQuoteParser:
} }
items.append(current_item) items.append(current_item)
elif current_item: elif current_item:
# Ligne de continuation de la description # Ligne de continuation potentielle de la description
if not line_str.startswith('http') and not line_str.startswith('https') and not '---' in line_str: has_url = bool(CATALOG_URL_PATTERN.search(line_str))
has_prices = bool(re.search(r'\d+[.,]\d{2}', line_str))
# Si la ligne est uniquement un lien de catalogue, ne pas polluer le libellé
if has_url and not has_prices:
continue
# Si la ligne contient un code article ET des valeurs chiffrées (prix, montants),
# ce n'est PAS une continuation mais un nouvel article non tabulé ou fusionné !
has_article_code = bool(ARTICLE_CODE_PATTERN.search(line_str))
if has_article_code and has_prices:
current_item = None
continuation = line_str.replace(' | ', ' ').strip() continuation = line_str.replace(' | ', ' ').strip()
continuation = CATALOG_URL_PATTERN.sub('', continuation).strip(' /.-')
items.append({
'reference': '',
'name': continuation,
'quantity': 1,
'unit': 'piece',
'raw_unit': 'Piece',
'gross_unit_price': 0.0,
'discount': '0%',
'unit_price': 0.0,
'total_amount': 0.0,
'vat_rate': 21.0,
})
continue
if not line_str.startswith('---'):
continuation = line_str.replace(' | ', ' ').strip()
continuation = CATALOG_URL_PATTERN.sub('', continuation).strip(' /.-')
if continuation and not any(k in continuation for k in ['RECUPEL', 'ING - BE95', 'BNP Paribas', '1/2', '2/2']): if continuation and not any(k in continuation for k in ['RECUPEL', 'ING - BE95', 'BNP Paribas', '1/2', '2/2']):
current_item['name'] += ' ' + continuation current_item['name'] += ' ' + continuation
# Nettoyage final des libellés # Détecter et scinder les articles potentiellement fusionnés dans la description
items = self._split_fused_items(items)
# Nettoyage final des libellés et filtrage des faux articles (pieds de page)
FOOTER_PATTERNS = [
'TVA-BTW', 'TVA - BTW', 'BNP Paribas', 'GEBABEBB', 'BBRUBEBB',
'ING-BE', 'ING - BE', 'BIC:', 'IBAN'
]
clean_items = []
for it in items: for it in items:
it['name'] = CATALOG_URL_PATTERN.sub('', it['name'])
it['name'] = re.sub(r'\s*\b[1-9]/[1-9]\b\s*$', '', it['name']).strip() it['name'] = re.sub(r'\s*\b[1-9]/[1-9]\b\s*$', '', it['name']).strip()
it['name'] = re.sub(r'categoryId=\d+.*', '', it['name']).strip() it['name'] = re.sub(r'categoryId=\d+.*', '', it['name']).strip()
it['name'] = re.sub(r'\s+', ' ', it['name']).strip(' /.-')
return items name_u = it.get('name', '').upper()
ref_u = it.get('reference', '').upper()
if any(fp.upper() in name_u or fp.upper() in ref_u for fp in FOOTER_PATTERNS):
continue
if it.get('total_amount', 0.0) == 0.0 and it.get('gross_unit_price', 0.0) == 0.0 and it.get('unit_price', 0.0) == 0.0:
continue
if not it.get('name') and not it.get('reference'):
continue
clean_items.append(it)
return clean_items
def _split_fused_items(self, items: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
"""
Détecte et scinde les articles fusionnés par mégarde dans la description d'une ligne précédente.
Identifie les patterns : [CODE ARTICLE] [DESCRIPTION] [PRIX / MONTANTS / TVA].
"""
out = []
for it in items:
name = it.get('name', '')
name_clean = CATALOG_URL_PATTERN.sub('', name)
name_clean = re.sub(r'(\d+);(\d{2})\b', r'\1,\2', name_clean)
name_clean = re.sub(r'(\d+[.,]\d{2})(\d{1,2}%)', r'\1 \2', name_clean)
name_clean = re.sub(r'\s+', ' ', name_clean).strip(' /.-')
matches = list(ARTICLE_CODE_PATTERN.finditer(name_clean))
# Filtrer les matches situés à l'intérieur de parenthèses (ex: '(EX CF E01-41-0180)')
valid_matches = []
for m in matches:
before = name_clean[:m.start()]
if before.count('(') > before.count(')'):
continue
valid_matches.append(m)
matches = valid_matches
# Si l'article n'a pas de référence et qu'un code article est présent au début
if not it.get('reference') and matches:
first_m = matches[0]
if first_m.start() < 5:
it['reference'] = first_m.group(0).strip()
name_clean = name_clean[first_m.end():].strip(' /.-')
matches = [m for m in matches[1:]]
# Vérifier si un des matches contient des montants après le code article
has_embedded = False
for m in matches:
if m.start() < 5 and it.get('reference') and m.group(0).replace(' ', '') == it['reference'].replace(' ', ''):
continue
after_match = name_clean[m.start():]
if re.search(r'\d+[.,]\d{2}', after_match):
has_embedded = True
break
if not has_embedded:
it['name'] = name_clean
out.append(it)
continue
# Découpage en segments sur les codes articles identifiés
segments = []
prev_idx = 0
for m in matches:
if m.start() < 5 and it.get('reference') and m.group(0).replace(' ', '') == it['reference'].replace(' ', ''):
continue
if m.start() > prev_idx:
segments.append(name_clean[prev_idx:m.start()].strip(' /.-'))
prev_idx = m.start()
if prev_idx < len(name_clean):
segments.append(name_clean[prev_idx:].strip(' /.-'))
if segments:
it['name'] = segments[0]
out.append(it)
for seg in segments[1:]:
if not seg:
continue
m_ref = ARTICLE_CODE_PATTERN.search(seg)
ref_val = m_ref.group(0) if m_ref else ''
rest = seg[m_ref.end():].strip(' /.-') if m_ref else seg
# Recherche des composantes chiffrées : [qté] [unité] [prix] [remise] [total] [tva]
m_num = re.search(
r'(?:(\d+[.,]\d+)\s*([A-Za-zÀ-ÿ]+)\s+)?(\d+[.,]\d+)\s*(?:(\d+[.,]?\d*%)\s+)?(\d+[.,]\d+)\s*(\d+%)?',
rest
)
if m_num:
clean_desc = (rest[:m_num.start()] + ' ' + rest[m_num.end():]).strip(' /.-')
clean_desc = re.sub(r'\s+', ' ', clean_desc)
raw_qty = m_num.group(1)
unit_str = m_num.group(2) or 'Piece'
gross_p = float(m_num.group(3).replace(',', '.'))
disc = m_num.group(4) or '0%'
tot = float(m_num.group(5).replace(',', '.'))
vat = float(m_num.group(6).replace('%', '')) if m_num.group(6) else 21.0
disc_pct = float(disc.replace('%', '').replace(',', '.')) / 100.0 if '%' in disc else 0.0
net_unit_p = round(gross_p * (1.0 - disc_pct), 2)
if raw_qty:
qty = float(raw_qty.replace(',', '.'))
else:
qty = round(tot / net_unit_p) if net_unit_p > 0 else 1.0
out.append({
'reference': ref_val,
'name': clean_desc,
'quantity': int(qty) if isinstance(qty, float) and qty.is_integer() else qty,
'unit': normalize_unit(unit_str),
'raw_unit': unit_str,
'gross_unit_price': gross_p,
'discount': disc,
'unit_price': net_unit_p,
'total_amount': tot,
'vat_rate': vat,
})
else:
if out:
out[-1]['name'] = (out[-1]['name'] + ' ' + seg).strip()
return out
def _match_supplier(self, metadata: Dict[str, Any]) -> Optional[Supplier]: def _match_supplier(self, metadata: Dict[str, Any]) -> Optional[Supplier]:
"""Tente de retrouver un fournisseur existant dans la base de données.""" """Tente de retrouver un fournisseur existant dans la base de données."""

View file

@ -2771,3 +2771,49 @@ class PurchaseOrderFromDocumentTests(TestCase):
self.assertIsNotNone(po) self.assertIsNotNone(po)
self.assertEqual(po.items.count(), 1) self.assertEqual(po.items.count(), 1)
self.assertEqual(po.items.first().quantity, 2) self.assertEqual(po.items.first().quantity, 2)
def test_normalize_ocr_text(self):
from .pdf_parser import normalize_ocr_text
raw = "1;00 Piece 89,80 20,00% 64,3021% 12,00Piece"
normalized = normalize_ocr_text(raw)
self.assertIn("1,00", normalized)
self.assertIn("64,30 21%", normalized)
self.assertIn("12,00 Piece", normalized)
def test_split_fused_items_recovers_merged_articles(self):
from .pdf_parser import PdfQuoteParser
parser = PdfQuoteParser("test_doc.pdf")
fused_items = [
{
'reference': 'SR 10361018',
'name': (
'Pince universelle-180 mm -DUOTECH-High Leverage(EX CF E01-41-0180) //www.prof-praxis.be/cat2026/page/245.pdf PM 503055 '
'Burin de macon avec poignee demolition 16,08 20,00% 64,3021% XTREME-300x22mm www.prof-praxis.be/page/340.pdf '
'JD101824 Crayon de menuisier PRO 101-laque rouge 24 1;00 Piece 89,80 20,00% 71,8421% cm-prix par 100 pcs'
),
'quantity': 5,
'unit': 'piece',
'raw_unit': 'Piece',
'gross_unit_price': 17.90,
'discount': '20%',
'unit_price': 14.32,
'total_amount': 71.60,
'vat_rate': 21.0,
}
]
split = parser._split_fused_items(fused_items)
self.assertEqual(len(split), 3)
self.assertEqual(split[0]['reference'], 'SR 10361018')
self.assertIn('Pince universelle', split[0]['name'])
self.assertNotIn('PM 503055', split[0]['name'])
self.assertEqual(split[1]['reference'], 'PM 503055')
self.assertIn('Burin de macon', split[1]['name'])
self.assertEqual(split[1]['quantity'], 5)
self.assertEqual(split[1]['total_amount'], 64.30)
self.assertEqual(split[2]['reference'], 'JD101824')
self.assertIn('Crayon de menuisier', split[2]['name'])
self.assertEqual(split[2]['quantity'], 1)
self.assertEqual(split[2]['total_amount'], 71.84)