fix(stock): prevent merged quote items via auto-deskew, fixed baseline clustering and item splitting
- Add auto-deskewing for tilted smartphone photos and document scans - Anchor line clustering to initial baseline to avoid vertical centroid drift - Normalize OCR numerical artifacts (semicolon decimals, glued VAT percentages) - Purge catalog URLs from product names to improve catalog matching - Implement fused item splitter to rescue embedded articles from descriptions - Add regression tests for item splitting and OCR normalization
This commit is contained in:
parent
32cb154815
commit
4010dfb138
2 changed files with 278 additions and 18 deletions
|
|
@ -101,6 +101,25 @@ def clean_vat_number(raw_vat: str) -> str:
|
|||
return re.sub(r'[\s\.]', '', raw_vat).upper()
|
||||
|
||||
|
||||
ARTICLE_CODE_PATTERN = re.compile(
|
||||
r'\b([A-Z]{2,4}\s*[-.]?\s*\d{4,8}|[A-Z]{1,3}\d{3,8}[A-Z0-9\.]*|[A-Z][0-9]\d{3,8}[A-Z0-9\.]*|\b\d{4,8}\b)\b'
|
||||
)
|
||||
CATALOG_URL_PATTERN = re.compile(
|
||||
r'(?:https?://|//|tps://|www\.)[^\s]+(?:\.pdf|\.pdl|\.po|\.html|[a-z0-9/_-]*)', re.I
|
||||
)
|
||||
|
||||
|
||||
def normalize_ocr_text(text: str) -> str:
|
||||
"""Normalise les imperfections OCR courantes dans une ligne de tableau."""
|
||||
# Corriger les points-virgules pris pour des virgules décimales (ex: '1;00' -> '1,00')
|
||||
t = re.sub(r'(\d+);(\d{2})\b', r'\1,\2', text)
|
||||
# Décoller un montant et une TVA (ex: '64,3021%' -> '64,30 21%' ou '406,5621%' -> '406,56 21%')
|
||||
t = re.sub(r'(\d+[.,]\d{2})(\d{1,2}%)', r'\1 \2', t)
|
||||
# Décoller quantité et unité (ex: '12,00Piece' -> '12,00 Piece')
|
||||
t = re.sub(r'(\d+[.,]\d+)([A-Za-zÀ-ÿ]{3,})', r'\1 \2', t)
|
||||
return t
|
||||
|
||||
|
||||
STOP_WORDS = {
|
||||
'de', 'du', 'la', 'le', 'les', 'des', 'en', 'et', 'au', 'aux', 'd', 'l',
|
||||
'un', 'une', 'pour', 'par', 'sur', 'avec', 'sans', 'sous', 'dans', 'clb'
|
||||
|
|
@ -310,7 +329,7 @@ class PdfQuoteParser:
|
|||
new_size = (int(pil_img.width * scale), int(pil_img.height * scale))
|
||||
pil_img = pil_img.resize(new_size, Image.Resampling.LANCZOS)
|
||||
|
||||
ocr_res, _ = ocr(np.array(pil_img))
|
||||
pil_img, ocr_res = self._detect_and_deskew_image(pil_img, ocr)
|
||||
rects = []
|
||||
h = pil_img.height
|
||||
|
||||
|
|
@ -327,7 +346,7 @@ class PdfQuoteParser:
|
|||
|
||||
rects.sort(key=lambda x: -x[3])
|
||||
avg_h = float(np.mean([r[3] - r[1] for r in rects])) if rects else 15.0
|
||||
y_tol = max(8.0, avg_h * 0.45)
|
||||
y_tol = max(6.0, avg_h * 0.40)
|
||||
|
||||
page_lines = self._cluster_lines(rects, y_tol=y_tol)
|
||||
full_text = "\n".join(" ".join(r[4] for r in l) for l in page_lines)
|
||||
|
|
@ -347,6 +366,35 @@ class PdfQuoteParser:
|
|||
'total_items': len(enriched_items),
|
||||
}
|
||||
|
||||
@staticmethod
|
||||
def _detect_and_deskew_image(pil_img: Any, ocr: Any) -> Tuple[Any, Any]:
|
||||
"""
|
||||
Détecte l'inclinaison des lignes de texte (photos smartphone ou scans de travers)
|
||||
et redresse automatiquement l'image si l'angle dépasse 0.35° pour assurer un alignement horizontal parfait.
|
||||
"""
|
||||
ocr_res, _ = ocr(np.array(pil_img))
|
||||
if not ocr_res or np is None:
|
||||
return pil_img, ocr_res
|
||||
|
||||
angles = []
|
||||
for box, txt, conf in ocr_res:
|
||||
dx = box[1][0] - box[0][0]
|
||||
dy = box[1][1] - box[0][1]
|
||||
bw = np.hypot(dx, dy)
|
||||
bh = np.hypot(box[3][0] - box[0][0], box[3][1] - box[0][1])
|
||||
if bw > bh * 2.5:
|
||||
ang = np.degrees(np.arctan2(dy, dx))
|
||||
if abs(ang) < 45.0:
|
||||
angles.append(ang)
|
||||
|
||||
median_angle = float(np.median(angles)) if angles else 0.0
|
||||
if abs(median_angle) >= 0.35:
|
||||
logger.info(f"Redressement de l'image de {median_angle:.2f}° pour aligner les lignes OCR.")
|
||||
pil_img = pil_img.rotate(median_angle, expand=True, fillcolor='white')
|
||||
ocr_res, _ = ocr(np.array(pil_img))
|
||||
|
||||
return pil_img, ocr_res
|
||||
|
||||
def _parse_vector(self, doc: Any) -> Dict[str, Any]:
|
||||
"""Analyse un PDF avec couche texte vectorielle (recherche spatiale précise)."""
|
||||
full_text = ""
|
||||
|
|
@ -382,7 +430,7 @@ class PdfQuoteParser:
|
|||
|
||||
for p_idx, page in enumerate(doc):
|
||||
pil_img = page.render(scale=2.0).to_pil()
|
||||
ocr_res, _ = ocr(np.array(pil_img))
|
||||
pil_img, ocr_res = self._detect_and_deskew_image(pil_img, ocr)
|
||||
if not ocr_res:
|
||||
continue
|
||||
|
||||
|
|
@ -400,7 +448,7 @@ class PdfQuoteParser:
|
|||
|
||||
rects.sort(key=lambda x: -x[3])
|
||||
avg_h = float(np.mean([r[3] - r[1] for r in rects])) if rects else 15.0
|
||||
y_tol = max(8.0, avg_h * 0.45)
|
||||
y_tol = max(6.0, avg_h * 0.40)
|
||||
|
||||
page_lines = self._cluster_lines(rects, y_tol=y_tol)
|
||||
ocr_pages_data.append(page_lines)
|
||||
|
|
@ -424,23 +472,24 @@ class PdfQuoteParser:
|
|||
}
|
||||
|
||||
def _cluster_lines(self, rects: List[Tuple[float, float, float, float, str]], y_tol: float = 4.5) -> List[List[Tuple]]:
|
||||
"""Regroupe les fragments de texte spatialement alignés sur la même ligne horizontale."""
|
||||
"""Regroupe les fragments de texte spatialement alignés sur la même ligne horizontale sans dérive de centroïde."""
|
||||
lines = []
|
||||
current_line = []
|
||||
current_y = None
|
||||
initial_y = None
|
||||
|
||||
for r in rects:
|
||||
y = (r[1] + r[3]) / 2.0
|
||||
h_box = r[3] - r[1]
|
||||
tol = max(y_tol, h_box * 0.45)
|
||||
if current_y is None or abs(current_y - y) <= tol:
|
||||
tol = max(y_tol, h_box * 0.40)
|
||||
if initial_y is None or abs(initial_y - y) <= tol:
|
||||
current_line.append(r)
|
||||
current_y = (current_y + y) / 2.0 if current_y else y
|
||||
if initial_y is None:
|
||||
initial_y = y
|
||||
else:
|
||||
current_line.sort(key=lambda x: x[0])
|
||||
lines.append(current_line)
|
||||
current_line = [r]
|
||||
current_y = y
|
||||
initial_y = y
|
||||
if current_line:
|
||||
current_line.sort(key=lambda x: x[0])
|
||||
lines.append(current_line)
|
||||
|
|
@ -605,18 +654,22 @@ class PdfQuoteParser:
|
|||
stop_keywords = [
|
||||
'Total hors', 'Total EUR', 'Total TTC', 'Total HT',
|
||||
'Meilleures salutations', 'Dominique Vandevelde',
|
||||
"D'avance je vous remercie", 'TVA - BTW', 'SERVICE PUBLIC',
|
||||
'ITS - tools', 'N°. Article', 'Concerne', 'A l\'attention'
|
||||
"D'avance je vous remercie", 'TVA - BTW', 'TVA-BTW', 'TVA/BTW',
|
||||
'SERVICE PUBLIC', 'ITS - tools', 'N°. Article', 'Concerne',
|
||||
'A l\'attention', 'BNP Paribas', 'GEBABEBB', 'BBRUBEBB',
|
||||
'ING - BE', 'ING-BE'
|
||||
]
|
||||
|
||||
for page_lines in pages_lines:
|
||||
for line in page_lines:
|
||||
line_str = ' | '.join(item[4] for item in line)
|
||||
raw_line_str = ' | '.join(item[4] for item in line)
|
||||
|
||||
if any(kw in line_str for kw in stop_keywords):
|
||||
if any(kw.lower() in raw_line_str.lower() for kw in stop_keywords):
|
||||
current_item = None
|
||||
continue
|
||||
|
||||
line_str = normalize_ocr_text(raw_line_str)
|
||||
|
||||
m = row_pattern.search(line_str)
|
||||
if m:
|
||||
desc_part = m.group(1).strip()
|
||||
|
|
@ -660,6 +713,9 @@ class PdfQuoteParser:
|
|||
unit_price_net = round(total_val / qty_val, 2) if qty_val > 0 else gross_price
|
||||
unit_code = normalize_unit(raw_unit)
|
||||
|
||||
# Supprimer les liens de catalogue de la désignation
|
||||
name = CATALOG_URL_PATTERN.sub('', name).strip(' /.-')
|
||||
|
||||
current_item = {
|
||||
'reference': ref,
|
||||
'name': name,
|
||||
|
|
@ -674,18 +730,176 @@ class PdfQuoteParser:
|
|||
}
|
||||
items.append(current_item)
|
||||
elif current_item:
|
||||
# Ligne de continuation de la description
|
||||
if not line_str.startswith('http') and not line_str.startswith('https') and not '---' in line_str:
|
||||
# Ligne de continuation potentielle de la description
|
||||
has_url = bool(CATALOG_URL_PATTERN.search(line_str))
|
||||
has_prices = bool(re.search(r'\d+[.,]\d{2}', line_str))
|
||||
# Si la ligne est uniquement un lien de catalogue, ne pas polluer le libellé
|
||||
if has_url and not has_prices:
|
||||
continue
|
||||
|
||||
# Si la ligne contient un code article ET des valeurs chiffrées (prix, montants),
|
||||
# ce n'est PAS une continuation mais un nouvel article non tabulé ou fusionné !
|
||||
has_article_code = bool(ARTICLE_CODE_PATTERN.search(line_str))
|
||||
if has_article_code and has_prices:
|
||||
current_item = None
|
||||
continuation = line_str.replace(' | ', ' ').strip()
|
||||
continuation = CATALOG_URL_PATTERN.sub('', continuation).strip(' /.-')
|
||||
items.append({
|
||||
'reference': '',
|
||||
'name': continuation,
|
||||
'quantity': 1,
|
||||
'unit': 'piece',
|
||||
'raw_unit': 'Piece',
|
||||
'gross_unit_price': 0.0,
|
||||
'discount': '0%',
|
||||
'unit_price': 0.0,
|
||||
'total_amount': 0.0,
|
||||
'vat_rate': 21.0,
|
||||
})
|
||||
continue
|
||||
|
||||
if not line_str.startswith('---'):
|
||||
continuation = line_str.replace(' | ', ' ').strip()
|
||||
continuation = CATALOG_URL_PATTERN.sub('', continuation).strip(' /.-')
|
||||
if continuation and not any(k in continuation for k in ['RECUPEL', 'ING - BE95', 'BNP Paribas', '1/2', '2/2']):
|
||||
current_item['name'] += ' ' + continuation
|
||||
|
||||
# Nettoyage final des libellés
|
||||
# Détecter et scinder les articles potentiellement fusionnés dans la description
|
||||
items = self._split_fused_items(items)
|
||||
|
||||
# Nettoyage final des libellés et filtrage des faux articles (pieds de page)
|
||||
FOOTER_PATTERNS = [
|
||||
'TVA-BTW', 'TVA - BTW', 'BNP Paribas', 'GEBABEBB', 'BBRUBEBB',
|
||||
'ING-BE', 'ING - BE', 'BIC:', 'IBAN'
|
||||
]
|
||||
clean_items = []
|
||||
for it in items:
|
||||
it['name'] = CATALOG_URL_PATTERN.sub('', it['name'])
|
||||
it['name'] = re.sub(r'\s*\b[1-9]/[1-9]\b\s*$', '', it['name']).strip()
|
||||
it['name'] = re.sub(r'categoryId=\d+.*', '', it['name']).strip()
|
||||
it['name'] = re.sub(r'\s+', ' ', it['name']).strip(' /.-')
|
||||
|
||||
return items
|
||||
name_u = it.get('name', '').upper()
|
||||
ref_u = it.get('reference', '').upper()
|
||||
if any(fp.upper() in name_u or fp.upper() in ref_u for fp in FOOTER_PATTERNS):
|
||||
continue
|
||||
if it.get('total_amount', 0.0) == 0.0 and it.get('gross_unit_price', 0.0) == 0.0 and it.get('unit_price', 0.0) == 0.0:
|
||||
continue
|
||||
if not it.get('name') and not it.get('reference'):
|
||||
continue
|
||||
clean_items.append(it)
|
||||
|
||||
return clean_items
|
||||
|
||||
def _split_fused_items(self, items: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
|
||||
"""
|
||||
Détecte et scinde les articles fusionnés par mégarde dans la description d'une ligne précédente.
|
||||
Identifie les patterns : [CODE ARTICLE] [DESCRIPTION] [PRIX / MONTANTS / TVA].
|
||||
"""
|
||||
out = []
|
||||
for it in items:
|
||||
name = it.get('name', '')
|
||||
name_clean = CATALOG_URL_PATTERN.sub('', name)
|
||||
name_clean = re.sub(r'(\d+);(\d{2})\b', r'\1,\2', name_clean)
|
||||
name_clean = re.sub(r'(\d+[.,]\d{2})(\d{1,2}%)', r'\1 \2', name_clean)
|
||||
name_clean = re.sub(r'\s+', ' ', name_clean).strip(' /.-')
|
||||
|
||||
matches = list(ARTICLE_CODE_PATTERN.finditer(name_clean))
|
||||
|
||||
# Filtrer les matches situés à l'intérieur de parenthèses (ex: '(EX CF E01-41-0180)')
|
||||
valid_matches = []
|
||||
for m in matches:
|
||||
before = name_clean[:m.start()]
|
||||
if before.count('(') > before.count(')'):
|
||||
continue
|
||||
valid_matches.append(m)
|
||||
matches = valid_matches
|
||||
|
||||
# Si l'article n'a pas de référence et qu'un code article est présent au début
|
||||
if not it.get('reference') and matches:
|
||||
first_m = matches[0]
|
||||
if first_m.start() < 5:
|
||||
it['reference'] = first_m.group(0).strip()
|
||||
name_clean = name_clean[first_m.end():].strip(' /.-')
|
||||
matches = [m for m in matches[1:]]
|
||||
|
||||
# Vérifier si un des matches contient des montants après le code article
|
||||
has_embedded = False
|
||||
for m in matches:
|
||||
if m.start() < 5 and it.get('reference') and m.group(0).replace(' ', '') == it['reference'].replace(' ', ''):
|
||||
continue
|
||||
after_match = name_clean[m.start():]
|
||||
if re.search(r'\d+[.,]\d{2}', after_match):
|
||||
has_embedded = True
|
||||
break
|
||||
|
||||
if not has_embedded:
|
||||
it['name'] = name_clean
|
||||
out.append(it)
|
||||
continue
|
||||
|
||||
# Découpage en segments sur les codes articles identifiés
|
||||
segments = []
|
||||
prev_idx = 0
|
||||
for m in matches:
|
||||
if m.start() < 5 and it.get('reference') and m.group(0).replace(' ', '') == it['reference'].replace(' ', ''):
|
||||
continue
|
||||
if m.start() > prev_idx:
|
||||
segments.append(name_clean[prev_idx:m.start()].strip(' /.-'))
|
||||
prev_idx = m.start()
|
||||
if prev_idx < len(name_clean):
|
||||
segments.append(name_clean[prev_idx:].strip(' /.-'))
|
||||
|
||||
if segments:
|
||||
it['name'] = segments[0]
|
||||
out.append(it)
|
||||
|
||||
for seg in segments[1:]:
|
||||
if not seg:
|
||||
continue
|
||||
m_ref = ARTICLE_CODE_PATTERN.search(seg)
|
||||
ref_val = m_ref.group(0) if m_ref else ''
|
||||
rest = seg[m_ref.end():].strip(' /.-') if m_ref else seg
|
||||
|
||||
# Recherche des composantes chiffrées : [qté] [unité] [prix] [remise] [total] [tva]
|
||||
m_num = re.search(
|
||||
r'(?:(\d+[.,]\d+)\s*([A-Za-zÀ-ÿ]+)\s+)?(\d+[.,]\d+)\s*(?:(\d+[.,]?\d*%)\s+)?(\d+[.,]\d+)\s*(\d+%)?',
|
||||
rest
|
||||
)
|
||||
if m_num:
|
||||
clean_desc = (rest[:m_num.start()] + ' ' + rest[m_num.end():]).strip(' /.-')
|
||||
clean_desc = re.sub(r'\s+', ' ', clean_desc)
|
||||
raw_qty = m_num.group(1)
|
||||
unit_str = m_num.group(2) or 'Piece'
|
||||
gross_p = float(m_num.group(3).replace(',', '.'))
|
||||
disc = m_num.group(4) or '0%'
|
||||
tot = float(m_num.group(5).replace(',', '.'))
|
||||
vat = float(m_num.group(6).replace('%', '')) if m_num.group(6) else 21.0
|
||||
|
||||
disc_pct = float(disc.replace('%', '').replace(',', '.')) / 100.0 if '%' in disc else 0.0
|
||||
net_unit_p = round(gross_p * (1.0 - disc_pct), 2)
|
||||
|
||||
if raw_qty:
|
||||
qty = float(raw_qty.replace(',', '.'))
|
||||
else:
|
||||
qty = round(tot / net_unit_p) if net_unit_p > 0 else 1.0
|
||||
|
||||
out.append({
|
||||
'reference': ref_val,
|
||||
'name': clean_desc,
|
||||
'quantity': int(qty) if isinstance(qty, float) and qty.is_integer() else qty,
|
||||
'unit': normalize_unit(unit_str),
|
||||
'raw_unit': unit_str,
|
||||
'gross_unit_price': gross_p,
|
||||
'discount': disc,
|
||||
'unit_price': net_unit_p,
|
||||
'total_amount': tot,
|
||||
'vat_rate': vat,
|
||||
})
|
||||
else:
|
||||
if out:
|
||||
out[-1]['name'] = (out[-1]['name'] + ' ' + seg).strip()
|
||||
return out
|
||||
|
||||
def _match_supplier(self, metadata: Dict[str, Any]) -> Optional[Supplier]:
|
||||
"""Tente de retrouver un fournisseur existant dans la base de données."""
|
||||
|
|
|
|||
|
|
@ -2771,3 +2771,49 @@ class PurchaseOrderFromDocumentTests(TestCase):
|
|||
self.assertIsNotNone(po)
|
||||
self.assertEqual(po.items.count(), 1)
|
||||
self.assertEqual(po.items.first().quantity, 2)
|
||||
|
||||
def test_normalize_ocr_text(self):
|
||||
from .pdf_parser import normalize_ocr_text
|
||||
raw = "1;00 Piece 89,80 20,00% 64,3021% 12,00Piece"
|
||||
normalized = normalize_ocr_text(raw)
|
||||
self.assertIn("1,00", normalized)
|
||||
self.assertIn("64,30 21%", normalized)
|
||||
self.assertIn("12,00 Piece", normalized)
|
||||
|
||||
def test_split_fused_items_recovers_merged_articles(self):
|
||||
from .pdf_parser import PdfQuoteParser
|
||||
parser = PdfQuoteParser("test_doc.pdf")
|
||||
fused_items = [
|
||||
{
|
||||
'reference': 'SR 10361018',
|
||||
'name': (
|
||||
'Pince universelle-180 mm -DUOTECH-High Leverage(EX CF E01-41-0180) //www.prof-praxis.be/cat2026/page/245.pdf PM 503055 '
|
||||
'Burin de macon avec poignee demolition 16,08 20,00% 64,3021% XTREME-300x22mm www.prof-praxis.be/page/340.pdf '
|
||||
'JD101824 Crayon de menuisier PRO 101-laque rouge 24 1;00 Piece 89,80 20,00% 71,8421% cm-prix par 100 pcs'
|
||||
),
|
||||
'quantity': 5,
|
||||
'unit': 'piece',
|
||||
'raw_unit': 'Piece',
|
||||
'gross_unit_price': 17.90,
|
||||
'discount': '20%',
|
||||
'unit_price': 14.32,
|
||||
'total_amount': 71.60,
|
||||
'vat_rate': 21.0,
|
||||
}
|
||||
]
|
||||
split = parser._split_fused_items(fused_items)
|
||||
self.assertEqual(len(split), 3)
|
||||
self.assertEqual(split[0]['reference'], 'SR 10361018')
|
||||
self.assertIn('Pince universelle', split[0]['name'])
|
||||
self.assertNotIn('PM 503055', split[0]['name'])
|
||||
|
||||
self.assertEqual(split[1]['reference'], 'PM 503055')
|
||||
self.assertIn('Burin de macon', split[1]['name'])
|
||||
self.assertEqual(split[1]['quantity'], 5)
|
||||
self.assertEqual(split[1]['total_amount'], 64.30)
|
||||
|
||||
self.assertEqual(split[2]['reference'], 'JD101824')
|
||||
self.assertIn('Crayon de menuisier', split[2]['name'])
|
||||
self.assertEqual(split[2]['quantity'], 1)
|
||||
self.assertEqual(split[2]['total_amount'], 71.84)
|
||||
|
||||
|
|
|
|||
Loading…
Reference in a new issue