"""Reconnaissance des indicateurs, des valeurs et de la semaine dans un tableau ou un texte OCR.

Fonctions pures (aucun accès à la base) : exécutées dans le sous-processus d'analyse isolé.
Rien n'est complété ni deviné : toute incertitude devient un avertissement à confirmer.
"""
from __future__ import annotations

import re
import unicodedata
from dataclasses import dataclass, field
from datetime import date
from difflib import SequenceMatcher
from typing import Any

LABEL_KEYWORDS = ("thematique", "theme", "infraction", "indicateur", "nature", "intitule", "rubrique", "libelle")
NUMBER_KEYWORDS = ("nombre", "nb", "nbre", "total", "quantite", "effectif", "resultat", "chiffre", "nombres")
OBS_KEYWORDS = ("observation", "observations", "commentaire", "commentaires", "remarque", "remarques", "precision",
                "precisions", "detail", "details", "obs")
GENERIC_LABELS = {"vitesse", "vitesses", "alcool", "alcoolemie", "afd", "stupefiants", "stup", "exces de vitesse"}
TOTAL_WORDS = ("total", "sous total", "totaux", "somme")
OCR_DIGIT_FIXES = str.maketrans({"O": "0", "o": "0", "Q": "0", "D": "0", "l": "1", "I": "1", "|": "1", "!": "1",
                                 "S": "5", "B": "8", "Z": "2"})


def norm(text: Any) -> str:
    s = "" if text is None else str(text)
    s = s.replace("≤", " <= ").replace("≥", " >= ").replace("œ", "oe").replace("’", "'")
    s = unicodedata.normalize("NFKD", s.lower())
    s = "".join(c for c in s if not unicodedata.combining(c))
    s = s.replace("km/h", " kmh ").replace("km / h", " kmh ")
    s = re.sub(r"[^a-z0-9+<>=]+", " ", s)
    return re.sub(r"\s+", " ", s).strip()


@dataclass
class IndicatorRef:
    id: int
    code: str
    label: str
    aliases: list[str] = field(default_factory=list)
    detail_kind: str | None = None


def _special(n: str) -> dict[str, float]:
    """Règles de désambiguïsation sur les mots clés des libellés initiaux (sans règle métier ajoutée)."""
    out: dict[str, float] = {}
    nums = set(re.findall(r"\d+", n))
    if "vitesse" in n or "exces" in n:
        if "50" in nums and ("+" in n or "plus" in n or ">" in n or "sup" in n or "au dela" in n):
            out["vitesse_plus_50"] = 0.96
        elif "40" in nums and "49" in nums:
            out["vitesse_40_49"] = 0.96
        elif "39" in nums or ("40" in nums and ("<" in n or "moins" in n or "inf" in n)):
            out["vitesse_39_moins"] = 0.96
    if "alcool" in n:
        if "delict" in n or "delit" in n:
            out["alcool_delictuelle"] = 0.95
        elif "contrav" in n:
            out["alcool_contraventionnelle"] = 0.95
    if re.search(r"\bafd\b", n):
        if "stup" in n:
            out["afd_stupefiants"] = 0.95
        elif "assur" in n:
            out["afd_assurance"] = 0.95
        elif "permis" in n:
            out["afd_permis"] = 0.95
    if "stup" in n and "conduite" in n and "afd" not in n:
        out["stupefiants"] = 0.93
    if "enquete" in n:
        out["enquetes_judiciaires"] = 0.9
    return out


def _sim(a: str, b: str) -> float:
    if not a or not b:
        return 0.0
    if a == b:
        return 1.0
    ratio = SequenceMatcher(None, a, b).ratio()
    ta, tb = set(a.split()), set(b.split())
    jac = len(ta & tb) / len(ta | tb) if ta | tb else 0.0
    return max(ratio, jac * 0.95)


def match_label(text: str, indicators: list[IndicatorRef]) -> tuple[IndicatorRef | None, float, list[str]]:
    """Renvoie (indicateur, score 0–1, avertissements)."""
    n = norm(text)
    if len(n) < 2:
        return None, 0.0, []
    special = _special(n)
    scored: list[tuple[float, IndicatorRef]] = []
    for ind in indicators:
        best = special.get(ind.code, 0.0)
        for cand in [ind.label, *ind.aliases]:
            best = max(best, _sim(n, norm(cand)))
        scored.append((best, ind))
    scored.sort(key=lambda t: t[0], reverse=True)
    if not scored or scored[0][0] < 0.55:
        return None, scored[0][0] if scored else 0.0, []
    score, ind = scored[0]
    warnings = []
    if n in GENERIC_LABELS:
        return ind, min(score, 0.6), ["Libellé trop général : plusieurs indicateurs possibles, choisissez le bon."]
    if len(scored) > 1 and score - scored[1][0] < 0.06 and score < 0.99:
        warnings.append(f"Correspondance ambiguë avec « {scored[1][1].label} ».")
        score = min(score, 0.7)
    if score < 0.75:
        warnings.append("Libellé reconnu avec une confiance faible : vérifiez l'indicateur.")
    return ind, round(score, 3), warnings


def parse_value(raw: Any, *, ocr: bool = False) -> tuple[int | None, str | None, list[str]]:
    """(valeur, erreur bloquante, avertissements). Vide → None (non renseigné), jamais 0."""
    warnings: list[str] = []
    if raw is None:
        return None, None, warnings
    if isinstance(raw, bool):
        return None, "Valeur booléenne inattendue.", warnings
    if isinstance(raw, (int, float)):
        if isinstance(raw, float) and raw != int(raw):
            return None, f"Valeur décimale ({str(raw).replace('.', ',')}) : un nombre entier est attendu.", warnings
        v = int(raw)
        if v < 0:
            return None, "Valeur négative interdite.", warnings
        return v, None, warnings
    s = str(raw).strip().replace(" ", " ").replace(" ", " ")
    if s == "":
        return None, None, warnings
    if s in {"-", "–", "—", "/", "x", "X", "néant", "neant", "NC", "nc", "n/c", "?"}:
        warnings.append(f"« {s} » interprété comme non renseigné : à confirmer (ou saisir 0 si aucune infraction).")
        return None, None, warnings
    compact = s.replace(" ", "")
    if ocr and not compact.isdigit():
        fixed = compact.translate(OCR_DIGIT_FIXES)
        if fixed.isdigit() and fixed != compact:
            warnings.append(f"Caractère ambigu lu « {compact} », interprété {int(fixed)} : vérifiez sur l'image.")
            compact = fixed
    if re.fullmatch(r"\d+", compact):
        v = int(compact)
        if v > 100000:
            return None, "Valeur trop grande.", warnings
        return v, None, warnings
    if re.fullmatch(r"-\d+", compact):
        return None, "Valeur négative interdite.", warnings
    if re.fullmatch(r"\d+[.,]\d+", compact):
        if re.fullmatch(r"\d+[.,]0+", compact):
            warnings.append(f"« {s} » écrit avec une décimale nulle : interprété comme entier.")
            return int(re.split(r"[.,]", compact)[0]), None, warnings
        return None, f"Valeur décimale ({s}) : un nombre entier est attendu.", warnings
    return None, f"« {s[:30]} » n'est pas un nombre.", warnings


WEEK_PATTERNS = [
    re.compile(r"semaine\s*(?:n\s*[°o]?\s*)?(\d{1,2})\b"),
    re.compile(r"\bsem\s*(?:n\s*[°o]?\s*)?(\d{1,2})\b"),
    re.compile(r"\bs\s?(\d{1,2})\b"),
]
DATE_RE = re.compile(r"\b(\d{1,2})[/.-](\d{1,2})[/.-](\d{2,4})\b")
YEAR_RE = re.compile(r"\b(20\d{2})\b")


def detect_week(texts: list[tuple[str, str]]) -> dict[str, Any] | None:
    """Cherche « Semaine 40 », « S40 » ou des dates. `texts` = [(texte, localisation)]."""
    found_week = None
    found_year = None
    first_date = None
    for text, where in texts:
        raw = str(text)
        n = norm(raw)
        if found_week is None:
            for pat in WEEK_PATTERNS:
                m = pat.search(n)
                if m and 1 <= int(m.group(1)) <= 53:
                    found_week = (int(m.group(1)), f"« {raw.strip()[:60]} » ({where})")
                    break
        if first_date is None:
            m = DATE_RE.search(raw)
            if m:
                d, mo, y = int(m.group(1)), int(m.group(2)), int(m.group(3))
                y = y + 2000 if y < 100 else y
                try:
                    first_date = (date(y, mo, d), f"« {m.group(0)} » ({where})")
                except ValueError:
                    pass
        if found_year is None:
            m = YEAR_RE.search(raw)
            if m:
                found_year = int(m.group(1))
    if first_date is not None:
        cal = first_date[0].isocalendar()
        result = {"iso_year": cal.year, "iso_week": cal.week, "evidence": f"date {first_date[1]}"}
        if found_week and found_week[0] != cal.week:
            result["warning"] = (f"Le numéro de semaine lu ({found_week[0]}) ne correspond pas à la date "
                                 f"({first_date[0].strftime('%d/%m/%Y')} = semaine {cal.week}).")
        return result
    if found_week:
        result = {"iso_year": found_year, "iso_week": found_week[0], "evidence": f"titre {found_week[1]}"}
        if found_year is None:
            result["warning"] = "Année non trouvée dans le document : vérifiez la semaine cible."
        return result
    return None


DECIMAL_RE = re.compile(r"(?<![\d/])(\d{1,2}[,.]\d{1,3})(?![\d/])")
SPEED_RE = re.compile(r"\b(\d{2,3})\s*(?:km\s*/?\s*h|kmh)\b", re.IGNORECASE)


def details_from_observation(kind: str | None, text: str) -> list[str]:
    if not kind or not text:
        return []
    if kind == "taux":
        return [m.replace(",", ".") for m in DECIMAL_RE.findall(text)][:20]
    if kind == "vitesse":
        return SPEED_RE.findall(text)[:20]
    return []


def is_total(text: str) -> bool:
    n = norm(text)
    return any(n == w or n.startswith(w + " ") for w in TOTAL_WORDS)


def header_columns(cells: list[str]) -> dict[str, int] | None:
    """Repère la ligne d'en-tête (Thématique / Nombre / Observations et variantes)."""
    cols: dict[str, int] = {}
    for idx, c in enumerate(cells):
        n = norm(c)
        if not n:
            continue
        words = set(n.split())
        if "label" not in cols and (words & set(LABEL_KEYWORDS) or n.startswith("thematique")):
            cols["label"] = idx
        elif "value" not in cols and (words & set(NUMBER_KEYWORDS)):
            cols["value"] = idx
        elif "obs" not in cols and (words & set(OBS_KEYWORDS)):
            cols["obs"] = idx
    if "label" in cols and "value" in cols:
        return cols
    return None


def col_letter(idx: int) -> str:
    s = ""
    idx += 1
    while idx:
        idx, r = divmod(idx - 1, 26)
        s = chr(65 + r) + s
    return s


# --- Construction des propositions à partir d'une grille (CSV / XLSX / ODS) ----------------------

def proposals_from_grid(grid: list[list[dict[str, Any]]], indicators: list[IndicatorRef], source_index: int
                        ) -> tuple[list[dict[str, Any]], list[str], dict[str, int] | None]:
    """Chaque cellule : {"v": valeur, "t": texte affiché, "f": formule?, "hidden_row": bool}."""
    warnings: list[str] = []
    header_row, cols = None, None
    for r, row in enumerate(grid[:60]):
        cols = header_columns([c.get("t", "") for c in row])
        if cols:
            header_row = r
            break
    start = (header_row + 1) if header_row is not None else 0
    if cols is None:
        # Pas d'en-tête : la colonne des libellés est celle qui contient le plus d'indicateurs reconnus.
        counts: dict[int, int] = {}
        for row in grid[:200]:
            for ci, c in enumerate(row[:15]):
                ind, score, _ = match_label(c.get("t", ""), indicators)
                if ind and score >= 0.75:
                    counts[ci] = counts.get(ci, 0) + 1
        if not counts:
            return [], ["Aucun tableau d'indicateurs reconnu dans cette feuille."], None
        label_col = max(counts, key=lambda k: counts[k])
        value_col = None
        for ci in range(label_col + 1, label_col + 6):
            numeric = sum(1 for row in grid[:200] if ci < len(row) and isinstance(row[ci].get("v"), (int, float)))
            if numeric >= 3:
                value_col = ci
                break
        if value_col is None:
            return [], ["Colonne « Nombre » introuvable : ajoutez une ligne d'en-tête."], None
        cols = {"label": label_col, "value": value_col}
        obs_col = value_col + 1
        if any(obs_col < len(row) and isinstance(row[obs_col].get("v"), str) for row in grid[:200]):
            cols["obs"] = obs_col
        warnings.append("Pas de ligne d'en-tête trouvée : colonnes déduites du contenu, à vérifier.")
    proposals = []
    for r in range(start, len(grid)):
        row = grid[r]

        def cell(key: str, row: list[dict[str, Any]] = row) -> dict[str, Any]:
            i = cols.get(key) if cols else None
            return row[i] if i is not None and i < len(row) else {}

        lab, val, obs = cell("label"), cell("value"), cell("obs")
        label_text = str(lab.get("t", "")).strip()
        if not label_text and val.get("v") in (None, ""):
            continue
        p: dict[str, Any] = {
            "source_index": source_index, "row_index": r, "raw_label": label_text,
            "label_ref": f"{col_letter(cols['label'])}{r + 1}",
            "value_ref": f"{col_letter(cols['value'])}{r + 1}",
            "observation_ref": f"{col_letter(cols['obs'])}{r + 1}" if "obs" in cols else None,
            "raw_value": str(val.get("t", ""))[:200], "observation": str(obs.get("t", "") or "").strip(),
            "warnings": [], "bbox": None, "value_bbox": None,
        }
        ind, score, w = match_label(label_text, indicators) if label_text else (None, 0.0, [])
        p["warnings"].extend(w)
        if label_text and is_total(label_text):
            p.update(indicator_code=None, match_score=0.0, value=None, value_error=None, details=[],
                     confidence=0.0, decision="ignore")
            p["warnings"].append("Ligne de total : ignorée par défaut (non importée).")
            proposals.append(p)
            continue
        value, err, vw = parse_value(val.get("v") if val.get("v") is not None else val.get("t"))
        p["warnings"].extend(vw)
        if val.get("f"):
            p["warnings"].append(f"Valeur issue d'une formule ({str(val['f'])[:60]}), non recalculée : vérifiez.")
            if val.get("v") is None:
                err = "Formule sans valeur calculée dans le fichier."
        if val.get("hidden_row") or lab.get("hidden_row"):
            p["warnings"].append("Ligne masquée dans le fichier source.")
        if lab.get("merged") or val.get("merged"):
            p["warnings"].append("Cellule fusionnée : vérifiez l'association libellé / valeur.")
        if not label_text:
            p["warnings"].append("Valeur sans libellé.")
        if ind is None and label_text:
            p["warnings"].append("Libellé non reconnu : choisissez l'indicateur ou ignorez la ligne.")
        details = details_from_observation(ind.detail_kind if ind else None, p["observation"])
        if details:
            p["warnings"].append("Taux / vitesses extraits de l'observation : à confirmer.")
        confidence = score if ind else 0.0
        if err or vw:
            confidence = min(confidence, 0.6)
        p.update(indicator_code=ind.code if ind else None, match_score=score, value=value, value_error=err,
                 details=details, confidence=round(confidence, 3))
        sure = ind is not None and score >= 0.9 and not err and not p["warnings"]
        p["decision"] = "accepte" if sure else "a_verifier"
        proposals.append(p)
    _flag_duplicates(proposals)
    return proposals, warnings, cols


def _flag_duplicates(proposals: list[dict[str, Any]]) -> None:
    seen: dict[str, list[dict[str, Any]]] = {}
    for p in proposals:
        if p.get("indicator_code") and p.get("decision") != "ignore":
            seen.setdefault(p["indicator_code"], []).append(p)
    for items in seen.values():
        if len(items) > 1:
            for p in items:
                p["warnings"].append(f"Doublon : cet indicateur apparaît {len(items)} fois dans le fichier.")
                p["decision"] = "a_verifier"
                p["duplicate"] = True


# --- Construction des propositions à partir de mots positionnés (OCR / PDF) ----------------------

@dataclass
class Word:
    text: str
    x0: float
    y0: float
    x1: float
    y1: float
    conf: float  # 0–1

    @property
    def yc(self) -> float:
        return (self.y0 + self.y1) / 2

    @property
    def h(self) -> float:
        return self.y1 - self.y0


def group_rows(words: list[Word]) -> list[list[Word]]:
    words = [w for w in words if w.text.strip()]
    if not words:
        return []
    heights = sorted(w.h for w in words if w.h > 0)
    tol = (heights[len(heights) // 2] if heights else 10) * 0.55
    rows: list[list[Word]] = []
    for w in sorted(words, key=lambda w: w.yc):
        if rows and abs(w.yc - sum(x.yc for x in rows[-1]) / len(rows[-1])) <= tol:
            rows[-1].append(w)
        else:
            rows.append([w])
    return [sorted(r, key=lambda w: w.x0) for r in rows]


def _bbox(words: list[Word]) -> dict[str, float] | None:
    if not words:
        return None
    x0, y0 = min(w.x0 for w in words), min(w.y0 for w in words)
    return {"x": round(x0, 1), "y": round(y0, 1), "w": round(max(w.x1 for w in words) - x0, 1),
            "h": round(max(w.y1 for w in words) - y0, 1)}


def _is_numberish(text: str) -> bool:
    t = text.strip(".:;,()[]")
    return bool(t) and (t.isdigit() or (len(t) <= 4 and t.translate(OCR_DIGIT_FIXES).isdigit()
                                          and any(ch.isdigit() for ch in t)))


def _median(values: list[float], default: float) -> float:
    values = sorted(v for v in values if v > 0)
    return values[len(values) // 2] if values else default


INDEX_RE = re.compile(r"^\d{1,2}[.)]?\s+")


def _finish(p: dict[str, Any], ind: IndicatorRef | None, score: float, label_words: list[Word],
            num_words: list[Word], ocr: bool) -> dict[str, Any]:
    raw_value = " ".join(w.text for w in num_words)
    value, err, vw = parse_value(raw_value if raw_value else None, ocr=ocr)
    num_conf = min((w.conf for w in num_words), default=1.0 if not ocr else 0.0)
    label_conf = min((w.conf for w in label_words), default=0.0) if ocr else 1.0
    p.update(raw_value=raw_value, value=value, value_error=err, value_bbox=_bbox(num_words),
             indicator_code=ind.code if ind else None, match_score=score)
    p["warnings"].extend(vw)
    if not num_words:
        p["warnings"].append("Aucun nombre lu sur cette ligne : valeur laissée non renseignée.")
    if ind is None:
        p["warnings"].append("Libellé non reconnu : choisissez l'indicateur ou ignorez la ligne.")
    if ocr and num_words and num_conf < 0.8:
        p["warnings"].append(f"Chiffre lu avec une confiance de {round(100 * num_conf)} % : vérifiez sur l'image.")
    if p["raw_label"] and is_total(p["raw_label"]):
        p["warnings"].append("Ligne de total : ignorée par défaut (non importée).")
        p["decision"] = "ignore"
        p["indicator_code"] = None
    p["details"] = details_from_observation(ind.detail_kind if ind else None, p["observation"])
    p["confidence"] = round(min(score, num_conf, max(label_conf, 0.5)) if ind else 0.0, 3)
    p.setdefault("decision", "a_verifier" if ocr else
                 ("accepte" if ind and score >= 0.9 and not err and not p["warnings"] else "a_verifier"))
    return p


def _new_prop(source_index: int, row_index: int, label: str, words: list[Word]) -> dict[str, Any]:
    return {"source_index": source_index, "row_index": row_index, "raw_label": label, "observation": "",
            "warnings": [], "bbox": _bbox(words), "label_ref": None, "value_ref": None, "observation_ref": None}


def proposals_from_words(words: list[Word], indicators: list[IndicatorRef], source_index: int, *, ocr: bool
                         ) -> tuple[list[dict[str, Any]], list[str], list[str]]:
    """Renvoie (propositions, avertissements, lignes non exploitées).

    Avec un en-tête « Nombre » repéré, le tableau est reconstruit par blocs : chaque libellé (éventuellement
    sur plusieurs lignes) forme une ligne du tableau ; les nombres et paragraphes d'observation sont rattachés
    au libellé le plus proche verticalement (cellules centrées ou alignées en haut)."""
    lines = group_rows(words)
    warnings: list[str] = []
    unused: list[str] = []
    value_x = None
    obs_x0 = None
    header_idx = None
    for i, row in enumerate(lines):
        texts = [norm(w.text) for w in row]
        if any(t in NUMBER_KEYWORDS for t in texts) and any(
                t.startswith(("thematique", "infraction", "indicateur", "nature", "observ")) for t in texts):
            for w, t in zip(row, texts):
                if t in NUMBER_KEYWORDS and value_x is None:
                    value_x = (w.x0, w.x1)
                if t.startswith("observ") and obs_x0 is None:
                    obs_x0 = w.x0
            header_idx = i
            break
    if value_x is None or header_idx is None:
        warnings.append("En-tête du tableau non repéré : colonnes déduites ligne par ligne, à vérifier.")
        props, unused = _per_line(lines, indicators, source_index, ocr)
        _flag_duplicates(props)
        return props, warnings, unused
    h = _median([w.h for w in words], 10.0)
    margin = (value_x[1] - value_x[0]) * 0.6 + h
    right = value_x[1] + margin
    if obs_x0 is not None and obs_x0 > value_x[1]:
        right = min(right, obs_x0 - 0.5 * h)
    vz = (value_x[0] - margin, right)
    blocks: list[dict[str, Any]] = []
    numbers: list[list[Word]] = []
    obs_lines: list[list[Word]] = []
    for line in lines[header_idx + 1:]:
        lw = [w for w in line if w.x1 <= vz[0] + 2]
        nw = [w for w in line if vz[0] - 2 < (w.x0 + w.x1) / 2 < vz[1] and w not in lw]
        ow = [w for w in line if w not in lw and w not in nw]
        if len(nw) > 2 or any(len(w.text) > 3 and not _is_numberish(w.text) for w in nw):
            unused.append(" ".join(w.text for w in line)[:120])  # texte pleine largeur (légende, pied de page)
            continue
        if nw:
            numbers.append(nw)
        if ow:
            obs_lines.append(ow)
        if not lw:
            continue
        text = " ".join(w.text for w in lw)
        ind, score, _ = match_label(INDEX_RE.sub("", text), indicators)
        rest = INDEX_RE.sub("", text)
        is_start = (ind is not None and score >= 0.6) or (bool(INDEX_RE.match(text)) and rest[:1].isupper())
        prev = blocks[-1] if blocks else None
        if prev is not None and not is_start and min(w.y0 for w in lw) - prev["y1"] <= 0.9 * h:
            prev["words"].extend(lw)
            prev["y1"] = max(prev["y1"], max(w.y1 for w in lw))
        else:
            blocks.append({"words": list(lw), "y0": min(w.y0 for w in lw), "y1": max(w.y1 for w in lw),
                           "nums": [], "obs": []})
    if not blocks:
        return [], ["Aucune ligne de tableau repérée sous l'en-tête."], [" ".join(w.text for w in ln) for ln in lines][:20]
    centers = [(b["y0"] + b["y1"]) / 2 for b in blocks]
    pitch = _median([b - a for a, b in zip(centers, centers[1:])], 3 * h)

    def nearest(yc: float) -> int | None:
        best = min(range(len(blocks)), key=lambda k: abs(centers[k] - yc))
        return best if abs(centers[best] - yc) <= 0.75 * pitch else None

    for nw in numbers:
        k = nearest(sum(w.yc for w in nw) / len(nw))
        if k is None:
            unused.append(" ".join(w.text for w in nw))
        else:
            blocks[k]["nums"].append(nw)
    paragraphs: list[list[Word]] = []
    for ow in obs_lines:
        if paragraphs and min(w.y0 for w in ow) - max(w.y1 for w in paragraphs[-1]) <= 0.6 * h:
            paragraphs[-1].extend(ow)
        else:
            paragraphs.append(list(ow))
    for para in paragraphs:
        k = nearest((min(w.y0 for w in para) + max(w.y1 for w in para)) / 2)
        if k is None:
            unused.append(" ".join(w.text for w in para)[:120])
        else:
            blocks[k]["obs"].append(para)
    proposals = []
    for i, b in enumerate(blocks):
        label = INDEX_RE.sub("", " ".join(w.text for w in sorted(b["words"], key=lambda w: (round(w.yc / h), w.x0))))
        ind, score, mw = match_label(label, indicators)
        if ind is None and not b["nums"]:
            unused.append(label[:120])
            continue
        nums = b["nums"][0] if b["nums"] else []
        all_words = b["words"] + [w for n in b["nums"] for w in n] + [w for o in b["obs"] for w in o]
        p = _new_prop(source_index, i, label, all_words)
        p["value_zone"] = [round(vz[0], 1), round(vz[1], 1)]
        p["warnings"].extend(mw)
        if len(b["nums"]) > 1:
            p["warnings"].append("Plusieurs nombres rattachés à cette ligne : vérifiez la valeur.")
        p["observation"] = " ".join(" ".join(w.text for w in sorted(o, key=lambda w: (round(w.yc / h), w.x0)))
                                    for o in b["obs"]).strip()
        proposals.append(_finish(p, ind, score, b["words"], nums, ocr))
    _flag_duplicates(proposals)
    return proposals, warnings, unused


def _per_line(rows: list[list[Word]], indicators: list[IndicatorRef], source_index: int, ocr: bool
              ) -> tuple[list[dict[str, Any]], list[str]]:
    """Sans en-tête : une ligne de texte = une ligne du tableau (libellé, puis premier nombre, puis observation)."""
    proposals: list[dict[str, Any]] = []
    unused: list[str] = []
    last: dict[str, Any] | None = None
    for i, row in enumerate(rows):
        label_words: list[Word] = []
        num_words: list[Word] = []
        obs_words: list[Word] = []
        for j, w in enumerate(row):
            if _is_numberish(w.text) and label_words:
                ind_with, s_with, _ = match_label(" ".join(x.text for x in label_words + [w]), indicators)
                _, s_without, _ = match_label(" ".join(x.text for x in label_words), indicators)
                if ind_with and s_with > s_without + 0.02:  # le nombre fait partie du libellé (ex. « 50 »)
                    label_words.append(w)
                    continue
                num_words, obs_words = [w], row[j + 1:]
                break
            label_words.append(w)
        label = INDEX_RE.sub("", " ".join(w.text for w in label_words).strip())
        ind, score, mw = match_label(label, indicators) if label else (None, 0.0, [])
        if ind is None and not num_words:
            if last is not None and not label_words and obs_words:
                last["observation"] = (last["observation"] + " " + " ".join(w.text for w in obs_words)).strip()
                continue
            unused.append(" ".join(w.text for w in row)[:120])
            continue
        p = _new_prop(source_index, i, label, row)
        p["warnings"].extend(mw)
        p["observation"] = " ".join(w.text for w in obs_words).strip()
        proposals.append(_finish(p, ind, score, label_words, num_words, ocr))
        last = p
    return proposals, unused
