#!/usr/bin/env python3
"""
FASE 2a: estrae dai .docx i pericoli, CORREGGENDO il bug che scartava le righe
col riferimento che termina col punto (1.1.2., 1.3.1.1., ...). Non tocca il DB:
produce un JSON { filename: [ {rif, col_c, col_d, col_e, col_f}, ... ] }.

  python3 import-tools/scripts/fix_docx_pericoli.py > /tmp/docx_pericoli.json
"""
import sys, os, re, json, glob
from docx import Document
from docx.table import _Cell

DVR_DIR = os.path.join(os.path.dirname(__file__), "..", "..", "-mat", "DVR")


def norm(s):
    if s is None:
        return None
    s = re.sub(r"[ \t\r\n]+", " ", str(s)).strip()
    return s or None


def parse_pericoli(doc):
    # tabella pericoli: >40 righe, prima cella '1'. Le colonne sono 6, ma in
    # qualche DVR le righe-titolo hanno una cella unita in più e python-docx
    # conta 7 colonne: le righe dati restano comunque 6 celle.
    table = None
    for t in doc.tables:
        if len(t.rows) > 40 and len(t.columns) in (6, 7):
            table = t
            break
    if table is None:
        return []
    rows = []
    for row in table.rows[1:]:
        # tc_lst e non row.cells: con le celle unite python-docx duplica il
        # contenuto e la riga sembra avere 7 colonne invece di 6.
        cells = [norm(_Cell(tc, table).text) for tc in row._tr.tc_lst]
        if len(cells) != 6:  # riga-titolo (2 celle) o struttura anomala
            continue
        rif, _req, col_c, col_d, col_e, col_f = cells
        if not rif:
            continue
        rif = rif.rstrip(".").strip()  # FIX: via il punto finale
        if not re.match(r"^\d+(\.\d+)*$", rif):
            continue
        if len(rif.split(".")) < 3:  # come l'import: solo punti foglia
            continue
        rows.append({"rif": rif, "col_c": col_c, "col_d": col_d, "col_e": col_e, "col_f": col_f})
    return rows


def main():
    files = sorted(glob.glob(os.path.join(DVR_DIR, "*.docx")))
    result = {}
    err = 0
    for f in files:
        name = os.path.basename(f)
        try:
            result[name] = parse_pericoli(Document(f))
        except Exception as e:
            err += 1
            sys.stderr.write(f"ERRORE {name}: {e}\n")
    sys.stderr.write(f"Elaborati {len(result)} docx, {err} errori.\n")
    json.dump(result, sys.stdout, ensure_ascii=False)


if __name__ == "__main__":
    main()
