#!/usr/bin/env python3
"""
Estrae TUTTO il contenuto dei DVR .docx, per confrontarlo con quello che sta a
database. Nato dopo due segnalazioni del cliente (Roma Centro senza Pericoli,
Castellammare senza punteggi) che venivano entrambe dallo stesso import.

Copre: inquadramento e testi lunghi, pericoli, valutazione, elenco
documentazione incendio, piano di miglioramento storico, allegati.

  python3 import-tools/scripts/audit_docx.py <master_elementi.json> > out.json

Il JSON ha una voce per file; il confronto col database lo fa
app/audit-docx-confronto.mjs.
"""
import sys, os, re, json, glob
from docx import Document
from docx.table import _Cell

DVR_DIR = os.path.join(os.path.dirname(__file__), "..", "..", "-mat", "DVR")

TITOLI = {
    "descrizione_immobile": re.compile(r"DESCRIZIONE IMMOBILE", re.I),
    "piano_emergenze_testo": re.compile(r"PIANO DI GESTIONE EMERGENZE", re.I),
    "rischio_incendio_testo": re.compile(r"RISCHIO INCENDIO", re.I),
    "conclusione_testo": re.compile(r"CONCLUSION", re.I),
}
# Qualsiasi titolo interrompe il blocco di testo precedente.
STOP = re.compile(
    r"^(DESCRIZIONE IMMOBILE|METODO DELLA MATRICE|DATI IDENTIFICATIVI|SOMMARIO|"
    r"PIANO DI GESTIONE|RISCHIO INCENDIO|VALUTAZIONE SECONDO|PIANO DI MIGLIORAMENTO|"
    r"INDIVIDUAZIONE DEI PERICOLI|ALLEGATI|CONCLUSION|ELENCO DOCUMENTAZIONE|"
    r"Azioni da intraprendere)",
    re.I,
)


def norm(s):
    if s is None:
        return None
    s = re.sub(r"[ \t\r\n]+", " ", str(s)).strip()
    return s or None


def key(s):
    return re.sub(r"\s+", " ", (s or "").lower()).strip().strip(".:;")


def righe(v):
    if v is None:
        return []
    return [x for x in (norm(r) for r in str(v).splitlines()) if x]


def celle(row, table):
    """Celle reali della riga: row.cells duplica quelle unite."""
    return [norm(_Cell(tc, table).text) for tc in row._tr.tc_lst]


def testi_lunghi(doc):
    paras = [p.text.strip() for p in doc.paragraphs]
    out = {k: None for k in TITOLI}
    for campo, rx in TITOLI.items():
        idx = next((i for i, t in enumerate(paras) if rx.search(t)), None)
        if idx is None:
            continue
        buf = []
        for t in paras[idx + 1:]:
            if not t:
                continue
            if STOP.search(t):
                break
            buf.append(t)
        out[campo] = norm(" ".join(buf))
    # luogo sicuro: didascalia dopo ALLEGATI
    idx = next((i for i, t in enumerate(paras) if t.strip().upper() == "ALLEGATI"), None)
    out["luogo_sicuro_esterno_descr"] = None
    if idx is not None:
        for t in paras[idx + 1:]:
            if t.strip() and not STOP.search(t):
                out["luogo_sicuro_esterno_descr"] = norm(t)
                break
    return out


def rischi(doc):
    out = {"rischio_sismico_descr": None, "rischio_idro_descr": None, "rischio_ambientale_descr": None}
    for t in doc.tables:
        for row in t.rows:
            c = celle(row, t)
            if not c or not c[0]:
                continue
            k = c[0].upper()
            val = norm(" ".join(x for x in c[1:] if x)) if len(c) > 1 else None
            if "SISMIC" in k and out["rischio_sismico_descr"] is None:
                out["rischio_sismico_descr"] = val
            elif "IDROGEOLOG" in k and out["rischio_idro_descr"] is None:
                out["rischio_idro_descr"] = val
            elif "AMBIENTAL" in k and out["rischio_ambientale_descr"] is None:
                out["rischio_ambientale_descr"] = val
    return out


def pericoli(doc):
    table = next((t for t in doc.tables if len(t.rows) > 40 and len(t.columns) in (6, 7)), None)
    if table is None:
        return []
    out = []
    for row in table.rows[1:]:
        c = celle(row, table)
        if len(c) != 6 or not c[0]:
            continue
        rif = c[0].rstrip(".").strip()
        if not re.match(r"^\d+(\.\d+)*$", rif) or len(rif.split(".")) < 3:
            continue
        out.append({"rif": rif, "col_c": c[2], "col_d": c[3], "col_e": c[4], "col_f": c[5]})
    return out


def valutazione(doc, master):
    seen, out = set(), []
    for t in doc.tables:
        for row in t.rows:
            grezze = [x.text for x in row.cells]  # qui servono le celle espanse
            c = [norm(x) for x in grezze]
            if not c or not c[0]:
                continue
            eid = master.get(key(c[0]))
            if eid is None or eid in seen:
                continue
            seen.add(eid)
            g = lambda i: c[i] if i < len(c) else None
            def n(v):
                v = norm(v)
                try:
                    return int(v) if v is not None and 0 <= int(v) <= 9 else None
                except (TypeError, ValueError):
                    return None
            out.append({
                "elemento_id": eid, "elemento": c[0],
                "figure_esposte": g(2), "rischi_descr": g(3), "misure_prevenzione": g(4),
                "p": n(g(6)), "g": n(g(7)),
                # una riga per ogni sotto-blocco (SEGNALETICA, PORTE INTERNE, …):
                # vanno tenute separate, non compresse in un unico testo
                "misure_h": righe(grezze[-1]) if len(c) > 8 else [],
            })
    return out


def incendio(doc):
    """Tabella ELENCO DOCUMENTAZIONE: documento → stato scritto nel Word."""
    for t in doc.tables:
        prima = celle(t.rows[0], t)
        if prima and prima[0] and "ELENCO DOCUMENTAZIONE" in prima[0].upper():
            out = []
            for row in t.rows[1:]:
                c = celle(row, t)
                if not c or not c[0]:
                    continue
                out.append({"documento": c[0], "stato": norm(" ".join(x for x in c[1:] if x))})
            return out
    return []


def piano(doc):
    """Tabella ATTIVITA' / RESPONSABILE / DATE del piano di miglioramento."""
    for t in doc.tables:
        prima = celle(t.rows[0], t)
        if prima and prima[0] and prima[0].upper().startswith("ATTIVITA"):
            out = []
            for row in t.rows[1:]:
                c = celle(row, t)
                if not c or not c[0]:
                    continue
                out.append({"attivita": c[0], "responsabile": c[1] if len(c) > 1 else None,
                            "date": c[2] if len(c) > 2 else None})
            return out
    return []


def apri(path):
    """python-docx rifiuta i file con un'immagine dal CRC rotto (Torre
    Annunziata): in quel caso si riscrive una copia senza quell'immagine, che
    per leggere i testi non serve."""
    try:
        return Document(path)
    except Exception:
        import zipfile, tempfile
        tmp = os.path.join(tempfile.gettempdir(), "docx_recuperato.docx")
        zin = zipfile.ZipFile(path)
        with zipfile.ZipFile(tmp, "w", zipfile.ZIP_DEFLATED) as zout:
            for it in zin.infolist():
                try:
                    dati = zin.read(it.filename)
                except Exception:
                    if not it.filename.startswith("word/media/"):
                        raise
                    dati = b""
                zout.writestr(it, dati)
        return Document(tmp)


def main():
    master = {key(k): v for k, v in json.load(open(sys.argv[1], encoding="utf-8")).items()}
    result, err = {}, 0
    for f in sorted(glob.glob(os.path.join(DVR_DIR, "*.docx"))):
        name = os.path.basename(f)
        try:
            doc = apri(f)
            rec = testi_lunghi(doc)
            rec.update(rischi(doc))
            rec["pericoli"] = pericoli(doc)
            rec["valutazione"] = valutazione(doc, master)
            rec["incendio"] = incendio(doc)
            rec["piano"] = piano(doc)
            result[name] = rec
        except Exception as e:
            err += 1
            sys.stderr.write(f"ERRORE {name}: {e}\n")
    sys.stderr.write(f"Elaborati {len(result)} docx, {err} errori.\n")
    json.dump(result, sys.stdout, ensure_ascii=False)


if __name__ == "__main__":
    main()
