"""
Riprocessa i metadati raw escludendo gioielli e orologi.
Tiene solo: vestiti, scarpe e accessori moda.

Input:  raw/meta_Clothing_Shoes_and_Jewelry.jsonl (17GB)
Output: artifacts/metadata_cleaned.json (sovrascrive il precedente)

Tempo stimato: 5-10 minuti.
"""
import json
import os
import sys

PROJECT_ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
SOURCE_FILE = os.path.join(PROJECT_ROOT, "raw", "meta_Clothing_Shoes_and_Jewelry.jsonl")
OUTPUT_FILE = os.path.join(PROJECT_ROOT, "artifacts", "metadata_cleaned.json")
TARGET_COUNT = 100000

# Categorie da escludere
EXCLUDED_CATEGORIES = {"jewelry", "watches", "watch accessories", "jewelry accessories"}


def has_excluded_category(categories):
    for c in categories:
        if isinstance(c, str) and c.lower() in EXCLUDED_CATEGORIES:
            return True
        if isinstance(c, list):
            for sub in c:
                if isinstance(sub, str) and sub.lower() in EXCLUDED_CATEGORIES:
                    return True
    return False


def main():
    print(f"--- RIPROCESSAMENTO METADATI (senza gioielli/orologi) ---")
    print(f"Target: {TARGET_COUNT:,} prodotti")

    if not os.path.exists(SOURCE_FILE):
        print(f"Errore: {SOURCE_FILE} non trovato!")
        sys.exit(1)

    count_saved = 0
    count_scanned = 0
    count_excluded = 0
    clean_products = []

    with open(SOURCE_FILE, "r", encoding="utf-8") as f_in:
        for line in f_in:
            count_scanned += 1

            if count_scanned % 100000 == 0:
                print(f"   Scansionati: {count_scanned:,} | Salvati: {count_saved:,} | Esclusi (jewelry): {count_excluded:,}")

            try:
                data = json.loads(line)

                asin = data.get("asin") or data.get("parent_asin")
                title = data.get("title")
                if not asin or not title or len(title) <= 3:
                    continue

                categories = data.get("categories", [])

                # Escludi gioielli e orologi
                if has_excluded_category(categories):
                    count_excluded += 1
                    continue

                # Recupero immagine
                image_url = ""
                if "images" in data and data["images"]:
                    img_entry = data["images"][0]
                    image_url = (
                        img_entry.get("large", "")
                        or img_entry.get("hi_res", "")
                        or img_entry.get("thumb", "")
                    )
                elif "imUrl" in data:
                    image_url = data["imUrl"]

                clean_products.append({
                    "asin": asin,
                    "title": title,
                    "brand": data.get("brand", "Generic"),
                    "price": data.get("price", "N/A"),
                    "categories": categories,
                    "image_url": image_url,
                })
                count_saved += 1

                if count_saved >= TARGET_COUNT:
                    break

            except Exception:
                continue

    print(f"\n--- RIEPILOGO ---")
    print(f"Righe scansionate: {count_scanned:,}")
    print(f"Prodotti esclusi (jewelry/watches): {count_excluded:,}")
    print(f"Prodotti salvati: {len(clean_products):,}")

    with open(OUTPUT_FILE, "w", encoding="utf-8") as f_out:
        json.dump(clean_products, f_out, indent=2)

    print(f"Output: {OUTPUT_FILE}")


if __name__ == "__main__":
    main()
