import json
import os
from datetime import datetime

# --- CONFIGURAZIONE ---
# Il nome esatto del tuo file gigante
SOURCE_FILE = "Clothing_Shoes_and_Jewelry.jsonl" 
# Il file snello che creeremo per il training
OUTPUT_FILE = "reviews_cleaned_final.json" 

# QUANTE RECENSIONI VOGLIAMO ALLA FINE?
# 500.000 sono ottime per una tesi su un Mac M1.
# Se ti senti coraggioso metti 1.000.000, ma il training sarà più lento.
TARGET_COUNT = 500000 

def process_heavy_file():
    print(f"--- 1. Analisi del Gigante: {SOURCE_FILE} (28GB) ---")
    print("⚠️  Nota: Leggerò il file riga per riga per non occupare RAM.")
    
    if not os.path.exists(SOURCE_FILE):
        print(f"❌ Errore: Non trovo {SOURCE_FILE} nel Desktop/tesi.")
        return

    count_saved = 0
    count_scanned = 0
    
    # Apriamo il file in output in modalità scrittura
    with open(OUTPUT_FILE, 'w', encoding='utf-8') as f_out:
        
        # Apriamo il file gigante in lettura (streaming)
        with open(SOURCE_FILE, 'r', encoding='utf-8') as f_in:
            
            for line in f_in:
                count_scanned += 1
                
                # Feedback a video ogni 100k righe scansionate
                if count_scanned % 100000 == 0:
                    print(f"   -> Scansionate: {count_scanned} | Salvate: {count_saved}...")
                
                try:
                    # Parsing della singola riga
                    data = json.loads(line)
                    
                    # --- FILTRI INTELLIGENTI ---
                    
                    # 1. Filtro Voto (Teniamo solo recensioni positive per il recommender)
                    # (Nei dataset nuovi a volte è 'rating', a volte 'overall')
                    rating = data.get('rating') or data.get('overall')
                    if not rating or float(rating) < 4.0:
                        continue
                        
                    # 2. Verifica Dati Essenziali
                    user = data.get('reviewerID') or data.get('user_id')
                    item = data.get('asin') or data.get('parent_asin')
                    
                    if not user or not item:
                        continue
                        
                    # 3. Pulizia Payload (Salviamo solo ciò che serve per risparmiare spazio)
                    clean_obj = {
                        "reviewerID": user,
                        "asin": item,
                        "overall": float(rating)
                    }
                    
                    # Scriviamo nel file pulito (una riga JSON alla volta)
                    json.dump(clean_obj, f_out)
                    f_out.write('\n')
                    
                    count_saved += 1
                    
                    # STOP APPENA RAGGIUNTO IL TARGET
                    if count_saved >= TARGET_COUNT:
                        print(f"\n🎯 OBIETTIVO RAGGIUNTO: {TARGET_COUNT} recensioni salvate!")
                        break
                        
                except Exception as e:
                    continue

    print("\n--- RIEPILOGO ---")
    print(f"Righe totali lette dal file gigante: {count_scanned}")
    print(f"Recensioni valide salvate in '{OUTPUT_FILE}': {count_saved}")
    print("✅ Ora puoi usare questo file 'piccolo' per addestrare LightFM velocemente!")

if __name__ == "__main__":
    process_heavy_file()