""" Evaluation of the Recommender Agent — Hybrid LightFM. Three groups of metrics: 1. **Category Coherence** (content-based): For a sample of anchors, checks whether the top-k recommended products share at least one category with the anchor. 2. **Leave-One-Out User Evaluation** (collaborative): For each user with >= 2 interactions in the model: - Hides 1 interaction (the last one) - Uses model.predict() to get the scores of all items for that user - Checks whether the hidden item appears in the top-k Metrics: Hit Rate@k, nDCG@k, MRR. Note: the model is retrained from scratch on the training interactions only (without the hidden ones), so the evaluation is correct and there is no data leakage. 3. **Category Depth** (sub-categories): Measures the coherence at the macro-category and sub-category level. """ import json import os import pickle import sys import warnings warnings.filterwarnings("ignore") import numpy as np from collections import defaultdict, Counter PROJECT_ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) sys.path.insert(0, PROJECT_ROOT) from utils.config import RECOMMENDER_MODEL_PATH, METADATA_PATH, RECOMMENDER_K def load_data(): """Loads the LightFM model, metadata and reviews.""" # Model print("Loading LightFM model...") with open(RECOMMENDER_MODEL_PATH, "rb") as f: package = pickle.load(f) model = package["model"] idx_to_item = package.get("idx_to_item", {}) item_features = package.get("item_features_matrix", None) print(f" Model loaded: {len(idx_to_item)} items") # Metadata print("Loading metadata...") with open(METADATA_PATH, "r", encoding="utf-8") as f: metadata_list = json.load(f) metadata_by_asin = {p["asin"]: p for p in metadata_list} print(f" {len(metadata_by_asin)} products") # Reviews reviews_path = os.path.join(PROJECT_ROOT, "artifacts", "reviews_cleaned.json") print("Loading reviews...") reviews = [] with open(reviews_path, "r", encoding="utf-8") as f: for line in f: reviews.append(json.loads(line.strip())) print(f" {len(reviews)} interactions") return model, idx_to_item, item_features, metadata_by_asin, reviews def get_top_k_similar(model, item_features, anchor_pos, valid_vectors, k): """Computes the top-k products similar to the anchor (same logic as the Recommender Agent).""" target_vector = valid_vectors[anchor_pos] all_scores = valid_vectors @ target_vector all_scores[anchor_pos] = -np.inf top_count = min(k, len(all_scores) - 1) top_pos = np.argpartition(all_scores, -top_count)[-top_count:] top_pos = top_pos[np.argsort(all_scores[top_pos])[::-1]] return top_pos # ===================================================================== # 1. CATEGORY COHERENCE # ===================================================================== def eval_category_coherence(model, idx_to_item, item_features, metadata_by_asin, k=RECOMMENDER_K, n_samples=500): """Measures how many of the top-k products share at least one category with the anchor.""" print(f"\n{'='*60}") print(f"CATEGORY COHERENCE (k={k}, {n_samples} anchors)") print(f"{'='*60}") if item_features is not None: _, precomputed_vectors = model.get_item_representations(features=item_features) else: precomputed_vectors = model.item_embeddings valid_indices = [] for idx, asin in idx_to_item.items(): if asin in metadata_by_asin: valid_indices.append(idx) valid_indices = np.array(valid_indices) valid_vectors = precomputed_vectors[valid_indices] rng = np.random.RandomState(42) sample_positions = rng.choice(len(valid_indices), size=min(n_samples, len(valid_indices)), replace=False) precisions = [] for pos in sample_positions: anchor_idx = valid_indices[pos] anchor_asin = idx_to_item[anchor_idx] anchor_meta = metadata_by_asin.get(anchor_asin, {}) anchor_cats = set(anchor_meta.get("categories", [])) if not anchor_cats: continue top_positions = get_top_k_similar(model, item_features, pos, valid_vectors, k) matches = 0 for rec_pos in top_positions: rec_idx = valid_indices[rec_pos] rec_asin = idx_to_item[rec_idx] rec_meta = metadata_by_asin.get(rec_asin, {}) rec_cats = set(rec_meta.get("categories", [])) if anchor_cats & rec_cats: matches += 1 precisions.append(matches / len(top_positions)) mean_coherence = np.mean(precisions) print(f"\nCategory Coherence@{k}: {mean_coherence:.4f}") print(f" (% of recommendations sharing at least 1 category with the anchor)") print(f" Sample: {len(precisions)} valid anchors") bins = [0, 0.2, 0.4, 0.6, 0.8, 1.01] hist, _ = np.histogram(precisions, bins=bins) print(f"\n Coherence distribution:") for i in range(len(bins)-1): pct = hist[i] / len(precisions) * 100 bar = "#" * int(pct / 2) print(f" {bins[i]:.1f}-{bins[i+1]:.1f}: {hist[i]:>4} ({pct:>5.1f}%) {bar}") return mean_coherence # ===================================================================== # 2. CATEGORY DEPTH # ===================================================================== def eval_category_depth(model, idx_to_item, item_features, metadata_by_asin, k=RECOMMENDER_K, n_samples=500): """Measures the coherence at the sub-category level (level 3-4 of the hierarchy).""" print(f"\n{'='*60}") print(f"CATEGORY DEPTH COHERENCE (k={k}, {n_samples} anchors)") print(f"{'='*60}") if item_features is not None: _, precomputed_vectors = model.get_item_representations(features=item_features) else: precomputed_vectors = model.item_embeddings valid_indices = [] for idx, asin in idx_to_item.items(): if asin in metadata_by_asin: valid_indices.append(idx) valid_indices = np.array(valid_indices) valid_vectors = precomputed_vectors[valid_indices] rng = np.random.RandomState(42) sample_positions = rng.choice(len(valid_indices), size=min(n_samples, len(valid_indices)), replace=False) coherence_by_level = defaultdict(list) for pos in sample_positions: anchor_idx = valid_indices[pos] anchor_asin = idx_to_item[anchor_idx] anchor_meta = metadata_by_asin.get(anchor_asin, {}) anchor_cats = anchor_meta.get("categories", []) if len(anchor_cats) < 3: continue top_positions = get_top_k_similar(model, item_features, pos, valid_vectors, k) for level in [2, 3]: if len(anchor_cats) <= level: continue anchor_cat_at_level = anchor_cats[level] matches = 0 counted = 0 for rec_pos in top_positions: rec_idx = valid_indices[rec_pos] rec_asin = idx_to_item[rec_idx] rec_meta = metadata_by_asin.get(rec_asin, {}) rec_cats = rec_meta.get("categories", []) if len(rec_cats) > level: counted += 1 if rec_cats[level] == anchor_cat_at_level: matches += 1 if counted > 0: coherence_by_level[level].append(matches / counted) for level in sorted(coherence_by_level.keys()): values = coherence_by_level[level] mean_val = np.mean(values) label = {2: "Macro-category (level 3)", 3: "Sub-category (level 4)"} print(f"\n {label.get(level, f'Level {level+1}')}: {mean_val:.4f}") print(f" Sample: {len(values)} anchors") return coherence_by_level # ===================================================================== # 3. LEAVE-ONE-OUT — Collaborative evaluation with model.predict() # ===================================================================== def eval_leave_one_out(metadata_by_asin, reviews, k_values=None): """Leave-one-out: retrains LightFM on train, evaluates on test with predict(). For each user with >= 2 interactions: - Hides the last interaction (test) - The model is trained only on the remaining interactions (train) - predict() computes the score of all items for that user - Checks whether the hidden item is in the top-k The model is retrained from scratch (it does not use the pre-trained model.pkl) to avoid data leakage: the pre-trained model has seen all the interactions. """ if k_values is None: k_values = [RECOMMENDER_K, 20, 50, 100] print(f"\n{'='*60}") print(f"LEAVE-ONE-OUT USER EVALUATION") print(f"{'='*60}") # --- Prepare interactions per user --- user_items = defaultdict(list) all_items = set() for r in reviews: user_items[r["reviewerID"]].append(r["asin"]) all_items.add(r["asin"]) # Filter users with >= 2 interactions eligible = {uid: items for uid, items in user_items.items() if len(items) >= 2} print(f" Total users: {len(user_items)}") print(f" Users with >= 2 interactions: {len(eligible)}") if not eligible: print(" No eligible users!") return {} # --- Train/test split --- print(" Leave-one-out split...") train_interactions = [] test_set = {} # {user_id: hidden_asin} all_users = set() for uid, items in eligible.items(): test_set[uid] = items[-1] # Hides the last one for item in items[:-1]: # Train: all except the last one train_interactions.append((uid, item)) all_users.add(uid) # Also add users with only 1 interaction to the train set (you don't test them but they help the model) for uid, items in user_items.items(): if uid not in eligible: for item in items: train_interactions.append((uid, item)) all_users.add(uid) print(f" Train interactions: {len(train_interactions):,}") print(f" Test users: {len(test_set):,}") # --- Retrain LightFM from scratch on the train set --- print(" Retraining LightFM on the train set...") from lightfm.data import Dataset as LFMDataset from lightfm import LightFM # Rebuild item features (same logic as 4_train_recommender.py) EXCLUDED_EXACT = {"Clothing, Shoes & Jewelry"} NOISE_KEYWORDS = { "sale", "off", "save", "deal", "prime", "gift", "clearance", "black friday", "holiday", "test", "exclusion", "shopbop", "cyber", "top 50", "top rated", "featured", "our brands" } MIN_CATEGORY_COUNT = 10 PRICE_BUCKETS = [ ("price:budget", 0, 15), ("price:affordable", 15, 30), ("price:mid", 30, 50), ("price:premium", 50, 100), ("price:luxury", 100, 10000), ] def flatten_categories(categories): result = [] for c in categories: if isinstance(c, str): result.append(c) elif isinstance(c, list): for sub in c: if isinstance(sub, str): result.append(sub) return result def get_price_feature(price): if price is None or price == "N/A": return None try: p = float(price) for name, lo, hi in PRICE_BUCKETS: if lo <= p < hi: return name return "price:luxury" except (ValueError, TypeError): return None # Category analysis on the items in the train set train_items = set(item for _, item in train_interactions) cat_counter = Counter() for asin in train_items: p = metadata_by_asin.get(asin) if p: for c in flatten_categories(p.get("categories", [])): if c not in EXCLUDED_EXACT: cat_counter[c] += 1 allowed_categories = set() for cat, count in cat_counter.items(): if count < MIN_CATEGORY_COUNT: continue lower = cat.lower() if any(kw in lower for kw in NOISE_KEYWORDS): continue allowed_categories.add(cat) price_feature_names = [name for name, _, _ in PRICE_BUCKETS] all_feature_names = sorted(allowed_categories) + price_feature_names # Build LightFM dataset dataset = LFMDataset() dataset.fit( users=list(all_users), items=list(all_items), item_features=all_feature_names, ) interactions_matrix, _ = dataset.build_interactions(train_interactions) # Build item features item_feature_tuples = [] for asin in all_items: p = metadata_by_asin.get(asin) features = [] if p: for c in flatten_categories(p.get("categories", [])): if c in allowed_categories: features.append(c) price_feat = get_price_feature(p.get("price")) if price_feat: features.append(price_feat) if features: item_feature_tuples.append((asin, features)) item_features_matrix = dataset.build_item_features(item_feature_tuples, normalize=True) # Train model = LightFM(loss="warp", no_components=10) model.fit(interactions_matrix, item_features=item_features_matrix, epochs=30, num_threads=2) print(" Model trained!") # --- Mappings --- user_id_map, _, item_id_map, _ = dataset.mapping() n_items = len(item_id_map) all_item_ids = np.arange(n_items) # --- Evaluation --- print(f" Evaluating on {len(test_set)} users...") # Sample max 1000 users for speed rng = np.random.RandomState(42) test_users = list(test_set.keys()) if len(test_users) > 1000: test_users = rng.choice(test_users, size=1000, replace=False).tolist() max_k = max(k_values) hits = {k: 0 for k in k_values} reciprocal_ranks = [] ndcg_values = {k: [] for k in k_values} total = 0 skipped = 0 for i, uid in enumerate(test_users): if (i + 1) % 200 == 0: print(f" {i+1}/{len(test_users)}...") hidden_asin = test_set[uid] # Convert to internal LightFM indices if uid not in user_id_map or hidden_asin not in item_id_map: skipped += 1 continue user_internal = user_id_map[uid] hidden_internal = item_id_map[hidden_asin] # Predict score for all items scores = model.predict(user_internal, all_item_ids, item_features=item_features_matrix) # Exclude items already in this user's train set train_asins = [item for item in eligible[uid][:-1]] for train_asin in train_asins: if train_asin in item_id_map: scores[item_id_map[train_asin]] = -np.inf # Ranking ranked_items = np.argsort(scores)[::-1] # Find the position of the hidden item in the ranking rank = np.where(ranked_items == hidden_internal)[0] if len(rank) == 0: skipped += 1 continue rank = rank[0] + 1 # 1-indexed total += 1 # MRR reciprocal_ranks.append(1.0 / rank) # Hit Rate and nDCG for each k for k in k_values: if rank <= k: hits[k] += 1 ndcg_values[k].append(1.0 / np.log2(rank + 1)) else: ndcg_values[k].append(0.0) if total == 0: print(" No valid tests!") return {} # --- Results --- print(f"\n Users evaluated: {total} (skipped: {skipped})") results = {} for k in k_values: hr = hits[k] / total ndcg = np.mean(ndcg_values[k]) results[k] = {"hit_rate": hr, "ndcg": ndcg, "hits": hits[k], "total": total} print(f" Hit Rate@{k}: {hr:.4f} ({hits[k]}/{total})") print(f" nDCG@{k}: {ndcg:.4f}") mrr = np.mean(reciprocal_ranks) results["mrr"] = mrr print(f" MRR: {mrr:.4f}") return results def main(): print("=" * 60) print("RECOMMENDER AGENT EVALUATION — Hybrid LightFM") print(f"Algorithm: WARP, no_components=10, 30 epochs") print(f"Features: categories + 5 price buckets") print("=" * 60) model, idx_to_item, item_features, metadata_by_asin, reviews = load_data() # 1. Category Coherence (uses the model pre-trained on all the data — fine for content-based) coherence = eval_category_coherence(model, idx_to_item, item_features, metadata_by_asin) # 2. Category Depth depth = eval_category_depth(model, idx_to_item, item_features, metadata_by_asin) # 3. Leave-One-Out (retrains from scratch on the train set — correct collaborative evaluation) loo_results = eval_leave_one_out(metadata_by_asin, reviews) # --- Summary --- print(f"\n{'='*60}") print(f"SUMMARY") print(f"{'='*60}") print(f" Category Coherence@{RECOMMENDER_K}: {coherence:.4f}") if 2 in depth: print(f" Macro-cat Coherence@{RECOMMENDER_K}: {np.mean(depth[2]):.4f}") if 3 in depth: print(f" Sub-cat Coherence@{RECOMMENDER_K}: {np.mean(depth[3]):.4f}") if loo_results: if RECOMMENDER_K in loo_results: print(f" Hit Rate@{RECOMMENDER_K}: {loo_results[RECOMMENDER_K]['hit_rate']:.4f}") print(f" nDCG@{RECOMMENDER_K}: {loo_results[RECOMMENDER_K]['ndcg']:.4f}") if "mrr" in loo_results: print(f" MRR: {loo_results['mrr']:.4f}") # --- Save to file --- results_path = os.path.join(PROJECT_ROOT, "evaluation", "recommender_results.txt") with open(results_path, "w", encoding="utf-8") as f: f.write(f"RECOMMENDER AGENT EVALUATION — Hybrid LightFM\n") f.write(f"Algorithm: WARP, no_components=10, 30 epochs\n") f.write(f"Features: categories + 5 price buckets\n\n") f.write(f"--- Content-Based (full model) ---\n") f.write(f"Category Coherence@{RECOMMENDER_K}: {coherence:.4f}\n") if 2 in depth: f.write(f"Macro-cat Coherence@{RECOMMENDER_K}: {np.mean(depth[2]):.4f}\n") if 3 in depth: f.write(f"Sub-cat Coherence@{RECOMMENDER_K}: {np.mean(depth[3]):.4f}\n") if loo_results: f.write(f"\n--- Collaborative (leave-one-out, model retrained on train) ---\n") for k in [RECOMMENDER_K, 20, 50, 100]: if k in loo_results: f.write(f"Hit Rate@{k}: {loo_results[k]['hit_rate']:.4f} ({loo_results[k]['hits']}/{loo_results[k]['total']})\n") f.write(f"nDCG@{k}: {loo_results[k]['ndcg']:.4f}\n") if "mrr" in loo_results: f.write(f"MRR: {loo_results['mrr']:.4f}\n") print(f"\nResults saved to: {results_path}") if __name__ == "__main__": main()