""" Evaluation of the Visual Agent — Visual Similarity with DinoV2. Metrics: 1. **Mean Cosine Similarity**: average score of the top-k FAISS results (how visually close the recommended products are) 2. **Self-retrieval accuracy**: for a sample of images present in the index, checks that the same image is result #1 (sanity check) 3. **Category coherence**: percentage of top-k results in the same macro-category and sub-category as the query image Note: it uses directly the vectors already computed in the FAISS index, without reloading the DinoV2 model (not needed for the evaluation). Original model: Trendyol DinoV2 (trendyol-dino-v2-ecommerce-256d, 256d embedding). Index: FAISS IndexFlatIP on L2-normalized vectors (= cosine similarity). """ import os os.environ["KMP_DUPLICATE_LIB_OK"] = "TRUE" import json import sys import warnings warnings.filterwarnings("ignore") import numpy as np import faiss PROJECT_ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) sys.path.insert(0, PROJECT_ROOT) from utils.config import FAISS_VISUAL_INDEX, METADATA_PATH, VISUAL_MODEL_NAME, VISUAL_SEARCH_K def main(): print("=" * 60) print("VISUAL AGENT EVALUATION — Visual Similarity DinoV2") print(f"Model: {VISUAL_MODEL_NAME}") print("=" * 60) # Load FAISS index print("\nLoading visual FAISS index...") index_path = os.path.join(FAISS_VISUAL_INDEX, "index.faiss") index = faiss.read_index(index_path) print(f" {index.ntotal} vectors, {index.d}d") # Load ASIN map map_path = os.path.join(FAISS_VISUAL_INDEX, "asin_map.json") with open(map_path, "r") as f: asin_map = json.load(f) # Load metadata print("Loading metadata...") with open(METADATA_PATH, "r", encoding="utf-8") as f: metadata_list = json.load(f) metadata_by_asin = {p["asin"]: p for p in metadata_list} print(f" {len(metadata_by_asin)} products") # Extract vectors from the FAISS index (IndexFlatIP keeps them in memory) print("Extracting vectors from the index...") vectors = faiss.rev_swig_ptr(index.get_xb(), index.ntotal * index.d) vectors = np.array(vectors).reshape(index.ntotal, index.d).astype("float32") # Sample N random images n_samples = 500 k = VISUAL_SEARCH_K rng = np.random.RandomState(42) sample_indices = rng.choice(index.ntotal, size=n_samples, replace=False) print(f"\nEvaluating on {n_samples} sample products (k={k})...") all_similarities = [] # Cosine similarity score for each result self_retrieval_hits = 0 # How many times the image itself is #1 top1_similarities = [] # Score of result #1 (after self-exclusion) cat_coherence_l3 = [] # Macro-category coherence (level 3) cat_coherence_l4 = [] # Sub-category coherence (level 4) for i, sample_idx in enumerate(sample_indices): if (i + 1) % 100 == 0: print(f" {i+1}/{n_samples}...") # Use the already-computed vector as the query query = vectors[sample_idx:sample_idx+1] # Search top-k+1 (the first one will be the image itself) scores, indices = index.search(query, k + 1) # Self-retrieval check if indices[0][0] == sample_idx: self_retrieval_hits += 1 result_scores = scores[0][1:k+1] result_indices = indices[0][1:k+1] else: result_scores = scores[0][:k] result_indices = indices[0][:k] # Similarity scores all_similarities.extend(result_scores.tolist()) if len(result_scores) > 0: top1_similarities.append(result_scores[0]) # Category coherence query_asin = asin_map[sample_idx] query_meta = metadata_by_asin.get(query_asin, {}) query_cats = query_meta.get("categories", []) if len(query_cats) >= 3: query_cat_l3 = query_cats[2] matches_l3 = 0 counted_l3 = 0 for idx in result_indices: if idx < 0 or idx >= len(asin_map): continue rec_asin = asin_map[idx] rec_meta = metadata_by_asin.get(rec_asin, {}) rec_cats = rec_meta.get("categories", []) if len(rec_cats) >= 3: counted_l3 += 1 if rec_cats[2] == query_cat_l3: matches_l3 += 1 if counted_l3 > 0: cat_coherence_l3.append(matches_l3 / counted_l3) if len(query_cats) >= 4: query_cat_l4 = query_cats[3] matches_l4 = 0 counted_l4 = 0 for idx in result_indices: if idx < 0 or idx >= len(asin_map): continue rec_asin = asin_map[idx] rec_meta = metadata_by_asin.get(rec_asin, {}) rec_cats = rec_meta.get("categories", []) if len(rec_cats) >= 4: counted_l4 += 1 if rec_cats[3] == query_cat_l4: matches_l4 += 1 if counted_l4 > 0: cat_coherence_l4.append(matches_l4 / counted_l4) # --- Report --- print(f"\n{'='*60}") print(f"RESULTS ({n_samples} products)") print(f"{'='*60}") mean_sim = np.mean(all_similarities) mean_top1 = np.mean(top1_similarities) self_rate = self_retrieval_hits / n_samples print(f"\n Mean Cosine Similarity (top-{k}): {mean_sim:.4f}") print(f" Mean Cosine Similarity (top-1): {mean_top1:.4f}") print(f" Self-retrieval accuracy: {self_rate:.4f} ({self_retrieval_hits}/{n_samples})") if cat_coherence_l3: mean_l3 = np.mean(cat_coherence_l3) print(f" Category coherence (macro): {mean_l3:.4f} ({len(cat_coherence_l3)} samples)") if cat_coherence_l4: mean_l4 = np.mean(cat_coherence_l4) print(f" Category coherence (sub-cat): {mean_l4:.4f} ({len(cat_coherence_l4)} samples)") # Similarity distribution print(f"\n Cosine similarity distribution (top-{k}):") bins = [0, 0.3, 0.4, 0.5, 0.6, 0.7, 0.8, 0.9, 1.01] hist, _ = np.histogram(all_similarities, bins=bins) for i in range(len(bins)-1): pct = hist[i] / len(all_similarities) * 100 bar = "#" * int(pct / 2) print(f" {bins[i]:.1f}-{bins[i+1]:.1f}: {hist[i]:>5} ({pct:>5.1f}%) {bar}") # --- Save to file --- results_path = os.path.join(PROJECT_ROOT, "evaluation", "visual_results.txt") with open(results_path, "w", encoding="utf-8") as f: f.write(f"VISUAL AGENT EVALUATION — Visual Similarity DinoV2\n") f.write(f"Model: {VISUAL_MODEL_NAME}\n") f.write(f"Sample: {n_samples} products\n\n") f.write(f"Mean Cosine Similarity (top-{k}): {mean_sim:.4f}\n") f.write(f"Mean Cosine Similarity (top-1): {mean_top1:.4f}\n") f.write(f"Self-retrieval accuracy: {self_rate:.4f}\n") if cat_coherence_l3: f.write(f"Category coherence (macro): {np.mean(cat_coherence_l3):.4f}\n") if cat_coherence_l4: f.write(f"Category coherence (sub-cat): {np.mean(cat_coherence_l4):.4f}\n") print(f"\nResults saved to: {results_path}") if __name__ == "__main__": main()