""" Evaluation of the Router Agent — Intent Classification. Method: simulated leave-one-out cross-validation. For each query in the dataset: 1. Search the top-(k+1) neighbors in the FAISS index 2. Exclude the first result (the query itself, distance ~0) 3. Classify with 2-level distance-weighted kNN (same logic as the Router) 4. Compare with the true label Metrics: - Level 1: 5 original classes (product, sales, support, account, bugs) - Level 2: 2 macro-groups (product vs support/default) - Classification report (precision, recall, F1-score) - Confusion matrix """ import json import os import sys import warnings warnings.filterwarnings("ignore") from collections import defaultdict, Counter import numpy as np # Add project root to the path to import utils PROJECT_ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) sys.path.insert(0, PROJECT_ROOT) from datasets import load_dataset from langchain_huggingface import HuggingFaceEmbeddings from langchain_community.vectorstores import FAISS from langchain_core.documents import Document from sklearn.metrics import classification_report, confusion_matrix from utils.config import EMBEDDING_MODEL_NAME, ROUTER_SEARCH_K # --- Constants (same as the Router) --- PRODUCT_CLASSES = {"product", "sales"} DEFAULT_CLASSES = {"support", "account", "bugs"} RECOMMENDER_KEYWORDS_EN = [ "recommend", "suggest", "best", "top rated", "ideas for", "what should i", "good for", "advice", "opinion", "popular", "trending", "look for", "matching", "outfit", "gift", "review" ] DATASET_NAME = "vansh-khaneja/ecommerce-intent-routing" FASHION_EXAMPLES_PATH = os.path.join(PROJECT_ROOT, "artifacts", "fashion_examples.json") def has_recommender_intent(text): """Same heuristic as the pipeline builder.""" text_lower = text.lower() return any(kw in text_lower for kw in RECOMMENDER_KEYWORDS_EN) def classify_query(query, results, k): """Exact replica of the Router's classification logic. Args: query: query text results: list of (Document, L2_distance) from the top-k FAISS k: number of neighbors to use Returns: (class_5, macro_2): tuple with 5-class and 2-macro-group classification """ epsilon = 1e-6 weighted_votes = defaultdict(float) # Use only the first k results for doc, distance in results[:k]: intent_class = doc.metadata["intent_class"] weight = 1.0 / (distance + epsilon) weighted_votes[intent_class] += weight # Class with the highest weight (level 1) if weighted_votes: class_5 = max(weighted_votes, key=weighted_votes.get) else: class_5 = "support" # Macro-groups (level 2) product_weight = sum(weighted_votes.get(c, 0) for c in PRODUCT_CLASSES) default_weight = sum(weighted_votes.get(c, 0) for c in DEFAULT_CLASSES) if product_weight > default_weight: macro_2 = "prodotto" else: macro_2 = "supporto" return class_5, macro_2 def get_macro_label(intent_class): """Converts an original class into the macro-group.""" if intent_class in PRODUCT_CLASSES: return "prodotto" else: return "supporto" def load_all_data(): """Loads all the data: HuggingFace dataset + fashion phrases.""" print(f"Downloading dataset '{DATASET_NAME}'...") dataset = load_dataset(DATASET_NAME, split="train") documents = [] for row in dataset: text = row['query'] raw_label = row['route'] label_text = str(raw_label) if isinstance(raw_label, int): try: features = dataset.features['route'] label_text = features.int2str(raw_label) except Exception: pass intent_class = label_text.lower().strip() recommender_hint = False if intent_class in ("product", "sales"): recommender_hint = has_recommender_intent(text) doc = Document( page_content=text, metadata={ "intent_class": intent_class, "recommender_hint": recommender_hint, } ) documents.append(doc) hf_count = len(documents) # Additional fashion phrases if os.path.exists(FASHION_EXAMPLES_PATH): print(f"Loading fashion phrases...") with open(FASHION_EXAMPLES_PATH, "r", encoding="utf-8") as f: fashion_data = json.load(f) for item in fashion_data: text = item["query"] intent_class = item["route"].lower().strip() recommender_hint = False if intent_class in ("product", "sales"): recommender_hint = has_recommender_intent(text) doc = Document( page_content=text, metadata={ "intent_class": intent_class, "recommender_hint": recommender_hint, } ) documents.append(doc) fashion_count = len(documents) - hf_count print(f"Dataset: {hf_count} HuggingFace + {fashion_count} fashion = {len(documents)} total") # Class distribution counts = Counter(d.metadata["intent_class"] for d in documents) for cls, count in sorted(counts.items(), key=lambda x: -x[1]): print(f" {cls}: {count} ({count/len(documents)*100:.1f}%)") return documents def evaluate_router(documents, embeddings, k=ROUTER_SEARCH_K): """Leave-one-out evaluation of the Router. Builds a single FAISS index with all the queries, then for each query: - Searches k+1 neighbors (the first one is the query itself) - Excludes results with distance < 0.01 (the query itself) - Classifies with the remaining k neighbors """ print(f"\nBuilding FAISS index ({len(documents)} documents)...") vector_store = FAISS.from_documents(documents, embeddings) y_true_5 = [] # True labels (5 classes) y_pred_5 = [] # Predictions (5 classes) y_true_2 = [] # True labels (2 macro-groups) y_pred_2 = [] # Predictions (2 macro-groups) errors = [] # Errors for qualitative analysis total = len(documents) print(f"\nLeave-one-out classification (k={k})...") for i, doc in enumerate(documents): if (i + 1) % 500 == 0: print(f" {i+1}/{total}...") query = doc.page_content true_class = doc.metadata["intent_class"] # Search k+1 neighbors (the first one will be the query itself) results = vector_store.similarity_search_with_score(query, k=k + 5) # Exclude results with near-zero distance (the query itself and any exact duplicates) filtered = [(d, dist) for d, dist in results if dist > 0.01] # Take only k results filtered = filtered[:k] if not filtered: # Rare case: all results are exact duplicates pred_5 = "support" pred_2 = "supporto" else: pred_5, pred_2 = classify_query(query, filtered, k) true_2 = get_macro_label(true_class) y_true_5.append(true_class) y_pred_5.append(pred_5) y_true_2.append(true_2) y_pred_2.append(pred_2) # Save errors for analysis if pred_2 != true_2: errors.append({ "query": query, "true": true_class, "pred_5": pred_5, "true_macro": true_2, "pred_macro": pred_2, }) return y_true_5, y_pred_5, y_true_2, y_pred_2, errors def print_confusion_matrix(y_true, y_pred, labels, title): """Prints the confusion matrix in a readable format.""" cm = confusion_matrix(y_true, y_pred, labels=labels) print(f"\n{'='*60}") print(f"CONFUSION MATRIX — {title}") print(f"{'='*60}") # Header max_label = max(len(l) for l in labels) header = " " * (max_label + 2) + " ".join(f"{l:>8}" for l in labels) print(f"\n{'Predicted →':>{max_label + 2}}{header[max_label+2:]}") print(f"{'True ↓'}") for i, label in enumerate(labels): row = " ".join(f"{cm[i][j]:>8}" for j in range(len(labels))) print(f" {label:<{max_label}} {row}") print() def main(): print("=" * 60) print("ROUTER EVALUATION — Intent Classification") print("Method: Leave-one-out cross-validation") print(f"Classifier: distance-weighted kNN (k={ROUTER_SEARCH_K})") print(f"Embedding: {EMBEDDING_MODEL_NAME}") print("=" * 60) # Load data documents = load_all_data() # Load embedding print(f"\nLoading embedding model '{EMBEDDING_MODEL_NAME}'...") embeddings = HuggingFaceEmbeddings(model_name=EMBEDDING_MODEL_NAME) # Evaluation y_true_5, y_pred_5, y_true_2, y_pred_2, errors = evaluate_router(documents, embeddings) # --- Report 5 classes --- labels_5 = sorted(set(y_true_5)) print(f"\n{'='*60}") print(f"CLASSIFICATION REPORT — 5 Original Classes") print(f"{'='*60}") print(classification_report(y_true_5, y_pred_5, labels=labels_5, digits=4)) print_confusion_matrix(y_true_5, y_pred_5, labels_5, "5 Classes") # --- Report 2 macro-groups --- labels_2 = ["prodotto", "supporto"] print(f"\n{'='*60}") print(f"CLASSIFICATION REPORT — 2 Macro-Groups (prodotto vs supporto)") print(f"{'='*60}") print(classification_report(y_true_2, y_pred_2, labels=labels_2, digits=4)) print_confusion_matrix(y_true_2, y_pred_2, labels_2, "2 Macro-Groups") # --- Summary accuracy --- acc_5 = sum(1 for t, p in zip(y_true_5, y_pred_5) if t == p) / len(y_true_5) acc_2 = sum(1 for t, p in zip(y_true_2, y_pred_2) if t == p) / len(y_true_2) print(f"\n{'='*60}") print(f"SUMMARY") print(f"{'='*60}") print(f"Accuracy 5 classes: {acc_5:.4f} ({acc_5*100:.2f}%)") print(f"Accuracy 2 macro: {acc_2:.4f} ({acc_2*100:.2f}%)") print(f"Total queries: {len(y_true_5)}") print(f"Errors (macro-group): {len(errors)}") # --- Error examples --- if errors: print(f"\n{'='*60}") print(f"MACRO-GROUP ERRORS (first 20)") print(f"{'='*60}") for e in errors[:20]: print(f" [{e['true']:>8} → {e['pred_5']:>8}] \"{e['query'][:80]}\"") # --- Save results to file --- results_path = os.path.join(PROJECT_ROOT, "evaluation", "router_results.txt") with open(results_path, "w", encoding="utf-8") as f: f.write(f"ROUTER EVALUATION — Intent Classification\n") f.write(f"Method: Leave-one-out cross-validation\n") f.write(f"k={ROUTER_SEARCH_K}, Embedding={EMBEDDING_MODEL_NAME}\n") f.write(f"Total queries: {len(y_true_5)}\n\n") f.write(f"ACCURACY\n") f.write(f" 5 classes: {acc_5:.4f} ({acc_5*100:.2f}%)\n") f.write(f" 2 macro: {acc_2:.4f} ({acc_2*100:.2f}%)\n\n") f.write(f"CLASSIFICATION REPORT — 5 Classes\n") f.write(classification_report(y_true_5, y_pred_5, labels=labels_5, digits=4)) f.write(f"\nCLASSIFICATION REPORT — 2 Macro-Groups\n") f.write(classification_report(y_true_2, y_pred_2, labels=labels_2, digits=4)) if errors: f.write(f"\nMACRO-GROUP ERRORS ({len(errors)} total)\n") for e in errors: f.write(f" [{e['true']:>8} -> {e['pred_5']:>8}] \"{e['query']}\"\n") print(f"\nResults saved to: {results_path}") if __name__ == "__main__": main()