""" Visual Recommender Agent — Product recommendation by visual similarity. Given the path of an image, finds the products in the catalog with the most similar appearance using DinoV2 embeddings + cosine similarity via FAISS. How it works: 1. Receives `image_path` from the state (set by the Router) 2. Loads the image and computes the embedding with Trendyol DinoV2 (256d) 3. Searches the top-k most similar products in the visual FAISS index 4. Returns the products with name, brand and price 0 LLM calls — all computation is offline (DinoV2 + FAISS + metadata lookup). Uses the factory pattern: create_visual_node(visual_model, visual_processor, visual_index, visual_asin_map, metadata_lookup, device) -> node function. """ import numpy as np import torch from PIL import Image from langchain_core.messages import AIMessage from utils.config import VISUAL_SEARCH_K from utils.i18n import t def create_visual_node(visual_model, visual_processor, visual_index, visual_asin_map, metadata_lookup, device): """Creates the Visual Recommender Agent node with injected dependencies. Args: visual_model: loaded DinoV2 (Trendyol) model visual_processor: preprocessor for the images visual_index: visual FAISS index (IndexFlatIP, 256d, normalized) visual_asin_map: list [ASIN] where the index corresponds to the FAISS position metadata_lookup: dictionary {ASIN: {title, brand, price, ...}} device: torch.device (cpu/mps/cuda) """ def compute_query_embedding(image_path): """Computes the DinoV2 embedding for a query image.""" img = Image.open(image_path).convert("RGB") inputs = visual_processor(images=img, return_tensors="pt") inputs = {k: v.to(device) for k, v in inputs.items()} with torch.no_grad(): outputs = visual_model(**inputs) # Trendyol DinoV2: output already projected to (batch, 256) and L2-normalized embedding = outputs.last_hidden_state embedding = embedding.cpu().numpy().astype("float32") return embedding def visual_node(state): print("--- [VISUAL AGENT] Visual similarity search... ---") image_path = state.get("image_path") if not image_path: return {"messages": [AIMessage(content=t("visual_no_image"))]} if not visual_index: return {"messages": [AIMessage(content=t("visual_offline"))]} try: # Compute the embedding of the query image print(f" Image: {image_path}") query_embedding = compute_query_embedding(image_path) # Search the top-k most similar in the FAISS index k = VISUAL_SEARCH_K scores, indices = visual_index.search(query_embedding, k) # Build structured product list (shown visually in the cards) products = [] for score, idx in zip(scores[0], indices[0]): if idx < 0 or idx >= len(visual_asin_map): continue asin = visual_asin_map[idx] p = metadata_lookup.get(asin, {}) if not p: continue title = p.get("title", asin) brand = p.get("brand", "") price = p.get("price", "") if brand == "Generic": brand = "" if price == "N/A": price = "" sim_pct = round(score * 100) # cosine similarity 0-1 → percentage products.append({ "asin": asin, "title": title, "brand": brand, "price": price, "similarity": sim_pct, }) print(f" [{score:.3f}] {title[:60]}") if not products: msg = t("visual_no_results") else: msg = t("visual_results") except FileNotFoundError: msg = t("visual_image_not_found", path=image_path) products = [] except Exception as e: msg = t("visual_error", error=e) products = [] return {"messages": [AIMessage(content=msg)], "products": products} return visual_node