""" Rebuilds the FAISS products index from the cleaned metadata. The page_content for each product is built as: title + brand (if not Generic) + categories (filtered) The categories are filtered to remove: - "Clothing, Shoes & Jewelry" (present in 99.98%, zero discriminating value) - Categories that are too generic (e.g. "Women", "Men" on their own) This improves the relevance of the semantic search by preventing irrelevant tokens from polluting the embedding. Input: artifacts/metadata_cleaned.json Output: artifacts/faiss_products/ Estimated time: 3-5 minutes (100k embeddings). """ import json import os import sys from langchain_huggingface import HuggingFaceEmbeddings from langchain_community.vectorstores import FAISS from langchain_core.documents import Document PROJECT_ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) INPUT_FILE = os.path.join(PROJECT_ROOT, "artifacts", "metadata_cleaned.json") INDEX_NAME = os.path.join(PROJECT_ROOT, "artifacts", "faiss_products") # Categories to exclude from page_content (too generic, they pollute the embedding) EXCLUDED_CATEGORIES = { "Clothing, Shoes & Jewelry", "Clothing, Shoes & Jewelry ", } def flatten_categories(categories): """Flattens categories (they can be strings or nested lists) and filters out the useless ones.""" result = [] for c in categories: if isinstance(c, str): if c not in EXCLUDED_CATEGORIES: result.append(c) elif isinstance(c, list): for sub in c: if isinstance(sub, str) and sub not in EXCLUDED_CATEGORIES: result.append(sub) return result def build_index(): print("--- BUILDING SEARCH INDEX (FAISS) ---") if not os.path.exists(INPUT_FILE): print(f"Error: {INPUT_FILE} not found!") sys.exit(1) print("-> Loading cleaned products...") with open(INPUT_FILE, "r", encoding="utf-8") as f: products = json.load(f) print(f"-> Preparing documents ({len(products)} items)...") docs = [] for p in products: # Title: always present parts = [p['title']] # Brand: only if not Generic (all Generic = zero value) brand = p.get('brand', 'Generic') if brand and brand != 'Generic': parts.append(brand) # Categories: filtered (removes "Clothing, Shoes & Jewelry") cats = flatten_categories(p.get('categories', [])) if cats: parts.append(' '.join(cats)) page_content = ' '.join(parts) meta = { "asin": p["asin"], "title": p["title"], "price": str(p["price"]), "brand": p["brand"], "image_url": p["image_url"], } docs.append(Document(page_content=page_content, metadata=meta)) print("-> Computing embeddings and indexing (may take a few minutes)...") embeddings = HuggingFaceEmbeddings(model_name="paraphrase-multilingual-MiniLM-L12-v2") vectorstore = FAISS.from_documents(docs, embeddings) vectorstore.save_local(INDEX_NAME) print(f"Index saved to: {INDEX_NAME}") if __name__ == "__main__": build_index()