""" Extracts 100k products from the raw catalog, selecting ONLY those that have: 1. At least one review with rating >= 4 2. A valid price (number > 0) 3. Are not jewelry/watches Strategy (2 passes): PASS 1: Scan the raw reviews (26GB) → collect the ASINs with rating >= 4 PASS 2: Scan the raw metadata (17GB) → take only products that satisfy all 3 constraints. Stops at 100k. This ensures that ALL 100k products in the catalog have both reviews and a valid price, making the "price bucket" feature useful in the LightFM model. Input: raw/Clothing_Shoes_and_Jewelry.jsonl (26GB reviews) raw/meta_Clothing_Shoes_and_Jewelry.jsonl (17GB metadata) Output: artifacts/metadata_cleaned.json (100k products, all with reviews and price) Estimated time: 15-25 minutes (scan of both raw files). """ import json import os import sys PROJECT_ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) RAW_REVIEWS = os.path.join(PROJECT_ROOT, "raw", "Clothing_Shoes_and_Jewelry.jsonl") RAW_METADATA = os.path.join(PROJECT_ROOT, "raw", "meta_Clothing_Shoes_and_Jewelry.jsonl") OUTPUT_FILE = os.path.join(PROJECT_ROOT, "artifacts", "metadata_cleaned.json") TARGET_COUNT = 100000 # Categories to exclude EXCLUDED_CATEGORIES = {"jewelry", "watches", "watch accessories", "jewelry accessories"} def has_excluded_category(categories): for c in categories: if isinstance(c, str) and c.lower() in EXCLUDED_CATEGORIES: return True if isinstance(c, list): for sub in c: if isinstance(sub, str) and sub.lower() in EXCLUDED_CATEGORIES: return True return False def is_valid_price(price): """Checks whether the price is a valid number > 0.""" if price is None or price == "N/A" or price == "": return False try: return float(price) > 0 except (ValueError, TypeError): return False def main(): print("=" * 60) print("METADATA EXTRACTION (only products with reviews)") print(f"Target: {TARGET_COUNT:,} products") print("=" * 60) # Check raw files if not os.path.exists(RAW_REVIEWS): print(f"Error: {RAW_REVIEWS} not found!") sys.exit(1) if not os.path.exists(RAW_METADATA): print(f"Error: {RAW_METADATA} not found!") sys.exit(1) # --- PASS 1: Scan raw reviews → ASINs with rating >= 4 --- print(f"\n--- STEP 1: Scanning raw reviews ---") print(f"File: {RAW_REVIEWS}") reviewed_asins = set() count_reviews_scanned = 0 with open(RAW_REVIEWS, "r", encoding="utf-8") as f: for line in f: count_reviews_scanned += 1 if count_reviews_scanned % 1000000 == 0: print(f" Reviews scanned: {count_reviews_scanned:,} | Unique ASINs with rating>=4: {len(reviewed_asins):,}") try: data = json.loads(line) rating = data.get("rating") or data.get("overall") if not rating or float(rating) < 4.0: continue asin = data.get("asin") or data.get("parent_asin") if asin: reviewed_asins.add(asin) except Exception: continue print(f"\n Total reviews scanned: {count_reviews_scanned:,}") print(f" Unique ASINs with at least 1 review (rating >= 4): {len(reviewed_asins):,}") # --- PASS 2: Scan raw metadata → take only products with reviews --- print(f"\n--- STEP 2: Scanning raw metadata ---") print(f"File: {RAW_METADATA}") count_meta_scanned = 0 count_saved = 0 count_excluded_jewelry = 0 count_no_reviews = 0 count_no_price = 0 clean_products = [] with open(RAW_METADATA, "r", encoding="utf-8") as f: for line in f: count_meta_scanned += 1 if count_meta_scanned % 100000 == 0: print(f" Metadata scanned: {count_meta_scanned:,} | Saved: {count_saved:,} | No reviews: {count_no_reviews:,} | No price: {count_no_price:,} | Jewelry: {count_excluded_jewelry:,}") try: data = json.loads(line) asin = data.get("asin") or data.get("parent_asin") title = data.get("title") if not asin or not title or len(title) <= 3: continue # Filter 1: must have reviews with rating >= 4 if asin not in reviewed_asins: count_no_reviews += 1 continue # Filter 2: must have a valid price price = data.get("price") if not is_valid_price(price): count_no_price += 1 continue # Filter 3: exclude jewelry and watches categories = data.get("categories", []) if has_excluded_category(categories): count_excluded_jewelry += 1 continue # Image retrieval image_url = "" if "images" in data and data["images"]: img_entry = data["images"][0] image_url = ( img_entry.get("large", "") or img_entry.get("hi_res", "") or img_entry.get("thumb", "") ) elif "imUrl" in data: image_url = data["imUrl"] clean_products.append({ "asin": asin, "title": title, "brand": data.get("brand", "Generic"), "price": float(price), "categories": categories, "image_url": image_url, }) count_saved += 1 if count_saved >= TARGET_COUNT: break except Exception: continue # --- SUMMARY --- print(f"\n{'='*60}") print(f"SUMMARY") print(f"{'='*60}") print(f"Reviews scanned: {count_reviews_scanned:,}") print(f"ASINs with reviews (rating>=4): {len(reviewed_asins):,}") print(f"Metadata scanned: {count_meta_scanned:,}") print(f"Products excluded (no reviews): {count_no_reviews:,}") print(f"Products excluded (no price): {count_no_price:,}") print(f"Products excluded (jewelry): {count_excluded_jewelry:,}") print(f"Products saved: {count_saved:,}") if count_saved < TARGET_COUNT: print(f"\n WARNING: found only {count_saved:,} products (target: {TARGET_COUNT:,})") print(f" You may need to lower the target or relax the filters.") with open(OUTPUT_FILE, "w", encoding="utf-8") as f_out: json.dump(clean_products, f_out, indent=2) print(f"\nOutput: {OUTPUT_FILE}") if __name__ == "__main__": main()