""" Downloads the product images from the Amazon Fashion catalog. Reads artifacts/metadata_cleaned.json and downloads each image_url into the artifacts/images/{ASIN}.jpg folder Features: - Parallel download with ThreadPoolExecutor (10 workers by default) - Resumes from where it stopped (skips already existing files) - Progress bar with percentage and speed - Robust error handling (retry on timeout, log of failures) - Space estimate: ~1.7 GB for ~100k images (thumbnails ~18 KB each) """ import json import os import sys import time import urllib.request import urllib.error from concurrent.futures import ThreadPoolExecutor, as_completed from pathlib import Path # --- CONFIGURATION --- PROJECT_ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) METADATA_PATH = os.path.join(PROJECT_ROOT, "artifacts", "metadata_cleaned.json") IMAGES_DIR = os.path.join(PROJECT_ROOT, "artifacts", "images") MAX_WORKERS = 10 # Parallel threads for the download TIMEOUT = 15 # Timeout per single request (seconds) MAX_RETRIES = 2 # Attempts per image def download_image(asin, url, output_dir): """Downloads a single image. Returns (asin, success, error_msg).""" # Determine extension from the URL if url.lower().endswith(".png"): ext = ".png" elif url.lower().endswith(".gif"): ext = ".gif" else: ext = ".jpg" filepath = os.path.join(output_dir, f"{asin}{ext}") # Skip if already downloaded if os.path.exists(filepath) and os.path.getsize(filepath) > 0: return (asin, True, "skipped") for attempt in range(MAX_RETRIES): try: req = urllib.request.Request(url) req.add_header("User-Agent", "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7)") resp = urllib.request.urlopen(req, timeout=TIMEOUT) data = resp.read() with open(filepath, "wb") as f: f.write(data) return (asin, True, None) except Exception as e: if attempt < MAX_RETRIES - 1: time.sleep(1) # Pause between retries else: return (asin, False, str(e)) return (asin, False, "max retries exceeded") def main(): print(f"--- Loading metadata... ---") with open(METADATA_PATH, "r", encoding="utf-8") as f: products = json.load(f) # Filter products with a valid URL to_download = [(p["asin"], p["image_url"]) for p in products if p.get("image_url")] print(f"Products with image: {len(to_download)}") # Create output folder os.makedirs(IMAGES_DIR, exist_ok=True) # Count already downloaded existing = set(Path(IMAGES_DIR).glob("*.*")) existing_asins = {f.stem for f in existing if f.stat().st_size > 0} already_done = sum(1 for asin, _ in to_download if asin in existing_asins) if already_done > 0: print(f"Already downloaded: {already_done} (will be skipped)") remaining = len(to_download) - already_done if remaining == 0: print("All images have already been downloaded!") return print(f"To download: {remaining}") print(f"Parallel workers: {MAX_WORKERS}") print(f"--- Starting download... ---\n") success = 0 failed = 0 skipped = 0 failed_asins = [] start_time = time.time() with ThreadPoolExecutor(max_workers=MAX_WORKERS) as executor: futures = { executor.submit(download_image, asin, url, IMAGES_DIR): asin for asin, url in to_download } total = len(futures) done_count = 0 for future in as_completed(futures): asin, ok, err = future.result() done_count += 1 if ok: if err == "skipped": skipped += 1 else: success += 1 else: failed += 1 failed_asins.append((asin, err)) # Progress every 500 or at the end if done_count % 500 == 0 or done_count == total: elapsed = time.time() - start_time rate = done_count / elapsed if elapsed > 0 else 0 pct = done_count / total * 100 eta = (total - done_count) / rate if rate > 0 else 0 sys.stdout.write( f"\r [{pct:5.1f}%] {done_count}/{total} | " f"OK: {success} | Skip: {skipped} | Fail: {failed} | " f"{rate:.0f} img/s | ETA: {eta/60:.1f}min" ) sys.stdout.flush() elapsed = time.time() - start_time print(f"\n\n--- DONE in {elapsed/60:.1f} minutes ---") print(f"Downloaded: {success}") print(f"Skipped (already existing): {skipped}") print(f"Failed: {failed}") if failed_asins: log_path = os.path.join(PROJECT_ROOT, "artifacts", "download_errors.log") with open(log_path, "w") as f: for asin, err in failed_asins: f.write(f"{asin}\t{err}\n") print(f"Error log saved to: {log_path}") # Space statistics total_size = sum( f.stat().st_size for f in Path(IMAGES_DIR).glob("*.*") ) print(f"Disk space used: {total_size / 1024 / 1024 / 1024:.2f} GB") if __name__ == "__main__": main()