{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfrom PIL import Image\nimport torch\nimport numpy as np\nfrom tqdm import tqdm\nfrom transformers import CLIPModel, CLIPProcessor\n\n# ===== CONFIG =====\nIMAGE_DIR = \"/kaggle/input/h-and-m-personalized-fashion-recommendations/images\"\nOUTPUT_FILE = \"clip_embeds.npz\"\nMODEL_NAME = \"openai/clip-vit-base-patch32\"\nBATCH_SIZE = 64\nDEVICE = \"cuda\" if torch.cuda.is_available() else \"cpu\"\nprint(DEVICE)\n# ===== LOAD MODEL =====\nmodel = CLIPModel.from_pretrained(MODEL_NAME).to(DEVICE)\nprocessor = CLIPProcessor.from_pretrained(MODEL_NAME)\nmodel.eval()\n\nprint(\"Model loaded\")\n\n# ===== LOAD IMAGES =====\nimage_files = []\nfor root, dirs, files in os.walk(IMAGE_DIR):\n    for f in files:\n        if f.lower().endswith((\"jpg\", \"jpeg\", \"png\", \"bmp\", \"webp\")):\n            image_files.append(os.path.join(root, f))\n\nprint(f\"Found {len(image_files)} images\")\n\nembeddings = []\nfilenames = []\n\n# ===== PROCESS IN BATCHES =====\nfor i in tqdm(range(0, len(image_files), BATCH_SIZE)):\n\n    batch_files = image_files[i:i+BATCH_SIZE]\n\n    images = [Image.open(f).convert(\"RGB\") for f in batch_files]\n\n    inputs = processor(\n        images=images,\n        return_tensors=\"pt\",\n        padding=True\n    ).to(DEVICE)\n\n    with torch.no_grad():\n        feats = model.get_image_features(**inputs)   # (B, 512)\n\n\n    embeddings.append(feats.cpu().numpy())\n    filenames.extend(batch_files)\n\n# ===== CONCATENATE ALL =====\nembeddings = np.concatenate(embeddings, axis=0)\n\nprint(\"Embedding shape:\", embeddings.shape)  # (num_images, 512)\n\n# ===== SAVE =====\nnp.savez(OUTPUT_FILE, filenames=filenames, embeddings=embeddings)\n\nprint(f\"Saved to {OUTPUT_FILE}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-26T07:51:33.060169Z","iopub.execute_input":"2025-12-26T07:51:33.060689Z","iopub.status.idle":"2025-12-26T07:53:56.760784Z","shell.execute_reply.started":"2025-12-26T07:51:33.060661Z","shell.execute_reply":"2025-12-26T07:53:56.759832Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}