{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":126777,"databundleVersionId":15314950,"sourceType":"competition"},{"sourceId":293816898,"sourceType":"kernelVersion"}],"dockerImageVersionId":31259,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras import layers, Model\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import normalize\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-14T15:12:07.575694Z","iopub.execute_input":"2026-02-14T15:12:07.576091Z","iopub.status.idle":"2026-02-14T15:12:07.582614Z","shell.execute_reply.started":"2026-02-14T15:12:07.576052Z","shell.execute_reply":"2026-02-14T15:12:07.581655Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_csv = \"/kaggle/input/jaguar-re-id/train.csv\"\ntest_csv  = \"/kaggle/input/jaguar-re-id/test.csv\"\n\ntrain_dir = \"/kaggle/input/jaguar-re-id/train/train\"\ntest_dir  = \"/kaggle/input/jaguar-re-id/test/test\"\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-14T15:12:27.267560Z","iopub.execute_input":"2026-02-14T15:12:27.267939Z","iopub.status.idle":"2026-02-14T15:12:27.273078Z","shell.execute_reply.started":"2026-02-14T15:12:27.267907Z","shell.execute_reply":"2026-02-14T15:12:27.271902Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-14T15:14:07.808203Z","iopub.execute_input":"2026-02-14T15:14:07.808575Z","iopub.status.idle":"2026-02-14T15:14:07.844867Z","shell.execute_reply.started":"2026-02-14T15:14:07.808543Z","shell.execute_reply":"2026-02-14T15:14:07.844045Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv(train_csv)\n\n# Encode string labels → numbers\nle = LabelEncoder()\ntrain_df[\"label\"] = le.fit_transform(train_df[\"ground_truth\"])\n\nprint(\"Total images:\", len(train_df))\nprint(\"Number of jaguars:\", train_df[\"label\"].nunique())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-14T15:15:07.508972Z","iopub.execute_input":"2026-02-14T15:15:07.509358Z","iopub.status.idle":"2026-02-14T15:15:07.526811Z","shell.execute_reply.started":"2026-02-14T15:15:07.509325Z","shell.execute_reply":"2026-02-14T15:15:07.525735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMG_SIZE = 224\nBATCH_SIZE = 32\n\ndef preprocess(path, label):\n    img = tf.io.read_file(path)\n    img = tf.image.decode_png(img, channels=3)\n    img = tf.image.resize(img, (IMG_SIZE, IMG_SIZE))\n    img = tf.keras.applications.efficientnet.preprocess_input(img)\n    return img, label\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-14T15:15:21.439287Z","iopub.execute_input":"2026-02-14T15:15:21.439625Z","iopub.status.idle":"2026-02-14T15:15:21.445688Z","shell.execute_reply.started":"2026-02-14T15:15:21.439596Z","shell.execute_reply":"2026-02-14T15:15:21.444434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_paths = train_df[\"filename\"].apply(\n    lambda x: os.path.join(train_dir, x)\n).values\n\nlabels = train_df[\"label\"].values\n\ndataset = tf.data.Dataset.from_tensor_slices((image_paths, labels))\ndataset = dataset.map(preprocess, num_parallel_calls=tf.data.AUTOTUNE)\ndataset = dataset.shuffle(2048)\ndataset = dataset.batch(BATCH_SIZE)\ndataset = dataset.prefetch(tf.data.AUTOTUNE)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-14T15:15:36.074923Z","iopub.execute_input":"2026-02-14T15:15:36.075767Z","iopub.status.idle":"2026-02-14T15:15:36.223839Z","shell.execute_reply.started":"2026-02-14T15:15:36.075730Z","shell.execute_reply":"2026-02-14T15:15:36.222736Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"base_model = EfficientNetB0(\n    weights=\"imagenet\",\n    include_top=False,\n    input_shape=(IMG_SIZE, IMG_SIZE, 3)\n)\n\nx = layers.GlobalAveragePooling2D()(base_model.output)\nx = layers.Dense(512)(x)\nx = layers.BatchNormalization()(x)\n\n# ✅ Proper L2 normalization for Keras 3\noutput = layers.Lambda(\n    lambda t: tf.math.l2_normalize(t, axis=1)\n)(x)\n\nmodel = Model(base_model.input, output)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-14T15:16:50.832666Z","iopub.execute_input":"2026-02-14T15:16:50.833424Z","iopub.status.idle":"2026-02-14T15:16:52.022928Z","shell.execute_reply.started":"2026-02-14T15:16:50.833377Z","shell.execute_reply":"2026-02-14T15:16:52.021860Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def supervised_contrastive_loss(labels, embeddings, temperature=0.1):\n    labels = tf.reshape(labels, [-1, 1])\n\n    similarity_matrix = tf.matmul(embeddings, embeddings, transpose_b=True)\n    logits = similarity_matrix / temperature\n\n    mask = tf.equal(labels, tf.transpose(labels))\n    mask = tf.cast(mask, tf.float32)\n\n    logits_mask = tf.ones_like(mask) - tf.eye(tf.shape(labels)[0])\n    mask = mask * logits_mask\n\n    exp_logits = tf.exp(logits) * logits_mask\n    log_prob = logits - tf.math.log(\n        tf.reduce_sum(exp_logits, axis=1, keepdims=True) + 1e-9\n    )\n\n    mean_log_prob_pos = tf.reduce_sum(mask * log_prob, axis=1) / (\n        tf.reduce_sum(mask, axis=1) + 1e-9\n    )\n\n    loss = -tf.reduce_mean(mean_log_prob_pos)\n    return loss\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-14T15:17:12.973825Z","iopub.execute_input":"2026-02-14T15:17:12.974181Z","iopub.status.idle":"2026-02-14T15:17:12.981537Z","shell.execute_reply.started":"2026-02-14T15:17:12.974150Z","shell.execute_reply":"2026-02-14T15:17:12.980568Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"optimizer = tf.keras.optimizers.Adam(1e-4)\n\n@tf.function\ndef train_step(images, labels):\n    with tf.GradientTape() as tape:\n        embeddings = model(images, training=True)\n        loss = supervised_contrastive_loss(labels, embeddings)\n\n    grads = tape.gradient(loss, model.trainable_variables)\n    optimizer.apply_gradients(zip(grads, model.trainable_variables))\n    return loss\n\nEPOCHS = 5\n\nfor epoch in range(EPOCHS):\n    print(f\"\\nEpoch {epoch+1}\")\n    for images, labels in dataset:\n        loss = train_step(images, labels)\n    print(\"Loss:\", float(loss))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-14T15:17:28.808390Z","iopub.execute_input":"2026-02-14T15:17:28.808752Z","iopub.status.idle":"2026-02-14T15:58:43.065623Z","shell.execute_reply.started":"2026-02-14T15:17:28.808721Z","shell.execute_reply":"2026-02-14T15:58:43.063451Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pairs = pd.read_csv(test_csv)\n\nunique_images = list(\n    set(pairs['query_image']).union(set(pairs['gallery_image']))\n)\n\ntest_paths = [os.path.join(test_dir, img) for img in unique_images]\n\ndef preprocess_test(path):\n    img = tf.io.read_file(path)\n    img = tf.image.decode_png(img, channels=3)\n    img = tf.image.resize(img, (IMG_SIZE, IMG_SIZE))\n    img = tf.keras.applications.efficientnet.preprocess_input(img)\n    return img\n\ntest_ds = tf.data.Dataset.from_tensor_slices(test_paths)\ntest_ds = test_ds.map(preprocess_test)\ntest_ds = test_ds.batch(BATCH_SIZE)\n\nembeddings = model.predict(test_ds)\nembeddings = normalize(embeddings)\n\nemb_dict = dict(zip(unique_images, embeddings))\n\nsimilarities = []\n\nfor _, row in pairs.iterrows():\n    q = emb_dict[row[\"query_image\"]]\n    g = emb_dict[row[\"gallery_image\"]]\n    sim = np.dot(q, g)\n    similarities.append(sim)\n\nsubmission = pd.DataFrame({\n    \"row_id\": pairs[\"row_id\"],\n    \"similarity\": similarities\n})\n\nsubmission.to_csv(\"submission_new.csv\", index=False)\nsubmission.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-14T16:03:33.746123Z","iopub.execute_input":"2026-02-14T16:03:33.746696Z","iopub.status.idle":"2026-02-14T16:04:31.465335Z","shell.execute_reply.started":"2026-02-14T16:03:33.746629Z","shell.execute_reply":"2026-02-14T16:04:31.463632Z"}},"outputs":[],"execution_count":null}]}