{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.11.11"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":29762,"databundleVersionId":2541532,"sourceType":"competition"},{"sourceId":11730964,"sourceType":"datasetVersion","datasetId":7364044},{"sourceId":11766825,"sourceType":"datasetVersion","datasetId":7374262},{"sourceId":11826352,"sourceType":"datasetVersion","datasetId":7393923},{"sourceId":11873425,"sourceType":"datasetVersion","datasetId":7393689},{"sourceId":11915250,"sourceType":"datasetVersion","datasetId":7419283}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\n\n# Load your CSVs\n# train_df = pd.read_csv('/kaggle/working/train.csv')\n# test_df = pd.read_csv('/kaggle/working/test.csv')\n# val_df = pd.read_csv('/kaggle/working/val.csv')\ntrain_df = pd.read_csv('/kaggle/input/landmark/train.csv')\ntest_df = pd.read_csv('/kaggle/input/landmark/test.csv')\nval_df = pd.read_csv('/kaggle/input/landmark/val.csv')\n# Build mapping\nlandmark_id_to_idx = {lid: idx for idx, lid in enumerate(sorted(train_df['landmark_id'].unique()))}\nNUM_CLASSES = len(landmark_id_to_idx) \n\n# Map class_idx\ntrain_df['class_idx'] = train_df['landmark_id'].map(landmark_id_to_idx)\ntest_df['class_idx'] = test_df['landmark_id'].map(landmark_id_to_idx)\nval_df['class_idx'] = val_df['landmark_id'].map(landmark_id_to_idx)\n\nNUM_CLASSES = len(landmark_id_to_idx)","metadata":{"execution":{"iopub.status.busy":"2025-05-25T18:14:45.499694Z","iopub.execute_input":"2025-05-25T18:14:45.500439Z","iopub.status.idle":"2025-05-25T18:14:46.300345Z","shell.execute_reply.started":"2025-05-25T18:14:45.500406Z","shell.execute_reply":"2025-05-25T18:14:46.299360Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_DIR = \"/kaggle/input/landmark-recognition-2021/train/\"","metadata":{"execution":{"iopub.status.busy":"2025-05-16T02:40:07.406767Z","iopub.execute_input":"2025-05-16T02:40:07.407152Z","iopub.status.idle":"2025-05-16T02:40:07.412751Z","shell.execute_reply.started":"2025-05-16T02:40:07.407118Z","shell.execute_reply":"2025-05-16T02:40:07.411263Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/landmark/train.csv', header=None, names=[\"id\", \"landmark_id\"])\nprint(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T02:40:09.773100Z","iopub.execute_input":"2025-05-16T02:40:09.773421Z","iopub.status.idle":"2025-05-16T02:40:09.839359Z","shell.execute_reply.started":"2025-05-16T02:40:09.773397Z","shell.execute_reply":"2025-05-16T02:40:09.838348Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# File paths\ntrain_csv_path = \"/kaggle/input/landmark/train.csv\"\nfixed_csv_path = \"/kaggle/input/id-2-names-landmark/ID_to_names/train_with_landmark_names_fixed.csv\"\n\n# Load datasets\ntrain_df = pd.read_csv(train_csv_path, header=None, names=[\"id\", \"landmark_id\"],encoding = \"latin1\")\nfixed_df = pd.read_csv(fixed_csv_path)\n\n# Convert landmark_id to string for consistency\ntrain_df[\"landmark_id\"] = train_df[\"landmark_id\"].astype(str)\nfixed_df[\"landmark_id\"] = fixed_df[\"landmark_id\"].astype(str)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T18:19:59.019413Z","iopub.execute_input":"2025-05-25T18:19:59.020191Z","iopub.status.idle":"2025-05-25T18:19:59.213794Z","shell.execute_reply.started":"2025-05-25T18:19:59.020162Z","shell.execute_reply":"2025-05-25T18:19:59.212549Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\nfrom PIL import Image\nimport matplotlib.pyplot as plt\n\n# Load CSV\ndf = pd.read_csv('/kaggle/input/landmark/train.csv', header=None, names=[\"id\", \"landmark_id\"])\n\n# Look up the landmark_id for the given image ID\nimage_id = '520a0aeb2d338a1c'\nrow = df[df['id'] == image_id]\n\nif not row.empty:\n    landmark_id = row.iloc[0]['landmark_id']\n    print(f\"Image ID: {image_id} has landmark ID: {landmark_id}\")\n    \n    # Build image path\n    path = f\"/kaggle/input/landmark-recognition-2021/train/{image_id[0]}/{image_id[1]}/{image_id[2]}/{image_id}.jpg\"\n\n    # Load and display the image\n    if os.path.exists(path):\n        img = Image.open(path)\n        plt.imshow(img)\n        plt.axis('off')\n        plt.title(f\"Landmark ID: {landmark_id}\")\n        plt.show()\n    else:\n        print(f\"Image file not found at {path}\")\nelse:\n    print(f\"Image ID {image_id} not found in CSV.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T16:52:52.795506Z","iopub.execute_input":"2025-05-14T16:52:52.795908Z","iopub.status.idle":"2025-05-14T16:52:53.203216Z","shell.execute_reply.started":"2025-05-14T16:52:52.795868Z","shell.execute_reply":"2025-05-14T16:52:53.201899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\nfrom PIL import Image\nimport matplotlib.pyplot as plt\n\n# Load CSV\ndf = pd.read_csv(\"/kaggle/input/landmark/train.csv\", header=0)  # contains columns: id, landmark_id\n\n# Filter for desired landmark ID\nlandmark_id = 138982\nfiltered_df = df[df['landmark_id'] == landmark_id]\n\nprint(f\"Found {len(filtered_df)} images for landmark ID {landmark_id}\")\n\n# Plot first 5 images\nfig, axes = plt.subplots(1, 5, figsize=(20, 5))\nfor i, (idx, row) in enumerate(filtered_df.head(5).iterrows()):\n    image_id = row['id']\n    # Extract subfolders based on first 3 characters\n    subfolder = f\"{image_id[0]}/{image_id[1]}/{image_id[2]}\"\n    image_path = f\"/kaggle/input/landmark-recognition-2021/train/{subfolder}/{image_id}.jpg\"\n\n    try:\n        img = Image.open(image_path)\n        axes[i].imshow(img)\n        axes[i].axis('off')\n        axes[i].set_title(image_id)\n    except FileNotFoundError:\n        print(f\"Image not found: {image_path}\")\n        axes[i].axis('off')\n        axes[i].set_title(\"Missing\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"execution_failed":"2025-05-12T13:51:01.069Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\n\n# Load the CSV\ndf = pd.read_csv('/kaggle/input/landmark-recognition-2021/train.csv')\n\nlandmark_29_df = df[df['landmark_id'] == 20102]\n\n# Create output directory if it doesn't exist\noutput_dir = '/kaggle/working/add_images'\nos.makedirs(output_dir, exist_ok=True)\n\n# Save to CSV\noutput_path = os.path.join(output_dir, 'landmark_20102_images.csv')\nlandmark_29_df.to_csv(output_path, index=False)\n\nprint(f\"✅ Saved {len(landmark_29_df)} entries to {output_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T04:33:01.333449Z","iopub.execute_input":"2025-05-12T04:33:01.334488Z","iopub.status.idle":"2025-05-12T04:33:02.604306Z","shell.execute_reply.started":"2025-05-12T04:33:01.334449Z","shell.execute_reply":"2025-05-12T04:33:02.602998Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/landmark-recognition-2021/train.csv')\nlandmark_counts = df['landmark_id'].value_counts()\ntop_100_to_200 = landmark_counts.iloc[100:110]\nprint(top_100_to_200)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T04:26:10.036615Z","iopub.execute_input":"2025-05-12T04:26:10.037025Z","iopub.status.idle":"2025-05-12T04:26:11.373232Z","shell.execute_reply.started":"2025-05-12T04:26:10.036998Z","shell.execute_reply":"2025-05-12T04:26:11.372066Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMAGE_SIZE = 224\nBATCH_SIZE = 32","metadata":{"execution":{"iopub.status.busy":"2025-05-12T02:33:52.069325Z","iopub.execute_input":"2025-05-12T02:33:52.069843Z","iopub.status.idle":"2025-05-12T02:33:52.075287Z","shell.execute_reply.started":"2025-05-12T02:33:52.069809Z","shell.execute_reply":"2025-05-12T02:33:52.073970Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torchvision import transforms\nfrom torch.utils.data import DataLoader\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\ntest_transform = A.Compose([\n    A.Resize(IMAGE_SIZE, IMAGE_SIZE),\n    A.Normalize(),\n    ToTensorV2()\n])","metadata":{"execution":{"iopub.status.busy":"2025-05-12T02:33:54.342304Z","iopub.execute_input":"2025-05-12T02:33:54.342644Z","iopub.status.idle":"2025-05-12T02:34:05.523658Z","shell.execute_reply.started":"2025-05-12T02:33:54.342618Z","shell.execute_reply":"2025-05-12T02:34:05.522258Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torchvision import models\nfrom torch.optim import Adam\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = models.resnet50(pretrained=True)\nmodel.fc = nn.Sequential(\n    nn.Linear(model.fc.in_features, 512),\n    nn.ReLU(),\n    nn.Dropout(0.5),\n    nn.Linear(512, 256),\n    nn.ReLU(),\n    nn.Dropout(0.5),\n    nn.Linear(256, NUM_CLASSES)\n)\nmodel = model.to(device)\ncheckpoint_path = \"/kaggle/input/model-effiecient/best_model_resnet_ver_2-2.pth\"\nmodel.load_state_dict(torch.load(checkpoint_path, map_location=device))\nmodel = model.to(device)","metadata":{"execution":{"iopub.status.busy":"2025-05-14T16:55:01.289280Z","iopub.execute_input":"2025-05-14T16:55:01.289588Z","iopub.status.idle":"2025-05-14T16:55:03.626444Z","shell.execute_reply.started":"2025-05-14T16:55:01.289564Z","shell.execute_reply":"2025-05-14T16:55:03.625556Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nsample = test_df.sample(90).iloc[0]\nimg_id = sample[\"id\"]\ntrue_label = sample[\"landmark_id\"]\nclass_idx = sample[\"class_idx\"]\n\nfolder = os.path.join(DATA_DIR, img_id[0], img_id[1], img_id[2])\nimg_path = os.path.join(folder, f\"{img_id}.jpg\")\n\nimage = Image.open(img_path).convert(\"RGB\")\nimage = test_transform(image=np.array(image))[\"image\"].unsqueeze(0).to(device)\n\nmodel.eval()\nwith torch.no_grad():\n    output = model(image)\n    pred_idx = output.argmax(dim=1).item()\n\n# Reverse map\nidx_to_landmark_id = {v: k for k, v in landmark_id_to_idx.items()}\npred_landmark = idx_to_landmark_id[pred_idx]\n\nprint(f\"🖼️ Image ID: {img_id}\")\nprint(f\"✅ Ground Truth Landmark ID: {true_label}\")\nprint(f\"🎯 Predicted Landmark ID: {pred_landmark}\")","metadata":{"execution":{"iopub.status.busy":"2025-05-16T01:19:05.090600Z","iopub.execute_input":"2025-05-16T01:19:05.090966Z","iopub.status.idle":"2025-05-16T01:19:05.108026Z","shell.execute_reply.started":"2025-05-16T01:19:05.090941Z","shell.execute_reply":"2025-05-16T01:19:05.106723Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install efficientnet_pytorch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T18:14:46.302509Z","iopub.execute_input":"2025-05-25T18:14:46.302883Z","iopub.status.idle":"2025-05-25T18:16:22.861238Z","shell.execute_reply.started":"2025-05-25T18:14:46.302852Z","shell.execute_reply":"2025-05-25T18:16:22.859958Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torchvision import models\nfrom efficientnet_pytorch import EfficientNet\n\nNUM_CLASSES = 100\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# -------- RESNET VERSION 1 --------\nresnet_v1 = models.resnet50(pretrained=False)\nin_features = resnet_v1.fc.in_features\nresnet_v1.fc = nn.Sequential(\n    nn.Linear(in_features, 512),\n    nn.ReLU(),\n    nn.Dropout(0.5),\n    nn.Linear(512, 256),\n    nn.ReLU(),\n    nn.Dropout(0.5),\n    nn.Linear(256, NUM_CLASSES)\n)\nresnet_v1.load_state_dict(torch.load(\"/kaggle/input/model-effiecient/best_model_resnet-5.pth\", map_location=device))\nresnet_v1 = resnet_v1.to(device)\n\n# -------- RESNET VERSION 2 --------\nresnet_v2 = models.resnet50(pretrained=False)\nin_features = resnet_v2.fc.in_features\nresnet_v2.fc = nn.Sequential(\n    nn.Linear(in_features, 512),\n    nn.ReLU(),\n    nn.Dropout(0.5),\n    nn.Linear(512, 256),\n    nn.ReLU(),\n    nn.Dropout(0.5),\n    nn.Linear(256, NUM_CLASSES)\n)\nresnet_v2.load_state_dict(torch.load(\"/kaggle/input/model-effiecient/best_model_resnet_ver_2-7.pth\", map_location=device))\nresnet_v2 = resnet_v2.to(device)\n\n\n# -------- EFFICIENTNET VERSION 1 --------\nefficientnet_v1 = EfficientNet.from_name('efficientnet-b2')\nin_features = efficientnet_v1._fc.in_features\nefficientnet_v1._fc = nn.Linear(in_features, NUM_CLASSES)\nefficientnet_v1.load_state_dict(torch.load(\"/kaggle/input/model-effiecient/best_model_efficientnet-2.pth\", map_location=device))\nefficientnet_v1 = efficientnet_v1.to(device)\n\n# -------- EFFICIENTNET VERSION 2 --------\nefficientnet_v2 = EfficientNet.from_name('efficientnet-b2')\nin_features = efficientnet_v2._fc.in_features\nefficientnet_v2._fc = nn.Linear(in_features, NUM_CLASSES)\nefficientnet_v2.load_state_dict(torch.load(\"/kaggle/input/model-effiecient/best_model_efficientnet_ver_2-2.pth\", map_location=device))\nefficientnet_v2 = efficientnet_v2.to(device)\n\n# -------- MODEL DICTIONARY --------\nmodels = {\n    \"resnet_v1\": resnet_v1,\n    \"resnet_v2\": resnet_v2,\n    \"efficientnet_v1\": efficientnet_v1,\n    \"efficientnet_v2\": efficientnet_v2\n}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T18:16:22.863302Z","iopub.execute_input":"2025-05-25T18:16:22.863752Z","iopub.status.idle":"2025-05-25T18:16:36.125380Z","shell.execute_reply.started":"2025-05-25T18:16:22.863715Z","shell.execute_reply":"2025-05-25T18:16:36.124580Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install gradio","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T18:16:36.126515Z","iopub.execute_input":"2025-05-25T18:16:36.126969Z","iopub.status.idle":"2025-05-25T18:16:48.016265Z","shell.execute_reply.started":"2025-05-25T18:16:36.126935Z","shell.execute_reply":"2025-05-25T18:16:48.015069Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gradio as gr\nfrom PIL import Image\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport numpy as np\nimport torch\n\n# Albumentations transform cho EfficientNet\ntransform_efficientnet = A.Compose([\n    A.Resize(260, 260),\n    A.Normalize(),  # mean/std mặc định giống ImageNet\n    ToTensorV2()\n])\n\n# Albumentations transform cho ResNet\ntransform_resnet = A.Compose([\n    A.Resize(224, 224),\n    A.Normalize(),\n    ToTensorV2()\n])\n\ndef predict(img: Image.Image, model_name: str):\n    # Chuyển PIL Image -> numpy\n    img = np.array(img)\n\n    # Chọn transform theo model\n    transform = transform_efficientnet if model_name.startswith(\"efficientnet\") else transform_resnet\n\n    # Apply transform\n    transformed = transform(image=img)\n    img_tensor = transformed[\"image\"].unsqueeze(0).to(device)\n\n    model = models[model_name]\n    model.eval()\n\n    with torch.no_grad():\n        output = model(img_tensor)\n        pred = output.argmax(dim=1).item()\n\n    return f\"Predicted class: {pred}\"\n\ninterface = gr.Interface(\n    fn=predict,\n    inputs=[\n        gr.Image(type=\"pil\"),\n        gr.Dropdown(choices=list(models.keys()), label=\"Choose Model\")\n    ],\n    outputs=\"text\",\n    title=\"Landmark Classification Demo (4 Models with Albumentations)\"\n)\n\ninterface.launch()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T01:29:36.806999Z","iopub.execute_input":"2025-05-16T01:29:36.807509Z","iopub.status.idle":"2025-05-16T01:29:36.854602Z","shell.execute_reply.started":"2025-05-16T01:29:36.807472Z","shell.execute_reply":"2025-05-16T01:29:36.853035Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gradio as gr\nfrom PIL import Image\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torchvision.models as tv_models\n\n# Device\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# Load pretrained classification models\nmodel_dict = {\n    \"resnet18\": tv_models.resnet18(pretrained=True).to(device),\n    \"resnet50\": tv_models.resnet50(pretrained=True).to(device),\n    \"efficientnet_b0\": tv_models.efficientnet_b0(pretrained=True).to(device),\n    \"efficientnet_b3\": tv_models.efficientnet_b3(pretrained=True).to(device)\n}\n\n# Transforms\ntransform_efficientnet = A.Compose([\n    A.Resize(260, 260),\n    A.Normalize(),\n    ToTensorV2()\n])\n\ntransform_resnet = A.Compose([\n    A.Resize(224, 224),\n    A.Normalize(),\n    ToTensorV2()\n])\n\ndef predict(img: Image.Image, model_name: str):\n    img_np = np.array(img)\n\n    # Step 1: Binary classification\n    binary_transformed = transform_mobilenet(image=img_np)\n    binary_tensor = binary_transformed[\"image\"].unsqueeze(0).to(device)\n\n    with torch.no_grad():\n        binary_output = mobilenet_binary(binary_tensor)\n        binary_confidence = torch.sigmoid(binary_output).item()\n\n    if binary_confidence <= 0.20:\n        return f\"The image does not appear to contain a landmark. (Binary confidence: {binary_confidence:.2f})\"\n\n    # Step 2: Main classification\n    transform = transform_efficientnet if model_name.startswith(\"efficientnet\") else transform_resnet\n    transformed = transform(image=img_np)\n    img_tensor = transformed[\"image\"].to(device)\n\n    model = model_dict[model_name]\n    model.eval()\n\n    # Step 3: ODIN Score\n    temperature = 10  # Use your validated hyperparameter\n    epsilon = 0.0014    # Use your validated hyperparameter\n\n    odin_confidence = odin_score(img_tensor, model, temperature, epsilon)\n    odin_threshold = 0.022769  # Youden's J\n\n    if odin_confidence < odin_threshold:\n        return (\n            f\"The image might contain a landmark. (Binary confidence: {binary_confidence:.2f})\\n\"\n            f\"⚠️ Rejected by ODIN (score: {odin_confidence:.4f} < threshold: {odin_threshold})\"\n        )\n\n    # Step 4: Final classification\n    img_tensor = img_tensor.unsqueeze(0)\n    with torch.no_grad():\n        output = model(img_tensor)\n        probabilities = torch.nn.functional.softmax(output, dim=1)\n        confidence, pred_class = torch.max(probabilities, dim=1)\n        confidence = confidence.item()\n        pred_class = pred_class.item()\n\n    return (\n        f\"The image might contain a landmark. (Binary confidence: {binary_confidence:.2f})\\n\"\n        f\"ODIN score: {odin_confidence:.4f} (above threshold: {odin_threshold})\\n\"\n        f\"Predicted class: {pred_class} (Confidence: {confidence:.2f})\"\n    )\n\n\n# Gradio interface\ninterface = gr.Interface(\n    fn=predict,\n    inputs=[\n        gr.Image(type=\"pil\"),\n        gr.Dropdown(choices=list(model_dict.keys()), label=\"Choose Model\")\n    ],\n    outputs=\"text\",\n    title=\"Landmark Classification Demo (Threshold-based Detection)\"\n)\n\ninterface.launch()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T18:20:38.973674Z","iopub.execute_input":"2025-05-25T18:20:38.974015Z","iopub.status.idle":"2025-05-25T18:20:43.414363Z","shell.execute_reply.started":"2025-05-25T18:20:38.973993Z","shell.execute_reply":"2025-05-25T18:20:43.413439Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Phần có ODIN ở đây ","metadata":{}},{"cell_type":"code","source":"import gradio as gr\nfrom PIL import Image\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torchvision.models as tv_models\n\n# Device\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# Load 4 pretrained classification models\nmodel_dict = {\n    \"resnet18\": tv_models.resnet18(pretrained=True).to(device),\n    \"resnet50\": tv_models.resnet50(pretrained=True).to(device),\n    \"efficientnet_b0\": tv_models.efficientnet_b0(pretrained=True).to(device),\n    \"efficientnet_b3\": tv_models.efficientnet_b3(pretrained=True).to(device)\n}\n\n# Load the binary MobileNet model (output=1)\nmobilenet_binary = tv_models.mobilenet_v2(pretrained=False)\nmobilenet_binary.classifier[1] = nn.Linear(mobilenet_binary.last_channel, 1)\nmobilenet_binary.load_state_dict(torch.load(\"/kaggle/input/mobilenet-landmark-binary-classification/mobilenet_landmark (1).pth\", map_location=device))\nmobilenet_binary.to(device)\nmobilenet_binary.eval()\n\n# Transforms\ntransform_efficientnet = A.Compose([\n    A.Resize(260, 260),\n    A.Normalize(),\n    ToTensorV2()\n])\n\ntransform_resnet = A.Compose([\n    A.Resize(224, 224),\n    A.Normalize(),\n    ToTensorV2()\n])\n\ntransform_mobilenet = A.Compose([\n    A.Resize(224, 224),\n    A.Normalize(),\n    ToTensorV2()\n])\n\ndef odin_score(image_tensor, model, temperature, epsilon):\n    image_tensor = image_tensor.unsqueeze(0).to(device)\n    image_tensor.requires_grad = True  # Needed for gradient computation\n\n    model.eval()  # Ensure in eval mode\n\n    # Forward pass\n    logits = model(image_tensor)\n    logits = logits / temperature\n    pred_class = logits.argmax(dim=1)\n\n    # Compute loss\n    loss = F.cross_entropy(logits, pred_class)\n    model.zero_grad()\n    loss.backward()\n\n    # Perturbation\n    gradient = torch.sign(image_tensor.grad.data)\n    perturbed = image_tensor - epsilon * gradient\n    perturbed = torch.clamp(perturbed, 0, 1)\n\n    # Forward with perturbed input\n    with torch.no_grad():\n        logits_perturbed = model(perturbed) / temperature\n        softmax_scores = F.softmax(logits_perturbed, dim=1)\n        score = torch.max(softmax_scores).item()\n\n    return score\n\ndef predict(img: Image.Image, model_name: str):\n    img_np = np.array(img)\n\n    # Step 1: Binary classification\n    binary_transformed = transform_mobilenet(image=img_np)\n    binary_tensor = binary_transformed[\"image\"].unsqueeze(0).to(device)\n\n    with torch.no_grad():\n        binary_output = mobilenet_binary(binary_tensor)\n        binary_confidence = torch.sigmoid(binary_output).item()\n\n    if binary_confidence <= 0.20:\n        return f\"The image does not appear to contain a landmark. (Binary confidence: {binary_confidence:.2f})\"\n\n    # Step 2: Main classification\n    transform = transform_efficientnet if model_name.startswith(\"efficientnet\") else transform_resnet\n    transformed = transform(image=img_np)\n    img_tensor = transformed[\"image\"].to(device)\n\n    model = model_dict[model_name]\n    model.eval()\n\n    # Step 3: ODIN Score\n    temperature = 10  # Use your validated hyperparameter\n    epsilon = 0.0014    # Use your validated hyperparameter\n\n    odin_confidence = odin_score(img_tensor, model, temperature, epsilon)\n    odin_threshold = 0.022769  # Youden's J\n\n    if odin_confidence < odin_threshold:\n        return (\n            f\"The image might contain a landmark. (Binary confidence: {binary_confidence:.2f})\\n\"\n            f\"⚠️ Rejected by ODIN (score: {odin_confidence:.4f} < threshold: {odin_threshold})\"\n        )\n\n    # Step 4: Final classification\n    img_tensor = img_tensor.unsqueeze(0)\n    with torch.no_grad():\n        output = model(img_tensor)\n        probabilities = torch.nn.functional.softmax(output, dim=1)\n        confidence, pred_class = torch.max(probabilities, dim=1)\n        confidence = confidence.item()\n        pred_class = pred_class.item()\n\n    return (\n        f\"The image might contain a landmark. (Binary confidence: {binary_confidence:.2f})\\n\"\n        f\"ODIN score: {odin_confidence:.4f} (above threshold: {odin_threshold})\\n\"\n        f\"Predicted class: {pred_class} (Confidence: {confidence:.2f})\"\n    )\n\n# Gradio interface\ninterface = gr.Interface(\n    fn=predict,\n    inputs=[\n        gr.Image(type=\"pil\"),\n        gr.Dropdown(choices=list(model_dict.keys()), label=\"Choose Model\")\n    ],\n    outputs=\"text\",\n    title=\"Landmark Classification Demo (with Binary Landmark Detection)\"\n)\n\ninterface.launch()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T18:25:27.139033Z","iopub.execute_input":"2025-05-25T18:25:27.139375Z","iopub.status.idle":"2025-05-25T18:25:31.318922Z","shell.execute_reply.started":"2025-05-25T18:25:27.139350Z","shell.execute_reply":"2025-05-25T18:25:31.317679Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## THêm def odin_score với chỉnh lại hàm predict là được","metadata":{}},{"cell_type":"code","source":"import gradio as gr\nfrom PIL import Image\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torchvision.models as tv_models\n\n# Device\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# Load 4 pretrained classification models\nmodel_dict = {\n    \"resnet18\": tv_models.resnet18(pretrained=True).to(device),\n    \"resnet50\": tv_models.resnet50(pretrained=True).to(device),\n    \"efficientnet_b0\": tv_models.efficientnet_b0(pretrained=True).to(device),\n    \"efficientnet_b3\": tv_models.efficientnet_b3(pretrained=True).to(device)\n}\n\n# Load the binary MobileNet model (output=1)\nmobilenet_binary = tv_models.mobilenet_v2(pretrained=False)\nmobilenet_binary.classifier[1] = nn.Linear(mobilenet_binary.last_channel, 1)\nmobilenet_binary.load_state_dict(torch.load(\"/kaggle/input/mobilenet-landmark-binary-classification/mobilenet_landmark (1).pth\", map_location=device))\nmobilenet_binary.to(device)\nmobilenet_binary.eval()\n\n# Transforms\ntransform_efficientnet = A.Compose([\n    A.Resize(260, 260),\n    A.Normalize(),\n    ToTensorV2()\n])\n\ntransform_resnet = A.Compose([\n    A.Resize(224, 224),\n    A.Normalize(),\n    ToTensorV2()\n])\n\ntransform_mobilenet = A.Compose([\n    A.Resize(224, 224),\n    A.Normalize(),\n    ToTensorV2()\n])\n\ndef predict(img: Image.Image, model_name: str):\n    img_np = np.array(img)\n\n    # Step 1: Run binary classifier\n    binary_transformed = transform_mobilenet(image=img_np)\n    binary_tensor = binary_transformed[\"image\"].unsqueeze(0).to(device)\n\n    with torch.no_grad():\n        binary_output = mobilenet_binary(binary_tensor)\n        prob = torch.sigmoid(binary_output).item()\n        is_landmark = prob >= 0.5  # You can lower this threshold if needed\n\n    if not is_landmark:\n        return f\"The image given does not contain a landmark. (Confidence: {prob:.2f})\"\n\n    # Step 2: Run selected model\n    transform = transform_efficientnet if model_name.startswith(\"efficientnet\") else transform_resnet\n    transformed = transform(image=img_np)\n    img_tensor = transformed[\"image\"].unsqueeze(0).to(device)\n\n    model = model_dict[model_name]\n    model.eval()\n\n    with torch.no_grad():\n        output = model(img_tensor)\n        pred = output.argmax(dim=1).item()\n\n    return f\"The image might contain a landmark. (Confidence: {prob:.2f})\\nPredicted class: {pred}\"\n\n# Gradio interface\ninterface = gr.Interface(\n    fn=predict,\n    inputs=[\n        gr.Image(type=\"pil\"),\n        gr.Dropdown(choices=list(model_dict.keys()), label=\"Choose Model\")\n    ],\n    outputs=\"text\",\n    title=\"Landmark Classification Demo (with Binary Landmark Detection)\"\n)\n\ninterface.launch()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T18:22:35.163344Z","iopub.execute_input":"2025-05-25T18:22:35.163721Z","iopub.status.idle":"2025-05-25T18:22:38.901682Z","shell.execute_reply.started":"2025-05-25T18:22:35.163695Z","shell.execute_reply":"2025-05-25T18:22:38.900836Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torchvision.models as models\nimport torch.nn.functional as F\n\n\nclass AttentionFusion(nn.Module):\n    def __init__(self, channels):\n        super().__init__()\n        self.attn = nn.Sequential(\n            nn.AdaptiveAvgPool2d(1),\n            nn.Conv2d(channels, channels, 1),\n            nn.ReLU(),\n            nn.Conv2d(channels, channels, 1),\n            nn.Sigmoid()\n        )\n\n    def forward(self, local_feat, global_feat):\n        b, c, h, w = local_feat.shape\n        global_feat_expanded = global_feat.view(b, c, 1, 1).expand(-1, -1, h, w)\n        attn = self.attn(local_feat + global_feat_expanded)\n        fused = local_feat * attn + global_feat_expanded * (1 - attn)\n        return fused","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T18:16:48.017803Z","iopub.execute_input":"2025-05-25T18:16:48.018082Z","iopub.status.idle":"2025-05-25T18:16:48.026569Z","shell.execute_reply.started":"2025-05-25T18:16:48.018059Z","shell.execute_reply":"2025-05-25T18:16:48.025460Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class DOLG_ArcFace(nn.Module):\n    def __init__(self, embedding_dim=512):\n        super().__init__()\n        resnet = models.resnet50(pretrained=True)\n\n        # Shared layers\n        self.backbone_common = nn.Sequential(\n            resnet.conv1, resnet.bn1, resnet.relu,\n            resnet.maxpool, resnet.layer1,\n            resnet.layer2, resnet.layer3\n        )\n\n        # ResNet layer4 expects input with 1024 channels (not from local_conv!)\n        self.backbone_global = resnet.layer4\n\n        self.global_pool = nn.AdaptiveAvgPool2d((1, 1))\n        self.global_fc = nn.Linear(2048, embedding_dim)\n\n        self.local_conv = nn.Conv2d(1024, embedding_dim, kernel_size=1)\n\n        self.fusion = AttentionFusion(embedding_dim)\n\n        self.head = nn.Sequential(\n            nn.Conv2d(embedding_dim, embedding_dim, kernel_size=3, padding=1),\n            nn.ReLU(),\n            nn.AdaptiveAvgPool2d(1),\n            nn.Flatten()\n        )\n\n    def forward(self, x):\n        shared_feat = self.backbone_common(x)  # Output: [B, 1024, H, W]\n\n        # Global branch\n        global_feat_map = self.backbone_global(shared_feat)  # Output: [B, 2048, H/2, W/2]\n        global_feat = self.global_pool(global_feat_map).view(x.size(0), -1)  # [B, 2048]\n        global_feat = self.global_fc(global_feat)  # [B, 512]\n\n        # Local branch\n        local_feat = self.local_conv(shared_feat)  # [B, 512, H, W]\n\n        # Fuse\n        fused_feat = self.fusion(local_feat, global_feat)  # [B, 512, H, W]\n        emb = self.head(fused_feat)  # [B, 512]\n        return emb","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T18:16:48.027755Z","iopub.execute_input":"2025-05-25T18:16:48.028108Z","iopub.status.idle":"2025-05-25T18:16:48.050476Z","shell.execute_reply.started":"2025-05-25T18:16:48.028075Z","shell.execute_reply":"2025-05-25T18:16:48.049346Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ArcFace(nn.Module):\n    def __init__(self, in_features, out_features, s=30.0, m=0.5):\n        super().__init__()\n        self.weight = nn.Parameter(torch.FloatTensor(out_features, in_features))\n        nn.init.xavier_uniform_(self.weight)\n        self.s = s\n        self.m = m\n\n    def forward(self, input, label):\n        # normalize inputs and weights\n        cosine = F.linear(F.normalize(input), F.normalize(self.weight))  # [B, C]\n\n        # compute cos(θ + m)\n        theta = torch.acos(torch.clamp(cosine, -1.0 + 1e-7, 1.0 - 1e-7))\n        phi = torch.cos(theta + self.m)\n\n        one_hot = F.one_hot(label, num_classes=cosine.size(1)).float().to(input.device)\n        logits = cosine * (1 - one_hot) + phi * one_hot\n        return logits * self.s\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T18:16:48.053187Z","iopub.execute_input":"2025-05-25T18:16:48.053473Z","iopub.status.idle":"2025-05-25T18:16:48.076675Z","shell.execute_reply.started":"2025-05-25T18:16:48.053450Z","shell.execute_reply":"2025-05-25T18:16:48.075645Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gradio as gr\nfrom PIL import Image\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torchvision.models as tv_models\n#from model import DOLG_ArcFace, ArcFace  # Make sure this file exists\n\n# ============ Device ============\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# ============ Pretrained Classification Models ============\nmodel_dict = {\n    \"resnet18\": tv_models.resnet18(pretrained=True).to(device),\n    \"resnet50\": tv_models.resnet50(pretrained=True).to(device),\n    \"efficientnet_b0\": tv_models.efficientnet_b0(pretrained=True).to(device),\n    \"efficientnet_b3\": tv_models.efficientnet_b3(pretrained=True).to(device)\n}\n\n# ============ Add DOLG Model ============\nembedding_dim = 512\nnum_classes = 102  # set this according to your dataset\n\ndolg_model = DOLG_ArcFace(embedding_dim=embedding_dim).to(device)\narcface_head = ArcFace(in_features=embedding_dim, out_features=num_classes).to(device)\n\n# Load weights\ncheckpoint = torch.load(\"/kaggle/input/resnet-dolg/kaggle/working/best_model.pth\", map_location=device)\ndolg_model.load_state_dict(checkpoint[\"model_state_dict\"])\narcface_head.load_state_dict(checkpoint[\"arcface_state_dict\"])\ndolg_model.eval()\narcface_head.eval()\n\nmodel_dict[\"resnet_dolg\"] = (dolg_model, arcface_head)  # special handling for this one\n\n# ============ Load Binary MobileNet ============\nmobilenet_binary = tv_models.mobilenet_v2(pretrained=False)\nmobilenet_binary.classifier[1] = nn.Linear(mobilenet_binary.last_channel, 1)\nmobilenet_binary.load_state_dict(torch.load(\"/kaggle/input/mobilenet-landmark-binary-classification/mobilenet_landmark (1).pth\", map_location=device))\nmobilenet_binary.to(device)\nmobilenet_binary.eval()\n\n# ============ Transforms ============\ntransform_efficientnet = A.Compose([\n    A.Resize(260, 260),\n    A.Normalize(),\n    ToTensorV2()\n])\n\ntransform_resnet = A.Compose([\n    A.Resize(224, 224),\n    A.Normalize(),\n    ToTensorV2()\n])\n\ntransform_mobilenet = A.Compose([\n    A.Resize(224, 224),\n    A.Normalize(),\n    ToTensorV2()\n])\n\n# ============ Inference Function ============\ndef predict(img: Image.Image, model_name: str):\n    img_np = np.array(img)\n\n    # Step 1: Run binary classifier\n    binary_transformed = transform_mobilenet(image=img_np)\n    binary_tensor = binary_transformed[\"image\"].unsqueeze(0).to(device)\n\n    with torch.no_grad():\n        binary_output = mobilenet_binary(binary_tensor)\n        prob = torch.sigmoid(binary_output).item()\n        is_landmark = prob >= 0.5\n\n    if not is_landmark:\n        return f\"The image does not contain a landmark. (Confidence: {prob:.2f})\"\n\n    # Step 2: Classify landmark\n    transform = transform_efficientnet if model_name.startswith(\"efficientnet\") else transform_resnet\n    transformed = transform(image=img_np)\n    img_tensor = transformed[\"image\"].unsqueeze(0).to(device)\n\n    # Handle ResNet + DOLG differently\n    if model_name == \"resnet_dolg\":\n        dolg_model, arcface_head = model_dict[\"resnet_dolg\"]\n        with torch.no_grad():\n            embedding = dolg_model(img_tensor)\n            logits = arcface_head(embedding, torch.zeros(1, dtype=torch.long).to(device))  # dummy label\n            prob_class = F.softmax(logits, dim=1)\n            conf, pred = torch.max(prob_class, dim=1)\n        return f\"Landmark detected (Confidence: {prob:.2f})\\nPredicted class (ResNet+DOLG): {pred.item()} (Conf: {conf.item():.2f})\"\n    \n    # Standard models\n    model = model_dict[model_name]\n    model.eval()\n    with torch.no_grad():\n        output = model(img_tensor)\n        pred = output.argmax(dim=1).item()\n\n    return f\"Landmark detected (Confidence: {prob:.2f})\\nPredicted class ({model_name}): {pred}\"\n\n# ============ Gradio Interface ============\ninterface = gr.Interface(\n    fn=predict,\n    inputs=[\n        gr.Image(type=\"pil\"),\n        gr.Dropdown(choices=list(model_dict.keys()), label=\"Choose Model\")\n    ],\n    outputs=\"text\",\n    title=\"Landmark Classification Demo (ResNet + DOLG + ArcFace Integrated)\"\n)\n\ninterface.launch()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T18:16:48.077695Z","iopub.execute_input":"2025-05-25T18:16:48.078008Z","iopub.status.idle":"2025-05-25T18:16:58.723924Z","shell.execute_reply.started":"2025-05-25T18:16:48.077966Z","shell.execute_reply":"2025-05-25T18:16:58.722842Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gradio as gr\nfrom PIL import Image\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torchvision.models as tv_models\n\n# Device\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# Load classification models\nmodel_dict = {\n    \"resnet18\": tv_models.resnet18(pretrained=True).to(device),\n    \"resnet50\": tv_models.resnet50(pretrained=True).to(device),\n    \"efficientnet_b0\": tv_models.efficientnet_b0(pretrained=True).to(device),\n    \"efficientnet_b3\": tv_models.efficientnet_b3(pretrained=True).to(device)\n}\n\n# Load binary classifier (MobileNet)\nmobilenet_binary = tv_models.mobilenet_v2(pretrained=False)\nmobilenet_binary.classifier[1] = nn.Linear(mobilenet_binary.last_channel, 1)\nmobilenet_binary.load_state_dict(torch.load(\"/kaggle/input/mobilenet-landmark-binary-classification/mobilenet_landmark (1).pth\", map_location=device))\nmobilenet_binary.to(device)\nmobilenet_binary.eval()\n\n# Transforms\ntransform_efficientnet = A.Compose([\n    A.Resize(260, 260),\n    A.Normalize(),\n    ToTensorV2()\n])\n\ntransform_resnet = A.Compose([\n    A.Resize(224, 224),\n    A.Normalize(),\n    ToTensorV2()\n])\n\ntransform_mobilenet = A.Compose([\n    A.Resize(224, 224),\n    A.Normalize(),\n    ToTensorV2()\n])\n\ndef predict(img: Image.Image, model_name: str):\n    img_np = np.array(img)\n\n    # Step 1: Binary classification\n    binary_transformed = transform_mobilenet(image=img_np)\n    binary_tensor = binary_transformed[\"image\"].unsqueeze(0).to(device)\n\n    with torch.no_grad():\n        binary_output = mobilenet_binary(binary_tensor)\n        binary_confidence = torch.sigmoid(binary_output).item()\n\n    if binary_confidence <= 0.20:\n        return f\"The image does not appear to contain a landmark. (Binary confidence: {binary_confidence:.2f})\"\n\n    # Step 2: Main classification\n    transform = transform_efficientnet if model_name.startswith(\"efficientnet\") else transform_resnet\n    transformed = transform(image=img_np)\n    img_tensor = transformed[\"image\"].unsqueeze(0).to(device)\n\n    model = model_dict[model_name]\n    model.eval()\n\n    with torch.no_grad():\n        output = model(img_tensor)\n        probabilities = torch.nn.functional.softmax(output, dim=1)\n        confidence, pred_class = torch.max(probabilities, dim=1)\n        confidence = confidence.item()\n        pred_class = pred_class.item()\n\n    return (\n        f\"The image might contain a landmark. (Binary confidence: {binary_confidence:.2f})\\n\"\n        f\"Predicted class: {pred_class} (Confidence: {confidence:.2f})\"\n    )\n\n# Gradio interface\ninterface = gr.Interface(\n    fn=predict,\n    inputs=[\n        gr.Image(type=\"pil\"),\n        gr.Dropdown(choices=list(model_dict.keys()), label=\"Choose Model\")\n    ],\n    outputs=\"text\",\n    title=\"Landmark Classification with Binary Filter (Threshold 0.20)\"\n)\n\ninterface.launch()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T01:35:12.089376Z","iopub.execute_input":"2025-05-16T01:35:12.090706Z","iopub.status.idle":"2025-05-16T01:35:14.966341Z","shell.execute_reply.started":"2025-05-16T01:35:12.090630Z","shell.execute_reply":"2025-05-16T01:35:14.965153Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gradio as gr\nfrom PIL import Image\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torchvision.models as tv_models\nimport random\nimport os\n\n# Device\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# Load classification models\nmodel_dict = {\n    \"resnet18\": tv_models.resnet18(pretrained=True).to(device),\n    \"resnet50\": tv_models.resnet50(pretrained=True).to(device),\n    \"efficientnet_b0\": tv_models.efficientnet_b0(pretrained=True).to(device),\n    \"efficientnet_b3\": tv_models.efficientnet_b3(pretrained=True).to(device)\n}\n\n# Load binary classifier (MobileNet)\nmobilenet_binary = tv_models.mobilenet_v2(pretrained=False)\nmobilenet_binary.classifier[1] = nn.Linear(mobilenet_binary.last_channel, 1)\nmobilenet_binary.load_state_dict(torch.load(\"/kaggle/input/mobilenet-landmark-binary-classification/mobilenet_landmark (1).pth\", map_location=device))\nmobilenet_binary.to(device)\nmobilenet_binary.eval()\n\n# Transforms\ntransform_efficientnet = A.Compose([\n    A.Resize(260, 260),\n    A.Normalize(),\n    ToTensorV2()\n])\n\ntransform_resnet = A.Compose([\n    A.Resize(224, 224),\n    A.Normalize(),\n    ToTensorV2()\n])\n\ntransform_mobilenet = A.Compose([\n    A.Resize(224, 224),\n    A.Normalize(),\n    ToTensorV2()\n])\n\n# Load CSV for \"I'm feeling lucky\"\ndf = pd.read_csv(\"/kaggle/input/id-2-names-landmark/ID_to_names/train_with_landmark_names_fixed.csv\", encoding=\"latin1\")\n\n# Drop rows with missing ID or parse errors\ndf = df.dropna(subset=['id'])\n\n# You may want to adjust the path below to point to actual images\nIMAGE_BASE_PATH = \"/kaggle/input/landmark-recognition-2021/train\"  # Change if needed\n\ndef predict(img: Image.Image, model_name: str):\n    img_np = np.array(img)\n\n    # Step 1: Binary classification\n    binary_transformed = transform_mobilenet(image=img_np)\n    binary_tensor = binary_transformed[\"image\"].unsqueeze(0).to(device)\n\n    with torch.no_grad():\n        binary_output = mobilenet_binary(binary_tensor)\n        binary_confidence = torch.sigmoid(binary_output).item()\n\n    if binary_confidence <= 0.20:\n        return f\"The image does not appear to contain a landmark. (Binary confidence: {binary_confidence:.2f})\"\n\n    # Step 2: Main classification\n    transform = transform_efficientnet if model_name.startswith(\"efficientnet\") else transform_resnet\n    transformed = transform(image=img_np)\n    img_tensor = transformed[\"image\"].to(device)\n\n    model = model_dict[model_name]\n    model.eval()\n\n    # Step 3: ODIN Score\n    temperature = 10  # Use your validated hyperparameter\n    epsilon = 0.0014    # Use your validated hyperparameter\n\n    odin_confidence = odin_score(img_tensor, model, temperature, epsilon)\n    odin_threshold = 0.022769  # Youden's J\n\n    if odin_confidence < odin_threshold:\n        return (\n            f\"The image might contain a landmark. (Binary confidence: {binary_confidence:.2f})\\n\"\n            f\"⚠️ Rejected by ODIN (score: {odin_confidence:.4f} < threshold: {odin_threshold})\"\n        )\n\n    # Step 4: Final classification\n    img_tensor = img_tensor.unsqueeze(0)\n    with torch.no_grad():\n        output = model(img_tensor)\n        probabilities = torch.nn.functional.softmax(output, dim=1)\n        confidence, pred_class = torch.max(probabilities, dim=1)\n        confidence = confidence.item()\n        pred_class = pred_class.item()\n\n    return (\n        f\"The image might contain a landmark. (Binary confidence: {binary_confidence:.2f})\\n\"\n        f\"ODIN score: {odin_confidence:.4f} (above threshold: {odin_threshold})\\n\"\n        f\"Predicted class: {pred_class} (Confidence: {confidence:.2f})\"\n    )\n\n\ndef build_image_path_from_id(img_id: str):\n    # Example: '1420523fe073af12' => /kaggle/input/landmark-recognition-2021/train/1/4/2/1420523fe073af12.jpg\n    return f\"/kaggle/input/landmark-recognition-2021/train/{img_id[0]}/{img_id[1]}/{img_id[2]}/{img_id}.jpg\"\n\ndef lucky_guess(model_name):\n    random_row = df.sample(1).iloc[0]\n    img_id = random_row['id']\n    landmark_name = random_row.get(\"landmark_name\", \"Unknown Landmark\")\n\n    image_path = build_image_path_from_id(img_id)\n\n    if not os.path.exists(image_path):\n        return None, f\"Image not found: {image_path}\"\n\n    img = Image.open(image_path).convert(\"RGB\")\n    prediction_text = predict(img, model_name)\n    \n    # Add landmark name from CSV if available\n    prediction_text += f\"\\nLandmark name (from CSV): {landmark_name}\"\n    return img, prediction_text\n\n\n\n# Gradio interface\nwith gr.Blocks() as interface:\n    gr.Markdown(\"# Landmark Classification with Binary Filter (Threshold 0.20)\")\n\n    with gr.Row():\n        image_input = gr.Image(type=\"pil\", label=\"Upload Image\")\n        model_dropdown = gr.Dropdown(choices=list(model_dict.keys()), label=\"Choose Model\", value=\"resnet18\")\n\n    output_text = gr.Textbox(label=\"Result\")\n\n    with gr.Row():\n        predict_btn = gr.Button(\"Predict\")\n        lucky_btn = gr.Button(\"I'm feeling lucky 🎲\")\n\n    lucky_image_output = gr.Image(label=\"Lucky Image\")\n\n    predict_btn.click(fn=predict, inputs=[image_input, model_dropdown], outputs=output_text)\n    lucky_btn.click(fn=lucky_guess, inputs=[model_dropdown], outputs=[lucky_image_output, output_text])\n\ninterface.launch()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T02:41:27.094384Z","iopub.execute_input":"2025-05-16T02:41:27.094776Z","iopub.status.idle":"2025-05-16T02:41:40.393796Z","shell.execute_reply.started":"2025-05-16T02:41:27.094734Z","shell.execute_reply":"2025-05-16T02:41:40.392753Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from PIL import Image\n\nimg_path = \"/kaggle/input/landmark-recognition-2021/train/1/6/b/16bbacee81a4230f.jpg\"\nimage = Image.open(img_path)\n\n# To display it in the notebook:\nimage.show()\n# or in Jupyter-compatible notebooks:\ndisplay(image)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T01:28:37.400258Z","iopub.execute_input":"2025-05-16T01:28:37.401142Z","iopub.status.idle":"2025-05-16T01:28:37.627214Z","shell.execute_reply.started":"2025-05-16T01:28:37.401115Z","shell.execute_reply":"2025-05-16T01:28:37.625433Z"}},"outputs":[],"execution_count":null}]}