{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.11.11"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":29762,"databundleVersionId":2541532,"sourceType":"competition"},{"sourceId":11730964,"sourceType":"datasetVersion","datasetId":7364044},{"sourceId":11766825,"sourceType":"datasetVersion","datasetId":7374262},{"sourceId":11786457,"sourceType":"datasetVersion","datasetId":7385034},{"sourceId":11826352,"sourceType":"datasetVersion","datasetId":7393923},{"sourceId":11859826,"sourceType":"datasetVersion","datasetId":7412797},{"sourceId":11873425,"sourceType":"datasetVersion","datasetId":7393689},{"sourceId":11915250,"sourceType":"datasetVersion","datasetId":7419283},{"sourceId":11933943,"sourceType":"datasetVersion","datasetId":7502656},{"sourceId":11940175,"sourceType":"datasetVersion","datasetId":7505289},{"sourceId":12011827,"sourceType":"datasetVersion","datasetId":7393981},{"sourceId":12037239,"sourceType":"datasetVersion","datasetId":7574305}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Demo ## ","metadata":{}},{"cell_type":"code","source":"!nvidia-smi","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Load your CSVs\n# train_df = pd.read_csv('/kaggle/working/train.csv')\n# test_df = pd.read_csv('/kaggle/working/test.csv')\n# val_df = pd.read_csv('/kaggle/working/val.csv')\ntrain_df = pd.read_csv('/kaggle/input/landmark/train.csv')\ntest_df = pd.read_csv('/kaggle/input/landmark/test.csv')\nval_df = pd.read_csv('/kaggle/input/landmark/val.csv')\n# Build mapping\nlandmark_id_to_idx_4_models = {lid: idx for idx, lid in enumerate(sorted(train_df['landmark_id'].unique()))}\n \n\n# Map class_idx\ntrain_df['class_idx'] = train_df['landmark_id'].map(landmark_id_to_idx_4_models)\ntest_df['class_idx'] = test_df['landmark_id'].map(landmark_id_to_idx_4_models)\nval_df['class_idx'] = val_df['landmark_id'].map(landmark_id_to_idx_4_models)\n\nNUM_CLASSES = len(landmark_id_to_idx_4_models)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-13T14:24:27.289875Z","iopub.execute_input":"2025-06-13T14:24:27.290494Z","iopub.status.idle":"2025-06-13T14:24:27.710126Z","shell.execute_reply.started":"2025-06-13T14:24:27.290471Z","shell.execute_reply":"2025-06-13T14:24:27.709505Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nfrom torch.utils.data import Dataset, DataLoader\n\nIMAGE_ROOT = \"/kaggle/input/landmark-recognition-2021/train\"\nCSV_PATH = \"/kaggle/input/landmark-labels/train_with_landmark_names_fixed.csv\"\nIMAGE_SIZE = 224\n\n# === Load CSV ===\ndf = pd.read_csv(CSV_PATH,encoding = \"Latin1\")\n\n# Drop missing or malformed rows\ndf = df.dropna(subset=['id', 'landmark_id'])\n\n# Ensure landmark_id is int\ndf['landmark_id'] = df['landmark_id'].astype(int)\n\n# === Map landmark_id to class indices ===\nlandmark_id_to_idx_dolg = {lid: idx for idx, lid in enumerate(sorted(df['landmark_id'].unique()))}\ndf['class_idx'] = df['landmark_id'].map(landmark_id_to_idx_dolg)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-13T14:24:29.824141Z","iopub.execute_input":"2025-06-13T14:24:29.824492Z","iopub.status.idle":"2025-06-13T14:24:32.946217Z","shell.execute_reply.started":"2025-06-13T14:24:29.824472Z","shell.execute_reply":"2025-06-13T14:24:32.945253Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install efficientnet_pytorch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-13T14:24:36.283687Z","iopub.execute_input":"2025-06-13T14:24:36.284182Z","iopub.status.idle":"2025-06-13T14:24:39.419099Z","shell.execute_reply.started":"2025-06-13T14:24:36.284158Z","shell.execute_reply":"2025-06-13T14:24:39.418252Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torchvision import models\nfrom efficientnet_pytorch import EfficientNet\n\nNUM_CLASSES = 100\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# -------- RESNET VERSION 1 --------\nresnet_v1 = models.resnet50(pretrained=False)\nin_features = resnet_v1.fc.in_features\nresnet_v1.fc = nn.Sequential(\n    nn.Linear(in_features, 512),\n    nn.ReLU(),\n    nn.Dropout(0.5),\n    nn.Linear(512, 256),\n    nn.ReLU(),\n    nn.Dropout(0.5),\n    nn.Linear(256, NUM_CLASSES)\n)\nresnet_v1.load_state_dict(torch.load(\"/kaggle/input/model-effiecient/best_model_resnet-5.pth\", map_location=device))\nresnet_v1 = resnet_v1.to(device)\n\n# -------- RESNET VERSION 2 --------\nresnet_v2 = models.resnet50(pretrained=False)\nin_features = resnet_v2.fc.in_features\nresnet_v2.fc = nn.Sequential(\n    nn.Linear(in_features, 512),\n    nn.ReLU(),\n    nn.Dropout(0.5),\n    nn.Linear(512, 256),\n    nn.ReLU(),\n    nn.Dropout(0.5),\n    nn.Linear(256, NUM_CLASSES)\n)\nresnet_v2.load_state_dict(torch.load(\"/kaggle/input/model-effiecient/best_model_resnet_ver_2-7.pth\", map_location=device))\nresnet_v2 = resnet_v2.to(device)\n\n\n# -------- EFFICIENTNET VERSION 1 --------\nefficientnet_v1 = EfficientNet.from_name('efficientnet-b2')\nin_features = efficientnet_v1._fc.in_features\nefficientnet_v1._fc = nn.Linear(in_features, NUM_CLASSES)\nefficientnet_v1.load_state_dict(torch.load(\"/kaggle/input/model-effiecient/best_model_efficientnet-2.pth\", map_location=device))\nefficientnet_v1 = efficientnet_v1.to(device)\n\n# -------- EFFICIENTNET VERSION 2 --------\nefficientnet_v2 = EfficientNet.from_name('efficientnet-b2')\nin_features = efficientnet_v2._fc.in_features\nefficientnet_v2._fc = nn.Linear(in_features, NUM_CLASSES)\nefficientnet_v2.load_state_dict(torch.load(\"/kaggle/input/model-effiecient/best_model_efficientnet_ver_2-2.pth\", map_location=device))\nefficientnet_v2 = efficientnet_v2.to(device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-13T14:24:41.720495Z","iopub.execute_input":"2025-06-13T14:24:41.721242Z","iopub.status.idle":"2025-06-13T14:24:44.748601Z","shell.execute_reply.started":"2025-06-13T14:24:41.721209Z","shell.execute_reply":"2025-06-13T14:24:44.747750Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install gradio","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-13T14:25:48.535633Z","iopub.execute_input":"2025-06-13T14:25:48.536117Z","iopub.status.idle":"2025-06-13T14:25:51.741342Z","shell.execute_reply.started":"2025-06-13T14:25:48.536094Z","shell.execute_reply":"2025-06-13T14:25:51.740291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gradio as gr\nfrom PIL import Image\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torchvision.models as tv_models\nimport math\nimport torch.nn.functional as F\nfrom torchvision import transforms\n# Device\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# ---- ArcFace + DOLG Definitions ----\nclass AttentionFusion(nn.Module):\n    def __init__(self, channels):\n        super().__init__()\n        self.attn = nn.Sequential(\n            nn.AdaptiveAvgPool2d(1),\n            nn.Conv2d(channels, channels, 1),\n            nn.ReLU(),\n            nn.Conv2d(channels, channels, 1),\n            nn.Sigmoid()\n        )\n\n    def forward(self, local_feat, global_feat):\n        b, c, h, w = local_feat.shape\n        global_feat_expanded = global_feat.view(b, c, 1, 1).expand(-1, -1, h, w)\n        attn = self.attn(local_feat + global_feat_expanded)\n        fused = local_feat * attn + global_feat_expanded * (1 - attn)\n        return fused\n\nclass ArcFace(nn.Module):\n    def __init__(self, in_features, out_features, s=30.0, m=0.3):\n        super().__init__()\n        self.weight = nn.Parameter(torch.FloatTensor(out_features, in_features))\n        nn.init.xavier_uniform_(self.weight)\n        self.s = s\n        self.m = m\n        self.cos_m = math.cos(m)\n        self.sin_m = math.sin(m)\n\n    def forward(self, input, labels):\n        cosine = F.linear(F.normalize(input), F.normalize(self.weight))\n        sine = torch.sqrt((1.0 - cosine ** 2).clamp(min=0.0))\n        phi = cosine * self.cos_m - sine * self.sin_m\n        one_hot = torch.zeros_like(cosine)\n        one_hot.scatter_(1, labels.view(-1, 1), 1.0)\n        logits = (one_hot * phi) + ((1.0 - one_hot) * cosine)\n        logits *= self.s\n        return logits\n\nclass DOLG_ArcFace(nn.Module):\n    def __init__(self, embedding_dim=512):\n        super().__init__()\n        resnet = tv_models.resnet50(pretrained=True)\n        self.backbone_common = nn.Sequential(\n            resnet.conv1, resnet.bn1, resnet.relu,\n            resnet.maxpool, resnet.layer1,\n            resnet.layer2, resnet.layer3\n        )\n        self.backbone_global = resnet.layer4\n        self.global_pool = nn.AdaptiveAvgPool2d((1, 1))\n        self.global_fc = nn.Linear(2048, embedding_dim)\n        self.local_conv = nn.Conv2d(1024, embedding_dim, kernel_size=1)\n        self.fusion = AttentionFusion(embedding_dim)\n        self.head = nn.Sequential(\n            nn.Conv2d(embedding_dim, embedding_dim, kernel_size=3, padding=1),\n            nn.ReLU(),\n            nn.AdaptiveAvgPool2d(1),\n            nn.Flatten()\n        )\n\n    def forward(self, x):\n        shared_feat = self.backbone_common(x)\n        global_feat_map = self.backbone_global(shared_feat)\n        global_feat = self.global_pool(global_feat_map).view(x.size(0), -1)\n        global_feat = self.global_fc(global_feat)\n        local_feat = self.local_conv(shared_feat)\n        fused_feat = self.fusion(local_feat, global_feat)\n        emb = self.head(fused_feat)\n        return emb","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-13T14:25:54.401844Z","iopub.execute_input":"2025-06-13T14:25:54.402618Z","iopub.status.idle":"2025-06-13T14:25:56.511782Z","shell.execute_reply.started":"2025-06-13T14:25:54.402582Z","shell.execute_reply":"2025-06-13T14:25:56.511216Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import collections\nimport pickle\nfrom sklearn.metrics.pairwise import cosine_similarity\nfrom scipy.spatial import cKDTree\nfrom skimage.measure import ransac\nfrom skimage.transform import AffineTransform\nimport tensorflow as tf\nimport tensorflow_hub as hub\n# Load train embeddings\nwith open(\"/kaggle/input/delf-embeddings/folder (1)/kaggle/working/train_embeddings_final.pkl\", \"rb\") as f:\n    train_embeddings = pickle.load(f)\n\n# Fix path\ncorrect_base_path = \"/kaggle/input/train-sub-delf-1/train_sub\"\ntrain_embeddings[\"images_paths\"] = [\n    os.path.join(correct_base_path, *os.path.normpath(p).split(os.sep)[-2:])\n    for p in train_embeddings[\"images_paths\"]\n]\n\n# TensorFlow DELF\ndelf = hub.load('https://tfhub.dev/google/delf/1').signatures['default']\n\n# EfficientNet as feature extractor\nfrom efficientnet_pytorch import EfficientNet\nNUM_CLASSES = 100 \nefficientnet_v2_delf = EfficientNet.from_name('efficientnet-b2')\nin_features = efficientnet_v2_delf._fc.in_features\nefficientnet_v2_delf._fc = nn.Linear(in_features, NUM_CLASSES)\n\n# Load weights\ncheckpoint_path = \"/kaggle/input/model-effiecient/best_model_efficientnet_ver_2-2.pth\"\nefficientnet_v2_delf.load_state_dict(torch.load(checkpoint_path, map_location=device))\nefficientnet_v2_delf._fc = nn.Identity()\nembedding_model = efficientnet_v2_delf.to(device)\nembedding_model.eval()\n\n# Helper: convert to tensor\ntransform_emb = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n    transforms.Normalize([0.485, 0.456, 0.406],\n                         [0.229, 0.224, 0.225])\n])\n\ndef get_embeddings(image: Image.Image):\n    image_tensor = transform_emb(image).unsqueeze(0).to(device)\n    with torch.no_grad():\n        emb = efficientnet_v2_delf(image_tensor).cpu().numpy()\n    return emb\n\ndef run_delf(image_np):\n    if isinstance(image_np, np.ndarray):\n        image_tf = tf.convert_to_tensor(image_np, dtype=tf.float32)\n    float_img = tf.image.convert_image_dtype(image_tf, tf.float32)\n    input_dict = {\n        'image': float_img,\n        'image_scales': tf.constant([1.0], dtype=tf.float32),\n        'max_feature_num': tf.constant(1000, dtype=tf.int32),\n        'score_threshold': tf.constant(100.0, dtype=tf.float32)\n    }\n    with tf.device('/CPU:0'):  # tránh lỗi cuDNN\n        return delf(**input_dict)\n\ndef delf_rerank(query_img, top_df):\n    query_np = np.array(query_img.resize((224, 224)))\n    delf_q = run_delf(query_np)\n\n    inliers_list = []\n    for path in top_df['image_paths']:\n        try:\n            img_np = np.array(Image.open(path).resize((224, 224)))\n            delf_k = run_delf(img_np)\n\n            d1_tree = cKDTree(delf_q['descriptors'])\n            _, indices = d1_tree.query(delf_k['descriptors'], distance_upper_bound=0.8)\n\n            locations_k = np.array([\n                delf_k['locations'][i]\n                for i in range(len(indices)) if indices[i] != len(delf_q['descriptors'])\n            ])\n            locations_q = np.array([\n                delf_q['locations'][indices[i]]\n                for i in range(len(indices)) if indices[i] != len(delf_q['descriptors'])\n            ])\n\n            _, inliers = ransac((locations_q, locations_k), AffineTransform,\n                                min_samples=3, residual_threshold=20, max_trials=1000)\n            total_inliers = sum(inliers)\n        except:\n            total_inliers = 1\n        inliers_list.append(total_inliers)\n\n    top_df['inliers'] = inliers_list\n    top_df['reranked_conf'] = np.sqrt(top_df['inliers']) * top_df['cos_similarity']\n    top_df = top_df.sort_values(\"reranked_conf\", ascending=False)\n    return top_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-13T14:25:59.021429Z","iopub.execute_input":"2025-06-13T14:25:59.021948Z","iopub.status.idle":"2025-06-13T14:26:12.102703Z","shell.execute_reply.started":"2025-06-13T14:25:59.021925Z","shell.execute_reply":"2025-06-13T14:26:12.101875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n# ---- Load Pretrained Models ----\n# Binary classification model\nmobilenet_binary = tv_models.mobilenet_v2(pretrained=False)\nmobilenet_binary.classifier[1] = nn.Linear(mobilenet_binary.last_channel, 1)\nmobilenet_binary.load_state_dict(torch.load(\"/kaggle/input/mobilenet-landmark-binary-classification/mobilenet_landmark.pth\", map_location=device))\nmobilenet_binary.to(device).eval()\n\n\ndolg_model = DOLG_ArcFace().to(device).eval()\narcface_head = ArcFace(in_features=512, out_features=102).to(device).eval()\nckpt = torch.load(\"/kaggle/input/resnet-dolg/ResNetDOLG.pth\", map_location=device)\ndolg_model.load_state_dict(ckpt['model_state_dict'])\narcface_head.load_state_dict(ckpt['arcface_state_dict'])\n\n# Models dictionary\nmodels = {\n    \"resnet_v1\": resnet_v1,\n    \"resnet_v2\": resnet_v2,\n    \"efficientnet_v1\": efficientnet_v1,\n    \"efficientnet_v2\": efficientnet_v2\n}\n\n# ---- Transforms ----\ntransform_mobilenet = A.Compose([\n    A.Resize(224, 224),\n    A.Normalize(),\n    ToTensorV2()\n])\n\ntransforms_dict = {\n    \"resnet_v1\": A.Compose([A.Resize(224, 224), A.Normalize(), ToTensorV2()]),\n    \"resnet_v2\": A.Compose([A.Resize(224, 224), A.Normalize(), ToTensorV2()]),\n    \"efficientnet_v1\": A.Compose([A.Resize(260, 260), A.Normalize(), ToTensorV2()]),\n    \"efficientnet_v2\": A.Compose([A.Resize(260, 260), A.Normalize(), ToTensorV2()]),\n    \"dolg\": A.Compose([A.Resize(256, 256), A.CenterCrop(224, 224), A.Normalize([0.5]*3, [0.5]*3), ToTensorV2()])\n}\nimport pandas as pd\n\ncsv_path = \"/kaggle/input/landmark-labels/train_with_landmark_names_fixed.csv\"\ndf = pd.read_csv(csv_path, encoding=\"latin1\")\n\nlandmark_id_to_name = (\n    df.drop_duplicates(\"landmark_id\")[[\"landmark_id\", \"landmark_name\"]]\n    .set_index(\"landmark_id\")[\"landmark_name\"]\n    .to_dict()\n)\n\n\nfrom PIL import Image\nimport os\n\ndef predict(img: Image.Image):\n    img = img.convert(\"RGB\")\n    img_np = np.array(img)\n    final_output = []\n\n    mobilenet_tensor = transform_mobilenet(image=img_np)[\"image\"].unsqueeze(0).to(device)\n    with torch.no_grad():\n        prob = torch.sigmoid(mobilenet_binary(mobilenet_tensor)).item()\n    final_output.append(f\"[Binary Classifier] Landmark probability: {prob:.2%}\")\n    \n    if prob < 0.1:\n        return \"\\n\".join(final_output) + \"\\n\\nThis image is NOT a landmark.\", []\n\n    final_output.append(\"This image IS a landmark. Running all models...\\n\")\n\n    top_images = [] \n\n    model_list = list(models.keys()) + [\"dolg\", \"retrieval\"]\n    for model_name in model_list:\n        debug_log = [f\"--- Result from {model_name.upper()} ---\"]\n        try:\n            if model_name == \"retrieval\":\n                query_emb = get_embeddings(img)\n                sim_matrix = cosine_similarity(query_emb, train_embeddings[\"embedded_images\"])\n                top5_idx = np.argsort(sim_matrix[0])[::-1][:5]\n                top_df = pd.DataFrame({\n                    \"image_paths\": [train_embeddings[\"images_paths\"][i] for i in top5_idx],\n                    \"cos_similarity\": [sim_matrix[0][i] for i in top5_idx],\n                    \"prediction\": [train_embeddings[\"labels\"][i] for i in top5_idx],\n                })\n                idx_to_landmark_id = {v: k for k, v in landmark_id_to_idx_4_models.items()}\n                top_df[\"landmark_id\"] = top_df[\"prediction\"].map(idx_to_landmark_id)\n                top_df = delf_rerank(img, top_df)\n                vote = collections.Counter(top_df[\"landmark_id\"])\n                final_landmark_id = vote.most_common(1)[0][0]\n                name = landmark_id_to_name.get(final_landmark_id, f\"Class {final_landmark_id}\")\n                debug_log.append(f\"Prediction: {name} (ID: {final_landmark_id})\")\n\n                # Load top 5 images as PIL\n                for path in top_df[\"image_paths\"]:\n                    if os.path.exists(path):\n                        top_images.append(Image.open(path).convert(\"RGB\"))\n            else:\n                transform = transforms_dict[model_name]\n                img_tensor = transform(image=img_np)[\"image\"].unsqueeze(0).to(device)\n                model = dolg_model if model_name == \"dolg\" else models[model_name]\n                model.eval()\n                with torch.no_grad():\n                    if model_name == \"dolg\":\n                        emb = model(img_tensor)\n                        logits = arcface_head(emb, torch.tensor([0], device=device))\n                        idx_map = landmark_id_to_idx_dolg\n                    else:\n                        logits = model(img_tensor)\n                        idx_map = landmark_id_to_idx_4_models\n                probs = F.softmax(logits, dim=1)\n                pred_idx = probs.argmax(dim=1).item()\n                confidence = probs[0, pred_idx].item()\n                idx_to_landmark_id = {v: k for k, v in idx_map.items()}\n                landmark_id = idx_to_landmark_id.get(pred_idx, pred_idx)\n                name = landmark_id_to_name.get(landmark_id, f\"Class {landmark_id}\")\n                debug_log.append(f\"Prediction: {name} (ID: {landmark_id}) - Confidence: {confidence:.2%}\")\n        except Exception as e:\n            debug_log.append(f\"[ERROR] Failed with {model_name}: {str(e)}\")\n\n        final_output.extend(debug_log)\n        final_output.append(\"\")\n\n    return \"\\n\".join(final_output), top_images\n# --- Launch Gradio ---\ninterface = gr.Interface(\n    fn=predict,\n    inputs=gr.Image(type=\"pil\"),\n    outputs=[\n        gr.Textbox(label=\"Model Predictions\"),\n        gr.Gallery(label=\"Top 5 Retrieved Images\", columns=5, height=\"auto\")\n    ],\n    title=\"Landmark Recognition System with Binary model Classification\",\n    description=\"Upload an image. If it's a landmark, the system predicts using 6 models and shows top 5 similar images from the database.\"\n)\ninterface.launch()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-13T14:26:21.567782Z","iopub.execute_input":"2025-06-13T14:26:21.568189Z","iopub.status.idle":"2025-06-13T14:26:24.098237Z","shell.execute_reply.started":"2025-06-13T14:26:21.568155Z","shell.execute_reply":"2025-06-13T14:26:24.097493Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n# ---- Landmark Non landmark ----\n# === Replace Mobilenet Binary Classifier with Feature Similarity Check ===\nimport numpy as np\nfrom PIL import Image\nimport torchvision.transforms as transforms\nimport torch\nimport torchvision.models as tv_models\n\n# ---- Load the precomputed features ----\nlandmark_features = np.load('/kaggle/input/feature-vector/landmark_features.npy')\nnon_landmark_features = np.load('/kaggle/input/feature-vector/non_landmark_features.npy')\n\n# ---- Define feature extractor ----\nfeature_model = tv_models.resnet50(pretrained=True)\nfeature_model.fc = torch.nn.Identity()\nfeature_model = feature_model.eval().to(device)\n\nfeature_transform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406],\n                         std=[0.229, 0.224, 0.225]),\n])\n\ndef extract_feature_tensor(image_pil):\n    img_tensor = feature_transform(image_pil).unsqueeze(0).to(device)\n    with torch.no_grad():\n        feature = feature_model(img_tensor).squeeze().cpu().numpy()\n        return feature / np.linalg.norm(feature)\n\ndef classify_landmark_similarity(image_pil, threshold=0.7):\n    new_feat = extract_feature_tensor(image_pil)\n    landmark_sims = np.dot(landmark_features, new_feat)\n    non_landmark_sims = np.dot(non_landmark_features, new_feat)\n    max_landmark_sim = landmark_sims.max() if landmark_sims.size > 0 else 0\n    max_non_landmark_sim = non_landmark_sims.max() if non_landmark_sims.size > 0 else 0\n\n    if max_landmark_sim > threshold and max_landmark_sim > max_non_landmark_sim:\n        return \"Landmark\", max_landmark_sim\n    elif max_non_landmark_sim > threshold:\n        return \"Non-Landmark\", max_non_landmark_sim\n    else:\n        return \"Unknown\", max(max_landmark_sim, max_non_landmark_sim)\n\n\ndolg_model = DOLG_ArcFace().to(device).eval()\narcface_head = ArcFace(in_features=512, out_features=102).to(device).eval()\nckpt = torch.load(\"/kaggle/input/resnet-dolg/ResNetDOLG.pth\", map_location=device)\ndolg_model.load_state_dict(ckpt['model_state_dict'])\narcface_head.load_state_dict(ckpt['arcface_state_dict'])\n\n# Models dictionary\nmodels = {\n    \"resnet_v1\": resnet_v1,\n    \"resnet_v2\": resnet_v2,\n    \"efficientnet_v1\": efficientnet_v1,\n    \"efficientnet_v2\": efficientnet_v2\n}\n\n# ---- Transforms ----\ntransform_mobilenet = A.Compose([\n    A.Resize(224, 224),\n    A.Normalize(),\n    ToTensorV2()\n])\n\ntransforms_dict = {\n    \"resnet_v1\": A.Compose([A.Resize(224, 224), A.Normalize(), ToTensorV2()]),\n    \"resnet_v2\": A.Compose([A.Resize(224, 224), A.Normalize(), ToTensorV2()]),\n    \"efficientnet_v1\": A.Compose([A.Resize(260, 260), A.Normalize(), ToTensorV2()]),\n    \"efficientnet_v2\": A.Compose([A.Resize(260, 260), A.Normalize(), ToTensorV2()]),\n    \"dolg\": A.Compose([A.Resize(256, 256), A.CenterCrop(224, 224), A.Normalize([0.5]*3, [0.5]*3), ToTensorV2()])\n}\nimport pandas as pd\n\ncsv_path = \"/kaggle/input/landmark-labels/train_with_landmark_names_fixed.csv\"\ndf = pd.read_csv(csv_path, encoding=\"latin1\")\n\nlandmark_id_to_name = (\n    df.drop_duplicates(\"landmark_id\")[[\"landmark_id\", \"landmark_name\"]]\n    .set_index(\"landmark_id\")[\"landmark_name\"]\n    .to_dict()\n)\n\nimport collections\nimport pickle\nfrom sklearn.metrics.pairwise import cosine_similarity\nfrom scipy.spatial import cKDTree\nfrom skimage.measure import ransac\nfrom skimage.transform import AffineTransform\nimport tensorflow as tf\nimport tensorflow_hub as hub\n\n\n\ndef predict(img: Image.Image):\n    img = img.convert(\"RGB\")\n    img_np = np.array(img)\n\n    final_output = []\n\n    # Step 1: Feature Similarity Check (Landmark vs Non-Landmark)\n    status, sim_score = classify_landmark_similarity(img)\n    final_output.append(f\"[Similarity Check] Status: {status} (Sim score: {sim_score:.2%})\")\n\n    if status != \"Landmark\":\n        return \"\\n\".join(final_output) + \"\\n\\nThis image is NOT a landmark.\", []\n\n    final_output.append(\"\\nThis image IS a landmark. Running all recognition models...\\n\")\n    model_list = list(models.keys()) + [\"dolg\", \"retrieval\"]\n    retrieved_images = []\n\n    for model_name in model_list:\n        debug_log = [f\"--- Result from {model_name.upper()} ---\"]\n\n        try:\n            if model_name == \"retrieval\":\n                query_emb = get_embeddings(img)\n                sim_matrix = cosine_similarity(query_emb, train_embeddings[\"embedded_images\"])\n                top5_idx = np.argsort(sim_matrix[0])[::-1][:5]\n                top_df = pd.DataFrame({\n                    \"image_paths\": [train_embeddings[\"images_paths\"][i] for i in top5_idx],\n                    \"cos_similarity\": [sim_matrix[0][i] for i in top5_idx],\n                    \"prediction\": [train_embeddings[\"labels\"][i] for i in top5_idx],\n                })\n                idx_to_landmark_id = {v: k for k, v in landmark_id_to_idx_4_models.items()}\n                top_df[\"landmark_id\"] = top_df[\"prediction\"].map(idx_to_landmark_id)\n                top_df = delf_rerank(img, top_df)\n                vote = collections.Counter(top_df[\"landmark_id\"])\n                final_landmark_id = vote.most_common(1)[0][0]\n                name = landmark_id_to_name.get(final_landmark_id, f\"Class {final_landmark_id}\")\n                debug_log.append(f\"Prediction: {name} (ID: {final_landmark_id})\")\n\n                for path in top_df[\"image_paths\"]:\n                    try:\n                        pil_img = Image.open(path).convert(\"RGB\")\n                        retrieved_images.append(pil_img)\n                    except Exception as e:\n                        debug_log.append(f\"[WARNING] Failed to load image {path}: {e}\")\n            else:\n                transform = transforms_dict[model_name]\n                img_tensor = transform(image=img_np)[\"image\"].unsqueeze(0).to(device)\n                model = dolg_model if model_name == \"dolg\" else models[model_name]\n                model.eval()\n                with torch.no_grad():\n                    if model_name == \"dolg\":\n                        emb = model(img_tensor)\n                        logits = arcface_head(emb, torch.tensor([0], device=device))\n                        idx_map = landmark_id_to_idx_dolg\n                    else:\n                        logits = model(img_tensor)\n                        idx_map = landmark_id_to_idx_4_models\n                probs = F.softmax(logits, dim=1)\n                pred_idx = probs.argmax(dim=1).item()\n                confidence = probs[0, pred_idx].item()\n                idx_to_landmark_id = {v: k for k, v in idx_map.items()}\n                landmark_id = idx_to_landmark_id.get(pred_idx, pred_idx)\n                name = landmark_id_to_name.get(landmark_id, f\"Class {landmark_id}\")\n                debug_log.append(f\"Prediction: {name} (ID: {landmark_id}) - Confidence: {confidence:.2%}\")\n        except Exception as e:\n            debug_log.append(f\"[ERROR] Failed with {model_name}: {str(e)}\")\n\n        final_output.extend(debug_log)\n        final_output.append(\"\")\n\n    return \"\\n\".join(final_output), retrieved_images\ninterface = gr.Interface(\n    fn=predict,\n    inputs=gr.Image(type=\"pil\"),\n    outputs=[\n        gr.Textbox(label=\"Recognition Results\"),\n        gr.Gallery(label=\"Top 5 Retrieved Images\", columns=5, height=\"auto\")\n    ],\n    title=\"Landmark Recognition System (Feature Similarity + All Models)\",\n    description=\"Upload an image. If it's a landmark, the system uses 6 models to predict and shows top 5 visually similar images.\"\n)\ninterface.launch()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-13T14:28:35.732531Z","iopub.execute_input":"2025-06-13T14:28:35.733364Z","iopub.status.idle":"2025-06-13T14:28:40.679686Z","shell.execute_reply.started":"2025-06-13T14:28:35.733337Z","shell.execute_reply":"2025-06-13T14:28:40.678674Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gradio as gr\nfrom PIL import Image\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport numpy as np\nimport torch\nimport torch.nn.functional as F\nimport pandas as pd\nimport collections\nfrom sklearn.metrics.pairwise import cosine_similarity\nfrom scipy.spatial import cKDTree\nfrom skimage.measure import ransac\nfrom skimage.transform import AffineTransform\nimport tensorflow as tf\nimport tensorflow_hub as hub\n\n# ---- ODIN ----\ndef odin_score(image_tensor, model, temperature=10, epsilon=0.0014):\n    image_tensor = image_tensor.unsqueeze(0).to(device)\n    image_tensor.requires_grad = True\n    model.eval()\n    logits = model(image_tensor)\n    logits = logits / temperature\n    pred_class = logits.argmax(dim=1)\n    loss = F.cross_entropy(logits, pred_class)\n    model.zero_grad()\n    loss.backward()\n    gradient = torch.sign(image_tensor.grad.data)\n    perturbed = image_tensor - epsilon * gradient\n    perturbed = torch.clamp(perturbed, 0, 1)\n    with torch.no_grad():\n        logits_perturbed = model(perturbed) / temperature\n        softmax_scores = F.softmax(logits_perturbed, dim=1)\n        score = torch.max(softmax_scores).item()\n    return score\n\n# ---- Main predict function ----\n# def predict(img: Image.Image, model_name: str):\n#     debug_log = []\n#     img = img.convert(\"RGB\")\n#     img_np = np.array(img)\n#     debug_log.append(f\"[DEBUG] Image shape: {img_np.shape}, dtype: {img_np.dtype}\")\n\n#     if model_name == \"retrieval\":\n#         query_emb = get_embeddings(img)\n#         sim_matrix = cosine_similarity(query_emb, train_embeddings[\"embedded_images\"])\n#         top5_idx = np.argsort(sim_matrix[0])[::-1][:5]\n#         top_df = pd.DataFrame({\n#             \"image_paths\": [train_embeddings[\"images_paths\"][i] for i in top5_idx],\n#             \"cos_similarity\": [sim_matrix[0][i] for i in top5_idx],\n#             \"prediction\": [train_embeddings[\"labels\"][i] for i in top5_idx],\n#         })\n#         idx_to_landmark_id = {v: k for k, v in landmark_id_to_idx_4_models.items()}\n#         top_df[\"landmark_id\"] = top_df[\"prediction\"].map(idx_to_landmark_id)\n#         top_df = delf_rerank(img, top_df)\n#         vote = collections.Counter(top_df[\"landmark_id\"])\n#         final_landmark_id = vote.most_common(1)[0][0]\n#         name = landmark_id_to_name.get(final_landmark_id, f\"Class {final_landmark_id}\")\n#         debug_log.append(f\"[DEBUG] Retrieval top prediction: {final_landmark_id} - {name}\")\n#         return f\"Prediction using retrieval:\\n{name} (ID: {final_landmark_id})\\n\\n\" \n\n#     # Classification\n#     transform = transforms_dict[model_name]\n#     img_tensor = transform(image=img_np)[\"image\"].to(device)\n#     debug_log.append(f\"[DEBUG] Input tensor shape: {img_tensor.shape}\")\n\n#     model = dolg_model if model_name == \"dolg\" else models[model_name]\n#     model.eval()\n\n#     # ODIN score\n#     odin_confidence = odin_score(img_tensor, model, temperature=10, epsilon=0.0014)\n#     debug_log.append(f\"[DEBUG] ODIN score: {odin_confidence:.4f}\")\n\n#     with torch.no_grad():\n#         img_tensor = img_tensor.unsqueeze(0)\n#         if model_name == \"dolg\":\n#             emb = model(img_tensor)\n#             logits = arcface_head(emb, torch.tensor([0], device=device))\n#         else:\n#             logits = model(img_tensor)\n\n#         probs = F.softmax(logits, dim=1)\n#         pred_idx = probs.argmax(dim=1).item()\n#         softmax_confidence = probs[0, pred_idx].item()\n\n#     # Fusion\n#     fusion_score = softmax_confidence * odin_confidence\n#     debug_log.append(f\"[DEBUG] Softmax confidence: {softmax_confidence:.4f}\")\n#     debug_log.append(f\"[DEBUG] Fusion score (Softmax * ODIN): {fusion_score:.4f}\")\n\n#     idx_to_landmark_id = (\n#         {v: k for k, v in landmark_id_to_idx_dolg.items()}\n#         if model_name == \"dolg\" else\n#         {v: k for k, v in landmark_id_to_idx_4_models.items()}\n#     )\n#     landmark_id = idx_to_landmark_id.get(pred_idx, pred_idx)\n#     name = landmark_id_to_name.get(landmark_id, f\"Class {landmark_id}\")\n#     debug_log.append(f\"[DEBUG] Final prediction: {landmark_id} - {name}\")\n\n#     return (\n#         f\"Prediction using {model_name}:\\n\"\n#         f\"{name} (ID: {landmark_id})\\n\"  + \"\\n\".join(debug_log)\n    # )\n\nfrom PIL import Image\nimport os\n\ndef predict(img: Image.Image):\n    img = img.convert(\"RGB\")\n    img_np = np.array(img)\n    final_output = [f\"[INFO] Image shape: {img_np.shape}\"]\n    gallery_images = []\n\n    model_list = list(models.keys()) + [\"dolg\", \"retrieval\"]\n\n    for model_name in model_list:\n        debug_log = [f\"--- Result from {model_name.upper()} ---\"]\n\n        try:\n            if model_name == \"retrieval\":\n                query_emb = get_embeddings(img)\n                sim_matrix = cosine_similarity(query_emb, train_embeddings[\"embedded_images\"])\n                top5_idx = np.argsort(sim_matrix[0])[::-1][:5]\n                top_df = pd.DataFrame({\n                    \"image_paths\": [train_embeddings[\"images_paths\"][i] for i in top5_idx],\n                    \"cos_similarity\": [sim_matrix[0][i] for i in top5_idx],\n                    \"prediction\": [train_embeddings[\"labels\"][i] for i in top5_idx],\n                })\n                idx_to_landmark_id = {v: k for k, v in landmark_id_to_idx_4_models.items()}\n                top_df[\"landmark_id\"] = top_df[\"prediction\"].map(idx_to_landmark_id)\n                top_df = delf_rerank(img, top_df)\n                vote = collections.Counter(top_df[\"landmark_id\"])\n                final_landmark_id = vote.most_common(1)[0][0]\n                name = landmark_id_to_name.get(final_landmark_id, f\"Class {final_landmark_id}\")\n                debug_log.append(f\"Prediction: {name} (ID: {final_landmark_id})\")\n\n                # Load top 5 retrieved images with caption\n                for i, row in top_df.iterrows():\n                    try:\n                        image_path = row[\"image_paths\"]\n                        pil_img = Image.open(image_path).convert(\"RGB\")\n                        caption = f\"ID: {row['landmark_id']} | Sim: {row['cos_similarity']:.3f}\"\n                        gallery_images.append((pil_img, caption))\n                    except Exception as e:\n                        debug_log.append(f\"[WARNING] Failed to load image {image_path}: {e}\")\n            else:\n                transform = transforms_dict[model_name]\n                img_tensor = transform(image=img_np)[\"image\"].to(device)\n\n                model = dolg_model if model_name == \"dolg\" else models[model_name]\n                model.eval()\n\n                # ODIN score\n                odin_confidence = odin_score(img_tensor, model, temperature=10, epsilon=0.0014)\n                debug_log.append(f\"ODIN confidence: {odin_confidence:.4f}\")\n\n                with torch.no_grad():\n                    img_tensor = img_tensor.unsqueeze(0)\n                    if model_name == \"dolg\":\n                        emb = model(img_tensor)\n                        logits = arcface_head(emb, torch.tensor([0], device=device))\n                        idx_map = landmark_id_to_idx_dolg\n                    else:\n                        logits = model(img_tensor)\n                        idx_map = landmark_id_to_idx_4_models\n\n                    probs = F.softmax(logits, dim=1)\n                    pred_idx = probs.argmax(dim=1).item()\n                    softmax_confidence = probs[0, pred_idx].item()\n                    fusion_score = softmax_confidence * odin_confidence\n\n                    idx_to_landmark_id = {v: k for k, v in idx_map.items()}\n                    landmark_id = idx_to_landmark_id.get(pred_idx, pred_idx)\n                    name = landmark_id_to_name.get(landmark_id, f\"Class {landmark_id}\")\n\n                    debug_log.append(f\"Softmax confidence: {softmax_confidence:.4f}\")\n                    debug_log.append(f\"Fusion score: {fusion_score:.4f}\")\n                    debug_log.append(f\"Prediction: {name} (ID: {landmark_id})\")\n        except Exception as e:\n            debug_log.append(f\"[ERROR] Failed on model {model_name}: {str(e)}\")\n\n        final_output.extend(debug_log)\n        final_output.append(\"\")\n\n    return \"\\n\".join(final_output), gallery_images\n# Launch\ninterface = gr.Interface(\n    fn=predict,\n    inputs=gr.Image(type=\"pil\"),\n    outputs=[\n        gr.Textbox(label=\"Recognition Results\"),\n        gr.Gallery(label=\"Top 5 Retrieved Images\", columns=5, height=\"auto\")\n    ],\n    title=\"Landmark Recognition with ODIN and Retrieval Gallery\",\n    description=\"Upload an image. The system predicts using 6 models. ODIN is used for OOD detection. The retrieval model shows the top 5 most similar images.\"\n)\ninterface.launch()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-13T14:30:53.722637Z","iopub.execute_input":"2025-06-13T14:30:53.723203Z","iopub.status.idle":"2025-06-13T14:30:55.180473Z","shell.execute_reply.started":"2025-06-13T14:30:53.723179Z","shell.execute_reply":"2025-06-13T14:30:55.179766Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class DOLG_ArcFace(nn.Module):\n    def __init__(self, embedding_dim=512):\n        super().__init__()\n        resnet = models.resnet50(pretrained=True)\n\n        # Shared layers\n        self.backbone_common = nn.Sequential(\n            resnet.conv1, resnet.bn1, resnet.relu,\n            resnet.maxpool, resnet.layer1,\n            resnet.layer2, resnet.layer3\n        )\n\n        # ResNet layer4 expects input with 1024 channels (not from local_conv!)\n        self.backbone_global = resnet.layer4\n\n        self.global_pool = nn.AdaptiveAvgPool2d((1, 1))\n        self.global_fc = nn.Linear(2048, embedding_dim)\n\n        self.local_conv = nn.Conv2d(1024, embedding_dim, kernel_size=1)\n\n        self.fusion = AttentionFusion(embedding_dim)\n\n        self.head = nn.Sequential(\n            nn.Conv2d(embedding_dim, embedding_dim, kernel_size=3, padding=1),\n            nn.ReLU(),\n            nn.AdaptiveAvgPool2d(1),\n            nn.Flatten()\n        )\n\n    def forward(self, x):\n        shared_feat = self.backbone_common(x)  # Output: [B, 1024, H, W]\n\n        # Global branch\n        global_feat_map = self.backbone_global(shared_feat)  # Output: [B, 2048, H/2, W/2]\n        global_feat = self.global_pool(global_feat_map).view(x.size(0), -1)  # [B, 2048]\n        global_feat = self.global_fc(global_feat)  # [B, 512]\n\n        # Local branch\n        local_feat = self.local_conv(shared_feat)  # [B, 512, H, W]\n\n        # Fuse\n        fused_feat = self.fusion(local_feat, global_feat)  # [B, 512, H, W]\n        emb = self.head(fused_feat)  # [B, 512]\n        return emb","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nfrom torchvision import transforms\nfrom PIL import Image\nimport torch.nn.functional as F\n\ndef predict_image(image_path, model, arcface_head, transform, device, idx_to_landmark_id=None, id_to_name=None):\n    model.eval()\n    arcface_head.eval()\n\n    image = Image.open(image_path).convert(\"RGB\")\n    input_tensor = transform(image).unsqueeze(0).to(device)\n\n    label_tensor = torch.tensor([0], device=device)\n\n    # Forward pass\n    with torch.no_grad():\n        embedding = model(input_tensor)\n        logits = arcface_head(embedding, label_tensor)\n        probs = F.softmax(logits, dim=1)\n        conf, pred = torch.max(probs, dim=1)\n\n    pred_idx = pred.item()\n    confidence = conf.item()\n\n    if idx_to_landmark_id:\n        pred_landmark_id = idx_to_landmark_id.get(pred_idx, \"Unknown ID\")\n    else:\n        pred_landmark_id = pred_idx\n\n    if id_to_name:\n        pred_name = id_to_name.get(pred_landmark_id, \"Unknown Landmark\")\n    else:\n        pred_name = f\"Class {pred_idx}\"\n\n    print(f\"Predicted: {pred_name} (ID: {pred_landmark_id}) | Confidence: {confidence:.2f}\")\n    return pred_idx, confidence","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ---- Setup ----\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# Load model\nNUM_CLASSES = 102\nmodel = DOLG_ArcFace(embedding_dim=512).to(device)\narcface_head = ArcFace(in_features=512, out_features=NUM_CLASSES).to(device)  # Replace NUM_CLASSES\n\n# Load checkpoint\ncheckpoint = torch.load(\"/kaggle/input/resnet-dolg/ResNetDOLG.pth\", map_location=device)\nmodel.load_state_dict(checkpoint['model_state_dict'])\narcface_head.load_state_dict(checkpoint['arcface_state_dict'])\n\n# Image transform (must match training)\ntransform = transforms.Compose([\n    transforms.Resize((256, 256)),\n    transforms.CenterCrop(224),\n    transforms.ToTensor(),\n    transforms.Normalize([0.5]*3, [0.5]*3)\n])\n\n# Optional mappings\nidx_to_landmark_id = {v: k for k, v in landmark_id_to_idx.items()}\nlandmark_id_to_name = {\n    27: \"Isa Khan Niyazi's tomb\",\n    # ... fill from your metadata\n}\n\n# ---- Predict ----\nimage_path = \"/kaggle/input/test-landmark-images/golden_gate_foggy.jpg\"\npredict_image(\n    image_path,\n    model,\n    arcface_head,\n    transform,\n    device,\n    idx_to_landmark_id=idx_to_landmark_id,\n    id_to_name=landmark_id_to_name\n)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gradio as gr\nfrom PIL import Image\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torchvision.models as tv_models\n#from model import DOLG_ArcFace, ArcFace  # Make sure this file exists\n\n# ============ Device ============\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# ============ Pretrained Classification Models ============\nmodel_dict = {\n    \"resnet18\": tv_models.resnet18(pretrained=True).to(device),\n    \"resnet50\": tv_models.resnet50(pretrained=True).to(device),\n    \"efficientnet_b0\": tv_models.efficientnet_b0(pretrained=True).to(device),\n    \"efficientnet_b3\": tv_models.efficientnet_b3(pretrained=True).to(device)\n}\n\n# ============ Add DOLG Model ============\nembedding_dim = 512\nnum_classes = 102  # set this according to your dataset\n\ndolg_model = DOLG_ArcFace(embedding_dim=embedding_dim).to(device)\narcface_head = ArcFace(in_features=embedding_dim, out_features=num_classes).to(device)\n\n# Load weights\ncheckpoint = torch.load(\"/kaggle/input/resnet-dolg/kaggle/working/best_model.pth\", map_location=device)\ndolg_model.load_state_dict(checkpoint[\"model_state_dict\"])\narcface_head.load_state_dict(checkpoint[\"arcface_state_dict\"])\ndolg_model.eval()\narcface_head.eval()\n\nmodel_dict[\"resnet_dolg\"] = (dolg_model, arcface_head)  # special handling for this one\n\n# ============ Load Binary MobileNet ============\nmobilenet_binary = tv_models.mobilenet_v2(pretrained=False)\nmobilenet_binary.classifier[1] = nn.Linear(mobilenet_binary.last_channel, 1)\nmobilenet_binary.load_state_dict(torch.load(\"/kaggle/input/mobilenet-landmark-binary-classification/mobilenet_landmark.pth\", map_location=device))\nmobilenet_binary.to(device)\nmobilenet_binary.eval()\n\n# ============ Transforms ============\ntransform_efficientnet = A.Compose([\n    A.Resize(260, 260),\n    A.Normalize(),\n    ToTensorV2()\n])\n\ntransform_resnet = A.Compose([\n    A.Resize(224, 224),\n    A.Normalize(),\n    ToTensorV2()\n])\n\ntransform_mobilenet = A.Compose([\n    A.Resize(224, 224),\n    A.Normalize(),\n    ToTensorV2()\n])\n\n# ============ Inference Function ============\ndef predict(img: Image.Image, model_name: str):\n    img_np = np.array(img)\n\n    # Step 1: Run binary classifier\n    binary_transformed = transform_mobilenet(image=img_np)\n    binary_tensor = binary_transformed[\"image\"].unsqueeze(0).to(device)\n\n    with torch.no_grad():\n        binary_output = mobilenet_binary(binary_tensor)\n        prob = torch.sigmoid(binary_output).item()\n        is_landmark = prob >= 0.5\n\n    if not is_landmark:\n        return f\"The image does not contain a landmark. (Confidence: {prob:.2f})\"\n\n    # Step 2: Classify landmark\n    transform = transform_efficientnet if model_name.startswith(\"efficientnet\") else transform_resnet\n    transformed = transform(image=img_np)\n    img_tensor = transformed[\"image\"].unsqueeze(0).to(device)\n\n    # Handle ResNet + DOLG differently\n    if model_name == \"resnet_dolg\":\n        dolg_model, arcface_head = model_dict[\"resnet_dolg\"]\n        with torch.no_grad():\n            embedding = dolg_model(img_tensor)\n            logits = arcface_head(embedding, torch.zeros(1, dtype=torch.long).to(device))  # dummy label\n            prob_class = F.softmax(logits, dim=1)\n            conf, pred = torch.max(prob_class, dim=1)\n        return f\"Landmark detected (Confidence: {prob:.2f})\\nPredicted class (ResNet+DOLG): {pred.item()} (Conf: {conf.item():.2f})\"\n    \n    # Standard models\n    model = model_dict[model_name]\n    model.eval()\n    with torch.no_grad():\n        output = model(img_tensor)\n        pred = output.argmax(dim=1).item()\n        idx_to_landmark_id = {v: k for k, v in landmark_id_to_idx.items()}\n        pred_landmark = idx_to_landmark_id[pred]\n\n    return f\"Landmark detected (Confidence: {prob:.2f})\\nPredicted class ({model_name}): {pred_landmark}\"\n\n# ============ Gradio Interface ============\ninterface = gr.Interface(\n    fn=predict,\n    inputs=[\n        gr.Image(type=\"pil\"),\n        gr.Dropdown(choices=list(model_dict.keys()), label=\"Choose Model\")\n    ],\n    outputs=\"text\",\n    title=\"Landmark Classification Demo (ResNet + DOLG + ArcFace Integrated)\"\n)\n\ninterface.launch()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T16:57:56.452412Z","iopub.execute_input":"2025-05-14T16:57:56.452723Z","iopub.status.idle":"2025-05-14T16:58:01.875821Z","shell.execute_reply.started":"2025-05-14T16:57:56.452684Z","shell.execute_reply":"2025-05-14T16:58:01.874587Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**DELF**","metadata":{}},{"cell_type":"code","source":"def get_image(path, resize = False, reshape = False, target_size = None):\n    img = cv2.imread(path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    if resize:\n        img = cv2.resize(img, dsize = (target_size, target_size))\n    if reshape:\n        img = tf.reshape(img, [1, target_size, target_size, 3])\n    return img","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T00:45:42.071115Z","iopub.execute_input":"2025-05-23T00:45:42.071406Z","iopub.status.idle":"2025-05-23T00:45:42.076325Z","shell.execute_reply.started":"2025-05-23T00:45:42.071384Z","shell.execute_reply":"2025-05-23T00:45:42.075787Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_query_image(img, resize=False, reshape=False, target_size=None):\n    # Nếu là PIL Image, chuyển sang numpy\n    if not isinstance(img, np.ndarray):\n        img = np.array(img)\n    \n    # Đảm bảo ảnh ở dạng RGB\n    if img.shape[-1] == 4:  # RGBA\n        img = img[:, :, :3]\n    \n    if resize:\n        img = cv2.resize(img, dsize=(target_size, target_size))\n    \n    if reshape:\n        img = tf.reshape(img, [1, target_size, target_size, 3])\n    \n    return img","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T00:45:43.817812Z","iopub.execute_input":"2025-05-23T00:45:43.818504Z","iopub.status.idle":"2025-05-23T00:45:43.823922Z","shell.execute_reply.started":"2025-05-23T00:45:43.818474Z","shell.execute_reply":"2025-05-23T00:45:43.823152Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_embeddings(model, image_paths, input_size, as_df = True):\n    embeddings = {}\n    embeddings['images_paths'] = []\n    embeddings['embedded_images'] = []\n    \n    target_dir = os.path.split(os.path.split(image_paths[0])[0])[0]\n    \n    print(f\"Retrieving embeddings for {target_dir} with {model.name}...\")\n    for image_path in tqdm(image_paths):\n        embeddings['images_paths'].append(image_path)\n        embedded_image = model.predict(get_image(image_path,\n                                                 resize = True,\n                                                 reshape = True,\n                                                 target_size = input_size))\n        embeddings['embedded_images'].append(embedded_image)\n    \n    if as_df:\n        embeddings = pd.DataFrame(embeddings)\n    \n    return embeddings","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T00:45:45.903232Z","iopub.execute_input":"2025-05-23T00:45:45.903988Z","iopub.status.idle":"2025-05-23T00:45:45.909286Z","shell.execute_reply.started":"2025-05-23T00:45:45.903961Z","shell.execute_reply":"2025-05-23T00:45:45.908547Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_query_embedding(model, images, input_size):\n    embeddings = []\n    for img in images:\n        processed_img = get_query_image(img, resize=True, reshape=True, target_size=input_size)\n        embedded = model.predict(processed_img)\n        embeddings.append(embedded)\n    return embeddings","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T00:45:48.145874Z","iopub.execute_input":"2025-05-23T00:45:48.146684Z","iopub.status.idle":"2025-05-23T00:45:48.151078Z","shell.execute_reply.started":"2025-05-23T00:45:48.146656Z","shell.execute_reply":"2025-05-23T00:45:48.150239Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get similarities between query key pair\ndef get_similarities(query, key):\n    '''\n    Get cosine similarity matrix between query and key pairs\n    Arguments:\n    query, key: embedded images\n    '''\n    query_array = np.stack(query.tolist()).reshape(query.shape[0],\n                                                   query[0].shape[1])\n    key_array = np.stack(key.tolist()).reshape(key.shape[0],\n                                               key[0].shape[1])\n    \n    # Initializing similarity matrix\n    similarity = np.zeros((query_array.shape[0], key_array.shape[0]))\n    \n    # Getting pairwise similarities\n    print(f\"Getting pairwise {query_array.shape[0]} query: {key_array.shape[0]} key similarities...\")\n    for query_index in tqdm(range(query_array.shape[0])):\n        similarity[query_index] = 1 - spatial.distance.cdist(query_array[np.newaxis, query_index, :],\n                                                             key_array,\n                                                             'cosine')[0]\n    return similarity","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T00:45:49.814509Z","iopub.execute_input":"2025-05-23T00:45:49.814798Z","iopub.status.idle":"2025-05-23T00:45:49.820719Z","shell.execute_reply.started":"2025-05-23T00:45:49.814776Z","shell.execute_reply":"2025-05-23T00:45:49.820119Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom tqdm.notebook import tqdm\ntrain_img_paths = []\ntrain_df = pd.read_csv(\"/kaggle/input/landmark/train.csv\", header=0)\n\n\nfor i, (idx, row) in enumerate(tqdm(train_df.iterrows(), total=len(train_df))):\n    image_id = row['id']\n    # Extract subfolders based on first 3 characters\n    subfolder = f\"{image_id[0]}/{image_id[1]}/{image_id[2]}\"\n    image_path = f\"/kaggle/input/landmark-recognition-2021/train/{subfolder}/{image_id}.jpg\"\n\n    # Chỉ thêm nếu file tồn tại\n    if os.path.exists(image_path):\n        train_img_paths.append(image_path)\n    else:\n        print(f\"❌ File not found: {image_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T00:46:43.360345Z","iopub.execute_input":"2025-05-23T00:46:43.361194Z","iopub.status.idle":"2025-05-23T00:51:46.753114Z","shell.execute_reply.started":"2025-05-23T00:46:43.361165Z","shell.execute_reply":"2025-05-23T00:51:46.752418Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -Uq tensorflow\nimport os; os._exit(00)  # khởi động lại kernel","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T01:03:18.833670Z","iopub.execute_input":"2025-05-23T01:03:18.834280Z","execution_failed":"2025-05-23T01:04:48.556Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.models import load_model\nembedding_model = load_model(\"/kaggle/input/delf-model/efficientnet_embedding_model-2.h5\", compile=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T01:05:06.563359Z","iopub.execute_input":"2025-05-23T01:05:06.564094Z","iopub.status.idle":"2025-05-23T01:05:11.046149Z","shell.execute_reply.started":"2025-05-23T01:05:06.564066Z","shell.execute_reply":"2025-05-23T01:05:11.045002Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.models import Model\nimport tensorflow as tf\n\nIMG_SIZE = 224\nNUM_CLASSES = 100  # <-- bạn cần điền đúng số class khi training\n\n# Rebuild model\nbase_model = EfficientNetB0(include_top=False, input_shape=(IMG_SIZE, IMG_SIZE, 3), weights='imagenet', pooling='avg')\nx = tf.keras.layers.Dense(512, activation='relu', name='embedding_512')(base_model.output)\noutput = tf.keras.layers.Dense(NUM_CLASSES, activation='softmax')(x)\nmodel = Model(inputs=base_model.input, outputs=output)\n\n# Embedding model\nembedding_model = tf.keras.Model(\n    inputs=model.input,\n    outputs=model.get_layer('embedding_512').output,\n    name=\"EfficientNetB0_embed512\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T00:44:24.405557Z","iopub.execute_input":"2025-05-23T00:44:24.406121Z","iopub.status.idle":"2025-05-23T00:44:32.462423Z","shell.execute_reply.started":"2025-05-23T00:44:24.406095Z","shell.execute_reply":"2025-05-23T00:44:32.461865Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nIMG_SIZE = 224\ntrain_embeddings = get_embeddings(model = embedding_model,\n                                 image_paths = train_img_paths,\n                                 input_size = IMG_SIZE)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculating confidence score per submission\ndef confidence_top(query = None, key = None, similarity = None, query_image_index = None, top = 5):\n    '''\n    Arguments:\n    query_image_index = index of query image on similarity matrix query axis\n    Return confidence scores for top N predictions\n    '''\n    query_paths = query['images_paths']\n    key_paths = key['images_paths']\n    \n    similar_n = np.argsort(similarity[query_image_index])[::-1][:top]\n    \n    confidence_df = {}    \n    confidence_df['top_similar'] = []\n    for similar in similar_n:\n        confidence_df['top_similar'].append(similar)\n\n    confidence_df['image_paths'] = []\n    for similar in similar_n:\n        similar_image_path = key_paths[similar]\n        confidence_df['image_paths'].append(similar_image_path)    \n        \n    confidence_df['prediction'] = []\n    for similar in similar_n:\n        similar_image_path = key_paths[similar]\n        y = int(os.path.split(os.path.split(similar_image_path)[0])[1])\n        idx_to_landmark_id = {v: k for k, v in landmark_id_to_idx.items()}\n        pred_landmark = idx_to_landmark_id[y]\n        confidence_df['prediction'].append(pred_landmark)  \n    \n    confidence_df['cos_similarity'] = []\n    for similar in similar_n:\n        confidence_df['cos_similarity'].append(similarity[query_image_index][similar]) \n    \n    return pd.DataFrame(confidence_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T03:21:38.990084Z","iopub.execute_input":"2025-05-22T03:21:38.990947Z","iopub.status.idle":"2025-05-22T03:21:38.998016Z","shell.execute_reply.started":"2025-05-22T03:21:38.990918Z","shell.execute_reply":"2025-05-22T03:21:38.997208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from absl import logging\nfrom PIL import Image, ImageOps\nfrom scipy.spatial import cKDTree\nfrom skimage.util import plot_matches\nfrom skimage.measure import ransac\nfrom skimage.transform import AffineTransform\nfrom six import BytesIO\n\nimport tensorflow_hub as hub\nfrom six.moves.urllib.request import urlopen\ndelf = hub.load('https://tfhub.dev/google/delf/1').signatures['default']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T03:22:52.329606Z","iopub.execute_input":"2025-05-22T03:22:52.330479Z","iopub.status.idle":"2025-05-22T03:22:52.382665Z","shell.execute_reply.started":"2025-05-22T03:22:52.330449Z","shell.execute_reply":"2025-05-22T03:22:52.381633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# DELF module\ndef run_delf(image):\n    '''\n    Apply DELF module to the input image\n    Arguments:\n    image: np.array resized image\n    '''\n    float_image = tf.image.convert_image_dtype(image, tf.float32)\n\n    return delf(\n      image = float_image,\n      score_threshold = tf.constant(100.0),\n      image_scales = tf.constant([0.25, 0.3536, 0.5, 0.7071, 1.0, 1.4142, 2.0]),\n      max_feature_num = tf.constant(1000))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DELF_IMG_SIZE = 600\ndef delf_rerank(img,query = None, key = None, query_image_index = None, confidence_df = None, re_sort = True):\n    distance_threshold = 0.8\n    query_paths = query['images_paths']\n    key_paths = key['images_paths']\n    \n    query_image = get_query_image(img,\n                            resize = True,\n                            target_size = DELF_IMG_SIZE)\n    \n    delf_result_query = run_delf(query_image)\n    \n    # Read query features\n    num_features_query = delf_result_query['locations'].shape[0]\n    \n    inliers_list = []\n    print(f\"Retrieving local features for top {len(confidence_df['image_paths'])} key images...\")\n    for image_path in tqdm(confidence_df['image_paths']):\n        key_image = get_image(image_path,\n                          resize = True,\n                          target_size = DELF_IMG_SIZE)\n        \n        delf_result_key = run_delf(key_image)\n    \n        # Read key features\n        num_features_key = delf_result_key['locations'].shape[0]\n\n        # Find nearest-neighbor matches using a KD tree.\n        d1_tree = cKDTree(delf_result_query['descriptors'])\n        _, indices = d1_tree.query(\n          delf_result_key['descriptors'],\n          distance_upper_bound=distance_threshold)\n\n        # Select feature locations for putative matches.\n        locations_k_to_use = np.array([\n          delf_result_key['locations'][i,]\n          for i in range(num_features_key)\n          if indices[i] != num_features_query\n        ])\n        locations_q_to_use = np.array([\n          delf_result_query['locations'][indices[i],]\n          for i in range(num_features_key)\n          if indices[i] != num_features_query\n        ])\n\n        # Perform geometric verification using RANSAC.\n        try:\n            _, inliers = ransac(\n              (locations_q_to_use, locations_k_to_use),\n              AffineTransform,\n              min_samples=3,\n              residual_threshold=20,\n              max_trials=1000)\n        except:\n            inliers = [0]\n        \n        # Handling 0 inliers\n        try:\n            total_inliers = sum(inliers)\n            inliers_list.append(total_inliers)\n        except:\n            inliers_list.append(1) # Appending inlier = 1 to avoid null confidence\n    \n    confidence_df['inliers'] = inliers_list\n    \n    original_confidence = confidence_df['inliers']\n    reranked_confidence = np.sqrt(original_confidence) * confidence_df['cos_similarity']\n    confidence_df['reranked_conf'] = reranked_confidence\n    \n    if re_sort:\n        confidence_df.sort_values('reranked_conf', ascending = False, inplace = True)\n    \n    return confidence_df","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom io import BytesIO\nimport PIL.Image as Image\n\ndef recognize_and_visualize_gradio(img, \n                                   train_embeddings, \n                                   model, \n                                   top_n=5, \n                                   use_rerank=True):\n    # Step 1: Get query embedding\n    query_embedding = get_query_embeddings(model=model,\n                                           img=img,\n                                           input_size=IMG_SIZE)\n\n    # Step 2: Cosine similarity\n    sim_matrix = get_similarities(query_embedding['embedded_images'],\n                                   train_embeddings['embedded_images'])\n\n    # Step 3: Top-N predictions\n    confidence_df = confidence_top(query=query_embedding,\n                                   key=train_embeddings,\n                                   similarity=sim_matrix,\n                                   query_image_index=0,\n                                   top=top_n)\n\n    # Step 4: Reranking (nếu cần)\n    if use_rerank:\n        confidence_df = delf_rerank(query=query_embedding,\n                                    key=train_embeddings,\n                                    query_image_index=0,\n                                    confidence_df=confidence_df,\n                                    re_sort=True)\n\n    # Step 5: Vẽ ảnh truy vấn\n    query_image = get_query_image(img, resize=True, target_size=DELF_IMG_SIZE)\n    fig1 = plt.figure(figsize=(4, 4))\n    plt.imshow(query_image)\n    plt.title(\"Query Image\")\n    plt.axis(\"off\")\n    \n    # Lưu ra buffer\n    buf1 = BytesIO()\n    fig1.savefig(buf1, format=\"png\", bbox_inches='tight')\n    plt.close(fig1)\n    buf1.seek(0)\n    query_image_pil = Image.open(buf1)\n\n    # Step 6: Vẽ ảnh tương tự\n    fig2, axs = plt.subplots(1, top_n, figsize=(15, 5))\n    for i in range(top_n):\n        img_path = confidence_df['image_paths'].iloc[i]\n        similar_img = get_image(img_path, resize=True, target_size=DELF_IMG_SIZE)\n        axs[i].imshow(similar_img)\n        axs[i].set_title(f\"ID: {confidence_df['prediction'].iloc[i]}\\nSim: {confidence_df['cos_similarity'].iloc[i]:.2f}\")\n        axs[i].axis(\"off\")\n\n    buf2 = BytesIO()\n    fig2.savefig(buf2, format=\"png\", bbox_inches='tight')\n    plt.close(fig2)\n    buf2.seek(0)\n    similar_images_pil = Image.open(buf2)\n\n    # Step 7: Trả kết quả\n    predicted_landmark = confidence_df['prediction'].iloc[0]\n    return predicted_landmark, query_image_pil, similar_images_pil","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"interface = gr.Interface(\n    fn=lambda img: recognize_and_visualize_gradio(img, train_embeddings, model, top_n=5),\n    inputs=gr.Image(type=\"pil\", label=\"Upload Query Image\"),\n    outputs=[\n        gr.Label(label=\"Predicted Landmark ID\"),\n        gr.Image(label=\"Query Image\"),\n        gr.Image(label=\"Top Similar Images\")\n    ],\n    title=\"🔍 Landmark Recognition\",\n    description=\"Upload an image to find its most similar landmarks\"\n)\n\ninterface.launch()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}