{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.10","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":29762,"databundleVersionId":2541532,"sourceType":"competition"},{"sourceId":2644751,"sourceType":"datasetVersion","datasetId":1596464}],"dockerImageVersionId":30124,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Import libraries and some directories ##","metadata":{}},{"cell_type":"code","source":"import pathlib\n\nimport torch\nimport torch.utils.data\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport numpy as np\nimport pandas as pd\n\nimport PIL.Image\nimport albumentations.pytorch\nimport cv2\nimport matplotlib.pyplot as plt\n\nfrom tqdm.notebook import tqdm\nfrom typing import List, Tuple\n\nIMAGE_SIZE = 320  \nBATCH_SIZE = 512\n\nMODEL_FILE = pathlib.Path('../input/google-landmark-2021-validation/model.pth')\nTRAIN_LABEL_FILE = pathlib.Path('train.csv')\nTRAIN_IMAGE_DIR = pathlib.Path('../input/landmark-recognition-2021/train')\nVALID_LABEL_FILE = pathlib.Path('valid.csv')\nVALID_IMAGE_DIR = pathlib.Path('../input/google-landmark-2021-validation/valid')\nTEST_LABEL_FILE = pathlib.Path('../input/landmark-recognition-2021/sample_submission.csv')\nTEST_IMAGE_DIR = pathlib.Path('../input/landmark-recognition-2021/test')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T01:31:01.266244Z","iopub.execute_input":"2025-05-06T01:31:01.266924Z","iopub.status.idle":"2025-05-06T01:31:01.272654Z","shell.execute_reply.started":"2025-05-06T01:31:01.266889Z","shell.execute_reply":"2025-05-06T01:31:01.271912Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Elimination of public training images\n(This code is updated at 2021/09/24 5:00AM GMT)\n\nIn order to reduce the processing time, only a subset of public training images are used for the feature extraction at saving the code.\nAt the submission, all private trainig images are used.","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/landmark-recognition-2021/train.csv')\n\nif len(train_df) == 1580470:\n    records = {}\n\n    for image_id, landmark_id in train_df.values:\n        if landmark_id in records:\n            records[landmark_id].append(image_id)\n        else:\n            records[landmark_id] = [image_id]\n        \n    image_ids = []\n    landmark_ids = []\n\n    for landmark_id, img_ids in records.items():\n        num = min(len(img_ids), 2)\n        image_ids.extend(records[landmark_id][:num])\n        landmark_ids.extend([landmark_id] * num)\n\n    train_df = pd.DataFrame({'id': image_ids, 'landmark_id': landmark_ids})\n\ntrain_df.to_csv(TRAIN_LABEL_FILE, index=False)\ntrain_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T01:21:43.058776Z","iopub.execute_input":"2025-05-06T01:21:43.059305Z","iopub.status.idle":"2025-05-06T01:21:46.898006Z","shell.execute_reply.started":"2025-05-06T01:21:43.059259Z","shell.execute_reply":"2025-05-06T01:21:46.897317Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### List of non-landmark images","metadata":{}},{"cell_type":"code","source":"valid_df = pd.read_csv('/kaggle/input/google-landmark-2021-validation/valid.csv')\nvalid_df = valid_df[valid_df['landmark_id'] == -1]\nvalid_df.to_csv(VALID_LABEL_FILE, index=False)\nvalid_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T01:21:46.899106Z","iopub.execute_input":"2025-05-06T01:21:46.899628Z","iopub.status.idle":"2025-05-06T01:21:47.156934Z","shell.execute_reply.started":"2025-05-06T01:21:46.899592Z","shell.execute_reply":"2025-05-06T01:21:47.156247Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Class and Functions for feature extraction","metadata":{}},{"cell_type":"code","source":"class Dataset(torch.utils.data.Dataset):\n    def __init__(self, label_file: pathlib.Path, image_dir: pathlib.Path) -> None:\n        super().__init__()\n        self.files = [\n            image_dir / n[0] / n[1] / n[2] / f'{n}.jpg'\n            for n in pd.read_csv(label_file)['id'].values]\n        \n        self.transformer = albumentations.Compose([\n            albumentations.SmallestMaxSize(IMAGE_SIZE, interpolation=cv2.INTER_CUBIC),\n            albumentations.CenterCrop(IMAGE_SIZE, IMAGE_SIZE),\n            albumentations.Normalize(),\n            albumentations.pytorch.ToTensorV2(),\n        ])\n\n    def __len__(self) -> int:\n        return len(self.files)\n\n    def __getitem__(self, index: int) -> Tuple[str, torch.Tensor]:\n        path = self.files[index]\n        image = PIL.Image.open(self.files[index])\n        image = self.transformer(image=np.array(image))['image']\n\n        return path.name[:-4], image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T01:21:47.158051Z","iopub.execute_input":"2025-05-06T01:21:47.158707Z","iopub.status.idle":"2025-05-06T01:21:47.165558Z","shell.execute_reply.started":"2025-05-06T01:21:47.15867Z","shell.execute_reply":"2025-05-06T01:21:47.16483Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"@torch.no_grad()\ndef get_features(\n    model: nn.Module,\n    label_file: pathlib.Path,\n    image_dir: pathlib.Path,\n) -> Tuple[List[str], torch.Tensor]:\n    loader = torch.utils.data.DataLoader(\n        Dataset(label_file, image_dir),\n        batch_size=BATCH_SIZE, shuffle=False, num_workers=2)\n\n    model = model.cuda()\n    model.eval()\n    \n    all_names = []\n    all_features = []\n\n    for names, images in tqdm(loader, desc=image_dir.name):\n        images = images.cuda()\n        features = model(images)\n        all_features.append(features)\n        all_names.extend(names)\n\n    return all_names, F.normalize(torch.cat(all_features, dim=0))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T01:21:47.166611Z","iopub.execute_input":"2025-05-06T01:21:47.16686Z","iopub.status.idle":"2025-05-06T01:21:47.180437Z","shell.execute_reply.started":"2025-05-06T01:21:47.166826Z","shell.execute_reply":"2025-05-06T01:21:47.179855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_similarity(model: nn.Module)-> Tuple[List[str], List[str]]:\n    # features\n    train_names, train_features = get_features(\n        model, TRAIN_LABEL_FILE, TRAIN_IMAGE_DIR)    \n    _, valid_features = get_features(\n        model, VALID_LABEL_FILE, VALID_IMAGE_DIR)\n    test_names, test_features = get_features(\n        model, TEST_LABEL_FILE, TEST_IMAGE_DIR)\n\n    # penalties\n    train_penalties_list = []\n    for i in range(0, train_features.shape[0], 128):\n        x = torch.mm(train_features[i:i + 128], valid_features.T)\n        x = torch.topk(x, k=5)[0].mean(dim=1)\n        train_penalties_list.append(x)\n    train_penalties = torch.cat(train_penalties_list, dim=0)\n\n    test_penalties_list = []\n    for i in range(0, test_features.shape[0], 128):\n        x = torch.mm(test_features[i:i + 128], valid_features.T)\n        x = torch.topk(x, k=10)[0].mean(dim=1)\n        test_penalties_list.append(x)\n    test_penalties = torch.cat(test_penalties_list, dim=0)\n\n    # neighbors\n    submit_ids = []\n    submit_landmark_ids = []\n    submit_confidences = []\n    \n    train_df = pd.read_csv(TRAIN_LABEL_FILE)\n    idmap = {n: v for n, v in train_df.values}\n\n    for i in range(0, test_features.shape[0], 128):\n        x = torch.mm(test_features[i:i + 128], train_features.T)\n        x -= train_penalties[None, :]\n        values, indexes = torch.topk(x, k=3)\n        \n        submit_ids.extend(test_names[i:i + 128])\n\n        for idxs, vals, penalty in zip(indexes, values, test_penalties[i:i + 128]):\n            scores = {}\n            for idx, val in zip(idxs, vals):\n                landmark_id = idmap[train_names[idx]]\n                if landmark_id in scores:\n                    scores[landmark_id] += float(val)\n                else:\n                    scores[landmark_id] = float(val)\n                    \n            landmark_id, confidence = max(\n                [(k, v) for k, v in scores.items()], key=lambda x: x[1])\n            submit_landmark_ids.append(landmark_id)\n            submit_confidences.append(confidence - penalty)\n\n    # standardize confidence values\n    max_conf = max(submit_confidences)\n    min_conf = min(submit_confidences)\n    submit_confidences = [\n        (v - min_conf) / (max_conf - min_conf) for v in submit_confidences]\n    \n    # make values for 'landmark' column\n    submit_landmarks = [\n        f'{i} {c:.8f}' for i, c in zip(submit_landmark_ids, submit_confidences)]\n    \n    return submit_ids, submit_landmarks","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T01:21:47.181614Z","iopub.execute_input":"2025-05-06T01:21:47.181901Z","iopub.status.idle":"2025-05-06T01:21:47.204004Z","shell.execute_reply.started":"2025-05-06T01:21:47.18187Z","shell.execute_reply":"2025-05-06T01:21:47.203333Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Inference and Submission","metadata":{}},{"cell_type":"code","source":"from torchvision.models import resnet50\nimport torch.nn as nn","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T01:21:47.205831Z","iopub.execute_input":"2025-05-06T01:21:47.206021Z","iopub.status.idle":"2025-05-06T01:21:47.218134Z","shell.execute_reply.started":"2025-05-06T01:21:47.206001Z","shell.execute_reply":"2025-05-06T01:21:47.217576Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"base_model = resnet50(pretrained=False)\n\n\n# Wrap model to flatten output and normalize\nclass ResNet50FeatureExtractor(nn.Module):\n    def __init__(self):\n        super().__init__()\n        base_model = resnet50(pretrained=False)  \n        self.features = nn.Sequential(*list(base_model.children())[:-1])  # remove fc\n        \n    def forward(self, x):\n        x = self.features(x)  # (B, 2048, 1, 1)\n        x = torch.flatten(x, 1)  # (B, 2048)\n        return x\n\nmodel = ResNet50FeatureExtractor()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T01:21:47.219216Z","iopub.execute_input":"2025-05-06T01:21:47.21981Z","iopub.status.idle":"2025-05-06T01:21:48.147973Z","shell.execute_reply.started":"2025-05-06T01:21:47.219772Z","shell.execute_reply":"2025-05-06T01:21:48.147175Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class LandmarkDataset(Dataset):\n    def __init__(self, dataframe, data_dir, transform=None, num_classes=81313):\n        self.dataframe = dataframe\n        self.data_dir = data_dir\n        self.transform = transform\n        self.num_classes = num_classes\n        \n    def __len__(self):\n        return len(self.dataframe)\n        \n    def __getitem__(self, idx):\n        img_id = self.dataframe.iloc[idx][\"id\"]\n        original_landmark_id = self.dataframe.iloc[idx][\"landmark_id\"]\n        \n        # Subtract 1 from the label to make it zero-indexed\n        landmark_id = original_landmark_id - 1  # This ensures 0-based labels\n        \n        # Create the full nested path using the first 3 characters\n        folder_path = os.path.join(\n            self.data_dir,\n            img_id[0],  # First character folder\n            img_id[1],  # Second character folder\n            img_id[2]   # Third character folder\n        )\n        \n        img_path = os.path.join(folder_path, f\"{img_id}.jpg\")\n        \n        if not os.path.exists(img_path):\n            raise FileNotFoundError(f\"❌ IMAGE NOT FOUND: {img_path}\")\n        \n        # Load and transform image\n        image = Image.open(img_path).convert(\"RGB\")\n        if self.transform:\n            image = self.transform(image)\n        \n        return image, landmark_id","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T01:21:48.149087Z","iopub.execute_input":"2025-05-06T01:21:48.149315Z","iopub.status.idle":"2025-05-06T01:21:48.15724Z","shell.execute_reply.started":"2025-05-06T01:21:48.14929Z","shell.execute_reply":"2025-05-06T01:21:48.156402Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#model = torch.jit.load(str(MODEL_FILE))\n#print(model)\n#submit_ids, submit_landmarks = get_similarity(model)\n#submit_df = pd.DataFrame({'id': submit_ids, 'landmarks': submit_landmarks})\n#submit_df.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T01:21:48.15838Z","iopub.execute_input":"2025-05-06T01:21:48.158943Z","iopub.status.idle":"2025-05-06T01:21:48.167648Z","shell.execute_reply.started":"2025-05-06T01:21:48.158906Z","shell.execute_reply":"2025-05-06T01:21:48.167007Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Check the submission\n\nFollowing code shows the inference results. Each figure shows a test image (LEFT), the estimated landmark image (RIGHT), landmark ID and confidence (TITLE).","metadata":{}},{"cell_type":"code","source":"submit_df = pd.read_csv('submission.csv')\nsubmit_df['landmark_id'] = submit_df['landmarks'].apply(lambda x: int(x.split()[0]))\nsubmit_df['confidence'] = submit_df['landmarks'].apply(lambda x: float(x.split()[1]))\ntrain_df = pd.read_csv(TRAIN_LABEL_FILE)\n\ndef get_image(path, name):\n    img = PIL.Image.open(path / name[0] / name[1] / name[2] / f'{name}.jpg')\n    if img.width > img.height:\n        img = img.resize((256, round(img.height / img.width * 256)))\n        new_img = PIL.Image.new(img.mode, (256, 256), (0, 0, 0))\n        new_img.paste(img, (0, (256 - img.height) // 2))\n    else:\n        img = img.resize((round(img.width / img.height * 256), 256))\n        new_img = PIL.Image.new(img.mode, (256, 256), (0, 0, 0))\n        new_img.paste(img, ((256 - img.width) // 2, 2))\n    return np.array(new_img)\n\nrows = 10\nfig = plt.figure(figsize=(15, 4 * rows))\nfor r in range(rows):\n    for c in range(3):\n        i = r * 3 + c\n        test_name, _, label, conf = submit_df.iloc[i].values\n        test_image = get_image(TEST_IMAGE_DIR, test_name)\n        train_name = train_df.query(f'landmark_id == {label}').iloc[0]['id']\n        train_image = get_image(TRAIN_IMAGE_DIR, train_name)\n        image = np.concatenate([test_image, train_image], axis=1)\n    \n        ax = fig.add_subplot(rows, 3, i + 1)        \n        ax.set_title(f'Label={label}, Confidence={conf:.2f}')\n        ax.axis('off')\n        ax.imshow(image)\nfig.tight_layout()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T01:21:48.16876Z","iopub.execute_input":"2025-05-06T01:21:48.169398Z","iopub.status.idle":"2025-05-06T01:21:48.320864Z","shell.execute_reply.started":"2025-05-06T01:21:48.169362Z","shell.execute_reply":"2025-05-06T01:21:48.319611Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Data Preprocessing ###","metadata":{}},{"cell_type":"code","source":"import albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport os","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T01:31:15.051431Z","iopub.execute_input":"2025-05-06T01:31:15.052157Z","iopub.status.idle":"2025-05-06T01:31:15.055444Z","shell.execute_reply.started":"2025-05-06T01:31:15.052122Z","shell.execute_reply":"2025-05-06T01:31:15.054683Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\n\n# Load full dataset\ntrain_df = pd.read_csv('/kaggle/input/landmark-recognition-2021/train.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T01:31:17.488686Z","iopub.execute_input":"2025-05-06T01:31:17.488942Z","iopub.status.idle":"2025-05-06T01:31:18.50064Z","shell.execute_reply.started":"2025-05-06T01:31:17.488915Z","shell.execute_reply":"2025-05-06T01:31:18.499842Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_DIR = \"/kaggle/input/landmark-recognition-2021/train/\"\ndef image_exists(row):\n    img_id = row[\"id\"]\n    img_path = os.path.join(DATA_DIR, img_id[0], img_id[1], img_id[2], f\"{img_id}.jpg\")\n    return os.path.exists(img_path)\n\ntrain_df = train_df[train_df.apply(image_exists, axis=1)]\nval_df = val_df[val_df.apply(image_exists, axis=1)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T01:31:22.100293Z","iopub.execute_input":"2025-05-06T01:31:22.101011Z","iopub.status.idle":"2025-05-06T01:39:20.94314Z","shell.execute_reply.started":"2025-05-06T01:31:22.100976Z","shell.execute_reply":"2025-05-06T01:39:20.942154Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 1: Get top 500 most frequent landmark_ids\ntop_landmarks = (\n    train_df['landmark_id']\n    .value_counts()\n    .head(50)\n    .index\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T01:39:24.98967Z","iopub.execute_input":"2025-05-06T01:39:24.990164Z","iopub.status.idle":"2025-05-06T01:39:25.020956Z","shell.execute_reply.started":"2025-05-06T01:39:24.990132Z","shell.execute_reply":"2025-05-06T01:39:25.020227Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 2: Filter to only those landmark_ids\nfiltered_df = train_df[train_df['landmark_id'].isin(top_landmarks)].copy()\n\n# Step 3: Group-wise stratified train-test split (80/20)\ntrain_list = []\ntest_list = []\n\nfor landmark_id, group in filtered_df.groupby('landmark_id'):\n    if len(group) < 2:\n        continue  # skip if not enough samples to split\n\n    train_part, test_part = train_test_split(\n        group,\n        test_size=0.2,\n        random_state=42,\n        shuffle=True\n    )\n    train_list.append(train_part)\n    test_list.append(test_part)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T01:39:27.38062Z","iopub.execute_input":"2025-05-06T01:39:27.380882Z","iopub.status.idle":"2025-05-06T01:39:27.446481Z","shell.execute_reply.started":"2025-05-06T01:39:27.380853Z","shell.execute_reply":"2025-05-06T01:39:27.445983Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 4: Concatenate final splits\nnew_train_df = pd.concat(train_list).reset_index(drop=True)\nnew_test_df = pd.concat(test_list).reset_index(drop=True)\n\nprint(f\"✅ New Train Size: {len(new_train_df)}\")\nprint(f\"✅ New Test Size:  {len(new_test_df)}\")\n\n# Optional: Save to CSV\nnew_train_df.to_csv(\"top50_train_split.csv\", index=False)\nnew_test_df.to_csv(\"top50_test_split.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T01:39:29.467589Z","iopub.execute_input":"2025-05-06T01:39:29.467858Z","iopub.status.idle":"2025-05-06T01:39:29.547316Z","shell.execute_reply.started":"2025-05-06T01:39:29.467829Z","shell.execute_reply":"2025-05-06T01:39:29.546643Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_image_path(image_id):\n    return os.path.join(DATA_DIR, image_id[0], image_id[1], image_id[2], f\"{image_id}.jpg\")\n\n# Only keep samples with existing files\ndef filter_existing_files(df):\n    df[\"filepath\"] = df[\"id\"].apply(build_image_path)\n    return df[df[\"filepath\"].apply(os.path.exists)].copy()\n\n# Filter for existing files\nnew_train_df = filter_existing_files(new_train_df)\nnew_test_df = filter_existing_files(new_test_df)\n\nprint(f\"✅ Filtered Train Size: {len(new_train_df)}\")\nprint(f\"✅ Filtered Val Size:   {len(new_test_df)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T02:26:20.735381Z","iopub.execute_input":"2025-05-06T02:26:20.736247Z","iopub.status.idle":"2025-05-06T02:27:21.192434Z","shell.execute_reply.started":"2025-05-06T02:26:20.736195Z","shell.execute_reply":"2025-05-06T02:27:21.191716Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"landmark_id_to_idx = {lid: idx for idx, lid in enumerate(sorted(new_train_df['landmark_id'].unique()))}\nnew_train_df['class_idx'] = new_train_df['landmark_id'].map(landmark_id_to_idx)\nnew_test_df['class_idx'] = new_test_df['landmark_id'].map(landmark_id_to_idx)\n\n# Final rename for clarity\ntrain_df = new_train_df\nval_df = new_test_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T02:27:29.418957Z","iopub.execute_input":"2025-05-06T02:27:29.419696Z","iopub.status.idle":"2025-05-06T02:27:29.427177Z","shell.execute_reply.started":"2025-05-06T02:27:29.419666Z","shell.execute_reply":"2025-05-06T02:27:29.4266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Load your CSVs\ntrain_df = pd.read_csv('/kaggle/working/top50_train_split.csv')\ntest_df = pd.read_csv('/kaggle/working/top50_test_split.csv')\nval_df = pd.read_csv('/kaggle/input/google-landmark-2021-validation/valid.csv')\n\n# Build mapping\nlandmark_id_to_idx = {lid: idx for idx, lid in enumerate(sorted(train_df['landmark_id'].unique()))}\nNUM_CLASSES = len(landmark_id_to_idx)\n\n# Map class_idx\ntrain_df['class_idx'] = train_df['landmark_id'].map(landmark_id_to_idx)\ntest_df['class_idx'] = test_df['landmark_id'].map(landmark_id_to_idx)\nval_df['class_idx'] = val_df['landmark_id'].map(landmark_id_to_idx)\n\nNUM_CLASSES = len(landmark_id_to_idx)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T02:10:06.645242Z","iopub.execute_input":"2025-05-06T02:10:06.645577Z","iopub.status.idle":"2025-05-06T02:10:06.793173Z","shell.execute_reply.started":"2025-05-06T02:10:06.645535Z","shell.execute_reply":"2025-05-06T02:10:06.792675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom PIL import Image\nfrom torch.utils.data import Dataset\n\nclass LandmarkDataset(Dataset):\n    def __init__(self, dataframe, data_dir, transform=None):\n        self.dataframe = dataframe.reset_index(drop=True)\n        self.data_dir = data_dir\n        self.transform = transform\n        \n    def __len__(self):\n        return len(self.dataframe)\n    \n    def __getitem__(self, idx):\n        row = self.dataframe.iloc[idx]\n        img_id = row[\"id\"]\n        class_idx = row[\"class_idx\"]\n        \n        folder = os.path.join(self.data_dir, img_id[0], img_id[1], img_id[2])\n        img_path = os.path.join(folder, f\"{img_id}.jpg\")\n        image = Image.open(img_path).convert(\"RGB\")\n        image = np.array(image)\n        \n        if self.transform:\n            image = self.transform(image=image)[\"image\"]\n        \n        return image, class_idx","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T02:27:39.692387Z","iopub.execute_input":"2025-05-06T02:27:39.693155Z","iopub.status.idle":"2025-05-06T02:27:39.700106Z","shell.execute_reply.started":"2025-05-06T02:27:39.693119Z","shell.execute_reply":"2025-05-06T02:27:39.699357Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Data Transform and Data Loader ###","metadata":{}},{"cell_type":"code","source":"from torchvision import transforms\nfrom torch.utils.data import DataLoader\n\nIMAGE_SIZE = 224\nBATCH_SIZE = 32\n\ntrain_transform = A.Compose([\n    A.RandomResizedCrop(IMAGE_SIZE, IMAGE_SIZE, scale=(0.8, 1.0)),\n    A.HorizontalFlip(p=0.5),\n    A.VerticalFlip(p=0.2),\n    A.ImageCompression(quality_lower=99, quality_upper=100),\n    A.RandomBrightnessContrast(p=0.2),\n    A.HueSaturationValue(p=0.2),\n    A.CLAHE(p=0.1),\n    A.GaussianBlur(p=0.1),\n    A.Normalize(),\n    ToTensorV2()\n])\n\nval_transform = A.Compose([\n    A.Resize(IMAGE_SIZE, IMAGE_SIZE),\n    A.Normalize(),\n    ToTensorV2()\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T02:27:42.375209Z","iopub.execute_input":"2025-05-06T02:27:42.375478Z","iopub.status.idle":"2025-05-06T02:27:42.382148Z","shell.execute_reply.started":"2025-05-06T02:27:42.375447Z","shell.execute_reply":"2025-05-06T02:27:42.381487Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset = LandmarkDataset(train_df, DATA_DIR, transform=train_transform)\nval_dataset = LandmarkDataset(val_df, DATA_DIR, transform=val_transform)\n\ntrain_loader = DataLoader(train_dataset, batch_size=BATCH_SIZE, shuffle=True, num_workers=2)\nval_loader = DataLoader(val_dataset, batch_size=BATCH_SIZE, shuffle=False, num_workers=2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T02:27:44.886649Z","iopub.execute_input":"2025-05-06T02:27:44.887363Z","iopub.status.idle":"2025-05-06T02:27:44.8951Z","shell.execute_reply.started":"2025-05-06T02:27:44.887331Z","shell.execute_reply":"2025-05-06T02:27:44.894598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset_1 = np.array(train_df)\nprint(train_dataset_1.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T01:21:48.335249Z","iopub.status.idle":"2025-05-06T01:21:48.335539Z","shell.execute_reply.started":"2025-05-06T01:21:48.335371Z","shell.execute_reply":"2025-05-06T01:21:48.335392Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Calling out ResNet model ##","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torchvision import models\nfrom torch.optim import Adam\nfrom tqdm import tqdm\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\nmodel = models.resnet50(pretrained=True)\nmodel.fc = nn.Sequential(\n    nn.Linear(model.fc.in_features, 512),\n    nn.ReLU(),\n    nn.Dropout(0.5),\n    nn.Linear(512, 256),\n    nn.ReLU(),\n    nn.Dropout(0.5),\n    nn.Linear(256, NUM_CLASSES)\n)\nmodel = model.to(device)\n\ncriterion = nn.CrossEntropyLoss()\noptimizer = Adam(model.parameters(), lr=1e-4)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T02:27:48.139453Z","iopub.execute_input":"2025-05-06T02:27:48.140224Z","iopub.status.idle":"2025-05-06T02:27:59.02691Z","shell.execute_reply.started":"2025-05-06T02:27:48.140191Z","shell.execute_reply":"2025-05-06T02:27:59.026351Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import classification_report, confusion_matrix\nimport numpy as np","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T02:18:21.009942Z","iopub.execute_input":"2025-05-06T02:18:21.01058Z","iopub.status.idle":"2025-05-06T02:18:21.014231Z","shell.execute_reply.started":"2025-05-06T02:18:21.010538Z","shell.execute_reply":"2025-05-06T02:18:21.013369Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def compute_gap(preds, confs, targets):\n    \"\"\"\n    Compute simplified GAP@20 assuming top-1 prediction per image.\n    \"\"\"\n    df = pd.DataFrame({\n        \"pred\": preds,\n        \"conf\": confs,\n        \"target\": targets\n    })\n\n    # Sort globally by confidence\n    df = df.sort_values(\"conf\", ascending=False).reset_index(drop=True)\n\n    correct = 0\n    total_precision = 0.0\n\n    for i, row in df.iterrows():\n        if row[\"pred\"] == row[\"target\"]:\n            correct += 1\n            total_precision += correct / (i + 1)\n\n    return total_precision / len(df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T03:07:55.143858Z","iopub.execute_input":"2025-05-06T03:07:55.14459Z","iopub.status.idle":"2025-05-06T03:07:55.150669Z","shell.execute_reply.started":"2025-05-06T03:07:55.144547Z","shell.execute_reply":"2025-05-06T03:07:55.149846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\nimport numpy as np\n\nEPOCHS = 2\n\nfor epoch in range(EPOCHS):\n    model.train()\n    train_loss = 0.0\n    \n    for images, labels in tqdm(train_loader, desc=f\"Epoch {epoch+1} - Training\"):\n        images, labels = images.to(device), labels.to(device)\n\n        optimizer.zero_grad()\n        outputs = model(images)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n        \n        train_loss += loss.item()\n    \n    avg_loss = train_loss / len(train_loader)\n    print(f\"✅ Epoch {epoch+1} | Train Loss: {avg_loss:.4f}\")\n    \n    # Validation\n    model.eval()\n    all_preds, all_labels, all_confs = [], [], []\n\n    for images, labels in tqdm(val_loader, desc=f\"Epoch {epoch+1} - Validation\"):\n        images, labels = images.to(device), labels.to(device)\n\n        with torch.no_grad():\n            outputs = model(images)\n            probs = torch.softmax(outputs, dim=1)\n            confs, preds = torch.max(probs, dim=1)\n\n        all_preds.extend(preds.cpu().numpy())\n        all_labels.extend(labels.cpu().numpy())\n        all_confs.extend(confs.cpu().numpy())\n\n    acc = accuracy_score(all_labels, all_preds)\n    gap = compute_gap(all_preds, all_confs, all_labels, k=20)\n\n    print(f\"🔍 Validation Accuracy: {acc:.4f}\")\n    print(f\"📈 GAP@20: {gap:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T03:07:57.212762Z","iopub.execute_input":"2025-05-06T03:07:57.213356Z","iopub.status.idle":"2025-05-06T03:13:42.732933Z","shell.execute_reply.started":"2025-05-06T03:07:57.213324Z","shell.execute_reply":"2025-05-06T03:13:42.731887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img_id = \"5551c2a604e9f9b5\"\nimg_path = f\"/kaggle/input/landmark-recognition-2021/train/{img_id[0]}/{img_id[1]}/{img_id[2]}/{img_id}.jpg\"\nos.path.exists(img_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T01:21:48.341767Z","iopub.status.idle":"2025-05-06T01:21:48.342002Z","shell.execute_reply.started":"2025-05-06T01:21:48.341882Z","shell.execute_reply":"2025-05-06T01:21:48.341894Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample = test_df.sample(1).iloc[0]\nimg_id = sample[\"id\"]\ntrue_label = sample[\"landmark_id\"]\nclass_idx = sample[\"class_idx\"]\n\nfolder = os.path.join(DATA_DIR, img_id[0], img_id[1], img_id[2])\nimg_path = os.path.join(folder, f\"{img_id}.jpg\")\n\nimage = Image.open(img_path).convert(\"RGB\")\nimage = val_transform(image=np.array(image))[\"image\"].unsqueeze(0).to(device)\n\nmodel.eval()\nwith torch.no_grad():\n    output = model(image)\n    pred_idx = output.argmax(dim=1).item()\n\n# Reverse map\nidx_to_landmark_id = {v: k for k, v in landmark_id_to_idx.items()}\npred_landmark = idx_to_landmark_id[pred_idx]\n\nprint(f\"🖼️ Image ID: {img_id}\")\nprint(f\"✅ Ground Truth Landmark ID: {true_label}\")\nprint(f\"🎯 Predicted Landmark ID: {pred_landmark}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T02:40:27.989439Z","iopub.execute_input":"2025-05-06T02:40:27.990318Z","iopub.status.idle":"2025-05-06T02:40:28.06506Z","shell.execute_reply.started":"2025-05-06T02:40:27.990258Z","shell.execute_reply":"2025-05-06T02:40:28.064342Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img_path = f\"/kaggle/input/landmark-recognition-2021/train/{img_id[0]}/{img_id[1]}/{img_id[2]}/{img_id}.jpg\"\nfig = show_data(img_path)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predict_single(model, test_df=val_df, data_dir=DATA_DIR, idx=5, transform=val_transform)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T01:21:48.343914Z","iopub.status.idle":"2025-05-06T01:21:48.344147Z","shell.execute_reply.started":"2025-05-06T01:21:48.344029Z","shell.execute_reply":"2025-05-06T01:21:48.344041Z"}},"outputs":[],"execution_count":null}]}