{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.11.11"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":29762,"databundleVersionId":2541532,"sourceType":"competition"},{"sourceId":11730964,"sourceType":"datasetVersion","datasetId":7364044},{"sourceId":11766825,"sourceType":"datasetVersion","datasetId":7374262}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\n\n# Load your CSVs\n# train_df = pd.read_csv('/kaggle/working/train.csv')\n# test_df = pd.read_csv('/kaggle/working/test.csv')\n# val_df = pd.read_csv('/kaggle/working/val.csv')\ntrain_df = pd.read_csv('/kaggle/input/landmark/train.csv')\ntest_df = pd.read_csv('/kaggle/input/landmark/test.csv')\nval_df = pd.read_csv('/kaggle/input/landmark/val.csv')\n# Build mapping\nlandmark_id_to_idx = {lid: idx for idx, lid in enumerate(sorted(train_df['landmark_id'].unique()))}\nNUM_CLASSES = len(landmark_id_to_idx) \n\n# Map class_idx\ntrain_df['class_idx'] = train_df['landmark_id'].map(landmark_id_to_idx)\ntest_df['class_idx'] = test_df['landmark_id'].map(landmark_id_to_idx)\nval_df['class_idx'] = val_df['landmark_id'].map(landmark_id_to_idx)\n\nNUM_CLASSES = len(landmark_id_to_idx)","metadata":{"execution":{"iopub.status.busy":"2025-05-12T04:56:39.438663Z","iopub.execute_input":"2025-05-12T04:56:39.439263Z","iopub.status.idle":"2025-05-12T04:56:41.855536Z","shell.execute_reply.started":"2025-05-12T04:56:39.439239Z","shell.execute_reply":"2025-05-12T04:56:41.854681Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_DIR = \"/kaggle/input/landmark-recognition-2021/train/\"","metadata":{"execution":{"iopub.status.busy":"2025-05-12T04:56:48.457766Z","iopub.execute_input":"2025-05-12T04:56:48.458174Z","iopub.status.idle":"2025-05-12T04:56:48.462803Z","shell.execute_reply.started":"2025-05-12T04:56:48.458140Z","shell.execute_reply":"2025-05-12T04:56:48.462020Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMAGE_SIZE = 224\nBATCH_SIZE = 32","metadata":{"execution":{"iopub.status.busy":"2025-05-12T04:04:15.833639Z","iopub.execute_input":"2025-05-12T04:04:15.834019Z","iopub.status.idle":"2025-05-12T04:04:15.838749Z","shell.execute_reply.started":"2025-05-12T04:04:15.833993Z","shell.execute_reply":"2025-05-12T04:04:15.837533Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torchvision import transforms\nfrom torch.utils.data import DataLoader\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\ntest_transform = A.Compose([\n    A.Resize(IMAGE_SIZE, IMAGE_SIZE),\n    A.Normalize(),\n    ToTensorV2()\n])","metadata":{"execution":{"iopub.status.busy":"2025-05-12T04:04:18.718608Z","iopub.execute_input":"2025-05-12T04:04:18.719581Z","iopub.status.idle":"2025-05-12T04:04:34.463963Z","shell.execute_reply.started":"2025-05-12T04:04:18.719544Z","shell.execute_reply":"2025-05-12T04:04:34.462470Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torchvision import models\nfrom torch.optim import Adam\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = models.resnet50(pretrained=True)\nmodel.fc = nn.Sequential(\n    nn.Linear(model.fc.in_features, 512),\n    nn.ReLU(),\n    nn.Dropout(0.5),\n    nn.Linear(512, 256),\n    nn.ReLU(),\n    nn.Dropout(0.5),\n    nn.Linear(256, NUM_CLASSES)\n)\nmodel = model.to(device)\ncheckpoint_path = \"/kaggle/input/model-effiecient/best_model_resnet_ver_2-2.pth\"\nmodel.load_state_dict(torch.load(checkpoint_path, map_location=device))\nmodel = model.to(device)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nfrom PIL import Image\nsample = test_df.sample(90).iloc[0]\nimg_id = sample[\"id\"]\ntrue_label = sample[\"landmark_id\"]\nclass_idx = sample[\"class_idx\"]\n\nfolder = os.path.join(DATA_DIR, img_id[0], img_id[1], img_id[2])\nimg_path = os.path.join(folder, f\"{img_id}.jpg\")\nprint(img_path)\nimage = Image.open(img_path).convert(\"RGB\")\nimage = test_transform(image=np.array(image))[\"image\"].unsqueeze(0).to(device)\n\nmodel.eval()\nwith torch.no_grad():\n    output = model(image)\n    pred_idx = output.argmax(dim=1).item()\n\n# Reverse map\nidx_to_landmark_id = {v: k for k, v in landmark_id_to_idx.items()}\npred_landmark = idx_to_landmark_id[pred_idx]\n\nprint(f\"🖼️ Image ID: {img_id}\")\nprint(f\"✅ Ground Truth Landmark ID: {true_label}\")\nprint(f\"🎯 Predicted Landmark ID: {pred_landmark}\")","metadata":{"execution":{"iopub.status.busy":"2025-05-12T04:08:48.953213Z","iopub.execute_input":"2025-05-12T04:08:48.953816Z","iopub.status.idle":"2025-05-12T04:08:49.119483Z","shell.execute_reply.started":"2025-05-12T04:08:48.953790Z","shell.execute_reply":"2025-05-12T04:08:49.118698Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install efficientnet_pytorch","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nimport torch\nimport torch.nn as nn\nfrom torchvision import models\nfrom efficientnet_pytorch import EfficientNet\n\nNUM_CLASSES = 100\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# -------- RESNET VERSION 1 --------\nresnet_v1 = models.resnet50(pretrained=False)\nin_features = resnet_v1.fc.in_features\nresnet_v1.fc = nn.Sequential(\n    nn.Linear(in_features, 512),\n    nn.ReLU(),\n    nn.Dropout(0.5),\n    nn.Linear(512, 256),\n    nn.ReLU(),\n    nn.Dropout(0.5),\n    nn.Linear(256, NUM_CLASSES)\n)\nresnet_v1.load_state_dict(torch.load(\"/kaggle/input/model-effiecient/best_model_resnet-5.pth\", map_location=device))\nresnet_v1 = resnet_v1.to(device)\n\n# -------- RESNET VERSION 2 --------\nresnet_v2 = models.resnet50(pretrained=False)\nin_features = resnet_v2.fc.in_features\nresnet_v2.fc = nn.Sequential(\n    nn.Linear(in_features, 512),\n    nn.ReLU(),\n    nn.Dropout(0.5),\n    nn.Linear(512, 256),\n    nn.ReLU(),\n    nn.Dropout(0.5),\n    nn.Linear(256, NUM_CLASSES)\n)\nresnet_v2.load_state_dict(torch.load(\"/kaggle/input/model-effiecient/best_model_resnet_ver_2-7.pth\", map_location=device))\nresnet_v2 = resnet_v2.to(device)\n\n\n# -------- EFFICIENTNET VERSION 1 --------\nefficientnet_v1 = EfficientNet.from_name('efficientnet-b2')\nin_features = efficientnet_v1._fc.in_features\nefficientnet_v1._fc = nn.Linear(in_features, NUM_CLASSES)\nefficientnet_v1.load_state_dict(torch.load(\"/kaggle/input/model-effiecient/best_model_efficientnet-2.pth\", map_location=device))\nefficientnet_v1 = efficientnet_v1.to(device)\n\n# -------- EFFICIENTNET VERSION 2 --------\nefficientnet_v2 = EfficientNet.from_name('efficientnet-b2')\nin_features = efficientnet_v2._fc.in_features\nefficientnet_v2._fc = nn.Linear(in_features, NUM_CLASSES)\nefficientnet_v2.load_state_dict(torch.load(\"/kaggle/input/model-effiecient/best_model_efficientnet_ver_2-2.pth\", map_location=device))\nefficientnet_v2 = efficientnet_v2.to(device)\n\n# -------- MODEL DICTIONARY --------\nmodels = {\n    \"resnet_v1\": resnet_v1,\n    \"resnet_v2\": resnet_v2,\n    \"efficientnet_v1\": efficientnet_v1,\n    \"efficientnet_v2\": efficientnet_v2\n}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T04:59:18.229754Z","iopub.execute_input":"2025-05-12T04:59:18.230546Z","iopub.status.idle":"2025-05-12T04:59:21.640622Z","shell.execute_reply.started":"2025-05-12T04:59:18.230520Z","shell.execute_reply":"2025-05-12T04:59:21.639727Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install gradio","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gradio as gr\nfrom PIL import Image\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport numpy as np\nimport torch\n\n# Albumentations transform cho EfficientNet\ntransform_efficientnet = A.Compose([\n    A.Resize(260, 260),\n    A.Normalize(),  # mean/std mặc định giống ImageNet\n    ToTensorV2()\n])\n\n# Albumentations transform cho ResNet\ntransform_resnet = A.Compose([\n    A.Resize(224, 224),\n    A.Normalize(),\n    ToTensorV2()\n])\n\ndef predict(img: Image.Image, model_name: str):\n    # Chuyển PIL Image -> numpy\n    img = np.array(img)\n\n    # Chọn transform theo model\n    transform = transform_efficientnet if model_name.startswith(\"efficientnet\") else transform_resnet\n\n    # Apply transform\n    transformed = transform(image=img)\n    img_tensor = transformed[\"image\"].unsqueeze(0).to(device)\n\n    model = models[model_name]\n    model.eval()\n\n    with torch.no_grad():\n        output = model(img_tensor)\n        pred = output.argmax(dim=1).item()\n    idx_to_landmark_id = {v: k for k, v in landmark_id_to_idx.items()}\n    pred_landmark = idx_to_landmark_id[pred]\n    return f\"Predicted class: {pred_landmark}\"\n\ninterface = gr.Interface(\n    fn=predict,\n    inputs=[\n        gr.Image(type=\"pil\"),\n        gr.Dropdown(choices=list(models.keys()), label=\"Choose Model\")\n    ],\n    outputs=\"text\",\n    title=\"Landmark Classification Demo (4 Models with Albumentations)\"\n)\n\ninterface.launch()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T05:00:17.547039Z","iopub.execute_input":"2025-05-12T05:00:17.547915Z","iopub.status.idle":"2025-05-12T05:00:27.066208Z","shell.execute_reply.started":"2025-05-12T05:00:17.547881Z","shell.execute_reply":"2025-05-12T05:00:27.065346Z"}},"outputs":[],"execution_count":null}]}