{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":19231,"databundleVersionId":1413778,"sourceType":"competition"},{"sourceId":8844081,"sourceType":"datasetVersion","datasetId":5323033},{"sourceId":70247,"sourceType":"modelInstanceVersion","modelInstanceId":58636},{"sourceId":70419,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":58790},{"sourceId":70683,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":59009},{"sourceId":70768,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":59083},{"sourceId":70985,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":59272},{"sourceId":71002,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":59288},{"sourceId":71705,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":59899}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"print(\"started\")\nimport torch\nfrom torchvision import transforms\nfrom torch.utils.data import DataLoader, Dataset\nimport pandas as pd\nimport os\nfrom PIL import Image\nimport numpy as np\nimport torch.nn as nn\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n# Check if GPU is available\nN_CLASSES = 10000\nBATCH_SIZE = 1300\ndevice = torch.device('cuda:0' if torch.cuda.is_available() else 'cpu')\n# device = 'cpu'\nprint(f'Using device: {device}')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-07-02T21:00:52.894462Z","iopub.execute_input":"2024-07-02T21:00:52.894932Z","iopub.status.idle":"2024-07-02T21:00:52.904160Z","shell.execute_reply.started":"2024-07-02T21:00:52.894895Z","shell.execute_reply":"2024-07-02T21:00:52.902812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load the data","metadata":{}},{"cell_type":"code","source":"class CustomImageDataset(Dataset):\n    def __init__(self, labels, root_dir, transform, top_classes=3):\n        labels_df = pd.read_csv(labels, dtype=({\"id\": str, \"landmark_id\": np.uint64}))\n        \n        label_counts = labels_df.value_counts(\"landmark_id\")\n        \n        with open(\"/kaggle/input/mapping-dict/mapping.dict\") as f:\n            text = f.read()\n            self.mapping = eval(text)\n            \n        used_pictures = labels_df.where(labels_df[\"landmark_id\"].isin(self.mapping.keys())).dropna()        \n        used_pictures = used_pictures.sample(frac=1, random_state=42).reset_index(drop=True)\n        self.labels_df = used_pictures[-70000:]\n        \n        self.root_dir = root_dir      \n        self.transform = transform\n    def __len__(self):\n        return len(self.labels_df)\n    \n    def __getitem__(self, idx):\n        img_name = self.labels_df.iloc[idx, 0]\n        label = int(self.labels_df.iloc[idx, 1])\n        img_path = os.path.join(self.root_dir, img_name[0], img_name[1], img_name[2], img_name + \".jpg\")\n        image = Image.open(img_path).convert(\"RGB\")\n        image = self.transform(image)\n        return image, self.mapping[label]","metadata":{"execution":{"iopub.status.busy":"2024-07-02T21:00:52.906376Z","iopub.execute_input":"2024-07-02T21:00:52.906768Z","iopub.status.idle":"2024-07-02T21:00:52.922957Z","shell.execute_reply.started":"2024-07-02T21:00:52.906738Z","shell.execute_reply":"2024-07-02T21:00:52.921527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Validation","metadata":{}},{"cell_type":"code","source":"import torch.nn.functional as F\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom torchvision.models import efficientnet_b0, EfficientNet_B0_Weights, EfficientNet, ResNet, resnet50, ResNet50_Weights, MobileNet_V3_Small_Weights, mobilenet_v3_small\n\nbase_path = \"/kaggle/input/\"\n\nfor model_name in (\"b0_10_000/pytorch/v3/1/b0-10_1000v3.pth\", \"resnet50-10_000/pytorch/v2/1/Resnet50-10_000v2.pth\", \"mobilenet_v3_small/pytorch/v2/1/MobileNet_V3_Small-10_000v2.pth\"):\n    print(f'Evaluating {model_name.split(\"/\")[0]}:')\n    \n    if model_name.startswith(\"b0\"):\n        weights = EfficientNet_B0_Weights.DEFAULT\n        model = efficientnet_b0()\n        model.classifier[1] = nn.Linear(model.classifier[1].in_features, N_CLASSES)\n        model.load_state_dict(torch.load(base_path + model_name, map_location=torch.device(\"cpu\")))\n    elif model_name.startswith(\"res\"):\n        weights = ResNet50_Weights.DEFAULT\n        model = resnet50()\n        model.fc = nn.Linear(model.fc.in_features, N_CLASSES)\n        model.load_state_dict(torch.load(base_path + model_name, map_location=torch.device(\"cpu\")))\n    elif model_name.startswith(\"mobile\"):\n        weights = MobileNet_V3_Small_Weights.DEFAULT\n        model = mobilenet_v3_small()\n        model.classifier[-1] = nn.Linear(model.classifier[-1].in_features, N_CLASSES)\n        model.load_state_dict(torch.load(base_path + model_name, map_location=torch.device(\"cpu\")))\n    model = model.to(device)\n    \n    validation_dataset = CustomImageDataset(labels=\"/kaggle/input/landmark-recognition-2020/train.csv\", root_dir=\"/kaggle/input/landmark-recognition-2020/train\", top_classes=N_CLASSES, transform=weights.transforms(antialias=True))\n    validation_loader = DataLoader(validation_dataset, batch_size=BATCH_SIZE, shuffle=True)\n\n    confusion_matrix = np.zeros((N_CLASSES, N_CLASSES))\n    number_of_top_5 = 0\n    print(f\"Doing {len(validation_loader)} batches\")\n\n    model.eval()\n    with torch.no_grad():\n        for i, (inputs, labels) in enumerate(validation_loader):\n            inputs = inputs.to(device)\n            output = model(inputs)\n\n            _, top_5_labels = torch.topk(output, 5, dim=1)\n            labels = labels.tolist()\n\n            for real_class, top_5_label in zip(labels, top_5_labels):\n                predicted_class = top_5_label[0]\n#                 print(real_class, predicted_class, top_5_label)\n                confusion_matrix[real_class, predicted_class] += 1\n                if real_class in top_5_label:\n                    number_of_top_5 += 1\n\n            if not i % 5:\n                print(f\"Batch {i} done. Doing {len(validation_loader)} batches\")\n                print(f\"Acc@top5 = {number_of_top_5 / ((i+1) * BATCH_SIZE)}\")\n                print(f\"Acc@top1 = {np.sum(np.diag(confusion_matrix)) / ((i+1) * BATCH_SIZE)} \")\n                print(\"-\"*50)\n    np.save(f'{model_name.split(\"/\")[0]}-confusion_matrix.npy', confusion_matrix)\n    print(f\"Acc@top5 = {number_of_top_5 / len(validation_dataset)}\")\n    print(f\"Acc@top1 = {np.sum(np.diag(confusion_matrix)) / len(validation_dataset)} \")\n","metadata":{"execution":{"iopub.status.busy":"2024-07-02T21:00:52.924756Z","iopub.execute_input":"2024-07-02T21:00:52.925173Z","iopub.status.idle":"2024-07-02T21:01:25.425904Z","shell.execute_reply.started":"2024-07-02T21:00:52.925143Z","shell.execute_reply":"2024-07-02T21:01:25.424662Z"},"trusted":true},"execution_count":null,"outputs":[]}]}