{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Essential Codes for Starter\n\n* We are going to follow previous guides using basic DL models and reduced train files.\n* We use Pytorch framework for DL training.\n    1. Create and train your own models (ref 2)\n    2. Load pre-trained models (ref 3, 4)\n* In this notebook, we practice just how to start and submit properly. Finding architectures for high score is out of the boundary of this notebook.\n\nreference\n1. [(Competition) Google Landmark Recognition 2021](https://www.kaggle.com/competitions/landmark-recognition-2021/overview)\n2. [(Notebook) Landmark_Recognition_2021_Starter](https://www.kaggle.com/code/drcapa/landmark-recognition-2021-starter)\n3. [(Notebook) Inference and Submission [PyTorch/ResNet34]](https://www.kaggle.com/code/takedarts/inference-and-submission-pytorch-resnet34/notebook)\n4. [(Dataset) Validation Data for Google Landmark 2021](https://www.kaggle.com/datasets/takedarts/google-landmark-2021-validation)","metadata":{}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport cv2\nfrom PIL import Image\n\nfrom sklearn.model_selection import train_test_split\n\nimport torch\nfrom torch import nn\nfrom torch.utils.data import Dataset, DataLoader\n\nimport torchvision\nfrom torchvision import transforms\nfrom torchinfo import summary\nfrom torchmetrics import Accuracy\n\nfrom tqdm.notebook import tqdm_notebook\n\nimport sys\nsys.path.append('/kaggle/input/python-scripts')\n\nfrom python_scripts import engine\n\nfrom python_scripts.models import Metrics\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:53:09.270738Z","iopub.execute_input":"2023-04-23T13:53:09.271404Z","iopub.status.idle":"2023-04-23T13:53:09.280806Z","shell.execute_reply.started":"2023-04-23T13:53:09.271182Z","shell.execute_reply":"2023-04-23T13:53:09.279727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if torch.cuda.is_available():\n    device = torch.device('cuda')\nelif torch.backends.mps.is_available():\n    device = torch.device('mps')\nelse:\n    device = torch.device('cpu')","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:53:09.283706Z","iopub.execute_input":"2023-04-23T13:53:09.284648Z","iopub.status.idle":"2023-04-23T13:53:09.365777Z","shell.execute_reply.started":"2023-04-23T13:53:09.284607Z","shell.execute_reply":"2023-04-23T13:53:09.364700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Path","metadata":{}},{"cell_type":"code","source":"path = '/kaggle/input/landmark-recognition-2021/'\nos.listdir(path)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:53:09.367302Z","iopub.execute_input":"2023-04-23T13:53:09.367681Z","iopub.status.idle":"2023-04-23T13:53:09.381465Z","shell.execute_reply.started":"2023-04-23T13:53:09.367649Z","shell.execute_reply":"2023-04-23T13:53:09.380277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load Data","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv(path + 'train.csv')\nsample_csv = pd.read_csv(path + 'sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:53:09.383743Z","iopub.execute_input":"2023-04-23T13:53:09.384102Z","iopub.status.idle":"2023-04-23T13:53:10.783954Z","shell.execute_reply.started":"2023-04-23T13:53:09.384066Z","shell.execute_reply":"2023-04-23T13:53:10.782924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_data), len(sample_csv)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:53:10.787145Z","iopub.execute_input":"2023-04-23T13:53:10.787523Z","iopub.status.idle":"2023-04-23T13:53:10.794649Z","shell.execute_reply.started":"2023-04-23T13:53:10.787487Z","shell.execute_reply":"2023-04-23T13:53:10.793639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:53:10.796610Z","iopub.execute_input":"2023-04-23T13:53:10.797483Z","iopub.status.idle":"2023-04-23T13:53:10.814599Z","shell.execute_reply.started":"2023-04-23T13:53:10.797436Z","shell.execute_reply":"2023-04-23T13:53:10.813508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_data)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:53:10.816411Z","iopub.execute_input":"2023-04-23T13:53:10.816748Z","iopub.status.idle":"2023-04-23T13:53:10.826531Z","shell.execute_reply.started":"2023-04-23T13:53:10.816714Z","shell.execute_reply":"2023-04-23T13:53:10.822175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['landmark_id'].value_counts()[:10]\n","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:53:10.828465Z","iopub.execute_input":"2023-04-23T13:53:10.828831Z","iopub.status.idle":"2023-04-23T13:53:10.874882Z","shell.execute_reply.started":"2023-04-23T13:53:10.828795Z","shell.execute_reply":"2023-04-23T13:53:10.873744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_data['landmark_id'].unique())","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:53:10.876460Z","iopub.execute_input":"2023-04-23T13:53:10.876899Z","iopub.status.idle":"2023-04-23T13:53:10.902869Z","shell.execute_reply.started":"2023-04-23T13:53:10.876862Z","shell.execute_reply":"2023-04-23T13:53:10.901794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_csv.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:53:10.904451Z","iopub.execute_input":"2023-04-23T13:53:10.904897Z","iopub.status.idle":"2023-04-23T13:53:10.915039Z","shell.execute_reply.started":"2023-04-23T13:53:10.904859Z","shell.execute_reply":"2023-04-23T13:53:10.913877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Functions","metadata":{}},{"cell_type":"code","source":"# Plot 5 (maximum) exmaples of images with the same landmark_id\ndef plot_examples(data=train_data, landmark_id=1):\n    indexes = data[data['landmark_id'] == landmark_id].index\n    num_pic = len(indexes) if len(indexes) < 5 else 5\n    if num_pic == 0:\n        print('No images available')\n        return None\n    \n    fig, axs = plt.subplots(1, num_pic, figsize=(5 * num_pic, 12))\n    fig.subplots_adjust(hspace=.2, wspace=.2)\n    axs = axs.ravel()\n      \n    for i in range(num_pic):\n        idx = indexes[i]\n        image_id = train_data.loc[idx]['id']\n        file = image_id + '.jpg'\n        subpath = '/'.join([char for char in image_id[0:3]])\n        img = cv2.imread(path + 'train/' + subpath + '/' + file)\n        axs[i].imshow(img)\n        axs[i].set_title('landmark_id: ' + str(landmark_id))\n        axs[i].set_xticklabels([])\n        axs[i].set_yticklabels([])","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:53:10.916453Z","iopub.execute_input":"2023-04-23T13:53:10.918018Z","iopub.status.idle":"2023-04-23T13:53:10.927452Z","shell.execute_reply.started":"2023-04-23T13:53:10.917989Z","shell.execute_reply":"2023-04-23T13:53:10.926243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_examples(train_data, 32349)\nplot_examples(train_data, 7)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:53:10.928707Z","iopub.execute_input":"2023-04-23T13:53:10.929619Z","iopub.status.idle":"2023-04-23T13:53:13.724678Z","shell.execute_reply.started":"2023-04-23T13:53:10.929583Z","shell.execute_reply":"2023-04-23T13:53:13.723790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Split Data","metadata":{}},{"cell_type":"code","source":"train_val_list, _ = train_test_split(list(train_data['id']), train_size=0.0005, random_state=42)\ntrain_list, val_list = train_test_split(train_val_list, test_size=0.1, random_state=42)\ntest_list = list(sample_csv['id'])\n\nlen(train_list), len(val_list), len(test_list)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:53:13.726291Z","iopub.execute_input":"2023-04-23T13:53:13.726970Z","iopub.status.idle":"2023-04-23T13:53:14.485584Z","shell.execute_reply.started":"2023-04-23T13:53:13.726932Z","shell.execute_reply":"2023-04-23T13:53:14.484266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Set","metadata":{}},{"cell_type":"code","source":"img_size = 224\nbatch_size = 64","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:53:14.494861Z","iopub.execute_input":"2023-04-23T13:53:14.498085Z","iopub.status.idle":"2023-04-23T13:53:14.506775Z","shell.execute_reply.started":"2023-04-23T13:53:14.498038Z","shell.execute_reply":"2023-04-23T13:53:14.504878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_transforms = transforms.Compose([\n    transforms.RandomRotation(30),\n    transforms.Resize(img_size),\n    transforms.RandomHorizontalFlip(),\n    transforms.ToTensor()\n])\n\ntest_transforms = transforms.Compose([\n    transforms.Resize(img_size),\n    transforms.ToTensor()\n])","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:53:14.509041Z","iopub.execute_input":"2023-04-23T13:53:14.509792Z","iopub.status.idle":"2023-04-23T13:53:14.517694Z","shell.execute_reply.started":"2023-04-23T13:53:14.509751Z","shell.execute_reply":"2023-04-23T13:53:14.516271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DatasetGenerator(Dataset):\n    def __init__(self, data_list, path, transforms, data=None, image_size=224):\n        self.data_list = data_list\n        self.path = path\n        self.transforms = transforms\n        self.image_size = image_size\n        self.data = data\n        self.classes = []\n        self.class_to_idx = {}\n        if self.data is not None and 'landmark_id' in self.data.columns:\n            self.classes = sorted(self.data['landmark_id'].unique())\n            self.class_to_idx = dict(zip(self.classes, range(len(self.classes))))\n        \n    def __len__(self):\n        return len(self.data_list)\n    \n    def __getitem__(self, index):\n        image_id = self.data_list[index]\n        file = image_id + '.jpg'\n        subpath = '/'.join([char for char in image_id[0:3]])\n        \n        image = cv2.imread(self.path + subpath + '/' + file)\n        image = cv2.resize(image, (self.image_size, self.image_size))\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB) # used because OpenCV follows BGR convention and PIL follows RGB convention\n        \n        X = self.transforms(Image.fromarray(image))\n        y = None\n        if self.data is not None and 'landmark_id' in self.data.columns:\n            c = self.data[self.data['id'] == image_id]['landmark_id'].values[0]\n            y = self.class_to_idx[c]\n        \n        return X, y","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:53:14.519875Z","iopub.execute_input":"2023-04-23T13:53:14.522057Z","iopub.status.idle":"2023-04-23T13:53:14.635261Z","shell.execute_reply.started":"2023-04-23T13:53:14.521987Z","shell.execute_reply":"2023-04-23T13:53:14.633851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = DatasetGenerator(\n    data_list=train_list,\n    path=path + 'train/',\n    transforms=train_transforms,\n    data=train_data,\n    image_size=img_size\n)\n\nval_dataset = DatasetGenerator(\n    data_list=val_list,\n    path=path + 'train/',\n    transforms=test_transforms,\n    data=train_data,\n    image_size=img_size\n)\n\ntest_dataset = DatasetGenerator(\n    data_list=test_list,\n    path=path + 'test/',\n    transforms=test_transforms,\n    image_size=img_size\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:53:14.637398Z","iopub.execute_input":"2023-04-23T13:53:14.638378Z","iopub.status.idle":"2023-04-23T13:53:14.747187Z","shell.execute_reply.started":"2023-04-23T13:53:14.638334Z","shell.execute_reply":"2023-04-23T13:53:14.746084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_dataset), len(val_dataset), len(test_dataset)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:53:14.752266Z","iopub.execute_input":"2023-04-23T13:53:14.754595Z","iopub.status.idle":"2023-04-23T13:53:14.764951Z","shell.execute_reply.started":"2023-04-23T13:53:14.754550Z","shell.execute_reply":"2023-04-23T13:53:14.764017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset[0][0].shape, val_dataset[0][0].shape, test_dataset[0][0].shape","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:53:14.769413Z","iopub.execute_input":"2023-04-23T13:53:14.771911Z","iopub.status.idle":"2023-04-23T13:53:15.210501Z","shell.execute_reply.started":"2023-04-23T13:53:14.771875Z","shell.execute_reply":"2023-04-23T13:53:15.209430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(train_dataset[0][0].permute(1, 2, 0)) # torch has shape of (3, 224, 224) while plt.imshow requires shape of (224, 224, 3)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:53:15.214714Z","iopub.execute_input":"2023-04-23T13:53:15.217012Z","iopub.status.idle":"2023-04-23T13:53:15.779915Z","shell.execute_reply.started":"2023-04-23T13:53:15.216974Z","shell.execute_reply":"2023-04-23T13:53:15.778859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_names = train_dataset.classes\nclass_names[0], len(class_names)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:53:15.781489Z","iopub.execute_input":"2023-04-23T13:53:15.782140Z","iopub.status.idle":"2023-04-23T13:53:15.790277Z","shell.execute_reply.started":"2023-04-23T13:53:15.782095Z","shell.execute_reply":"2023-04-23T13:53:15.789140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Loader","metadata":{}},{"cell_type":"code","source":"train_dataloader = DataLoader(\n    dataset=train_dataset,\n    batch_size=batch_size,\n    shuffle=False,\n    num_workers=1,\n    pin_memory=True\n)\n\nval_dataloader = DataLoader(\n    dataset=val_dataset,\n    batch_size=batch_size,\n    shuffle=False,\n    num_workers=1,\n    pin_memory=True\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:53:15.791857Z","iopub.execute_input":"2023-04-23T13:53:15.792836Z","iopub.status.idle":"2023-04-23T13:53:15.801121Z","shell.execute_reply.started":"2023-04-23T13:53:15.792796Z","shell.execute_reply":"2023-04-23T13:53:15.800102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# this may take few minutes\nf, ax = plt.subplots(2, 3, figsize=(15, 12))\nax = ax.ravel()\n\nfor i, (data, label) in enumerate(train_dataloader):\n    img = torchvision.utils.make_grid(data).numpy()\n    img = np.transpose(img, (1, 2, 0))\n    img += np.array([1, 1, 1])\n    img *= 127.5\n    img = img.astype(np.uint8)\n    img = img[:, :, [2, 1, 0]]\n    \n    ax[i].imshow(img)\n    if i == 5:\n        break\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:53:15.803651Z","iopub.execute_input":"2023-04-23T13:53:15.804074Z","iopub.status.idle":"2023-04-23T13:54:07.041748Z","shell.execute_reply.started":"2023-04-23T13:53:15.804045Z","shell.execute_reply":"2023-04-23T13:54:07.040309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Create and train your own models\nWe would generate a ResNet50 model with metric learning method (cosine similarity) resulting 81313 classes of output.\n\nUnder limitation of GPU resource, this example only used 0.05% of images of whole training data, so the accuracy is pool.\n\nMaybe you can train the model on Colab or local desktop, and then load the trained model into kaggle notebook as implemented in section 2 below.","metadata":{}},{"cell_type":"markdown","source":"## Model","metadata":{}},{"cell_type":"code","source":"learning_rate_list = [1e-3]\nweight_decay_list = [0]\nepochs_list = [10]\nbatch_size_list = [64]","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:54:07.043753Z","iopub.execute_input":"2023-04-23T13:54:07.044803Z","iopub.status.idle":"2023-04-23T13:54:07.050154Z","shell.execute_reply.started":"2023-04-23T13:54:07.044757Z","shell.execute_reply":"2023-04-23T13:54:07.049116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checkout dataset 'python-scripts'\nmodel = torchvision.models.resnet50()\nmodel.fc = nn.Linear(\n    in_features=2048,\n    out_features=512,\n    bias=True\n)\nsummary(model)\n\nclass Resnet50_Cos(nn.Module):\n    def __init__(self) -> None:\n        super().__init__()\n        weights = torchvision.models.ResNet50_Weights.DEFAULT # remove pre-trained weight if you turn-off the internet option.\n        self.backbone = torchvision.models.resnet50(weights=weights)\n        self.backbone.fc = nn.Linear(\n            in_features=2048,\n            out_features=512,\n            bias=True\n        )\n        self.metric = Metrics.AddMarginProduct(\n            in_features=512,\n            out_features=len(class_names),\n            m=0.4\n        )\n\n    def forward(self, X, label):\n        X = self.backbone(X)\n        output = self.metric(X, label)\n        return output\n\nmodel = Resnet50_Cos()\nsummary(model)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:54:16.170904Z","iopub.execute_input":"2023-04-23T13:54:16.171415Z","iopub.status.idle":"2023-04-23T13:54:18.451036Z","shell.execute_reply.started":"2023-04-23T13:54:16.171372Z","shell.execute_reply":"2023-04-23T13:54:18.449697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checkout dataset 'python-scripts'\ntuning_results = engine.HP_tune_train(\n    model=model,\n    model_generator=None,\n    model_weights=None,\n    model_name='ResNet50_Cos_Google_landmark',\n    train_dataset=train_dataset,\n    test_dataset=val_dataset,\n    learning_rate_list=learning_rate_list,\n    weight_decay_list=weight_decay_list,\n    epochs_list=epochs_list,\n    batch_size_list=batch_size_list,\n    is_tensorboard_writer=False,\n    device=device,\n    gradient_accumulation_num=1,\n    metric_learning=True\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:54:23.495232Z","iopub.execute_input":"2023-04-23T13:54:23.495807Z","iopub.status.idle":"2023-04-23T13:57:26.067043Z","shell.execute_reply.started":"2023-04-23T13:54:23.495762Z","shell.execute_reply":"2023-04-23T13:57:26.064432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predict Test Data","metadata":{}},{"cell_type":"code","source":"model.eval()\ntrain_features = []\ntrain_labels = []\n\nwith torch.inference_mode():\n    for X_batch_train, y_batch_train in tqdm_notebook(train_dataloader, desc='predict_1', leave=True):\n        X_batch_train, y_batch_train = X_batch_train.to(device), y_batch_train.to(device)\n\n        train_features.append(model.backbone(X_batch_train).detach().cpu())\n        train_labels.append(y_batch_train.cpu())\n\n    train_features = torch.cat(train_features, dim=0)\n    train_labels = torch.cat(train_labels, dim=0)\n    \n    for i in tqdm_notebook(range(len(test_dataset)), desc='predict_2', leave=True):\n        X_test = test_dataset[i][0].to(device)\n\n        test_features = model.backbone(X_test.unsqueeze(0)).detach().cpu()\n        cos_sim = torch.mm(nn.functional.normalize(test_features), nn.functional.normalize(train_features).T)\n        test_pred = train_labels[torch.argmax(cos_sim, dim=1)]\n        test_score = torch.max(nn.functional.softmax(cos_sim), dim=1)\n        \n        category = class_names[test_pred[0].numpy()]\n        score = test_score[0][0].numpy()\n        sample_csv.loc[i]['landmarks'] = str(category) + ' ' + str(score)\n        ","metadata":{"execution":{"iopub.status.busy":"2023-04-23T13:57:31.798006Z","iopub.execute_input":"2023-04-23T13:57:31.798621Z","iopub.status.idle":"2023-04-23T14:03:56.800022Z","shell.execute_reply.started":"2023-04-23T13:57:31.798575Z","shell.execute_reply":"2023-04-23T14:03:56.798668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_csv.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-23T14:03:56.802404Z","iopub.execute_input":"2023-04-23T14:03:56.803137Z","iopub.status.idle":"2023-04-23T14:03:56.814183Z","shell.execute_reply.started":"2023-04-23T14:03:56.803090Z","shell.execute_reply":"2023-04-23T14:03:56.813136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Export","metadata":{}},{"cell_type":"code","source":"sample_csv.to_csv('submission.csv', index=False)\nprint('submission completed!')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualizing","metadata":{}},{"cell_type":"code","source":"import pathlib\nimport PIL\n\nTRAIN_IMAGE_DIR = pathlib.Path('../input/landmark-recognition-2021/train')\nTEST_IMAGE_DIR = pathlib.Path('../input/landmark-recognition-2021/test')\n\nsubmit_df = pd.read_csv('submission.csv')\nsubmit_df['landmark_id'] = submit_df['landmarks'].apply(lambda x: int(x.split()[0]))\nsubmit_df['confidence'] = submit_df['landmarks'].apply(lambda x: float(x.split()[1]))\ntrain_df = train_data\n\ndef get_image(path, name):\n    img = PIL.Image.open(path / name[0] / name[1] / name[2] / f'{name}.jpg')\n    if img.width > img.height:\n        img = img.resize((256, round(img.height / img.width * 256)))\n        new_img = PIL.Image.new(img.mode, (256, 256), (0, 0, 0))\n        new_img.paste(img, (0, (256 - img.height) // 2))\n    else:\n        img = img.resize((round(img.width / img.height * 256), 256))\n        new_img = PIL.Image.new(img.mode, (256, 256), (0, 0, 0))\n        new_img.paste(img, ((256 - img.width) // 2, 2))\n    return np.array(new_img)\n\nrows = 10\nfig = plt.figure(figsize=(15, 4 * rows))\nfor r in range(rows):\n    for c in range(3):\n        i = r * 3 + c\n        test_name, _, label, conf = submit_df.iloc[i].values\n        test_image = get_image(TEST_IMAGE_DIR, test_name)\n        train_name = train_df.query(f'landmark_id == {label}').iloc[0]['id']\n        train_image = get_image(TRAIN_IMAGE_DIR, train_name)\n        image = np.concatenate([test_image, train_image], axis=1)\n    \n        ax = fig.add_subplot(rows, 3, i + 1)        \n        ax.set_title(f'Label={label}, Confidence={conf:.2f}')\n        ax.axis('off')\n        ax.imshow(image)\nfig.tight_layout()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Load pre-trained models\nWe would follow a notebook with Landmark 2020 competition's 1st place architecture.","metadata":{}},{"cell_type":"markdown","source":"## Import and Constants","metadata":{}},{"cell_type":"code","source":"import pathlib\n\nimport torch\nimport torch.utils.data\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport numpy as np\nimport pandas as pd\n\nimport PIL.Image\nimport albumentations.pytorch\nimport cv2\nimport matplotlib.pyplot as plt\n\nfrom tqdm.notebook import tqdm\nfrom typing import List, Tuple\n\nIMAGE_SIZE = 320  # (2021/09/24 5:00AM) Updated.\nBATCH_SIZE = 512\n\nMODEL_FILE = pathlib.Path('../input/google-landmark-2021-validation/model.pth')\nTRAIN_LABEL_FILE = pathlib.Path('train.csv')\nTRAIN_IMAGE_DIR = pathlib.Path('../input/landmark-recognition-2021/train')\nVALID_LABEL_FILE = pathlib.Path('valid.csv')\nVALID_IMAGE_DIR = pathlib.Path('../input/google-landmark-2021-validation/valid')\nTEST_LABEL_FILE = pathlib.Path('../input/landmark-recognition-2021/sample_submission.csv')\nTEST_IMAGE_DIR = pathlib.Path('../input/landmark-recognition-2021/test')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# A subset of training images for time reduction","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('../input/landmark-recognition-2021/train.csv')\n\nif len(train_df) == 1580470:\n    records = {}\n\n    for image_id, landmark_id in train_df.values:\n        if landmark_id in records:\n            records[landmark_id].append(image_id)\n        else:\n            records[landmark_id] = [image_id]\n        \n    image_ids = []\n    landmark_ids = []\n\n    for landmark_id, img_ids in records.items():\n        num = min(len(img_ids), 2)\n        image_ids.extend(records[landmark_id][:num])\n        landmark_ids.extend([landmark_id] * num)\n\n    train_df = pd.DataFrame({'id': image_ids, 'landmark_id': landmark_ids})\n\ntrain_df.to_csv(TRAIN_LABEL_FILE, index=False)\ntrain_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Filtering only non-landmark images","metadata":{}},{"cell_type":"code","source":"valid_df = pd.read_csv('../input/google-landmark-2021-validation/valid.csv')\nvalid_df = valid_df[valid_df['landmark_id'] == -1]\nvalid_df.to_csv(VALID_LABEL_FILE, index=False)\nvalid_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Class and Function for feature extraction","metadata":{}},{"cell_type":"code","source":"class Dataset(torch.utils.data.Dataset):\n    def __init__(self, label_file: pathlib.Path, image_dir: pathlib.Path) -> None:\n        super().__init__()\n        self.files = [\n            image_dir / n[0] / n[1] / n[2] / f'{n}.jpg'\n            for n in pd.read_csv(label_file)['id'].values]\n        \n        self.transformer = albumentations.Compose([\n            albumentations.SmallestMaxSize(IMAGE_SIZE, interpolation=cv2.INTER_CUBIC),\n            albumentations.CenterCrop(IMAGE_SIZE, IMAGE_SIZE),\n            albumentations.Normalize(),\n            albumentations.pytorch.ToTensorV2(),\n        ])\n\n    def __len__(self) -> int:\n        return len(self.files)\n\n    def __getitem__(self, index: int) -> Tuple[str, torch.Tensor]:\n        path = self.files[index]\n        image = PIL.Image.open(self.files[index])\n        image = self.transformer(image=np.array(image))['image']\n\n        return path.name[:-4], image","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@torch.no_grad()\ndef get_features(\n    model: nn.Module,\n    label_file: pathlib.Path,\n    image_dir: pathlib.Path,\n) -> Tuple[List[str], torch.Tensor]:\n    loader = torch.utils.data.DataLoader(\n        Dataset(label_file, image_dir),\n        batch_size=BATCH_SIZE, shuffle=False, num_workers=2)\n\n    model = model.cuda()\n    model.eval()\n    \n    all_names = []\n    all_features = []\n\n    for names, images in tqdm(loader, desc=image_dir.name):\n        images = images.cuda()\n        features = model(images)\n        all_features.append(features)\n        all_names.extend(names)\n\n    return all_names, F.normalize(torch.cat(all_features, dim=0))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_similarity(model: nn.Module)-> Tuple[List[str], List[str]]:\n    # features\n    train_names, train_features = get_features(\n        model, TRAIN_LABEL_FILE, TRAIN_IMAGE_DIR)    \n    _, valid_features = get_features(\n        model, VALID_LABEL_FILE, VALID_IMAGE_DIR)\n    test_names, test_features = get_features(\n        model, TEST_LABEL_FILE, TEST_IMAGE_DIR)\n\n    # penalties\n    train_penalties_list = []\n    for i in range(0, train_features.shape[0], 128):\n        x = torch.mm(train_features[i:i + 128], valid_features.T)\n        x = torch.topk(x, k=5)[0].mean(dim=1)\n        train_penalties_list.append(x)\n    train_penalties = torch.cat(train_penalties_list, dim=0)\n\n    test_penalties_list = []\n    for i in range(0, test_features.shape[0], 128):\n        x = torch.mm(test_features[i:i + 128], valid_features.T)\n        x = torch.topk(x, k=10)[0].mean(dim=1)\n        test_penalties_list.append(x)\n    test_penalties = torch.cat(test_penalties_list, dim=0)\n\n    # neighbors\n    submit_ids = []\n    submit_landmark_ids = []\n    submit_confidences = []\n    \n    train_df = pd.read_csv(TRAIN_LABEL_FILE)\n    idmap = {n: v for n, v in train_df.values}\n\n    for i in range(0, test_features.shape[0], 128):\n        x = torch.mm(test_features[i:i + 128], train_features.T)\n        x -= train_penalties[None, :]\n        values, indexes = torch.topk(x, k=3)\n        \n        submit_ids.extend(test_names[i:i + 128])\n\n        for idxs, vals, penalty in zip(indexes, values, test_penalties[i:i + 128]):\n            scores = {}\n            for idx, val in zip(idxs, vals):\n                landmark_id = idmap[train_names[idx]]\n                if landmark_id in scores:\n                    scores[landmark_id] += float(val)\n                else:\n                    scores[landmark_id] = float(val)\n                    \n            landmark_id, confidence = max(\n                [(k, v) for k, v in scores.items()], key=lambda x: x[1])\n            submit_landmark_ids.append(landmark_id)\n            submit_confidences.append(confidence - penalty)\n\n    # standardize confidence values\n    max_conf = max(submit_confidences)\n    min_conf = min(submit_confidences)\n    submit_confidences = [\n        (v - min_conf) / (max_conf - min_conf) for v in submit_confidences]\n    \n    # make values for 'landmark' column\n    submit_landmarks = [\n        f'{i} {c:.8f}' for i, c in zip(submit_landmark_ids, submit_confidences)]\n    \n    return submit_ids, submit_landmarks","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Loading, Inference and Submission","metadata":{}},{"cell_type":"code","source":"model = torch.jit.load(str(MODEL_FILE))\n#print(model)\nsubmit_ids, submit_landmarks = get_similarity(model)\nsubmit_df = pd.DataFrame({'id': submit_ids, 'landmarks': submit_landmarks})\nsubmit_df.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualizing","metadata":{}},{"cell_type":"code","source":"submit_df = pd.read_csv('submission.csv')\nsubmit_df['landmark_id'] = submit_df['landmarks'].apply(lambda x: int(x.split()[0]))\nsubmit_df['confidence'] = submit_df['landmarks'].apply(lambda x: float(x.split()[1]))\ntrain_df = pd.read_csv(TRAIN_LABEL_FILE)\n\ndef get_image(path, name):\n    img = PIL.Image.open(path / name[0] / name[1] / name[2] / f'{name}.jpg')\n    if img.width > img.height:\n        img = img.resize((256, round(img.height / img.width * 256)))\n        new_img = PIL.Image.new(img.mode, (256, 256), (0, 0, 0))\n        new_img.paste(img, (0, (256 - img.height) // 2))\n    else:\n        img = img.resize((round(img.width / img.height * 256), 256))\n        new_img = PIL.Image.new(img.mode, (256, 256), (0, 0, 0))\n        new_img.paste(img, ((256 - img.width) // 2, 2))\n    return np.array(new_img)\n\nrows = 10\nfig = plt.figure(figsize=(15, 4 * rows))\nfor r in range(rows):\n    for c in range(3):\n        i = r * 3 + c\n        test_name, _, label, conf = submit_df.iloc[i].values\n        test_image = get_image(TEST_IMAGE_DIR, test_name)\n        train_name = train_df.query(f'landmark_id == {label}').iloc[0]['id']\n        train_image = get_image(TRAIN_IMAGE_DIR, train_name)\n        image = np.concatenate([test_image, train_image], axis=1)\n    \n        ax = fig.add_subplot(rows, 3, i + 1)        \n        ax.set_title(f'Label={label}, Confidence={conf:.2f}')\n        ax.axis('off')\n        ax.imshow(image)\nfig.tight_layout()","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}