{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"In this notebook we will see one of the possible ways to perform ensambling for this competition. \nThe submission should contain one single model files, which must be **jittable** by Pytorch (meaning that it can only contains other nn.Module or jittable functions).\n\nI will use a create a class *Ensemble* which can accept a varying number of different encoder. For this notebook, the results will be the just the average of the predictions.","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torchvision import models\nfrom torchvision import transforms","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-21T08:09:52.191213Z","iopub.execute_input":"2022-07-21T08:09:52.191626Z","iopub.status.idle":"2022-07-21T08:09:54.617414Z","shell.execute_reply.started":"2022-07-21T08:09:52.191592Z","shell.execute_reply":"2022-07-21T08:09:54.616124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In this notebook, I will be using different pre-trained EfficientNet models, but feel free to customize it to your needs","metadata":{}},{"cell_type":"code","source":"class EfficientNet(nn.Module):\n    def __init__(self, encoder_fn, resize, size):\n        super().__init__()\n        encoder = encoder_fn(pretrained=True)\n        encoder.classifier = nn.Sequential(\n            encoder.classifier,\n            nn.AdaptiveAvgPool1d(64)\n        )\n        self.encoder = encoder\n        self.resize = resize \n        self.size = size\n\n    def forward(self, x):\n        x = transforms.functional.resize(x, self.resize)\n        x = transforms.functional.center_crop(x, self.size)\n        x = x/255.\n        x = transforms.functional.normalize(x, mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n        return self.encoder(x)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T08:09:54.619635Z","iopub.execute_input":"2022-07-21T08:09:54.620553Z","iopub.status.idle":"2022-07-21T08:09:54.631146Z","shell.execute_reply.started":"2022-07-21T08:09:54.620501Z","shell.execute_reply":"2022-07-21T08:09:54.629949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we define the different variants for EfficientNet, along with their required input shape","metadata":{}},{"cell_type":"code","source":"encoders_fn = {\n    'b0': models.efficientnet_b0,\n    'b1': models.efficientnet_b1,\n    'b2': models.efficientnet_b2,\n    'b3': models.efficientnet_b3,\n    'b4': models.efficientnet_b4,\n    'b5': models.efficientnet_b5,\n    'b6': models.efficientnet_b6,\n    'b7': models.efficientnet_b7,\n}\nsizes = {\n    'b0': (256, 224), 'b1': (256, 240), 'b2': (288, 288), 'b3': (320, 300),\n    'b4': (384, 380), 'b5': (489, 456), 'b6': (561, 528), 'b7': (633, 600),\n}\n\nmodel = EfficientNet(encoder_fn=encoders_fn['b0'], resize=(sizes['b0'][0], sizes['b0'][0]), size=(sizes['b0'][0], sizes['b0'][1]))\nmodel.eval()\nmodel(torch.randn((1, 3, 256, 256))).shape","metadata":{"execution":{"iopub.status.busy":"2022-07-21T08:09:54.632519Z","iopub.execute_input":"2022-07-21T08:09:54.633104Z","iopub.status.idle":"2022-07-21T08:09:57.188167Z","shell.execute_reply.started":"2022-07-21T08:09:54.633062Z","shell.execute_reply":"2022-07-21T08:09:57.187159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Ensemble(nn.Module):\n    def __init__(self, encoders):\n        super().__init__()\n        for idx, encoder in enumerate(encoders):\n            setattr(self, f'encoder{idx}', encoder)\n        self.num_encoders = len(encoders)\n    \n    def forward(self, x):\n        y = []\n        for name, encoder in self.named_children():\n            print(name)\n            y.append(encoder(x))\n        y = torch.cat(y, dim=0)\n        y = torch.nn.functional.normalize(y)\n        y = y.mean(dim=0).unsqueeze(0)\n        y = torch.nn.functional.normalize(y)\n        return y","metadata":{"execution":{"iopub.status.busy":"2022-07-21T08:09:57.190469Z","iopub.execute_input":"2022-07-21T08:09:57.190820Z","iopub.status.idle":"2022-07-21T08:09:57.201060Z","shell.execute_reply.started":"2022-07-21T08:09:57.190790Z","shell.execute_reply":"2022-07-21T08:09:57.199835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we can build the ensemble by selecting a few encoders","metadata":{}},{"cell_type":"code","source":"selected_encoders = ['b4', 'b5', 'b6', 'b7']\nencoders = []\n\nfor encoder_name in selected_encoders:\n    size = sizes[encoder_name]\n    encoders.append(EfficientNet(encoder_fn=encoders_fn[encoder_name], resize=(size[0], size[0]), size=size))\n\nensemble = Ensemble(encoders)\nensemble.eval()\nensemble(torch.randn((1, 3, 256, 256))).shape","metadata":{"execution":{"iopub.status.busy":"2022-07-21T08:09:57.202467Z","iopub.execute_input":"2022-07-21T08:09:57.202804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"saved_model = torch.jit.script(ensemble)\nsaved_model.save('saved_model.pt')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"And that's it. Check out the inference notebook: https://www.kaggle.com/code/carloalbertobarbano/pytorch-ensemble-pretrained-baseline-inference","metadata":{}},{"cell_type":"code","source":"!ls -lh","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}