{"cells":[{"metadata":{},"cell_type":"markdown","source":"this notebook will load weights from a saved model and do the inference step for a submission.\n\nthis notebook (and the training notebook) copied from original work here - https://www.kaggle.com/debarshichanda/cassava-bitempered-logistic-loss\n\n## preparation steps\n\n### upload the timm module to load since internet must be off during submission\n- in order to use the timm module to load the proper model, you should download it to your local system and then upload to this notebook as an input.\n- download from here - https://pypi.org/project/timm/#files\n- upload using '+ Add data'\n\n### add the output from the saved training notebook as input in this notebook\n- use '+ Add data' to add 'output from a notebook' as an input\n- choose the name of the notebook you used to train the model (and set to save the model)"},{"metadata":{"trusted":true},"cell_type":"code","source":"# verify GPU?\n!nvidia-smi","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# !pip install timm from the uploaded files\n!pip install ../input/timmmodels/dist/timm-0.3.4.tar","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import os\nimport cv2\n\nimport numpy as np\nimport pandas as pd\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\nimport torchvision\nfrom torchvision import models\nfrom torch.utils.data import DataLoader, Dataset\nfrom torch.cuda import amp\n\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\n\nimport timm","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ROOT_DIR = \"../input/cassava-leaf-disease-classification\"\nTEST_DIR = \"../input/cassava-leaf-disease-classification/test_images\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# just configuration items used throughout notebook\nclass CFG:\n    model_name = 'tf_efficientnet_b4_ns'\n    img_size = 512\n    loadmodelpath = '/kaggle/input/cassava-bitempered-logistic-loss/bitemp-01.pth'\n    num_classes = 5\n    device = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class CassavaLeafDataset(nn.Module):\n    def __init__(self, root_dir, df, transforms=None):\n        self.root_dir = root_dir\n        self.df = df\n        self.transforms = transforms\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, index):\n        img_path = os.path.join(self.root_dir, self.df.iloc[index, 0])\n        img = cv2.imread(img_path)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        label = self.df.iloc[index, 1]\n        \n        if self.transforms:\n            img = self.transforms(image=img)[\"image\"]\n            \n        return img, label","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# these are left in from the training notebook\n# 'valid' is used in inference, but not sure if that is necessary\ndata_transforms = {\n    \"train\": A.Compose([\n        A.RandomResizedCrop(CFG.img_size, CFG.img_size),\n        A.Transpose(p=0.5),\n        A.HorizontalFlip(p=0.5),\n        A.VerticalFlip(p=0.5),\n        A.ShiftScaleRotate(p=0.5),\n        A.HueSaturationValue(\n                hue_shift_limit=0.2, \n                sat_shift_limit=0.2, \n                val_shift_limit=0.2, \n                p=0.5\n            ),\n        A.RandomBrightnessContrast(\n                brightness_limit=(-0.1,0.1), \n                contrast_limit=(-0.1, 0.1), \n                p=0.5\n            ),\n        A.Normalize(\n                mean=[0.485, 0.456, 0.406], \n                std=[0.229, 0.224, 0.225], \n                max_pixel_value=255.0, \n                p=1.0\n            ),\n        A.CoarseDropout(p=0.5),\n        A.Cutout(p=0.5),\n        ToTensorV2()], p=1.),\n    \n    \"valid\": A.Compose([\n        A.CenterCrop(CFG.img_size, CFG.img_size, p=0.8),\n        A.Resize(CFG.img_size, CFG.img_size),\n        A.Normalize(\n                mean=[0.485, 0.456, 0.406], \n                std=[0.229, 0.224, 0.225], \n                max_pixel_value=255.0, \n                p=1.0\n            ),\n        ToTensorV2()], p=1.)\n}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# load the model definition\nmodel = timm.create_model(CFG.model_name, pretrained=False)\nnum_features = model.classifier.in_features\nmodel.classifier = nn.Linear(num_features, CFG.num_classes)\nmodel.to(CFG.device);","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# load the model weights from the input and then set it for inference\nmodel = torch.load(CFG.loadmodelpath)\nmodel.eval()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# set up for the test submission\n\n# set the folder used in the Dataset Class\nT_DIR = TEST_DIR\n# read the submission sample into a dataframe\nt_df = pd.read_csv(f\"{ROOT_DIR}/sample_submission.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# instantiate the dataset and the dataloader for inference processing\nt_data = CassavaLeafDataset(T_DIR, t_df, transforms=data_transforms[\"valid\"])\nt_loader = DataLoader(dataset=t_data, batch_size=1, num_workers=4, pin_memory=True, shuffle=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# make a copy of the dataframe used to read the inputs\n# not sure about modifying dataframe used in the dataloader\nsubmit_df = pd.DataFrame(t_df).copy(deep=True)\n\n# iterate through all the entries fed from the dataloader\nfor i, (inputs, _) in enumerate(t_loader):\n    inputs = inputs.to(CFG.device)\n    # make the prediction\n    outputs = model(inputs).detach().cpu().numpy()\n    # choose the largest to store as the label\n    pred_label = np.argmax(outputs)\n    # add that to the dataframe which will be saved as the submission\n    submit_df.iloc[i] = [t_df.iloc[i]['image_id'], pred_label]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# save the submission dataframe as the submission file\nsubmit_df.to_csv(\"/kaggle/working/submission.csv\", index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}