{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":13836,"databundleVersionId":1718836,"sourceType":"competition"},{"sourceId":8271301,"sourceType":"datasetVersion","datasetId":4856757},{"sourceId":779,"sourceType":"modelInstanceVersion","modelInstanceId":646}],"dockerImageVersionId":30699,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Intelisys Inference Notebook\nThis notebook is dedicated towards collating the models to be used to create the submission document for the kaggle competition.","metadata":{}},{"cell_type":"markdown","source":"# Imports and Variable Declarations","metadata":{}},{"cell_type":"code","source":"! pip install \"/kaggle/input/efficientnet-b5/torch_summary-1.4.5-py3-none-any.whl\"","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:01:30.382986Z","iopub.execute_input":"2024-04-30T13:01:30.383813Z","iopub.status.idle":"2024-04-30T13:02:02.487298Z","shell.execute_reply.started":"2024-04-30T13:01:30.383772Z","shell.execute_reply":"2024-04-30T13:02:02.486114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install \"/kaggle/input/efficientnet-b5/timm-0.9.16-py3-none-any.whl\"","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:02:02.489314Z","iopub.execute_input":"2024-04-30T13:02:02.489651Z","iopub.status.idle":"2024-04-30T13:02:34.628726Z","shell.execute_reply.started":"2024-04-30T13:02:02.489618Z","shell.execute_reply":"2024-04-30T13:02:34.627569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install /kaggle/input/efficientnet-b5/tez-0.7.2-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:02:34.630026Z","iopub.execute_input":"2024-04-30T13:02:34.630323Z","iopub.status.idle":"2024-04-30T13:03:06.841556Z","shell.execute_reply.started":"2024-04-30T13:02:34.630292Z","shell.execute_reply":"2024-04-30T13:03:06.840427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install \"/kaggle/input/efficientnet-b5/torch_summary-1.4.5-py3-none-any.whl\" ","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:03:06.844051Z","iopub.execute_input":"2024-04-30T13:03:06.844365Z","iopub.status.idle":"2024-04-30T13:03:38.815538Z","shell.execute_reply.started":"2024-04-30T13:03:06.844328Z","shell.execute_reply":"2024-04-30T13:03:38.814336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install /kaggle/input/efficientnet-b5/efficientnet_pytorch-0.7.1-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:03:38.817139Z","iopub.execute_input":"2024-04-30T13:03:38.817458Z","iopub.status.idle":"2024-04-30T13:04:11.055959Z","shell.execute_reply.started":"2024-04-30T13:03:38.817427Z","shell.execute_reply":"2024-04-30T13:04:11.054951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport datetime\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\nimport tensorflow as tf\nfrom tensorflow.keras import models, layers\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\nfrom tensorflow.keras.applications import ResNet50, DenseNet121, EfficientNetB0\nfrom keras.optimizers import Adam\n\n# ignoring warnings\nimport warnings\nwarnings.simplefilter(\"ignore\")\n\nimport os, cv2, json\nfrom PIL import Image\n\n\nimport matplotlib.pyplot as plt\nfrom torch.utils.data import Dataset,DataLoader\nfrom torch.utils.data.sampler import SequentialSampler, RandomSampler\nfrom  torch.cuda.amp import autocast, GradScaler\n\nimport sklearn\nimport warnings\nimport joblib\nfrom sklearn.metrics import roc_auc_score, log_loss\nfrom sklearn import metrics\nimport warnings\nimport cv2\nimport pydicom\nimport timm #from efficientnet_pytorch import EfficientNet\nfrom scipy.ndimage.interpolation import zoom\nfrom sklearn.metrics import log_loss","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-30T13:04:11.057395Z","iopub.execute_input":"2024-04-30T13:04:11.057715Z","iopub.status.idle":"2024-04-30T13:04:11.068623Z","shell.execute_reply.started":"2024-04-30T13:04:11.057685Z","shell.execute_reply":"2024-04-30T13:04:11.067654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from albumentations import (\n    HorizontalFlip, VerticalFlip, ShiftScaleRotate, CLAHE, RandomRotate90,\n    Transpose, ShiftScaleRotate, Blur, OpticalDistortion, GridDistortion, HueSaturationValue,\n     GaussNoise, MotionBlur, MedianBlur, RandomResizedCrop, RandomBrightnessContrast, Flip, OneOf, Compose, Normalize, CoarseDropout, ShiftScaleRotate, CenterCrop, Resize\n)\n\nfrom albumentations.pytorch import ToTensorV2\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:11.069784Z","iopub.execute_input":"2024-04-30T13:04:11.070111Z","iopub.status.idle":"2024-04-30T13:04:11.083706Z","shell.execute_reply.started":"2024-04-30T13:04:11.070080Z","shell.execute_reply":"2024-04-30T13:04:11.082965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torchsummary import summary\nimport torch\nimport torchvision.models as models  # Or import your model class\n","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:11.085169Z","iopub.execute_input":"2024-04-30T13:04:11.085844Z","iopub.status.idle":"2024-04-30T13:04:11.098298Z","shell.execute_reply.started":"2024-04-30T13:04:11.085812Z","shell.execute_reply":"2024-04-30T13:04:11.097526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport albumentations\nimport pandas as pd\n\nimport tez\nfrom tez.datasets import ImageDataset\nfrom tez.callbacks import EarlyStopping\nfrom tensorflow.keras.callbacks import ModelCheckpoint\n\nimport torch\nimport torch.nn as nn\nfrom torch.nn import functional as F\n\nfrom efficientnet_pytorch import EfficientNet\nfrom sklearn import metrics, model_selection, preprocessing\nfrom tez import Tez, TezConfig","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:11.099649Z","iopub.execute_input":"2024-04-30T13:04:11.099915Z","iopub.status.idle":"2024-04-30T13:04:11.109538Z","shell.execute_reply.started":"2024-04-30T13:04:11.099892Z","shell.execute_reply":"2024-04-30T13:04:11.108737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from glob import glob\nfrom sklearn.model_selection import GroupKFold, StratifiedKFold\nimport cv2\nfrom skimage import io\nimport torch\nfrom torch import nn\nimport os\nfrom datetime import datetime\nimport time\nimport random\nimport cv2\nimport torchvision\nfrom torchvision import transforms\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\n\nimport matplotlib.pyplot as plt\nfrom torch.utils.data import Dataset,DataLoader\nfrom torch.utils.data.sampler import SequentialSampler, RandomSampler\nfrom  torch.cuda.amp import autocast, GradScaler\n\nimport sklearn\nimport warnings\nimport joblib\nfrom sklearn.metrics import roc_auc_score, log_loss\nfrom sklearn import metrics\nimport warnings\nimport cv2\nimport pydicom\nimport timm #from efficientnet_pytorch import EfficientNet\nfrom sklearn.metrics import log_loss","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:11.113542Z","iopub.execute_input":"2024-04-30T13:04:11.113834Z","iopub.status.idle":"2024-04-30T13:04:11.121565Z","shell.execute_reply.started":"2024-04-30T13:04:11.113811Z","shell.execute_reply":"2024-04-30T13:04:11.120718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"WORK_DIR = '../input/cassava-leaf-disease-classification'\nTARGET_SIZE = 224\nos.listdir(WORK_DIR)","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:11.122823Z","iopub.execute_input":"2024-04-30T13:04:11.123384Z","iopub.status.idle":"2024-04-30T13:04:11.135651Z","shell.execute_reply.started":"2024-04-30T13:04:11.123347Z","shell.execute_reply":"2024-04-30T13:04:11.134634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CFG = {\n    'fold_num': 5,\n    'seed': 719,\n    'model_arch': 'tf_efficientnet_b4_ns',\n    'img_size': 512,\n    'epochs': 10,\n    'train_bs': 32,\n    'valid_bs': 32,\n    'lr': 1e-4,\n    'num_workers': 4,\n    'accum_iter': 1, # suppoprt to do batch accumulation for backprop with effectively larger batch size\n    'verbose_step': 1,\n    'device': 'cuda:0',\n    'tta': 3,\n    'used_epochs': [6,7,8,9],\n    'weights': [1,1,1,1]\n}\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:11.136734Z","iopub.execute_input":"2024-04-30T13:04:11.137059Z","iopub.status.idle":"2024-04-30T13:04:11.143227Z","shell.execute_reply.started":"2024-04-30T13:04:11.137029Z","shell.execute_reply":"2024-04-30T13:04:11.142343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_images_path = \"/kaggle/input/cassava-leaf-disease-classification/test_images\"","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:11.144520Z","iopub.execute_input":"2024-04-30T13:04:11.144819Z","iopub.status.idle":"2024-04-30T13:04:11.157041Z","shell.execute_reply.started":"2024-04-30T13:04:11.144797Z","shell.execute_reply":"2024-04-30T13:04:11.156196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Helper Functions","metadata":{}},{"cell_type":"code","source":"def seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = True\n    \ndef get_img(path):\n    im_bgr = cv2.imread(path)\n    im_rgb = im_bgr[:, :, ::-1]\n    #print(im_rgb)\n    return im_rgb","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:11.158128Z","iopub.execute_input":"2024-04-30T13:04:11.158410Z","iopub.status.idle":"2024-04-30T13:04:11.167473Z","shell.execute_reply.started":"2024-04-30T13:04:11.158382Z","shell.execute_reply":"2024-04-30T13:04:11.166766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_inference_transforms():\n    return Compose([\n            RandomResizedCrop(CFG['img_size'], CFG['img_size']),\n            Transpose(p=0.5),\n            HorizontalFlip(p=0.5),\n            VerticalFlip(p=0.5),\n            HueSaturationValue(hue_shift_limit=0.2, sat_shift_limit=0.2, val_shift_limit=0.2, p=0.5),\n            RandomBrightnessContrast(brightness_limit=(-0.1,0.1), contrast_limit=(-0.1, 0.1), p=0.5),\n            Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225], max_pixel_value=255.0, p=1.0),\n            ToTensorV2(p=1.0),\n        ], p=1.)","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:11.168474Z","iopub.execute_input":"2024-04-30T13:04:11.168802Z","iopub.status.idle":"2024-04-30T13:04:11.178301Z","shell.execute_reply.started":"2024-04-30T13:04:11.168780Z","shell.execute_reply":"2024-04-30T13:04:11.177531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Loading","metadata":{}},{"cell_type":"code","source":"with open(os.path.join(WORK_DIR, \"label_num_to_disease_map.json\")) as file:\n    print(json.dumps(json.loads(file.read()), indent=4))","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:11.179543Z","iopub.execute_input":"2024-04-30T13:04:11.179808Z","iopub.status.idle":"2024-04-30T13:04:11.189988Z","shell.execute_reply.started":"2024-04-30T13:04:11.179786Z","shell.execute_reply":"2024-04-30T13:04:11.189137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the file directory location\nbase_dir = \"../input/cassava-leaf-disease-classification\"","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:11.191056Z","iopub.execute_input":"2024-04-30T13:04:11.191313Z","iopub.status.idle":"2024-04-30T13:04:11.197730Z","shell.execute_reply.started":"2024-04-30T13:04:11.191292Z","shell.execute_reply":"2024-04-30T13:04:11.196856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss = pd.read_csv(os.path.join(WORK_DIR, \"sample_submission.csv\"))\nss","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:11.198664Z","iopub.execute_input":"2024-04-30T13:04:11.198944Z","iopub.status.idle":"2024-04-30T13:04:11.217163Z","shell.execute_reply.started":"2024-04-30T13:04:11.198918Z","shell.execute_reply":"2024-04-30T13:04:11.216271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define Model Classes","metadata":{}},{"cell_type":"code","source":"class CassvaImgClassifier(nn.Module):\n    def __init__(self, model_arch, n_class, pretrained=False):\n        super().__init__()\n        self.model = timm.create_model(model_arch, pretrained=False)\n        n_features = self.model.classifier.in_features\n        self.model.classifier = nn.Linear(n_features, n_class)\n        \n    def forward(self, x):\n        x = self.model(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:11.218165Z","iopub.execute_input":"2024-04-30T13:04:11.218409Z","iopub.status.idle":"2024-04-30T13:04:11.224798Z","shell.execute_reply.started":"2024-04-30T13:04:11.218388Z","shell.execute_reply":"2024-04-30T13:04:11.223892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CassavaDataset(Dataset):\n    def __init__(\n        self, df, data_root, transforms=None, output_label=True\n    ):\n        \n        super().__init__()\n        self.df = df.reset_index(drop=True).copy()\n        self.transforms = transforms\n        self.data_root = data_root\n        self.output_label = output_label\n    \n    def __len__(self):\n        return self.df.shape[0]\n    \n    def __getitem__(self, index: int):\n        \n        # get labels\n        if self.output_label:\n            target = self.df.iloc[index]['label']\n          \n        path = \"{}/{}\".format(self.data_root, self.df.iloc[index]['image_id'])\n        \n        img  = get_img(path)\n        \n        if self.transforms:\n            img = self.transforms(image=img)['image']\n            \n        # do label smoothing\n        if self.output_label == True:\n            return img, target\n        else:\n            return img","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:11.225996Z","iopub.execute_input":"2024-04-30T13:04:11.226333Z","iopub.status.idle":"2024-04-30T13:04:11.239381Z","shell.execute_reply.started":"2024-04-30T13:04:11.226304Z","shell.execute_reply":"2024-04-30T13:04:11.238542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ResNext(nn.Module):\n    def __init__(self, model_name='resnext50_32x4d', pretrained=False):\n        super().__init__()\n        self.model = timm.create_model(model_name, pretrained=pretrained)\n        n_features = self.model.fc.in_features\n        self.model.fc = nn.Linear(n_features, 5)\n\n    def forward(self, x):\n        x = self.model(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:11.240397Z","iopub.execute_input":"2024-04-30T13:04:11.240694Z","iopub.status.idle":"2024-04-30T13:04:11.250609Z","shell.execute_reply.started":"2024-04-30T13:04:11.240670Z","shell.execute_reply":"2024-04-30T13:04:11.249807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  EfficientNet B4 Model - K Folds\n","metadata":{}},{"cell_type":"markdown","source":"## Load Data","metadata":{}},{"cell_type":"code","source":"submission = pd.read_csv('../input/cassava-leaf-disease-classification/sample_submission.csv')\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:11.251529Z","iopub.execute_input":"2024-04-30T13:04:11.251806Z","iopub.status.idle":"2024-04-30T13:04:11.266760Z","shell.execute_reply.started":"2024-04-30T13:04:11.251784Z","shell.execute_reply":"2024-04-30T13:04:11.265914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = get_img('../input/cassava-leaf-disease-classification/train_images/1000015157.jpg')\nplt.imshow(img)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:11.268069Z","iopub.execute_input":"2024-04-30T13:04:11.268575Z","iopub.status.idle":"2024-04-30T13:04:11.668945Z","shell.execute_reply.started":"2024-04-30T13:04:11.268544Z","shell.execute_reply":"2024-04-30T13:04:11.667986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.DataFrame() ","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:11.670140Z","iopub.execute_input":"2024-04-30T13:04:11.670498Z","iopub.status.idle":"2024-04-30T13:04:11.675461Z","shell.execute_reply.started":"2024-04-30T13:04:11.670459Z","shell.execute_reply":"2024-04-30T13:04:11.674620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.DataFrame()\ntest['image_id'] = list(os.listdir(test_images_path))\ntest_ds = CassavaDataset(test, test_images_path, transforms=get_inference_transforms(), output_label=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:11.676412Z","iopub.execute_input":"2024-04-30T13:04:11.676686Z","iopub.status.idle":"2024-04-30T13:04:11.691989Z","shell.execute_reply.started":"2024-04-30T13:04:11.676662Z","shell.execute_reply":"2024-04-30T13:04:11.691217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ds.__len__()","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:11.693631Z","iopub.execute_input":"2024-04-30T13:04:11.693981Z","iopub.status.idle":"2024-04-30T13:04:11.701960Z","shell.execute_reply.started":"2024-04-30T13:04:11.693948Z","shell.execute_reply":"2024-04-30T13:04:11.701103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set up DataLoader\ntest_loader = DataLoader(\n    test_ds,\n    batch_size=CFG['valid_bs'],\n    num_workers=CFG['num_workers'],\n    shuffle=False,  # No need to shuffle during inference\n    pin_memory=False\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:11.703332Z","iopub.execute_input":"2024-04-30T13:04:11.703621Z","iopub.status.idle":"2024-04-30T13:04:11.711163Z","shell.execute_reply.started":"2024-04-30T13:04:11.703573Z","shell.execute_reply":"2024-04-30T13:04:11.710381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_loader","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:11.718070Z","iopub.execute_input":"2024-04-30T13:04:11.718307Z","iopub.status.idle":"2024-04-30T13:04:11.724020Z","shell.execute_reply.started":"2024-04-30T13:04:11.718286Z","shell.execute_reply":"2024-04-30T13:04:11.723152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load Model","metadata":{}},{"cell_type":"code","source":"state_dict = torch.load('/kaggle/input/efficientnet-b5/tf_efficientnet_b4_ns_fold_4_4', map_location=torch.device('cpu'))","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:11.725176Z","iopub.execute_input":"2024-04-30T13:04:11.725508Z","iopub.status.idle":"2024-04-30T13:04:11.816995Z","shell.execute_reply.started":"2024-04-30T13:04:11.725479Z","shell.execute_reply":"2024-04-30T13:04:11.816024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Instantiate the corresponding model (must match the architecture of the state dict)\nb4_model = CassvaImgClassifier(CFG['model_arch'], 5, pretrained=True).to(CFG[\"device\"])  # or your custom model class\n\n# Load the state dictionary into the model\nb4_model.load_state_dict(state_dict)","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:11.818220Z","iopub.execute_input":"2024-04-30T13:04:11.818504Z","iopub.status.idle":"2024-04-30T13:04:12.319107Z","shell.execute_reply.started":"2024-04-30T13:04:11.818480Z","shell.execute_reply":"2024-04-30T13:04:12.318153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"b4_model.eval()","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:12.320345Z","iopub.execute_input":"2024-04-30T13:04:12.320713Z","iopub.status.idle":"2024-04-30T13:04:12.339334Z","shell.execute_reply.started":"2024-04-30T13:04:12.320685Z","shell.execute_reply":"2024-04-30T13:04:12.338642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"summary(b4_model, (3, 512, 512))\n","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:12.340817Z","iopub.execute_input":"2024-04-30T13:04:12.341087Z","iopub.status.idle":"2024-04-30T13:04:12.452504Z","shell.execute_reply.started":"2024-04-30T13:04:12.341064Z","shell.execute_reply":"2024-04-30T13:04:12.451627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predictions","metadata":{}},{"cell_type":"code","source":"# Inference\ntest_preds = []\n\nwith torch.no_grad():  # Disable gradient tracking for inference\n    for batch in test_loader:\n        inputs = batch  # Assuming the data loader yields images only\n        outputs = b4_model(inputs.to(torch.device(CFG['device'])))  # Run the model\n        probabilities = F.softmax(outputs, dim=1)  # Convert to probabilities\n        test_preds.append(probabilities.cpu().numpy())  # Store results\n\n# Flatten list of predictions and perform further processing\ntest_preds = np.concatenate(test_preds, axis=0)  # Combine predictions into a single array\n\n# Example: converting probabilities to labels\npredicted_labels = np.argmax(test_preds, axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:12.453911Z","iopub.execute_input":"2024-04-30T13:04:12.454549Z","iopub.status.idle":"2024-04-30T13:04:12.718069Z","shell.execute_reply.started":"2024-04-30T13:04:12.454513Z","shell.execute_reply":"2024-04-30T13:04:12.716938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"b4_predictions = probabilities","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:12.719498Z","iopub.execute_input":"2024-04-30T13:04:12.720879Z","iopub.status.idle":"2024-04-30T13:04:12.725876Z","shell.execute_reply.started":"2024-04-30T13:04:12.720839Z","shell.execute_reply":"2024-04-30T13:04:12.724941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['label'] = predicted_labels\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:12.727025Z","iopub.execute_input":"2024-04-30T13:04:12.727294Z","iopub.status.idle":"2024-04-30T13:04:12.743429Z","shell.execute_reply.started":"2024-04-30T13:04:12.727262Z","shell.execute_reply":"2024-04-30T13:04:12.742543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PreTrained CropNet","metadata":{}},{"cell_type":"markdown","source":"## Load Model","metadata":{}},{"cell_type":"code","source":"import os\nos.environ['TF_USE_LEGACY_KERAS'] = '1'\nimport re\nfrom datetime import datetime\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nfrom kaggle_datasets import KaggleDatasets\n\nimport tensorflow as tf\n\nimport tf_keras as keras\nimport kagglehub\n\n# Download latest version\nfrom tensorflow.keras.applications.efficientnet import EfficientNetB0\nfrom tensorflow.keras.applications.efficientnet import preprocess_input as effnet_preprocess_input\nimport tensorflow_datasets as tfds\nimport tensorflow_hub as hub","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:12.744692Z","iopub.execute_input":"2024-04-30T13:04:12.745017Z","iopub.status.idle":"2024-04-30T13:04:12.755358Z","shell.execute_reply.started":"2024-04-30T13:04:12.744994Z","shell.execute_reply":"2024-04-30T13:04:12.754519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = kagglehub.model_download(\"google/cropnet/tensorFlow2/classifier-cassava-disease-v1\")\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:12.756369Z","iopub.execute_input":"2024-04-30T13:04:12.756654Z","iopub.status.idle":"2024-04-30T13:04:13.709753Z","shell.execute_reply.started":"2024-04-30T13:04:12.756625Z","shell.execute_reply":"2024-04-30T13:04:13.708940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.environ[\"TFHUB_CACHE_DIR\"] = \"/kaggle/working/cassava-layer/\"","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:13.710950Z","iopub.execute_input":"2024-04-30T13:04:13.711226Z","iopub.status.idle":"2024-04-30T13:04:13.715487Z","shell.execute_reply.started":"2024-04-30T13:04:13.711202Z","shell.execute_reply":"2024-04-30T13:04:13.714509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cassava = hub.KerasLayer(path)\nmodel = keras.Sequential([keras.Input(shape=(224,224,3)),\n                             cassava])","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:13.716469Z","iopub.execute_input":"2024-04-30T13:04:13.716732Z","iopub.status.idle":"2024-04-30T13:04:16.778312Z","shell.execute_reply.started":"2024-04-30T13:04:13.716709Z","shell.execute_reply":"2024-04-30T13:04:16.777533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.load_weights(\"/kaggle/input/efficientnet-b5/preTrainedCropnet5.keras\")","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:16.779439Z","iopub.execute_input":"2024-04-30T13:04:16.779733Z","iopub.status.idle":"2024-04-30T13:04:18.170771Z","shell.execute_reply.started":"2024-04-30T13:04:16.779708Z","shell.execute_reply":"2024-04-30T13:04:18.169716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\nBATCH_SIZE = 32\nIMAGE_SIZE = [512, 512]","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:18.172130Z","iopub.execute_input":"2024-04-30T13:04:18.173891Z","iopub.status.idle":"2024-04-30T13:04:18.178714Z","shell.execute_reply.started":"2024-04-30T13:04:18.173847Z","shell.execute_reply":"2024-04-30T13:04:18.177803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def _parse_function(proto):\n    # feature_description needs to be defined since datasets use graph-execution\n    # - its used to build their shape and type signature\n    feature_description = {\n        'image': tf.io.FixedLenFeature([], tf.string, default_value=''),\n        'image_name': tf.io.FixedLenFeature([], tf.string, default_value=''),\n        'target': tf.io.FixedLenFeature([], tf.int64, default_value=-1)\n    }\n\n    parsed_features = tf.io.parse_single_example(proto, feature_description)\n    image = tf.image.decode_jpeg(parsed_features['image'], channels=3)\n    image = tf.cast(image, tf.float32) # :: [0.0, 255.0]\n    image = tf.reshape(image, [*IMAGE_SIZE, 3])\n    target = tf.one_hot(parsed_features['target'], depth=5)\n    image_id = parsed_features['image_name']\n    \n    return image, target, image_id","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:18.180028Z","iopub.execute_input":"2024-04-30T13:04:18.180307Z","iopub.status.idle":"2024-04-30T13:04:18.192043Z","shell.execute_reply.started":"2024-04-30T13:04:18.180284Z","shell.execute_reply":"2024-04-30T13:04:18.191092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def _preprocess_fn(image, label, image_id):\n    image = image / 255.0\n    image = tf.image.resize(image, (224, 224))\n    label = tf.concat([label, [0]], axis=0)\n    return image, label, image_id","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:18.193196Z","iopub.execute_input":"2024-04-30T13:04:18.193462Z","iopub.status.idle":"2024-04-30T13:04:18.202883Z","shell.execute_reply.started":"2024-04-30T13:04:18.193439Z","shell.execute_reply":"2024-04-30T13:04:18.202000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_dataset(tfrecords_fnames):\n    raw_ds = tf.data.TFRecordDataset(tfrecords_fnames, num_parallel_reads=AUTO)\n    parsed_ds = raw_ds.map(_parse_function, num_parallel_calls=AUTO)\n    parsed_ds = parsed_ds.map(_preprocess_fn, num_parallel_calls=AUTO)\n    return parsed_ds","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:18.204053Z","iopub.execute_input":"2024-04-30T13:04:18.204397Z","iopub.status.idle":"2024-04-30T13:04:18.215303Z","shell.execute_reply.started":"2024-04-30T13:04:18.204365Z","shell.execute_reply":"2024-04-30T13:04:18.214427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_valid_ds(valid_fnames):\n    ds = load_dataset(valid_fnames)\n    ds = ds.batch(BATCH_SIZE).prefetch(AUTO)\n    return ds","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:18.216496Z","iopub.execute_input":"2024-04-30T13:04:18.216837Z","iopub.status.idle":"2024-04-30T13:04:18.230510Z","shell.execute_reply.started":"2024-04-30T13:04:18.216806Z","shell.execute_reply":"2024-04-30T13:04:18.229716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TEST_PATH = '../input/cassava-leaf-disease-classification/test_tfrecords/'\nvalid_fnames = [TEST_PATH + fname for fname in os.listdir(TEST_PATH)]","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:18.232296Z","iopub.execute_input":"2024-04-30T13:04:18.232805Z","iopub.status.idle":"2024-04-30T13:04:18.241621Z","shell.execute_reply.started":"2024-04-30T13:04:18.232772Z","shell.execute_reply":"2024-04-30T13:04:18.240784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ds = build_valid_ds(valid_fnames)","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:18.242755Z","iopub.execute_input":"2024-04-30T13:04:18.243049Z","iopub.status.idle":"2024-04-30T13:04:18.461970Z","shell.execute_reply.started":"2024-04-30T13:04:18.243025Z","shell.execute_reply":"2024-04-30T13:04:18.461026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = model.predict(test_ds)\nprint(preds)\nlabels = tf.argmax(preds, axis=-1)\nlabels = labels.numpy()","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:18.463224Z","iopub.execute_input":"2024-04-30T13:04:18.463567Z","iopub.status.idle":"2024-04-30T13:04:19.184105Z","shell.execute_reply.started":"2024-04-30T13:04:18.463535Z","shell.execute_reply":"2024-04-30T13:04:19.183201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ds = build_valid_ds(valid_fnames)","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:19.185203Z","iopub.execute_input":"2024-04-30T13:04:19.185507Z","iopub.status.idle":"2024-04-30T13:04:19.237937Z","shell.execute_reply.started":"2024-04-30T13:04:19.185481Z","shell.execute_reply":"2024-04-30T13:04:19.237011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"names = []\nfor item in test_ds:\n    names.append(item[2].numpy())\nnames = np.concatenate(names)\nnames = [name.decode() for name in names]","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:19.239448Z","iopub.execute_input":"2024-04-30T13:04:19.239745Z","iopub.status.idle":"2024-04-30T13:04:19.283580Z","shell.execute_reply.started":"2024-04-30T13:04:19.239720Z","shell.execute_reply":"2024-04-30T13:04:19.282792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cropnet_predictions = preds","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:19.284664Z","iopub.execute_input":"2024-04-30T13:04:19.285816Z","iopub.status.idle":"2024-04-30T13:04:19.289941Z","shell.execute_reply.started":"2024-04-30T13:04:19.285791Z","shell.execute_reply":"2024-04-30T13:04:19.288882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = pd.DataFrame({'image_id':names, 'label':labels})\nsubmission_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:19.291018Z","iopub.execute_input":"2024-04-30T13:04:19.291305Z","iopub.status.idle":"2024-04-30T13:04:19.302361Z","shell.execute_reply.started":"2024-04-30T13:04:19.291280Z","shell.execute_reply":"2024-04-30T13:04:19.301514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:19.303567Z","iopub.execute_input":"2024-04-30T13:04:19.303965Z","iopub.status.idle":"2024-04-30T13:04:19.315471Z","shell.execute_reply.started":"2024-04-30T13:04:19.303939Z","shell.execute_reply":"2024-04-30T13:04:19.314637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Resnext 50","metadata":{}},{"cell_type":"code","source":"state_dict = torch.load('/kaggle/input/efficientnet-b5/resnext_model.pth', map_location = torch.device('cpu'))","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:19.316434Z","iopub.execute_input":"2024-04-30T13:04:19.316696Z","iopub.status.idle":"2024-04-30T13:04:19.381880Z","shell.execute_reply.started":"2024-04-30T13:04:19.316673Z","shell.execute_reply":"2024-04-30T13:04:19.381070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"resnext_model = ResNext(pretrained = False).to(CFG[\"device\"])\n\nresnext_model.load_state_dict(state_dict)","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:19.383307Z","iopub.execute_input":"2024-04-30T13:04:19.384023Z","iopub.status.idle":"2024-04-30T13:04:19.914186Z","shell.execute_reply.started":"2024-04-30T13:04:19.383988Z","shell.execute_reply":"2024-04-30T13:04:19.913239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"resnext_model.eval()","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:19.915470Z","iopub.execute_input":"2024-04-30T13:04:19.916851Z","iopub.status.idle":"2024-04-30T13:04:19.929004Z","shell.execute_reply.started":"2024-04-30T13:04:19.916821Z","shell.execute_reply":"2024-04-30T13:04:19.928125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"summary(resnext_model, (3, 256, 256))","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:19.930132Z","iopub.execute_input":"2024-04-30T13:04:19.930434Z","iopub.status.idle":"2024-04-30T13:04:20.001076Z","shell.execute_reply.started":"2024-04-30T13:04:19.930411Z","shell.execute_reply":"2024-04-30T13:04:20.000189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def inference(model, state, test_loader, device):\n    model.to(torch.device(CFG['device']))\n    tk0 = tqdm(enumerate(test_loader), total=len(test_loader))\n    probs = []\n    for i, (images) in tk0:\n        images = images.to(device)\n        avg_preds = []\n        model.load_state_dict(state['model'])\n        model.eval()\n        with torch.no_grad():\n            y_preds = model(images)\n        avg_preds.append(y_preds.softmax(1).to('cpu').numpy())\n        avg_preds = np.mean(avg_preds, axis=0)\n        probs.append(avg_preds)\n    probs = np.concatenate(probs)\n    return probs\n\nstates = torch.load('/kaggle/input/efficientnet-b5/resnext50_32x4d_fold1_best.pth')\nresnext_predictions = inference(resnext_model, states, test_loader,CFG['device'])\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:20.002317Z","iopub.execute_input":"2024-04-30T13:04:20.002620Z","iopub.status.idle":"2024-04-30T13:04:20.428815Z","shell.execute_reply.started":"2024-04-30T13:04:20.002574Z","shell.execute_reply":"2024-04-30T13:04:20.427578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission Code","metadata":{}},{"cell_type":"markdown","source":"Get Probabilities from CropNet Model","metadata":{}},{"cell_type":"code","source":"# Drop the last element using slicing\nc_array = cropnet_predictions[:, :-1]\ncropnet_predictions_final = c_array\ncropnet_predictions_final","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:20.430482Z","iopub.execute_input":"2024-04-30T13:04:20.431276Z","iopub.status.idle":"2024-04-30T13:04:20.439179Z","shell.execute_reply.started":"2024-04-30T13:04:20.431230Z","shell.execute_reply":"2024-04-30T13:04:20.438169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Get Probabilites from EfficientNet B4 Model","metadata":{}},{"cell_type":"code","source":"tensor_gpu = b4_predictions\n\n# Move the tensor to the CPU\ntensor_cpu = tensor_gpu.cpu()\n\n# Convert to NumPy array\nb4_predictions_final = tensor_cpu.numpy()\n\nb4_predictions","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:20.440575Z","iopub.execute_input":"2024-04-30T13:04:20.440892Z","iopub.status.idle":"2024-04-30T13:04:20.456881Z","shell.execute_reply.started":"2024-04-30T13:04:20.440865Z","shell.execute_reply":"2024-04-30T13:04:20.455854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Get Probabilities from Resnext50 Model","metadata":{}},{"cell_type":"code","source":"resnext_predictions_final = resnext_predictions","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:20.458112Z","iopub.execute_input":"2024-04-30T13:04:20.458386Z","iopub.status.idle":"2024-04-30T13:04:20.464430Z","shell.execute_reply.started":"2024-04-30T13:04:20.458361Z","shell.execute_reply":"2024-04-30T13:04:20.463620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Combine Probability Predictions","metadata":{}},{"cell_type":"code","source":"result = cropnet_predictions_final + b4_predictions_final + resnext_predictions_final\n\nprint(result)","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:20.465694Z","iopub.execute_input":"2024-04-30T13:04:20.466239Z","iopub.status.idle":"2024-04-30T13:04:20.475059Z","shell.execute_reply.started":"2024-04-30T13:04:20.466207Z","shell.execute_reply":"2024-04-30T13:04:20.474000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Averaging the predictions- wasn't necessary as argmax will select the highest value anyways ","metadata":{}},{"cell_type":"code","source":"labels = tf.argmax(result, axis=-1)\nlabels = labels.numpy()\nlabels","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:20.476348Z","iopub.execute_input":"2024-04-30T13:04:20.476663Z","iopub.status.idle":"2024-04-30T13:04:20.487910Z","shell.execute_reply.started":"2024-04-30T13:04:20.476613Z","shell.execute_reply":"2024-04-30T13:04:20.486871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Format the Predictions in submission format","metadata":{}},{"cell_type":"code","source":"test['label'] = labels\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:20.489263Z","iopub.execute_input":"2024-04-30T13:04:20.489570Z","iopub.status.idle":"2024-04-30T13:04:20.501313Z","shell.execute_reply.started":"2024-04-30T13:04:20.489533Z","shell.execute_reply":"2024-04-30T13:04:20.500442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:04:20.512974Z","iopub.execute_input":"2024-04-30T13:04:20.513486Z","iopub.status.idle":"2024-04-30T13:04:20.523824Z","shell.execute_reply.started":"2024-04-30T13:04:20.513450Z","shell.execute_reply":"2024-04-30T13:04:20.523074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}