{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport cv2\nimport glob\nimport matplotlib.pyplot as plt\nimport gc\nimport albumentations as A\nimport torchmetrics\nimport seaborn as sns\nfrom torch.utils.data import Dataset, DataLoader\nimport torch\nimport torchvision\nfrom albumentations.pytorch import ToTensorV2\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom sklearn.model_selection import train_test_split\nimport torchvision.models as models","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-15T15:52:53.822613Z","iopub.execute_input":"2021-07-15T15:52:53.822979Z","iopub.status.idle":"2021-07-15T15:52:53.831388Z","shell.execute_reply.started":"2021-07-15T15:52:53.822947Z","shell.execute_reply":"2021-07-15T15:52:53.829891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torch.cuda.empty_cache()\ngc.collect()\nDEBUG = False\nDIMENTION = (256, 171) # image size\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\ntrain_data = pd.read_csv('/kaggle/input/plant-pathology-2021-fgvc8/train.csv')\n# train_data = train_data.head(300)\ntest_images_path = '/kaggle/input/plant-pathology-2021-fgvc8/test_images/'\ntest_images_names = glob.glob(test_images_path + '*.jpg')\n\ntrain_data['labels'] = train_data['labels'].apply(lambda string: string.split(' ')) # Create a list of labels\ns = list(train_data['labels'])\nmlb = MultiLabelBinarizer() \ntrain_labels = pd.DataFrame(mlb.fit_transform(s), columns=mlb.classes_, index=train_data.index) # onehot encoding the labels in the list in 6 columns\n\nlabels_size = len(train_labels.columns)","metadata":{"execution":{"iopub.status.busy":"2021-07-15T15:52:53.833471Z","iopub.execute_input":"2021-07-15T15:52:53.834237Z","iopub.status.idle":"2021-07-15T15:52:54.026007Z","shell.execute_reply.started":"2021-07-15T15:52:53.834191Z","shell.execute_reply":"2021-07-15T15:52:54.024795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class PlantDataSet(Dataset):\n    def __init__(self, dataset, images_path, transform=None):\n        super(PlantDataSet, self).__init__() \n        self.dataset = dataset\n        self.images_path = images_path\n        self.transform = transform\n\n    def __getitem__(self, idx):  \n        if self.images_path is not None:\n            image = cv2.imread(self.images_path+self.dataset.image[idx])\n            labels = torch.tensor(self.dataset.loc[idx, self.dataset.columns != 'image'].tolist())\n        else:\n            image = cv2.imread(self.dataset[idx])\n            labels = np.array([])\n            \n        image  = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        image = cv2.resize(image, DIMENTION)\n        \n        if self.transform:\n            image = self.transform(image=image)['image']\n        \n        return image, labels\n\n    def __len__(self):\n        return len(self.dataset)","metadata":{"execution":{"iopub.status.busy":"2021-07-15T15:52:54.028646Z","iopub.execute_input":"2021-07-15T15:52:54.029104Z","iopub.status.idle":"2021-07-15T15:52:54.041234Z","shell.execute_reply.started":"2021-07-15T15:52:54.029059Z","shell.execute_reply":"2021-07-15T15:52:54.039577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transform = A.Compose([A.Normalize(mean=(0.485, 0.456, 0.406), std=(0.229, 0.224, 0.225)),\n                       ToTensorV2()])\ntest_dataset = PlantDataSet(test_images_names, None, transform)\nBS = 30\nplants_test_data_loader = DataLoader(dataset=test_dataset, batch_size=BS, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2021-07-15T15:52:54.043871Z","iopub.execute_input":"2021-07-15T15:52:54.044402Z","iopub.status.idle":"2021-07-15T15:52:54.054321Z","shell.execute_reply.started":"2021-07-15T15:52:54.044348Z","shell.execute_reply":"2021-07-15T15:52:54.053032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def test(test_dataloader, model):\n    predictions = None\n    model.eval()\n    model = model.to(device)\n    with torch.no_grad():\n        for i, (images, _) in enumerate(test_dataloader):\n            images = images.float()    \n            images = images.to(device)\n            \n            output = model(images)\n            probabilities = torch.sigmoid(output)\n            if i == 0:\n                predictions = probabilities.detach().cpu().numpy()\n            else:\n                predictions = np.concatenate((predictions, probabilities.detach().cpu().numpy()), axis=0)\n\n            del images\n            torch.cuda.empty_cache()\n            gc.collect()\n            \n    return np.array(predictions)","metadata":{"execution":{"iopub.status.busy":"2021-07-15T15:52:54.056307Z","iopub.execute_input":"2021-07-15T15:52:54.056880Z","iopub.status.idle":"2021-07-15T15:52:54.068143Z","shell.execute_reply.started":"2021-07-15T15:52:54.056834Z","shell.execute_reply":"2021-07-15T15:52:54.066756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_submission(test_images_path ,predictions):\n    submission_df = pd.DataFrame(columns=['image', 'labels'])\n    for image_name, prediction in zip(test_images_path, predictions):\n        name = image_name.split('/')[-1]\n        arr = [name for pred, name in zip(prediction, train_labels.columns) if pred > 0.75]\n        if len(arr) == 0:\n            arr = ['healthy']\n            \n        prediction_labels = \" \".join(arr)\n\n        row = pd.DataFrame([[name, prediction_labels]], columns=['image', 'labels'])\n        submission_df = submission_df.append(row)\n                \n\n    submission_df = submission_df.reset_index(drop=True)\n    display(submission_df)\n    submission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-07-15T15:52:54.069993Z","iopub.execute_input":"2021-07-15T15:52:54.070621Z","iopub.status.idle":"2021-07-15T15:52:54.083207Z","shell.execute_reply.started":"2021-07-15T15:52:54.070575Z","shell.execute_reply":"2021-07-15T15:52:54.082139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ResNet50 Inference Submission","metadata":{}},{"cell_type":"code","source":"# Load model\n# resnet50 = models.resnet50(pretrained=False, num_classes=labels_size)\n# resnet50.load_state_dict(torch.load('../input/resnet50-final/resnet50_final.pth'))","metadata":{"execution":{"iopub.status.busy":"2021-07-15T15:52:54.084785Z","iopub.execute_input":"2021-07-15T15:52:54.085362Z","iopub.status.idle":"2021-07-15T15:52:54.094353Z","shell.execute_reply.started":"2021-07-15T15:52:54.085319Z","shell.execute_reply":"2021-07-15T15:52:54.093028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_predictions = test(plants_test_data_loader, resnet50)\n# print(test_predictions)\n# create_submission(test_images_names, test_predictions)","metadata":{"execution":{"iopub.status.busy":"2021-07-15T15:52:54.097684Z","iopub.execute_input":"2021-07-15T15:52:54.098212Z","iopub.status.idle":"2021-07-15T15:52:54.105056Z","shell.execute_reply.started":"2021-07-15T15:52:54.098168Z","shell.execute_reply":"2021-07-15T15:52:54.103703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ResNext50_32x4d Inference Submission","metadata":{}},{"cell_type":"code","source":"# Load model\nresnext50 = models.resnext50_32x4d(pretrained=False, num_classes=labels_size)\nresnext50.load_state_dict(torch.load('../input/resnext50-32x4d-final/resnext50_32x4d_final.pth'))","metadata":{"execution":{"iopub.status.busy":"2021-07-15T15:52:54.107011Z","iopub.execute_input":"2021-07-15T15:52:54.107569Z","iopub.status.idle":"2021-07-15T15:53:01.639168Z","shell.execute_reply.started":"2021-07-15T15:52:54.107524Z","shell.execute_reply":"2021-07-15T15:53:01.638056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_predictions = test(plants_test_data_loader, resnext50)\nprint(test_predictions)\ncreate_submission(test_images_names, test_predictions)","metadata":{"execution":{"iopub.status.busy":"2021-07-15T15:53:01.640669Z","iopub.execute_input":"2021-07-15T15:53:01.641117Z","iopub.status.idle":"2021-07-15T15:53:03.726931Z","shell.execute_reply.started":"2021-07-15T15:53:01.641085Z","shell.execute_reply":"2021-07-15T15:53:03.725882Z"},"trusted":true},"execution_count":null,"outputs":[]}]}