{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-24T08:30:09.602249Z","iopub.execute_input":"2022-12-24T08:30:09.603326Z","iopub.status.idle":"2022-12-24T08:30:21.739747Z","shell.execute_reply.started":"2022-12-24T08:30:09.603176Z","shell.execute_reply":"2022-12-24T08:30:21.738466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:30:21.742397Z","iopub.execute_input":"2022-12-24T08:30:21.743079Z","iopub.status.idle":"2022-12-24T08:30:21.748357Z","shell.execute_reply.started":"2022-12-24T08:30:21.743043Z","shell.execute_reply":"2022-12-24T08:30:21.747203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/understanding_cloud_organization/train.csv\")\ntrain_df = train_df.dropna()\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:30:21.750029Z","iopub.execute_input":"2022-12-24T08:30:21.750546Z","iopub.status.idle":"2022-12-24T08:30:26.353259Z","shell.execute_reply.started":"2022-12-24T08:30:21.7505Z","shell.execute_reply":"2022-12-24T08:30:26.352077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:30:26.356551Z","iopub.execute_input":"2022-12-24T08:30:26.357013Z","iopub.status.idle":"2022-12-24T08:30:26.369734Z","shell.execute_reply.started":"2022-12-24T08:30:26.356968Z","shell.execute_reply":"2022-12-24T08:30:26.368258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv(\"/kaggle/input/understanding_cloud_organization/train.csv\")\n\n# Split Image_Label into ImageId and Label\nsplit = train_data['Image_Label'].str.split('_', n = 1, expand = True)\ntrain_data['id'] = split[0]\ntrain_data['label'] = split[1]\n\n# Select columns \nselected_features = [cname for cname in train_data.columns if cname not in ['Image_Label']]\n\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:30:26.371731Z","iopub.execute_input":"2022-12-24T08:30:26.372224Z","iopub.status.idle":"2022-12-24T08:30:28.476404Z","shell.execute_reply.started":"2022-12-24T08:30:26.372169Z","shell.execute_reply":"2022-12-24T08:30:28.475187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# count unique labels \ntrain_data['label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:30:28.478204Z","iopub.execute_input":"2022-12-24T08:30:28.478654Z","iopub.status.idle":"2022-12-24T08:30:28.492522Z","shell.execute_reply.started":"2022-12-24T08:30:28.478615Z","shell.execute_reply":"2022-12-24T08:30:28.491675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:32:01.012843Z","iopub.execute_input":"2022-12-24T08:32:01.013291Z","iopub.status.idle":"2022-12-24T08:32:01.036128Z","shell.execute_reply.started":"2022-12-24T08:32:01.013243Z","shell.execute_reply":"2022-12-24T08:32:01.034922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe().T","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:32:31.077623Z","iopub.execute_input":"2022-12-24T08:32:31.078113Z","iopub.status.idle":"2022-12-24T08:32:31.307055Z","shell.execute_reply.started":"2022-12-24T08:32:31.078076Z","shell.execute_reply":"2022-12-24T08:32:31.306041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dropping old Image_Label columns \ndf = train_data[['id', 'label']]","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:30:28.493426Z","iopub.execute_input":"2022-12-24T08:30:28.49374Z","iopub.status.idle":"2022-12-24T08:30:28.51039Z","shell.execute_reply.started":"2022-12-24T08:30:28.493695Z","shell.execute_reply":"2022-12-24T08:30:28.509362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:32:47.591733Z","iopub.execute_input":"2022-12-24T08:32:47.592166Z","iopub.status.idle":"2022-12-24T08:32:47.606261Z","shell.execute_reply.started":"2022-12-24T08:32:47.59213Z","shell.execute_reply":"2022-12-24T08:32:47.605033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**SPECIFY MODEL**","metadata":{}},{"cell_type":"code","source":"from tensorflow.python.keras.models import Sequential\nfrom tensorflow.python.keras.layers import Dense","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:33:20.194937Z","iopub.execute_input":"2022-12-24T08:33:20.196311Z","iopub.status.idle":"2022-12-24T08:33:20.20167Z","shell.execute_reply.started":"2022-12-24T08:33:20.19625Z","shell.execute_reply":"2022-12-24T08:33:20.200537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Preprocessing**","metadata":{}},{"cell_type":"code","source":"train_df = train_df[~train_df['EncodedPixels'].isnull()]\ntrain_df['Image'] = train_df['Image_Label'].map(lambda x: x.split('_')[0])\ntrain_df['Class'] = train_df['Image_Label'].map(lambda x: x.split('_')[1])\nclasses = train_df['Class'].unique()\ntrain_df = train_df.groupby('Image')['Class'].agg(set).reset_index()\nfor class_name in classes:\n    train_df[class_name] = train_df['Class'].map(lambda x: 1 if class_name in x else 0)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:33:25.220077Z","iopub.execute_input":"2022-12-24T08:33:25.221598Z","iopub.status.idle":"2022-12-24T08:33:25.484755Z","shell.execute_reply.started":"2022-12-24T08:33:25.221541Z","shell.execute_reply":"2022-12-24T08:33:25.483715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport keras\nimport random\nimport warnings\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport albumentations as albu\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom keras import optimizers\nfrom keras import backend as K\nfrom keras.models import Model\nfrom keras.losses import binary_crossentropy\nfrom keras.callbacks import EarlyStopping, ReduceLROnPlateau\nfrom keras.layers import Input, Conv2D, Conv2DTranspose, MaxPooling2D, Concatenate","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:33:35.077669Z","iopub.execute_input":"2022-12-24T08:33:35.078154Z","iopub.status.idle":"2022-12-24T08:33:37.447489Z","shell.execute_reply.started":"2022-12-24T08:33:35.078099Z","shell.execute_reply":"2022-12-24T08:33:37.446301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_train = len(os.listdir(\"/kaggle/input/understanding_cloud_organization/train_images\"))\nn_test = len(os.listdir(\"/kaggle/input/understanding_cloud_organization/test_images\"))\nprint(f'There are {n_train} images in train dataset')\nprint(f'There are {n_test} images in test dataset')","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:33:38.673415Z","iopub.execute_input":"2022-12-24T08:33:38.67413Z","iopub.status.idle":"2022-12-24T08:33:38.687451Z","shell.execute_reply.started":"2022-12-24T08:33:38.67408Z","shell.execute_reply":"2022-12-24T08:33:38.686159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.layers import Convolution2D, MaxPooling2D\nfrom keras.layers import Dense, Dropout, Activation, Flatten\nfrom tensorflow.keras import utils\nfrom sklearn.model_selection import train_test_split\nfrom fastai import *\nfrom fastai.vision import *\nfrom fastai.metrics import error_rate\nimport os\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:33:42.129406Z","iopub.execute_input":"2022-12-24T08:33:42.129829Z","iopub.status.idle":"2022-12-24T08:33:44.694639Z","shell.execute_reply.started":"2022-12-24T08:33:42.129795Z","shell.execute_reply":"2022-12-24T08:33:44.693484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Explore the correlation between different cloud types.\n#Using the dataframe with labels, we can try to find the correlation between different types of clouds.","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:30:33.928445Z","iopub.status.idle":"2022-12-24T08:30:33.92929Z","shell.execute_reply.started":"2022-12-24T08:30:33.928938Z","shell.execute_reply":"2022-12-24T08:30:33.928968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:33:49.422669Z","iopub.execute_input":"2022-12-24T08:33:49.424339Z","iopub.status.idle":"2022-12-24T08:33:49.440075Z","shell.execute_reply.started":"2022-12-24T08:33:49.424278Z","shell.execute_reply":"2022-12-24T08:33:49.438742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['Gravel'].hist(bins=100)","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:36:33.915956Z","iopub.execute_input":"2022-12-24T08:36:33.917103Z","iopub.status.idle":"2022-12-24T08:36:34.308806Z","shell.execute_reply.started":"2022-12-24T08:36:33.917061Z","shell.execute_reply":"2022-12-24T08:36:34.307581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Data Generator**","metadata":{}},{"cell_type":"code","source":"img_2_ohe_vector = {img:vec for img, vec in zip(train_df['Image'], train_df.iloc[:, 2:].values)}","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:45:46.218108Z","iopub.execute_input":"2022-12-24T08:45:46.218564Z","iopub.status.idle":"2022-12-24T08:45:46.22746Z","shell.execute_reply.started":"2022-12-24T08:45:46.218532Z","shell.execute_reply":"2022-12-24T08:45:46.226258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_imgs, val_imgs = train_test_split(train_df['Image'].values, \n                                        test_size=0.2, \n                                        stratify=train_df['Class'].map(lambda x: str(sorted(list(x)))),\n                                        random_state=5)\n","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:45:49.373701Z","iopub.execute_input":"2022-12-24T08:45:49.374115Z","iopub.status.idle":"2022-12-24T08:45:49.411278Z","shell.execute_reply.started":"2022-12-24T08:45:49.37408Z","shell.execute_reply":"2022-12-24T08:45:49.40976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from copy import deepcopy\n\nclass DataGenenerator(utils.Sequence):\n    def __init__(self, images_list=None, folder_imgs=f'/kaggle/input/understanding_cloud_organization/train_images', \n                 batch_size=32, shuffle=True, augmentation=None,\n                 resized_height=224, resized_width=224, num_channels=3):\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n        self.augmentation = augmentation\n        if images_list is None:\n            self.images_list = os.listdir(folder_imgs)\n        else:\n            self.images_list = deepcopy(images_list)\n        self.folder_imgs = folder_imgs\n        self.len = len(self.images_list) // self.batch_size\n        self.resized_height = resized_height\n        self.resized_width = resized_width\n        self.num_channels = num_channels\n        self.num_classes = 4\n        self.is_test = not 'train' in folder_imgs\n        if not shuffle and not self.is_test:\n            self.labels = [img_2_ohe_vector[img] for img in self.images_list[:self.len*self.batch_size]]\n\n    def __len__(self):\n        return self.len\n    \n    def on_epoch_start(self):\n        if self.shuffle:\n            random.shuffle(self.images_list)\n\n    def __getitem__(self, idx):\n        current_batch = self.images_list[idx * self.batch_size: (idx + 1) * self.batch_size]\n        X = np.empty((self.batch_size, self.resized_height, self.resized_width, self.num_channels))\n        y = np.empty((self.batch_size, self.num_classes))\n\n        for i, image_name in enumerate(current_batch):\n            path = os.path.join(self.folder_imgs, image_name)\n            img = cv2.resize(cv2.imread(path), (self.resized_height, self.resized_width)).astype(np.float32)\n            if not self.augmentation is None:\n                augmented = self.augmentation(image=img)\n                img = augmented['image']\n            X[i, :, :, :] = img/255.0\n            if not self.is_test:\n                y[i, :] = img_2_ohe_vector[image_name]\n        return X, y\n\n    def get_labels(self):\n        if self.shuffle:\n            images_current = self.images_list[:self.len*self.batch_size]\n            labels = [img_2_ohe_vector[img] for img in images_current]\n        else:\n            labels = self.labels\n        return np.array(labels)","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:45:52.94525Z","iopub.execute_input":"2022-12-24T08:45:52.945632Z","iopub.status.idle":"2022-12-24T08:45:52.961909Z","shell.execute_reply.started":"2022-12-24T08:45:52.945603Z","shell.execute_reply":"2022-12-24T08:45:52.960367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_generator_train = DataGenenerator(train_imgs)\ndata_generator_train_eval = DataGenenerator(train_imgs, shuffle=False)\ndata_generator_val = DataGenenerator(val_imgs, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:45:56.840432Z","iopub.execute_input":"2022-12-24T08:45:56.841119Z","iopub.status.idle":"2022-12-24T08:45:56.855324Z","shell.execute_reply.started":"2022-12-24T08:45:56.841081Z","shell.execute_reply":"2022-12-24T08:45:56.854523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**CNN Model**","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom keras.models import Sequential\nfrom keras.layers import Convolution2D, MaxPooling2D\nfrom keras.layers import Dense, Dropout, Activation, Flatten\nfrom tensorflow.keras import utils\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nimport os","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:51:59.013941Z","iopub.execute_input":"2022-12-24T08:51:59.01438Z","iopub.status.idle":"2022-12-24T08:51:59.021898Z","shell.execute_reply.started":"2022-12-24T08:51:59.014344Z","shell.execute_reply":"2022-12-24T08:51:59.020439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from fastai.vision.all import *","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:53:23.795914Z","iopub.execute_input":"2022-12-24T08:53:23.796305Z","iopub.status.idle":"2022-12-24T08:53:23.802124Z","shell.execute_reply.started":"2022-12-24T08:53:23.796275Z","shell.execute_reply":"2022-12-24T08:53:23.800892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential()\nmodel.add(Convolution2D(32, (3, 3), activation='relu', input_shape=(224,224,3)))\nmodel.add(Convolution2D(32, (3, 3), activation='relu'))\nmodel.add(MaxPooling2D(pool_size=(2,2)))\nmodel.add(Dropout(0.25)) \nmodel.add(Dense(128, activation='relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(4, activation='softmax'))\nmodel.compile(loss='categorical_crossentropy',metrics=['accuracy'], optimizer='adam')\n\nprint(model.summary())","metadata":{"execution":{"iopub.status.busy":"2022-12-24T09:01:43.788436Z","iopub.execute_input":"2022-12-24T09:01:43.788894Z","iopub.status.idle":"2022-12-24T09:01:43.884872Z","shell.execute_reply.started":"2022-12-24T09:01:43.788853Z","shell.execute_reply":"2022-12-24T09:01:43.883687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#We now use a pre-trained ResNet18 Convolutional Neural Net model, and use transfer learning to learn weights of only the last layer of the network.\n#Why Transfer learning? Because with transfer learning, you begin with an existing (trained) neural network used for image recognition — and then tweak it a bit (or more) here and there to train a model for your particular use case. And why do we do that? Training a reasonable neural network would mean needing approximately 300,000 image samples, and to achieve really good performance, we’re going to need at least a million images.\n#In our case, we have approximately 2500 images in our training set — you have one guess to decide if that would have been enough if were to train a neural net from scratch.\n#We use the create_cnn() function for loading a pre-trained ResNet18 network, that was trained on around a million images from the ImageNet database.","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:30:33.934793Z","iopub.status.idle":"2022-12-24T08:30:33.935183Z","shell.execute_reply.started":"2022-12-24T08:30:33.934986Z","shell.execute_reply":"2022-12-24T08:30:33.935003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**CNN Training**","metadata":{}},{"cell_type":"code","source":"from multiprocessing import cpu_count\nimport cv2\nnum_cores = cpu_count()\nhistory = model.fit_generator(generator=data_generator_train,\n                              validation_data=data_generator_val,\n                              epochs=10,\n                              workers=num_cores,\n                              verbose=1\n                             )","metadata":{"execution":{"iopub.status.busy":"2022-12-24T09:05:02.930067Z","iopub.status.idle":"2022-12-24T09:05:02.931688Z","shell.execute_reply.started":"2022-12-24T09:05:02.931349Z","shell.execute_reply":"2022-12-24T09:05:02.931379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy = history.history['accuracy']\nloss = history.history['loss']\nepochs = range(len(accuracy))\nplt.plot(epochs, accuracy, 'b-', label='Training accuracy')\nplt.title('Training and validation accuracy')\nplt.legend()\nplt.figure()\nplt.plot(epochs, loss, 'r-', label='Training loss')\nplt.title('Training and validation loss')\nplt.legend()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**CNN Testing**","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import multilabel_confusion_matrix\n\ny_pred = model.predict_generator(data_generator_val, workers=num_cores)\ny_true = data_generator_val.get_labels()\n\nprint(multilabel_confusion_matrix(y_true, y_pred))","metadata":{"execution":{"iopub.status.busy":"2022-12-24T09:04:21.958857Z","iopub.execute_input":"2022-12-24T09:04:21.959851Z","iopub.status.idle":"2022-12-24T09:05:02.928856Z","shell.execute_reply.started":"2022-12-24T09:04:21.959805Z","shell.execute_reply":"2022-12-24T09:05:02.927012Z"},"trusted":true},"execution_count":null,"outputs":[]}]}