{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle'):\n    print(dirname)\n    \n    for filename in filenames:\n        if filename[-3:] != 'jpg':\n            print(os.path.join(dirname, filename))\n\n            \n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-08T10:22:57.535067Z","iopub.execute_input":"2022-11-08T10:22:57.536060Z","iopub.status.idle":"2022-11-08T10:23:00.754677Z","shell.execute_reply.started":"2022-11-08T10:22:57.535995Z","shell.execute_reply":"2022-11-08T10:23:00.753688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport tensorflow as tf\nfrom tensorflow import keras\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nfrom tensorflow.keras.layers import Dense,Activation,Flatten, Conv2D, MaxPooling2D,Activation\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.metrics import categorical_crossentropy\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator","metadata":{"execution":{"iopub.status.busy":"2022-11-08T10:28:41.243849Z","iopub.execute_input":"2022-11-08T10:28:41.244563Z","iopub.status.idle":"2022-11-08T10:28:41.252353Z","shell.execute_reply.started":"2022-11-08T10:28:41.244525Z","shell.execute_reply":"2022-11-08T10:28:41.251075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1. Import and view the data ","metadata":{}},{"cell_type":"code","source":"train_path = \"/kaggle/input/plant-pathology-2021-fgvc8/train.csv\"\ntest_path = \"/kaggle/input/plant-pathology-2021-fgvc8/sample_submission.csv\"","metadata":{"execution":{"iopub.status.busy":"2022-11-08T10:23:01.142370Z","iopub.execute_input":"2022-11-08T10:23:01.143053Z","iopub.status.idle":"2022-11-08T10:23:01.147687Z","shell.execute_reply.started":"2022-11-08T10:23:01.142996Z","shell.execute_reply":"2022-11-08T10:23:01.146586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv(train_path)\nprint(data.shape)\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-08T10:23:59.079182Z","iopub.execute_input":"2022-11-08T10:23:59.079668Z","iopub.status.idle":"2022-11-08T10:23:59.120554Z","shell.execute_reply.started":"2022-11-08T10:23:59.079625Z","shell.execute_reply":"2022-11-08T10:23:59.119566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv(test_path)\ntest","metadata":{"execution":{"iopub.status.busy":"2022-11-08T10:23:59.234934Z","iopub.execute_input":"2022-11-08T10:23:59.235420Z","iopub.status.idle":"2022-11-08T10:23:59.247485Z","shell.execute_reply.started":"2022-11-08T10:23:59.235393Z","shell.execute_reply":"2022-11-08T10:23:59.246557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_labels=set()\nfor i in data['labels'].unique():\n    for j in i.split():        \n        unique_labels.add(j)\nunique_labels\nprint(\"All unique labels are : \",unique_labels)\nnum_classes = len(unique_labels)\nprint(\"No of classes : \",num_classes)\nimg_size = 224","metadata":{"execution":{"iopub.status.busy":"2022-11-08T10:23:59.413053Z","iopub.execute_input":"2022-11-08T10:23:59.413766Z","iopub.status.idle":"2022-11-08T10:23:59.423131Z","shell.execute_reply.started":"2022-11-08T10:23:59.413731Z","shell.execute_reply":"2022-11-08T10:23:59.421741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#convert each label to a encoded vector for data analysis\ndata[\"list_labels\"] = data[\"labels\"].apply(lambda x:x.split(\" \")) \ndata[list(unique_labels)]=0\ndata","metadata":{"execution":{"iopub.status.busy":"2022-11-08T10:23:59.589479Z","iopub.execute_input":"2022-11-08T10:23:59.589786Z","iopub.status.idle":"2022-11-08T10:23:59.623881Z","shell.execute_reply.started":"2022-11-08T10:23:59.589758Z","shell.execute_reply":"2022-11-08T10:23:59.623000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(data)):\n    for item in data.iloc[i,2]:\n        data.loc[i,item] = 1\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-08T10:23:59.739799Z","iopub.execute_input":"2022-11-08T10:23:59.740364Z","iopub.status.idle":"2022-11-08T10:24:04.837664Z","shell.execute_reply.started":"2022-11-08T10:23:59.740326Z","shell.execute_reply":"2022-11-08T10:24:04.836526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Analysing the data , Exploratory data analysis, Feature Engineering","metadata":{}},{"cell_type":"code","source":"print(data.info())","metadata":{"execution":{"iopub.status.busy":"2022-11-08T10:24:04.839708Z","iopub.execute_input":"2022-11-08T10:24:04.840133Z","iopub.status.idle":"2022-11-08T10:24:04.861438Z","shell.execute_reply.started":"2022-11-08T10:24:04.840079Z","shell.execute_reply":"2022-11-08T10:24:04.860304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(data.describe())","metadata":{"execution":{"iopub.status.busy":"2022-11-08T10:24:06.350591Z","iopub.execute_input":"2022-11-08T10:24:06.350953Z","iopub.status.idle":"2022-11-08T10:24:06.381764Z","shell.execute_reply.started":"2022-11-08T10:24:06.350922Z","shell.execute_reply":"2022-11-08T10:24:06.380709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cheking if there are any null values present in the dataset.\n#If any, we have to use data imputation techniques to make it suitable for training.\ndata.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-11-08T10:24:07.794737Z","iopub.execute_input":"2022-11-08T10:24:07.795114Z","iopub.status.idle":"2022-11-08T10:24:07.809316Z","shell.execute_reply.started":"2022-11-08T10:24:07.795080Z","shell.execute_reply":"2022-11-08T10:24:07.808077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# count plot for a binary variable\nsns.countplot(data = data, x = list(unique_labels)[0])","metadata":{"execution":{"iopub.status.busy":"2022-11-08T10:24:07.992760Z","iopub.execute_input":"2022-11-08T10:24:07.993580Z","iopub.status.idle":"2022-11-08T10:24:08.268628Z","shell.execute_reply.started":"2022-11-08T10:24:07.993538Z","shell.execute_reply":"2022-11-08T10:24:08.265075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for i in unique_labels:\n#     print(data[i].value_counts())","metadata":{"execution":{"iopub.status.busy":"2022-11-08T10:24:08.270514Z","iopub.execute_input":"2022-11-08T10:24:08.271254Z","iopub.status.idle":"2022-11-08T10:24:08.275648Z","shell.execute_reply.started":"2022-11-08T10:24:08.271203Z","shell.execute_reply":"2022-11-08T10:24:08.274517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"c=1\nfor i in unique_labels:\n    plt.subplot(3, 2, c)\n    plt.xlabel(i)\n    sns.countplot(data[i])\n    plt.title(i)\n    c = c + 1\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-08T10:24:10.381350Z","iopub.execute_input":"2022-11-08T10:24:10.381727Z","iopub.status.idle":"2022-11-08T10:24:10.817812Z","shell.execute_reply.started":"2022-11-08T10:24:10.381695Z","shell.execute_reply":"2022-11-08T10:24:10.816953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3.Slipt the data into train,validation sets","metadata":{}},{"cell_type":"code","source":"training_data = data[:11000]\nvalidation_data = data[11000:]","metadata":{"execution":{"iopub.status.busy":"2022-11-08T10:30:21.535809Z","iopub.execute_input":"2022-11-08T10:30:21.536207Z","iopub.status.idle":"2022-11-08T10:30:21.544055Z","shell.execute_reply.started":"2022-11-08T10:30:21.536171Z","shell.execute_reply":"2022-11-08T10:30:21.542838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_imgs_path = '/kaggle/input/plant-pathology-2021-fgvc8/train_images'\ntest_imgs_path = '/kaggle/input/plant-pathology-2021-fgvc8/test_images'\ndatagen = ImageDataGenerator(rescale = 1/255)\n#                                   ,rotation_range=20,\n#                                     width_shift_range=0.2,\n#                                     height_shift_range=0.2,\n#                                     horizontal_flip=True,\n#                                     validation_split = 0.2,\n#                                     zoom_range = 0.2,\n#                                     shear_range = 0.2,\n#                                     vertical_flip = False)\n\ntrain_dataset = datagen.flow_from_dataframe(\n    training_data,\n    directory = train_imgs_path,\n    x_col = \"image\",\n    y_col = 'list_labels',\n    target_size = (img_size,img_size),\n    class_mode='categorical',\n    batch_size = 64,\n    shuffle = True,\n)\n\n\nval_dataset = datagen.flow_from_dataframe(\n    validation_data,\n    directory = train_imgs_path,\n    x_col = \"image\",\n    y_col = 'list_labels',\n    target_size = (img_size,img_size),\n    class_mode='categorical',\n    batch_size = 64,\n    shuffle = True,\n)\n\n\ntest_dataset = datagen.flow_from_dataframe(\n    test,\n    directory = test_imgs_path,\n    x_col = \"image\",\n    y_col = 'labels',\n    target_size = (img_size,img_size),\n    class_mode='categorical',\n    batch_size = 64,\n    shuffle = True,\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4 .Model building","metadata":{}},{"cell_type":"code","source":"model=keras.models.Sequential()\nmodel.add(Conv2D(32,(3,3),activation='relu',padding='same',input_shape=(img_size,img_size,3)))\nmodel.add(MaxPooling2D(2,2))\nmodel.add(Conv2D(64,(3,3),activation='relu',padding='same'))\nmodel.add(MaxPooling2D(2,2))\nmodel.add(Conv2D(64,(3,3),activation='relu',padding='same'))\nmodel.add(MaxPooling2D(2,2))\nmodel.add(Conv2D(128,(3,3),activation='relu',padding='same'))\nmodel.add(MaxPooling2D(2,2))\nmodel.add(Flatten())\nmodel.add(Dense(6,activation='softmax'))","metadata":{"execution":{"iopub.status.busy":"2022-11-08T10:30:34.492189Z","iopub.execute_input":"2022-11-08T10:30:34.492643Z","iopub.status.idle":"2022-11-08T10:30:34.557053Z","shell.execute_reply.started":"2022-11-08T10:30:34.492606Z","shell.execute_reply":"2022-11-08T10:30:34.556046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model = tf.keras.Sequential([\n#     hub.KerasLayer(\"https://tfhub.dev/agripredict/disease-classification/1\",trainable=False),\n#     tf.keras.layers.Dense(256, activation='relu'),\n#     tf.keras.layers.Dense(num_classes, activation='softmax')\n# ])\n# model.build([None, img_size,img_size, 3])\n\n#---------------------------------------------------------------------------------------------\n# from keras.applications.vgg16 import VGG16\n\n# vgg_model = VGG16(weights='imagenet', include_top=False, input_shape=(224, 224, 3))\n# vgg_model.trainable = False ## Not trainable weights\n\n# x = vgg_model.output\n# x = Flatten()(x) # Flatten dimensions to for use in FC layers\n# x = Dense(512, activation='relu')(x)\n# x = Dropout(0.7)(x) # Dropout layer to reduce overfitting\n# x = Dense(256, activation='relu')(x)\n# x = Dense(6, activation='softmax')(x) # Softmax for multiclass\n# model= tf.keras.models.Model(inputs=vgg_model.input, outputs=x)\n# model.save(\"my_model.h5\")\n#---------------------------------------------------------------------------------------------\n\n","metadata":{"execution":{"iopub.status.busy":"2022-11-08T10:30:34.558384Z","iopub.execute_input":"2022-11-08T10:30:34.559340Z","iopub.status.idle":"2022-11-08T10:30:34.564529Z","shell.execute_reply.started":"2022-11-08T10:30:34.559300Z","shell.execute_reply":"2022-11-08T10:30:34.563307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model = keras.models.load_model(\"../input/final-mod/my_model.h5\")","metadata":{"execution":{"iopub.status.busy":"2022-11-08T10:30:34.566406Z","iopub.execute_input":"2022-11-08T10:30:34.566698Z","iopub.status.idle":"2022-11-08T10:30:34.576429Z","shell.execute_reply.started":"2022-11-08T10:30:34.566672Z","shell.execute_reply":"2022-11-08T10:30:34.575233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.predict(test_dataset)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T10:30:34.577541Z","iopub.execute_input":"2022-11-08T10:30:34.577805Z","iopub.status.idle":"2022-11-08T10:30:35.941076Z","shell.execute_reply.started":"2022-11-08T10:30:34.577780Z","shell.execute_reply":"2022-11-08T10:30:35.939996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#compiling the model\nmodel.compile(optimizer=Adam(learning_rate=0.001),loss=categorical_crossentropy,metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-11-08T10:30:35.944001Z","iopub.execute_input":"2022-11-08T10:30:35.944327Z","iopub.status.idle":"2022-11-08T10:30:35.954201Z","shell.execute_reply.started":"2022-11-08T10:30:35.944299Z","shell.execute_reply":"2022-11-08T10:30:35.953151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#training the model\nmodel.fit(train_dataset, epochs=4,validation_data=val_dataset)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T10:30:35.957034Z","iopub.execute_input":"2022-11-08T10:30:35.957506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Save the model\nmodel.save(\"cnn_mod.h5\")\n#load the model back\nmodel = keras.models.load_model('cnn_mod.h5')","metadata":{"execution":{"iopub.status.busy":"2022-11-08T10:06:55.797008Z","iopub.status.idle":"2022-11-08T10:06:55.802981Z","shell.execute_reply.started":"2022-11-08T10:06:55.802741Z","shell.execute_reply":"2022-11-08T10:06:55.802764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#making predictions\npredictions = model.predict(test_dataset)\npredictions","metadata":{"execution":{"iopub.status.busy":"2022-11-08T10:06:55.803825Z","iopub.status.idle":"2022-11-08T10:06:55.804216Z","shell.execute_reply.started":"2022-11-08T10:06:55.804007Z","shell.execute_reply":"2022-11-08T10:06:55.804025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = list(train_dataset.class_indices.keys())\nlabels","metadata":{"execution":{"iopub.status.busy":"2022-11-08T10:06:55.804963Z","iopub.status.idle":"2022-11-08T10:06:55.805339Z","shell.execute_reply.started":"2022-11-08T10:06:55.805147Z","shell.execute_reply":"2022-11-08T10:06:55.805165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label=[]\nfor i in predictions:\n    answer = []\n    for ind,j in enumerate(i):\n        if j > 0.2:\n            answer.append(labels[ind])\n    answer = ' '.join(answer)\n    label.append(answer)\ntest['labels'] = label\ntest","metadata":{"execution":{"iopub.status.busy":"2022-11-08T10:06:55.810921Z","iopub.status.idle":"2022-11-08T10:06:55.811313Z","shell.execute_reply.started":"2022-11-08T10:06:55.811117Z","shell.execute_reply":"2022-11-08T10:06:55.811135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T10:06:55.812082Z","iopub.status.idle":"2022-11-08T10:06:55.812442Z","shell.execute_reply.started":"2022-11-08T10:06:55.812251Z","shell.execute_reply":"2022-11-08T10:06:55.812268Z"},"trusted":true},"execution_count":null,"outputs":[]}]}