{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#IMPORT REQUIRED LIBRARIES:\n\nimport numpy as np\nimport pandas as pd\nimport os\nfrom re import search\nimport shutil\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nimport cv2\nimport seaborn as sns\nfrom pathlib import Path\n\nfrom sklearn.preprocessing import MultiLabelBinarizer\n\nimport tensorflow_addons as tfa\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential, Model\nfrom tensorflow.keras.applications import ResNet50, ResNet50V2\nfrom tensorflow.keras.callbacks import ModelCheckpoint,EarlyStopping\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.layers import Input, Dense, ZeroPadding2D, Dropout, Activation, Flatten, Conv2D, MaxPooling2D, ReLU, BatchNormalization, AveragePooling2D, GlobalAveragePooling2D, add","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a list with the filepaths for training and testing\ntrain_img_Path = '../input/plant-pathology-2021-fgvc8/train_images'\n\ntest_img_Path = '../input/plant-pathology-2021-fgvc8/test_images'\n\nimg_Path = '../input/resized-plant2021/img_sz_256'\n\ntrain = pd.read_csv(r'../input/plant-pathology-2021-fgvc8/train.csv')\n\nsample_submission = pd.read_csv(r'../input/plant-pathology-2021-fgvc8/sample_submission.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Number of pictures in the training dataset: {train.shape[0]}\\n')\nprint(f'Number of different labels: {len(train.labels.unique())}\\n')\nprint(f'Labels: {train.labels.unique()}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['labels'].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(14,7))\nb = sns.countplot(x='labels', data=train, order=sorted(train['labels'].unique()))\nfor item in b.get_xticklabels():\n    item.set_rotation(90)\nplt.title('Label Distribution', weight='bold')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,40))\ni=1\nfor idx,s in train.head(9).iterrows():\n    img_path = os.path.join(img_Path,s['image'])\n    img=cv2.imread(img_path)\n    img=cv2.cvtColor(img,cv2.COLOR_BGR2RGB)\n    fig=plt.subplot(9,3,i)\n    fig.imshow(img)\n    fig.set_title(s['labels'])\n    i+=1","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CLASSES = train['labels'].unique().tolist()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n# Preprocessing the Training set\ntrain_datagen = ImageDataGenerator(rescale=1./255,\n                                   shear_range = 0.1,\n                                   zoom_range = 0.1,\n                                   horizontal_flip = True,\n                                   validation_split=0.25)\n\ntrain_data = train_datagen.flow_from_dataframe(train,\n                                              directory=img_Path,\n                                              classes=CLASSES,\n                                              x_col=\"image\",\n                                              y_col=\"labels\",\n                                              target_size=(150, 150),\n                                              subset='training')\n\nval_data = train_datagen.flow_from_dataframe(train,\n                                            directory=img_Path,\n                                            classes=CLASSES,\n                                            x_col=\"image\",\n                                            y_col=\"labels\",\n                                            target_size=(150, 150),\n                                            subset='validation')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dict_classes = train_data.class_indices\ndict_classes","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\nfrom keras import Sequential\nfrom keras.applications import ResNet50V2, ResNet152V2, InceptionResNetV2\nfrom tensorflow.keras.layers import Conv2D, Dropout,MaxPooling2D,Flatten,Dense, BatchNormalization, Dropout","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_ResNet50V2 = ResNet50V2(include_top = False, \n                         weights = '../input/keras-pretrained-models/ResNet50V2_NoTop_ImageNet.h5', \n                         input_shape = train_data.image_shape, \n                         pooling='avg',\n                         classes = CLASSES)\n\n#Adding the final layers to the above base models where the actual classification is done in the dense layers\nmodel_ResNet = Sequential()\nmodel_ResNet.add(base_ResNet50V2)\nmodel_ResNet.add(BatchNormalization())\nmodel_ResNet.add(Dense(64, activation=('relu')))\nmodel_ResNet.add(Dropout(0.2))\nmodel_ResNet.add(BatchNormalization())\nmodel_ResNet.add(Dense(12, activation=('softmax')))\n\nmodel_ResNet.compile(optimizer = tf.keras.optimizers.Adam(learning_rate=0.001, decay=0.0001), \n                     loss = 'categorical_crossentropy', \n                     metrics = ['accuracy'])\n\nmodel_ResNet.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"checkpoint=ModelCheckpoint(r'Models\\ResNet50_Model.h5',\n                          monitor='val_loss',\n                          mode='min',\n                          save_best_only=False,\n                          verbose=1)\nearlystop=EarlyStopping(monitor='val_loss',\n                       min_delta=0,\n                       patience=5,\n                       verbose=1,\n                       restore_best_weights=True)\n\ncallbacks=[checkpoint,earlystop]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training the CNN on the Train data and evaluating it on the val data\nr = model_ResNet.fit(train_data, \n                     validation_data = val_data, \n                     epochs = 12, \n                     batch_size=32,\n                     steps_per_epoch=train_data.samples//64,\n                     validation_steps=val_data.samples//64,\n                     shuffle=True,\n                     callbacks=callbacks)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_history = r.history\n\nplt.figure()\nplt.plot(model_history['accuracy'])\nplt.plot(model_history['val_accuracy'])\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'])\nplt.savefig('accuracy')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure()\nplt.plot(model_history['loss'])\nplt.plot(model_history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'])\nplt.savefig('loss')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dir = '/kaggle/input/plant-pathology-2021-fgvc8/test_images/'\ntest_df = pd.DataFrame()\ntest_df['image'] = os.listdir(test_dir)\n\ntest_data = train_datagen.flow_from_dataframe(dataframe=test_df,\n                                    directory=test_dir,\n                                    x_col=\"image\",\n                                    y_col=None,\n                                    batch_size=32,\n                                    seed=42,\n                                    shuffle=False,\n                                    class_mode=None,\n                                    target_size=(150, 150))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_resnet = model_ResNet.predict(test_data)\npred_resnet","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = (pred_resnet+pred_resnet+pred_resnet).tolist()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(pred)):\n    pred[i] = np.argmax(pred[i])\n\n    \ndef get_key(val):\n    for key, value in dict_classes.items():\n        if val == value:\n            return key\n        \n\nfor i in range(len(pred)):\n    pred[i] = get_key(pred[i])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['labels'] = pred\ntest_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.to_csv('submission.csv',index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}