{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#IMPORT REQUIRED LIBRARIES:\n\nimport numpy as np\nimport pandas as pd\nimport os\nfrom re import search\nimport shutil\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nimport cv2\nimport seaborn as sns\n\nfrom sklearn.preprocessing import MultiLabelBinarizer\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.callbacks import ModelCheckpoint,EarlyStopping\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.layers import Dropout, Dense, Activation, Flatten, Conv2D, MaxPooling2D, BatchNormalization","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#IMAGE PATH & DATAFRAME:\n\ntrain_dir= '../input/plant-pathology-2021-fgvc8/train_images'\ntest_dir =  '../input/plant-pathology-2021-fgvc8/test_images'\ntrain = pd.read_csv('../input/plant-pathology-2021-fgvc8/train.csv')\ntrain.head","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.DataFrame(train,columns = ['image','labels'])\ntrain['labels'].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(35,12))\nplt.xticks(fontsize = 30)\nplt.yticks(fontsize = 30)\nlabels = sns.barplot(x=train.labels.value_counts().index,y=train.labels.value_counts())\nfor item in labels.get_xticklabels():\n    item.set_rotation(45)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['labels'] = train['labels'].apply(lambda s: s.split(' '))\ntrain[:10]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use the Image Data Generator to import the images from the dataset\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\ndatagen = ImageDataGenerator(rescale = 1/255.,\n    rotation_range = 10,#Performing Rotation\n    width_shift_range = 0.2,\n    height_shift_range = 0.2,\n    brightness_range = [0.2,1.0],\n    shear_range = 0.2,\n    zoom_range = 0.2,\n    horizontal_flip=True,\n    vertical_flip=True,\n    validation_split= 0.2)\n\n\nHEIGHT = 248\nWIDTH=248\nSEED = 143\nBATCH_SIZE=32\ntrain_ds = datagen.flow_from_dataframe(\n    train,\n    directory = '../input/resized-plant2021/img_sz_256',# We are using the resized images otherwise it will take a lot of time to train \n    x_col = 'image',\n    y_col = 'labels',\n    subset=\"training\",\n    color_mode=\"rgb\",\n    target_size = (HEIGHT,WIDTH),\n    class_mode=\"categorical\",\n    batch_size=BATCH_SIZE,\n    shuffle=True,\n    seed=SEED,\n)\n\n\nval_ds = datagen.flow_from_dataframe(\n    train,\n    directory = '../input/resized-plant2021/img_sz_256',# We are using the resized images otherwise it will take a lot of time to train \n    x_col = 'image',\n    y_col = 'labels',\n    subset=\"validation\",\n    color_mode=\"rgb\",\n    target_size = (HEIGHT,WIDTH),\n    class_mode=\"categorical\",\n    batch_size=BATCH_SIZE,\n    shuffle=True,\n    seed=SEED,\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example = next(train_ds)\nprint(example[0].shape)\nplt.imshow(example[0][0,:,:,:])\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model=Sequential()\n\nmodel.add(Conv2D(32, (3, 3), padding=\"same\", activation='relu', input_shape=(HEIGHT, WIDTH,3)))\nmodel.add(BatchNormalization(axis=3))\nmodel.add(MaxPooling2D(pool_size=(3, 3)))\nmodel.add(Dropout(0.25))\n        \nmodel.add(Conv2D(64, (3, 3), padding=\"same\", activation='relu'))\nmodel.add(BatchNormalization(axis=3))\nmodel.add(Conv2D(64, (3, 3), padding=\"same\", activation='relu'))\nmodel.add(BatchNormalization(axis=1))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\n\nmodel.add(Conv2D(128, (3, 3), padding=\"same\", activation='relu'))\nmodel.add(BatchNormalization(axis=3))\nmodel.add(Conv2D(128, (3, 3), padding=\"same\", activation='relu'))\nmodel.add(BatchNormalization(axis=3))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\n\nmodel.add(Flatten())\nmodel.add(Dense(64))\nmodel.add(Activation(\"relu\"))\nmodel.add(Dropout(0.25))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.25))\nmodel.add(Dense(6))\nmodel.add(Activation(\"softmax\"))\n\n# model.add(Conv2D(64,(3,3),activation='relu',padding='same', strides=(2,2), input_shape=(HEIGHT,WIDTH,3)))\n# model.add(MaxPooling2D(2,2))\n# model.add(Conv2D(64,(3,3),activation='relu',padding='same'))\n# model.add(MaxPooling2D(2,2))\n# model.add(Conv2D(64,(3,3),activation='relu',padding='same'))\n# model.add(MaxPooling2D(2,2))\n# model.add(Conv2D(128,(3,3),activation='relu',padding='same'))\n# model.add(MaxPooling2D(2,2))\n# model.add(Flatten())\n# model.add(Dropout(0.3))\n# model.add(Dense(6,activation='softmax'))\n\n# Compile the Model\nmodel.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=0.01, decay=0.01/30),\n    loss='binary_crossentropy',\n    metrics=['accuracy'])\nmodel.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"checkpoint=ModelCheckpoint(r'Models\\CNN_Model.h5',\n                          monitor='val_loss',\n                          mode='min',\n                          save_best_only=True,\n                          verbose=1)\nearlystop=EarlyStopping(monitor='val_loss',\n                       min_delta=0,\n                       patience=10,\n                       verbose=1,\n                       restore_best_weights=True)\n\ncallbacks=[checkpoint,earlystop]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cnn_model=model.fit(train_ds,\n                    validation_data=val_ds,\n                    epochs=30,\n                    shuffle=True,\n                    verbose=1,\n                    batch_size=BATCH_SIZE,\n#                     steps_per_epoch=train_ds.samples//128,\n#                     validation_steps=val_ds.samples//128,\n                    callbacks=callbacks)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_history = cnn_model.history\n\nplt.figure()\nplt.plot(model_history['accuracy'])\nplt.plot(model_history['val_accuracy'])\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'])\nplt.savefig('accuracy')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure()\nplt.plot(model_history['loss'])\nplt.plot(model_history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'])\nplt.savefig('loss')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('/kaggle/input/plant-pathology-2021-fgvc8/sample_submission.csv')\nsubmission.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_datagen = ImageDataGenerator(\n    rescale = 1./255\n)\nINPUT_SIZE = (HEIGHT,WIDTH,3)\ntest_generator =  test_datagen.flow_from_dataframe(\n    submission,\n    directory=\"../input/plant-pathology-2021-fgvc8/test_images\",\n    x_col='image',\n    y_col=None,\n    class_mode=None,\n    target_size=INPUT_SIZE[:2]\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = model.predict(test_generator)\nprint(preds)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = preds.tolist()\nindices = []\nfor pred in preds:\n    temp = []\n    for category in pred:\n        if category>=0.23:\n            temp.append(pred.index(category))\n    if temp!=[]:\n        indices.append(temp)\n    else:\n        temp.append(np.argmax(pred))\n        indices.append(temp)\n    \nprint(indices)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = (train_ds.class_indices)\nlabels = dict((v,k) for k,v in labels.items())\nprint(labels)\n\ntestlabels = []\n\n\nfor image in indices:\n    temp = []\n    for i in image:\n        temp.append(str(labels[i]))\n    testlabels.append(' '.join(temp))\n\nprint(testlabels)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['labels'] = testlabels\nsubmission.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}