{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Libraries"},{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport tensorflow as tf\nimport seaborn as sns\n\nimport os\nimport json\n\nimport matplotlib.pyplot as plt\n\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout, Flatten, MaxPool2D, Conv2D, BatchNormalization, ReLU\nfrom tensorflow import keras\n\nfrom sklearn.model_selection import StratifiedKFold","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Tensorflow Version:\", tf.__version__)\nprint(\"Is GPU Available:\", tf.test.is_gpu_available())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Exploring Data"},{"metadata":{"trusted":true},"cell_type":"code","source":"label_map = json.load(open(\"../input/cassava-leaf-disease-classification/label_num_to_disease_map.json\"))\nlabel_map","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"train_dir = \"../input/cassava-leaf-disease-classification/train_images\"\ntest_dir = \"../input/cassava-leaf-disease-classification/test_images\"\ntrain_file = \"../input/cassava-leaf-disease-classification/train.csv\"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Read the train csv file"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(train_file)\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Create a new column called kfold and fill it with -1"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['kfold'] = -1","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Randomize the rows of the data"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = train_df.sample(frac=1).reset_index(drop=True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Initiate the kfold class"},{"metadata":{"trusted":true},"cell_type":"code","source":"kf = StratifiedKFold(n_splits=5)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Fill the new kfold column"},{"metadata":{"trusted":true},"cell_type":"code","source":"for f, (t_, v_) in enumerate(kf.split(X=train_df, y=train_df.label.values)):\n    train_df.loc[v_, 'kfold'] = f","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.tail()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Function to get training and validation generators"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_datagen = keras.preprocessing.image.ImageDataGenerator(\n    rescale=1./255,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    rotation_range=30,\n)\n\ntest_datagen = keras.preprocessing.image.ImageDataGenerator(rescale=1./255)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_generators(df, k):\n    val = df[df.kfold==k]\n    train = df[df.kfold!=k]\n    \n    train_gen = train_datagen.flow_from_dataframe(\n        dataframe=train,\n        directory=train_dir,\n        x_col='image_id',\n        y_col='label',\n        target_size=(224,224),\n        batch_size=16,\n        shuffle=True,\n        class_mode='raw'\n    )\n    \n    val_gen = test_datagen.flow_from_dataframe(\n        dataframe=val,\n        directory=train_dir,\n        x_col='image_id',\n        y_col='label',\n        target_size=(224,224),\n        batch_size=16,\n        shuffle=False,\n        class_mode='raw'\n    )\n    \n    return train_gen, val_gen","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_gen, val_gen = get_generators(train_df, 0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"images, labels = next(train_gen)\n\ntotal_imgs = len(images)\ncols = 5\nrows = total_imgs / cols + 1\nplt.figure(figsize=(25,rows*5))\nfor i in range(total_imgs):\n    plt.subplot(rows,cols,i+1)\n    plt.xticks([])\n    plt.yticks([])\n    plt.imshow(images[i])\n    plt.xlabel(label_map[str(labels[i])])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"images, labels = next(val_gen)\n\ntotal_imgs = len(images)\ncols = 5\nrows = total_imgs / cols + 1\nplt.figure(figsize=(25,rows*5))\nfor i in range(total_imgs):\n    plt.subplot(rows,cols,i+1)\n    plt.xticks([])\n    plt.yticks([])\n    plt.imshow(images[i])\n    plt.xlabel(label_map[str(labels[i])])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":" # Model"},{"metadata":{"trusted":true},"cell_type":"code","source":"model = Sequential([\n    Conv2D(16, 3, strides=2, use_bias=False, input_shape=images[0].shape),\n    BatchNormalization(),\n    ReLU(),\n    Conv2D(32, 3, strides=2, use_bias=False),\n    BatchNormalization(),\n    ReLU(),\n    Conv2D(64, 3, strides=2, use_bias=False),\n    BatchNormalization(),\n    ReLU(),\n    Conv2D(128, 3, strides=2, use_bias=False),\n    BatchNormalization(),\n    ReLU(),\n    Conv2D(256, 3, strides=2),\n    ReLU(),\n    Flatten(),\n    Dense(512, activation='relu'),\n    Dropout(0.2),\n    Dense(256, activation='relu'),\n    Dropout(0.2),\n    Dense(5, activation='softmax')\n])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.compile(loss=\"sparse_categorical_crossentropy\", optimizer='adam',metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"EPOCHS=20\nif not tf.test.is_gpu_available():\n    EPOCHS=5","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"checkpoint_filepath = './checkpoint'\n\nmodel_checkpoint_callback = tf.keras.callbacks.ModelCheckpoint(\n    filepath=checkpoint_filepath,\n    save_weights_only=True,\n    monitor='val_loss',\n    mode='min',\n    save_best_only=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_gen, val_gen = get_generators(train_df, 4)\nhistory = model.fit_generator(\n    train_gen,\n    epochs=EPOCHS,\n    validation_data=val_gen,\n    verbose=2,\n    callbacks=[model_checkpoint_callback],\n    max_queue_size=100,\n    workers=10,\n    use_multiprocessing=True\n)\n    \nmodel.load_weights(checkpoint_filepath)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"epochs_range = list(range(1, EPOCHS+1))\n\nplt.plot(epochs_range, history.history['loss'], label=\"Training Loss\")\nplt.plot(epochs_range, history.history['val_loss'], '--', label=\"Validation Loss\")\nplt.title(\"Loss Graphs\")\nplt.legend()\nplt.show()\n\nplt.plot(epochs_range, history.history['accuracy'], label=\"Training Accuracy\")\nplt.plot(epochs_range, history.history['val_accuracy'], '--', label=\"Validation Accuracy\")\nplt.title(\"Accuracy Graphs\")\nplt.legend()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Evaluation"},{"metadata":{"trusted":true},"cell_type":"code","source":"model.evaluate(val_gen)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_preds = model.predict(val_gen)\ny_preds = tf.argmax(y_preds, axis=1)\n\ny_labels = val_gen.labels\nsns.heatmap(tf.math.confusion_matrix(y_labels, y_preds), annot=True, fmt='d', cmap='Blues')\nplt.xlabel('Predicted Label')\nplt.ylabel('True Label')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Prediction"},{"metadata":{"trusted":true},"cell_type":"code","source":"test_image_id = os.listdir(test_dir)\ntest_df = pd.DataFrame(test_image_id, columns=['image_id'])\ntest_df['label'] = -1\ntest_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_gen = test_datagen.flow_from_dataframe(\n        dataframe=test_df,\n        directory=test_dir,\n        x_col='image_id',\n        y_col='image_id',\n        target_size=(224,224),\n        batch_size=16,\n        shuffle=False,\n        class_mode='raw'\n    )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_pred = model.predict(test_gen)\ny_pred = tf.argmax(y_pred, axis=1)\nimage_ids = test_gen.labels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"result = pd.DataFrame(np.stack([image_ids, y_pred], axis=1), columns=['image_id','label'])\nresult.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"result.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}