{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# import the requisite packages\n\nimport json\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport os\nimport pandas as pd\nfrom PIL import Image\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom keras import preprocessing","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# credentials\nFILEPATH = '../input/cassava-leaf-disease-classification'\nRESULTSPATH = './'\n\n# list of image file names\ntrain_imgs_dir = os.path.join(FILEPATH, 'train_images')\nimg_names = os.listdir(train_imgs_dir)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# file mapping image file names to encoded labels\nlabels_df = pd.read_csv(os.path.join(FILEPATH, 'train.csv'))\n\n# dictionary mapping encoded labels to disease classifications\nwith open(os.path.join(FILEPATH, 'label_num_to_disease_map.json')) as json_file: \n    labels_map = json.load(json_file)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels_df['label'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Code for a baseline majority classifier. This code serves as a proof of concept but is not included in our submission. Code for this classifier came from https://www.kaggle.com/mohneesh7/what-happens-if-i-predict-the-majority-class"},{"metadata":{"trusted":true},"cell_type":"code","source":"# def baseline_classifier_1(img): return 3\n\n# test_images = os.listdir(os.path.join(FILEPATH, 'test_images'))\n# predictions = []\n\n# for image in test_images:\n#     predictions.append(baseline_classifier_1(image))\n\n# sub = pd.DataFrame({'image_id': test_images, 'label': predictions})\n# sub.to_csv(os.path.join(RESULTSPATH, 'submission_baseline_1.csv'), index = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# changing the labels to the class names because the data loader needs string\nlabels_map_short = {0: 'CBB', 1: 'CBSD', 2: 'CGM', 3: 'CMD', 4: 'Healthy'}\n\nlabels_df = labels_df.replace({\"label\": labels_map_short})\nlabels_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"base_train_datagen = preprocessing.image.ImageDataGenerator(validation_split = 0.2)\nbase_train_gen = base_train_datagen.flow_from_dataframe(dataframe=labels_df,\n                                                        directory=train_imgs_dir,\n                                                        subset = 'training',\n                                                        x_col='image_id',\n                                                        y_col='label',\n                                                        target_size=(600, 800),\n                                                        batch_size=128,\n                                                        labels=list(labels_df['label']),\n                                                        class_mode='categorical')\n\nbase_val_datagen = preprocessing.image.ImageDataGenerator(validation_split = 0.2)\nbase_val_gen = base_val_datagen.flow_from_dataframe(dataframe=labels_df,\n                                                    directory=train_imgs_dir,\n                                                    subset='validation',\n                                                    x_col='image_id',\n                                                    y_col='label',\n                                                    target_size=(600, 800),\n                                                    batch_size=128,\n                                                    labels=list(labels_df['label']),\n                                                    class_mode='categorical')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Our baseline simple CNN model with two convolutional layers and three fully connected layers"},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout, Activation, Flatten\nfrom tensorflow.keras.layers import Conv2D, AvgPool2D\nfrom tensorflow.keras import callbacks\nfrom tensorflow import optimizers\n\ndef baseline_net():\n    \n    model = Sequential()\n    \n    model.add(Conv2D(filters=6, \n                     kernel_size=5, \n                     activation='relu',\n                     padding='same', \n                     input_shape = [600, 800, 3]))\n    model.add(AvgPool2D(pool_size=2, strides=2))\n    \n    model.add(Conv2D(filters=16,\n                     kernel_size=5,\n                     activation='relu'))\n    model.add(AvgPool2D(pool_size=2, strides=2))\n    \n    model.add(Flatten())\n    model.add(Dense(60, activation='relu'))\n    model.add(Dense(42, activation='relu'))\n    model.add(Dense(5, activation='softmax'))\n    \n    return model\n\nbaseline_cnn = baseline_net()\nbaseline_cnn.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"baseline_cnn.compile(optimizer = optimizers.Adam(lr = 0.001),\n                     loss = \"categorical_crossentropy\",\n                     metrics = [\"accuracy\"])\n\nearly_stop = callbacks.EarlyStopping(monitor = 'val_loss', \n                                     min_delta = 0.001, \n                                     patience = 5, \n                                     mode = 'min', \n                                     verbose = 1,\n                                     restore_best_weights = True)\n\nbase_history = baseline_cnn.fit(base_train_gen,\n                                epochs=3,\n                                callbacks=early_stop,\n                                verbose=1,\n                                validation_data=base_val_gen)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.plot(base_history.history['accuracy'])\nplt.plot(base_history.history['val_accuracy'])\nplt.title('Base Model Accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.legend(['train', 'val'], loc='upper left')\nplt.grid()\nplt.savefig('base_cnn_acc.png', dpi=100)\nplt.show()\n\nplt.plot(base_history.history['loss'])\nplt.plot(base_history.history['val_loss'])\nplt.title('Base Model Loss')\nplt.ylabel('Loss')\nplt.xlabel('Epoch')\nplt.legend(['train', 'val'], loc='upper left')\nplt.grid()\nplt.savefig('base_cnn_loss.png', dpi=100)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_path = '../input/cassava-leaf-disease-classification/test_images'\n\ntest_images = os.listdir(test_path)\npredictions = []\n\nfor image_id in test_images:\n    \n    image = Image.open(os.path.join(test_path, image_id))\n    image = np.array(image)\n    image = np.expand_dims(image, axis=0)\n#     print(image.shape)\n    predictions.append(np.argmax(baseline_cnn.predict(image)))\n\nsub = pd.DataFrame({'image_id': test_images, 'label': predictions})\nsub.to_csv(os.path.join(RESULTSPATH, 'submission.csv'), index = False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}