{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport json\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport seaborn as sns\nimport cv2\n\nsns.set(style=\"darkgrid\")\n\nbase_dir = \"../input/cassava-leaf-disease-classification\"\n\nwith open(os.path.join(base_dir, \"label_num_to_disease_map.json\")) as f:\n    diseases = json.loads(f.read())\n    \nprint(json.dumps(diseases, indent=5))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train_input = os.listdir(os.path.join(base_dir, \"train_images\"))\ntest_input = os.listdir(os.path.join(base_dir, \"test_images\"))\nprint(f\"Total Number of Train Images: {len(train_input)}\")\nprint(f\"Total Number of Test Images: {len(test_input)}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data = pd.read_csv(os.path.join(base_dir,\"train.csv\"))\ndata['Disease'] = data['label'].astype(str).map(diseases)\ndata.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data.info(memory_usage='deep')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(10,5))\ng = sns.countplot(data=data, x=\"label\")\nfor p in g.patches:\n    g.annotate(format(p.get_height(), '.2f'), (p.get_x() + p.get_width() / 2., \n                                               p.get_height()), ha = 'center', \n                                               va = 'center', xytext = (0, 10), \n                                               textcoords = 'offset points')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def select_images(n, label):\n    t = data[data['label'] == label]\n    img_ids = t.sample(n = n, random_state = 0)['image_id']\n    return list(img_ids)\n\ndef plot_images(df, ids, label=None):\n    n = len(ids)\n    fig, ax = plt.subplots(2,n//2, figsize=(14,7))\n    for i, image_id in enumerate(ids):\n        img = mpimg.imread(os.path.join(os.path.join(base_dir, \"train_images\"), image_id))\n        ax[i//(n//2)][i%(n//2)].imshow(img)\n        ax[i//(n//2)][i%(n//2)].axis('off')\n    plt.tight_layout()\n    if label is not None:\n        plt.suptitle(diseases[str(label)])\n        \n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_images(data, select_images(8, 0), 0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_images(data, select_images(8, 1), 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_images(data, select_images(8, 2), 2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_images(data, select_images(8, 3), 3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_images(data, select_images(8, 4), 4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ndata.label = data.label.astype('str')\ntraining, validation = train_test_split(data, test_size=0.2, \n                                        random_state=42, shuffle=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, (ax1,ax2) = plt.subplots(1,2,figsize=(12,5))\nplt.suptitle(\"Training VS Validation\")\n\nx = sns.countplot(training.label, ax=ax1)\ny = sns.countplot(validation.label, ax=ax2)\nax1.set_title('TRAINING')\nax2.set_title('VALIDATION')\nfor p in x.patches:\n    x.annotate(format(p.get_height(), '.2f'), (p.get_x() + p.get_width() / 2., \n                                               p.get_height()), ha = 'center', \n                                               va = 'center', xytext = (0, 10), \n                                               textcoords = 'offset points')\n    \nfor p in y.patches:\n    y.annotate(format(p.get_height(), '.2f'), (p.get_x() + p.get_width() / 2., \n                                               p.get_height()), ha = 'center', \n                                               va = 'center', xytext = (0, 10), \n                                               textcoords = 'offset points')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import tensorflow as tf\n\ngenerator = tf.keras.preprocessing.image.ImageDataGenerator(rescale=1./255,\n                                                            fill_mode='nearest',\n                                                            rotation_range=10,\n                                                            zoom_range = 0.2,\n                                                            width_shift_range=0.2,\n                                                            height_shift_range=0.2,\n                                                            horizontal_flip=True,\n                                                            vertical_flip=True)\n\ntrain_generator = generator.flow_from_dataframe(training,\n                                                 directory = os.path.join(base_dir, \"train_images\"),\n                                                 x_col = \"image_id\",\n                                                 y_col = \"label\",\n                                                 target_size = (150, 150),\n                                                 batch_size = 100,\n                                                 class_mode = \"categorical\")\n\nvalid_generator = generator.flow_from_dataframe(validation,\n                                                target_size=(150,150),\n                                                directory = os.path.join(base_dir,\"train_images\"),\n                                                x_col='image_id',\n                                                y_col='label',\n                                                batch_size=100,\n                                                class_mode='categorical')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = tf.keras.models.Sequential([\n    tf.keras.layers.Conv2D(32,(3,3),strides=1,padding='same',activation='relu',input_shape=(150,150,3)),\n    tf.keras.layers.MaxPooling2D((2,2),strides=2,padding='same'),\n    tf.keras.layers.Conv2D(64,(3,3),strides=2,padding='same',activation='relu'),\n    tf.keras.layers.Dropout(0.2),\n    tf.keras.layers.MaxPooling2D((2,2),strides=2,padding='same'),\n    tf.keras.layers.Conv2D(128,(3,3),strides=1,padding='same',activation='relu'),\n    tf.keras.layers.MaxPooling2D((2,2),strides=2,padding='same'),\n    tf.keras.layers.Flatten(),\n    tf.keras.layers.Dense(512, activation='relu'),\n    tf.keras.layers.Dense(5, activation='softmax')\n])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.compile(optimizer='adam',\n             loss='categorical_crossentropy',\n             metrics=['accuracy'])\n\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learning_rate_reduction = tf.keras.callbacks.ReduceLROnPlateau(monitor='val_accuracy', \n                                            patience = 2, \n                                            verbose=1,\n                                            factor=0.5, \n                                            min_lr=0.00001)\n\nhistory = model.fit_generator(train_generator,\n                             steps_per_epoch=len(training)/100,\n                             epochs=2,\n                             validation_data = valid_generator,\n                             validation_steps = len(validation)/100,\n                             callbacks=[learning_rate_reduction])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"accuracy = history.history['accuracy']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\nval_acc = history.history['val_accuracy']\nepochs = range(1, len(accuracy)+1)\n\nfig, (ax1,ax2) = plt.subplots(1,2,figsize=(12,5))\n\nax1.plot(epochs, accuracy, 'r-', label='Training Accuracy')\nax1.plot(epochs, val_acc, 'bo-', label = 'Validation Accuracy')\nax1.set_title(\"Training Accuracy VS Validation Accuracy\")\nax1.legend()\n\nax2.plot(epochs, loss, 'r-', label='Training Loss')\nax2.plot(epochs, val_loss, 'bo-', label = 'Validation Loss')\nax2.set_title(\"Training Loss VS Validation Loss\")\nax2.legend()\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.save('./models.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.load_weights('./model.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission = pd.read_csv(os.path.join(base_dir, \"sample_submission.csv\"))\nsubmission.label = submission.label.astype('str')\nsubmission","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"generator = tf.keras.preprocessing.image.ImageDataGenerator(rescale=1./255)\n\ntest_generator = generator.flow_from_dataframe(submission,\n                                              directory=os.path.join(base_dir, \"test_images\"),\n                                              x_col='image_id',\n                                              y_col='label',\n                                              target_size=(150,150),\n                                              batch_size=100,\n                                              class_mode='categorical')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predict = model.predict_generator(test_generator)\nsubmission.label = predict.argmax(axis=1)\nsubmission","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.to_csv('submission.csv',index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}