{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-07T04:49:06.092739Z","iopub.execute_input":"2021-07-07T04:49:06.093212Z","iopub.status.idle":"2021-07-07T04:49:08.53922Z","shell.execute_reply.started":"2021-07-07T04:49:06.093106Z","shell.execute_reply":"2021-07-07T04:49:08.538447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os, shutil\nfrom keras import layers, models, optimizers\nimport pandas as pd\nimport numpy as np\nimport cv2\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing import image_dataset_from_directory\nfrom keras.preprocessing.image import ImageDataGenerator, load_img, img_to_array, smart_resize\nfrom keras.callbacks import EarlyStopping, LearningRateScheduler, ReduceLROnPlateau, ModelCheckpoint\nfrom keras.applications import VGG16, ResNet50\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2021-07-07T04:49:08.541895Z","iopub.execute_input":"2021-07-07T04:49:08.542148Z","iopub.status.idle":"2021-07-07T04:49:10.140166Z","shell.execute_reply.started":"2021-07-07T04:49:08.542116Z","shell.execute_reply":"2021-07-07T04:49:10.139332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = pd.read_csv(\"../input/cassava-leaf-disease-classification/train.csv\")\nlabels.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-07T04:49:10.142044Z","iopub.execute_input":"2021-07-07T04:49:10.142375Z","iopub.status.idle":"2021-07-07T04:49:10.172067Z","shell.execute_reply.started":"2021-07-07T04:49:10.142338Z","shell.execute_reply":"2021-07-07T04:49:10.171143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels.info()","metadata":{"execution":{"iopub.status.busy":"2021-07-07T04:49:10.173746Z","iopub.execute_input":"2021-07-07T04:49:10.174102Z","iopub.status.idle":"2021-07-07T04:49:10.186988Z","shell.execute_reply.started":"2021-07-07T04:49:10.174066Z","shell.execute_reply":"2021-07-07T04:49:10.185918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels['label']=labels['label'].astype(str)","metadata":{"execution":{"iopub.status.busy":"2021-07-07T04:49:10.18841Z","iopub.execute_input":"2021-07-07T04:49:10.188907Z","iopub.status.idle":"2021-07-07T04:49:10.21838Z","shell.execute_reply.started":"2021-07-07T04:49:10.188867Z","shell.execute_reply":"2021-07-07T04:49:10.21763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dir = '../input/cassava-leaf-disease-classification/train_images'\ntest_dir = '../input/cassava-leaf-disease-classification/test_images'","metadata":{"execution":{"iopub.status.busy":"2021-07-07T04:49:10.219551Z","iopub.execute_input":"2021-07-07T04:49:10.219883Z","iopub.status.idle":"2021-07-07T04:49:10.229117Z","shell.execute_reply.started":"2021-07-07T04:49:10.219848Z","shell.execute_reply":"2021-07-07T04:49:10.228279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"ImageDataGenerator --> Generate batches of tensor image data with real-time data augmentation.","metadata":{}},{"cell_type":"code","source":"#Defining the preprocessing function\ndef preprocess(image):\n    #Converting to numpy array from numpy tensor with rank 3\n    image = np.array(image, dtype=np.uint8)\n    #Converting to RGB\n    #img = cv2.cvtCoor(img, cv2.COLOR_BGR2RGB)\n    #Gaussian Blur\n    gaussian_blur = cv2.GaussianBlur(image,(5,5),0)\n    img = np.asarray(gaussian_blur, dtype=np.float64)\n    return img","metadata":{"execution":{"iopub.status.busy":"2021-07-07T04:49:10.232208Z","iopub.execute_input":"2021-07-07T04:49:10.232561Z","iopub.status.idle":"2021-07-07T04:49:10.239301Z","shell.execute_reply.started":"2021-07-07T04:49:10.232534Z","shell.execute_reply":"2021-07-07T04:49:10.238401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dengan augmentasi\ntrain_datagen = ImageDataGenerator(rescale=1./225,\n                                   rotation_range=40,\n                                   width_shift_range=0.2,\n                                   height_shift_range=0.2,\n                                   shear_range=0.2,\n                                   zoom_range=0.2,\n                                   validation_split=0.2,\n                                   horizontal_flip=True,\n                                   preprocessing_function=preprocess)\n","metadata":{"execution":{"iopub.status.busy":"2021-07-07T04:49:10.242664Z","iopub.execute_input":"2021-07-07T04:49:10.243105Z","iopub.status.idle":"2021-07-07T04:49:10.252015Z","shell.execute_reply.started":"2021-07-07T04:49:10.24307Z","shell.execute_reply":"2021-07-07T04:49:10.251144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # tanpa augmentasi\n# train_datagen = ImageDataGenerator(rescale=1./225,\n#                                    validation_split=0.3,)","metadata":{"execution":{"iopub.status.busy":"2021-07-07T04:49:10.254211Z","iopub.execute_input":"2021-07-07T04:49:10.254645Z","iopub.status.idle":"2021-07-07T04:49:10.261009Z","shell.execute_reply.started":"2021-07-07T04:49:10.254602Z","shell.execute_reply":"2021-07-07T04:49:10.260196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_datagen = ImageDataGenerator(rescale=1./255)","metadata":{"execution":{"iopub.status.busy":"2021-07-07T04:49:10.262647Z","iopub.execute_input":"2021-07-07T04:49:10.263094Z","iopub.status.idle":"2021-07-07T04:49:10.27114Z","shell.execute_reply.started":"2021-07-07T04:49:10.263055Z","shell.execute_reply":"2021-07-07T04:49:10.270311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator = train_datagen.flow_from_dataframe(dataframe=labels,\n                                                    directory=train_dir,\n                                                    subset='training',\n                                                    x_col=\"image_id\",\n                                                    y_col=\"label\",\n                                                    shuffle=True,\n                                                    target_size=(150,150),\n                                                    batch_size=32,\n                                                    class_mode='categorical')","metadata":{"execution":{"iopub.status.busy":"2021-07-07T04:49:10.272409Z","iopub.execute_input":"2021-07-07T04:49:10.272956Z","iopub.status.idle":"2021-07-07T04:49:17.861621Z","shell.execute_reply.started":"2021-07-07T04:49:10.272861Z","shell.execute_reply":"2021-07-07T04:49:17.860142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"flow_from_dataframe --> Takes the dataframe and the path to a directory + generates batches. The generated batches contain augmented/normalized data.","metadata":{}},{"cell_type":"code","source":"valid_generator = train_datagen.flow_from_dataframe(dataframe=labels,\n                                                    directory=train_dir,\n                                                    subset='validation',\n                                                    x_col=\"image_id\",\n                                                    y_col=\"label\",\n                                                    shuffle=True,\n                                                    target_size=(150,150),\n                                                    batch_size=32,\n                                                    class_mode='categorical')","metadata":{"execution":{"iopub.status.busy":"2021-07-07T04:49:17.862897Z","iopub.execute_input":"2021-07-07T04:49:17.863239Z","iopub.status.idle":"2021-07-07T04:49:18.051923Z","shell.execute_reply.started":"2021-07-07T04:49:17.8632Z","shell.execute_reply":"2021-07-07T04:49:18.05115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Create the base model from the pre-trained convnets","metadata":{}},{"cell_type":"code","source":"conv_base = ResNet50(weights='imagenet', include_top=False, input_shape=(150,150,3))","metadata":{"execution":{"iopub.status.busy":"2021-07-07T04:49:18.053261Z","iopub.execute_input":"2021-07-07T04:49:18.053629Z","iopub.status.idle":"2021-07-07T04:49:20.447193Z","shell.execute_reply.started":"2021-07-07T04:49:18.053591Z","shell.execute_reply":"2021-07-07T04:49:20.446342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This feature extractor converts each 150x150x3 image into a 5x5x2048 block of features. Let's see what it does to an example batch of images:","metadata":{}},{"cell_type":"code","source":"image_batch, label_batch = next(iter(train_generator))\nfeature_batch = conv_base(image_batch)\nprint(feature_batch.shape)","metadata":{"execution":{"iopub.status.busy":"2021-07-07T04:49:20.448491Z","iopub.execute_input":"2021-07-07T04:49:20.448858Z","iopub.status.idle":"2021-07-07T04:49:21.800462Z","shell.execute_reply.started":"2021-07-07T04:49:20.448821Z","shell.execute_reply":"2021-07-07T04:49:21.799513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It is important to freeze the convolutional base before you compile and train the model. Freezing (by setting layer.trainable = False) prevents the weights in a given layer from being updated during training.","metadata":{}},{"cell_type":"code","source":"conv_base.trainable = False","metadata":{"execution":{"iopub.status.busy":"2021-07-07T04:49:21.802059Z","iopub.execute_input":"2021-07-07T04:49:21.802422Z","iopub.status.idle":"2021-07-07T04:49:21.811741Z","shell.execute_reply.started":"2021-07-07T04:49:21.80238Z","shell.execute_reply":"2021-07-07T04:49:21.810908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"conv_base.summary()","metadata":{"execution":{"iopub.status.busy":"2021-07-07T04:49:21.812925Z","iopub.execute_input":"2021-07-07T04:49:21.81342Z","iopub.status.idle":"2021-07-07T04:49:21.900386Z","shell.execute_reply.started":"2021-07-07T04:49:21.813383Z","shell.execute_reply":"2021-07-07T04:49:21.899663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Q = tf.keras.layers.Flatten()","metadata":{"execution":{"iopub.status.busy":"2021-07-07T04:49:21.901419Z","iopub.execute_input":"2021-07-07T04:49:21.90177Z","iopub.status.idle":"2021-07-07T04:49:21.908859Z","shell.execute_reply.started":"2021-07-07T04:49:21.901734Z","shell.execute_reply":"2021-07-07T04:49:21.90799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fp1_layer = tf.keras.layers.Dense(128,activation=\"relu\",kernel_initializer=tf.keras.initializers.he_normal())\nprediction_layer = tf.keras.layers.Dense(1)","metadata":{"execution":{"iopub.status.busy":"2021-07-07T04:49:21.909977Z","iopub.execute_input":"2021-07-07T04:49:21.910443Z","iopub.status.idle":"2021-07-07T04:49:21.919569Z","shell.execute_reply.started":"2021-07-07T04:49:21.910384Z","shell.execute_reply":"2021-07-07T04:49:21.918711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs = tf.keras.Input(shape=(150, 150, 3))\n# x = data_augmentation(inputs)\n# x = preprocess_input(x)\nx = conv_base(inputs, training=False)\nx = Q(x)\nx = fp1_layer(x)\n#x = global_average_layer(x)\n#x = tf.keras.layers.Dropout(0.2)(x)\noutputs = prediction_layer(x)\nmodel = tf.keras.Model(inputs, outputs)","metadata":{"execution":{"iopub.status.busy":"2021-07-07T04:49:21.921284Z","iopub.execute_input":"2021-07-07T04:49:21.921597Z","iopub.status.idle":"2021-07-07T04:49:22.291992Z","shell.execute_reply.started":"2021-07-07T04:49:21.921572Z","shell.execute_reply":"2021-07-07T04:49:22.291208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_learning_rate = 0.0005\nmodel.compile(optimizer=tf.keras.optimizers.Adam(lr=base_learning_rate),\n              loss=tf.keras.losses.BinaryCrossentropy(from_logits=True),\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2021-07-07T04:49:22.294209Z","iopub.execute_input":"2021-07-07T04:49:22.294721Z","iopub.status.idle":"2021-07-07T04:49:22.311302Z","shell.execute_reply.started":"2021-07-07T04:49:22.294682Z","shell.execute_reply":"2021-07-07T04:49:22.310476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2021-07-07T04:49:22.312537Z","iopub.execute_input":"2021-07-07T04:49:22.312897Z","iopub.status.idle":"2021-07-07T04:49:22.331959Z","shell.execute_reply.started":"2021-07-07T04:49:22.31286Z","shell.execute_reply":"2021-07-07T04:49:22.331059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(model.trainable_variables)","metadata":{"execution":{"iopub.status.busy":"2021-07-07T04:49:22.333139Z","iopub.execute_input":"2021-07-07T04:49:22.333495Z","iopub.status.idle":"2021-07-07T04:49:22.338995Z","shell.execute_reply.started":"2021-07-07T04:49:22.333459Z","shell.execute_reply":"2021-07-07T04:49:22.338155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filepath=\"/kaggle/working/\"+\"pre_models_best_k.hdf5\"\nchkp = tf.keras.callbacks.ModelCheckpoint(filepath, save_best_only = True)\ncallbackslist = [chkp]","metadata":{"execution":{"iopub.status.busy":"2021-07-07T04:49:22.342963Z","iopub.execute_input":"2021-07-07T04:49:22.343465Z","iopub.status.idle":"2021-07-07T04:49:22.348277Z","shell.execute_reply.started":"2021-07-07T04:49:22.343411Z","shell.execute_reply":"2021-07-07T04:49:22.347339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"initial_epochs=10\nhistory = model.fit(train_generator,\n                    validation_data=valid_generator,\n                    callbacks=callbackslist,\n                    epochs=initial_epochs\n                   )","metadata":{"execution":{"iopub.status.busy":"2021-07-07T04:49:22.349957Z","iopub.execute_input":"2021-07-07T04:49:22.350378Z","iopub.status.idle":"2021-07-07T05:34:26.930852Z","shell.execute_reply.started":"2021-07-07T04:49:22.350344Z","shell.execute_reply":"2021-07-07T05:34:26.930044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\n\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\nepochs = range(1, len(acc)+1)","metadata":{"execution":{"iopub.status.busy":"2021-07-07T05:34:26.932261Z","iopub.execute_input":"2021-07-07T05:34:26.932625Z","iopub.status.idle":"2021-07-07T05:34:26.937279Z","shell.execute_reply.started":"2021-07-07T05:34:26.932587Z","shell.execute_reply":"2021-07-07T05:34:26.936463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure()\n\nplt.plot(epochs, loss, 'bo', label='Train Loss')\nplt.plot(epochs, val_loss, 'b', label='Validation Loss')\nplt.title('Loss vs Epochs')\nplt.ylabel('Loss')\nplt.xlabel('Epochs')\nplt.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-07T05:34:26.93848Z","iopub.execute_input":"2021-07-07T05:34:26.938962Z","iopub.status.idle":"2021-07-07T05:34:27.097886Z","shell.execute_reply.started":"2021-07-07T05:34:26.938923Z","shell.execute_reply":"2021-07-07T05:34:27.097109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure()\n\nplt.plot(epochs, acc, 'bo', label='Training Accuracy')\nplt.plot(epochs, val_acc, 'b', label='Validation Accuracy')\nplt.title('Accuracy vs Epochs')\nplt.ylabel('Accuracy')\nplt.xlabel('Epochs')\nplt.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-07T05:34:27.098981Z","iopub.execute_input":"2021-07-07T05:34:27.099296Z","iopub.status.idle":"2021-07-07T05:34:27.24053Z","shell.execute_reply.started":"2021-07-07T05:34:27.09926Z","shell.execute_reply":"2021-07-07T05:34:27.239778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Predict using test set images from kaggle**","metadata":{}},{"cell_type":"code","source":"submission = pd.read_csv(\"../input/cassava-leaf-disease-classification/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-07-07T05:34:27.241705Z","iopub.execute_input":"2021-07-07T05:34:27.242041Z","iopub.status.idle":"2021-07-07T05:34:27.25523Z","shell.execute_reply.started":"2021-07-07T05:34:27.242004Z","shell.execute_reply":"2021-07-07T05:34:27.254509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-07T05:34:27.256397Z","iopub.execute_input":"2021-07-07T05:34:27.256858Z","iopub.status.idle":"2021-07-07T05:34:27.265132Z","shell.execute_reply.started":"2021-07-07T05:34:27.25682Z","shell.execute_reply":"2021-07-07T05:34:27.264215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = load_img(\"../input/cassava-leaf-disease-classification/test_images/2216849948.jpg\")\nimg = img_to_array(img)\nimg = smart_resize(img, (150,150))\nimg = tf.reshape(img, (-1, 150, 150, 3))","metadata":{"execution":{"iopub.status.busy":"2021-07-07T05:34:27.266444Z","iopub.execute_input":"2021-07-07T05:34:27.26711Z","iopub.status.idle":"2021-07-07T05:34:27.307106Z","shell.execute_reply.started":"2021-07-07T05:34:27.267045Z","shell.execute_reply":"2021-07-07T05:34:27.306409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model.predict(img/255.)\npred = np.argmax(pred)\nsubmission_result = pd.DataFrame({'image_id' : submission.image_id, 'label' : pred})\nsubmission_result","metadata":{"execution":{"iopub.status.busy":"2021-07-07T05:34:27.309163Z","iopub.execute_input":"2021-07-07T05:34:27.30964Z","iopub.status.idle":"2021-07-07T05:34:28.2372Z","shell.execute_reply.started":"2021-07-07T05:34:27.309603Z","shell.execute_reply":"2021-07-07T05:34:28.236403Z"},"trusted":true},"execution_count":null,"outputs":[]}]}