{"cells":[{"metadata":{"execution":{"iopub.execute_input":"2021-01-03T18:03:53.074503Z","iopub.status.busy":"2021-01-03T18:03:53.073807Z","iopub.status.idle":"2021-01-03T18:03:59.273877Z","shell.execute_reply":"2021-01-03T18:03:59.272351Z"},"id":"8b-taxqLMShr","papermill":{"duration":6.21974,"end_time":"2021-01-03T18:03:59.274014","exception":false,"start_time":"2021-01-03T18:03:53.054274","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nimport tensorflow as tf\nfrom kaggle_datasets import KaggleDatasets\nfrom sklearn.metrics import accuracy_score\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras import datasets, layers, models\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout, LSTM, BatchNormalization\nfrom tensorflow.keras.callbacks import TensorBoard\nfrom tensorflow.keras.callbacks import ModelCheckpoint\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau\nfrom PIL import Image \nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nfrom tensorflow.keras.layers import Conv2D , MaxPool2D , Flatten\n\nfrom tensorflow.keras.layers import Input, Lambda, Dense, Flatten\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.applications.vgg16 import VGG16\nfrom tensorflow.keras.applications.vgg16 import preprocess_input\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator,load_img\nfrom tensorflow.keras.models import Sequential\nimport numpy as np\nfrom glob import glob\nimport os, cv2, json","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"try:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Device:', tpu.master())\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nexcept:\n    strategy = tf.distribute.get_strategy()\nprint('Number of replicas:', strategy.num_replicas_in_sync)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-03T18:03:59.298889Z","iopub.status.busy":"2021-01-03T18:03:59.298269Z","iopub.status.idle":"2021-01-03T18:03:59.308905Z","shell.execute_reply":"2021-01-03T18:03:59.308384Z"},"id":"DPmMc6dbsXTp","outputId":"34d3e06c-8ce0-444a-8b62-3a407f077809","papermill":{"duration":0.024635,"end_time":"2021-01-03T18:03:59.309007","exception":false,"start_time":"2021-01-03T18:03:59.284372","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"# For easy acces to files\nWORK_DIR = \"../input/cassava-leaf-disease-classification/\"\nos.listdir(WORK_DIR)\n","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-03T18:03:59.337565Z","iopub.status.busy":"2021-01-03T18:03:59.33692Z","iopub.status.idle":"2021-01-03T18:03:59.378403Z","shell.execute_reply":"2021-01-03T18:03:59.377489Z"},"id":"kaKs505TQfBT","outputId":"9bf60ffb-1f4d-424d-b515-38a3b27f83df","papermill":{"duration":0.05893,"end_time":"2021-01-03T18:03:59.378513","exception":false,"start_time":"2021-01-03T18:03:59.319583","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"train_labels = pd.read_csv(os.path.join(WORK_DIR, \"train.csv\"))\ntrain_labels.head()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-03T18:03:59.413225Z","iopub.status.busy":"2021-01-03T18:03:59.412661Z","iopub.status.idle":"2021-01-03T18:03:59.585163Z","shell.execute_reply":"2021-01-03T18:03:59.584682Z"},"id":"K_ZjrinlR9xh","outputId":"fe3a99b0-657f-4581-ca27-1059ebce6243","papermill":{"duration":0.195917,"end_time":"2021-01-03T18:03:59.585266","exception":false,"start_time":"2021-01-03T18:03:59.389349","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"sns.countplot(train_labels.label, edgecolor = 'black',\n              palette = sns.color_palette(\"viridis\", 5))\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Describing Image Size,Batch Size,Epoch,Validations Steps and Steps Per Epcohs"},{"metadata":{"execution":{"iopub.execute_input":"2021-01-03T18:03:59.664526Z","iopub.status.busy":"2021-01-03T18:03:59.663781Z","iopub.status.idle":"2021-01-03T18:03:59.66642Z","shell.execute_reply":"2021-01-03T18:03:59.666906Z"},"id":"EQqSpaEiVBGy","papermill":{"duration":0.020147,"end_time":"2021-01-03T18:03:59.667014","exception":false,"start_time":"2021-01-03T18:03:59.646867","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"BATCH_SIZE = 64\nSTEPS_PER_EPOCH = len(train_labels)*0.8 / BATCH_SIZE\nVALIDATION_STEPS = len(train_labels)*0.3 / BATCH_SIZE\nEPOCHS = 5\nTARGET_SIZE = 512","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Perparing Dataset"},{"metadata":{"execution":{"iopub.execute_input":"2021-01-03T18:03:59.734189Z","iopub.status.busy":"2021-01-03T18:03:59.732012Z","iopub.status.idle":"2021-01-03T18:04:22.860485Z","shell.execute_reply":"2021-01-03T18:04:22.861518Z"},"id":"UckWV8fEVMdv","outputId":"a05bbe06-138e-4b2f-cf73-b8d6bfc4d8b9","papermill":{"duration":23.181568,"end_time":"2021-01-03T18:04:22.861694","exception":false,"start_time":"2021-01-03T18:03:59.680126","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"train_labels.label = train_labels.label.astype('str')\n\ntrain_generator = ImageDataGenerator(validation_split = 0.2,\n                                     preprocessing_function = None,\n                                     zoom_range = 0.15,\n                                     cval = 0.,\n                                     horizontal_flip = True,\n                                     vertical_flip = True,\n                                     fill_mode = 'nearest',\n                                     shear_range = 0.15,\n                                     height_shift_range = 0.15,\n                                     width_shift_range = 0.15) \\\n    .flow_from_dataframe(train_labels,\n                         directory = os.path.join(WORK_DIR, \"train_images\"),\n                         subset = \"training\",\n                         x_col = \"image_id\",\n                         y_col = \"label\",\n                         target_size = (TARGET_SIZE, TARGET_SIZE),\n                         batch_size = BATCH_SIZE,\n                         class_mode = \"sparse\")\n\nvalidation_generator = ImageDataGenerator(validation_split = 0.2) \\\n    .flow_from_dataframe(train_labels,\n                         directory = os.path.join(WORK_DIR, \"train_images\"),\n                         subset = \"validation\",\n                         x_col = \"image_id\",\n                         y_col = \"label\",\n                         target_size = (TARGET_SIZE, TARGET_SIZE),\n                         batch_size = BATCH_SIZE,\n                         class_mode = \"sparse\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Importing VGG16 using Transfer Learning"},{"metadata":{"trusted":true},"cell_type":"code","source":"IMAGE_SIZE = [512, 512]\nvgg16 = VGG16(input_shape=IMAGE_SIZE + [3], weights='imagenet', include_top=False)\n\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Adding a softmax layers for classification.You can add add more layers for more accuracy."},{"metadata":{"trusted":true},"cell_type":"code","source":"# don't train existing weights\nfor layer in vgg16.layers:\n    layer.trainable = False\n    \nx = Flatten()(vgg16.output)\nprediction = Dense(5, activation='softmax')(x)\n\n# create a model object\nmodel = Model(inputs=vgg16.input, outputs=prediction)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# tell the model what cost and optimization method to use\nmodel.compile(\n  loss='sparse_categorical_crossentropy',\n  optimizer='adam',\n  metrics=['accuracy']\n)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model_save = ModelCheckpoint('./baseline_model_vgg16.h5', \n                             save_best_only = True, \n                             save_weights_only = True,\n                             monitor = 'val_loss', \n                             mode = 'min', verbose = 1)\nearly_stop = EarlyStopping(monitor = 'val_loss', min_delta = 0.001, \n                           patience = 5, mode = 'min', verbose = 1,\n                           restore_best_weights = True)\nreduce_lr = ReduceLROnPlateau(monitor = 'val_loss', factor = 0.1, \n                              patience = 2, min_delta = 0.001, \n                              mode = 'min', verbose = 1)\n\n\n\nhistory = model.fit_generator(\n    train_generator,\n    steps_per_epoch = STEPS_PER_EPOCH,\n    epochs = EPOCHS,\n    validation_data = validation_generator,\n    validation_steps = VALIDATION_STEPS,\n    callbacks = [model_save, early_stop, reduce_lr]\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"s = pd.read_csv(os.path.join(WORK_DIR, \"sample_submission.csv\"))\ns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds = []\n\nfor image_id in s.image_id:\n    image = Image.open(os.path.join(WORK_DIR,  \"test_images\", image_id))\n    image = image.resize((TARGET_SIZE, TARGET_SIZE))\n    image = np.expand_dims(image, axis = 0)\n    preds.append(np.argmax(model.predict(image)))\n\ns['label'] = preds\ns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"s.to_csv('submission_5.csv', index = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"s.to_csv(\"/kaggle/working/submission.csv\", index=False,header=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}