{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        os.path.join(dirname, filename)\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-02-19T04:58:27.898651Z","iopub.execute_input":"2022-02-19T04:58:27.899022Z","iopub.status.idle":"2022-02-19T04:58:33.170667Z","shell.execute_reply.started":"2022-02-19T04:58:27.898975Z","shell.execute_reply":"2022-02-19T04:58:33.169924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-02-19T04:58:33.173944Z","iopub.execute_input":"2022-02-19T04:58:33.174165Z","iopub.status.idle":"2022-02-19T04:58:33.177603Z","shell.execute_reply.started":"2022-02-19T04:58:33.174140Z","shell.execute_reply":"2022-02-19T04:58:33.176871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.layers import Input,Lambda,Dense,Flatten,Conv2D\nfrom keras.models import Model\nfrom keras.preprocessing import image\nfrom keras.applications.vgg16 import VGG16\nfrom keras.preprocessing.image import ImageDataGenerator,load_img\nfrom keras.models import Sequential\nfrom glob import glob\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport cv2\nimport tensorflow as tf\nfrom PIL import Image\nimport os\nimport random","metadata":{"execution":{"iopub.status.busy":"2022-02-19T04:58:33.179246Z","iopub.execute_input":"2022-02-19T04:58:33.179755Z","iopub.status.idle":"2022-02-19T04:58:33.190645Z","shell.execute_reply.started":"2022-02-19T04:58:33.179720Z","shell.execute_reply":"2022-02-19T04:58:33.189888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(r\"../input/cassava-leaf-disease-classification/train.csv\")\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-19T04:58:33.193166Z","iopub.execute_input":"2022-02-19T04:58:33.193403Z","iopub.status.idle":"2022-02-19T04:58:33.221166Z","shell.execute_reply.started":"2022-02-19T04:58:33.193372Z","shell.execute_reply":"2022-02-19T04:58:33.220525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,6))\nsns.countplot(y='label',data=df)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-19T04:58:33.222454Z","iopub.execute_input":"2022-02-19T04:58:33.222902Z","iopub.status.idle":"2022-02-19T04:58:33.395442Z","shell.execute_reply.started":"2022-02-19T04:58:33.222867Z","shell.execute_reply":"2022-02-19T04:58:33.394808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(22,25))\npath = '../input/cassava-leaf-disease-classification/train_images/'\ntemp = df[df['label']==1]['image_id']  #all the images of dogs will be stored in temp\nstart = random.randint(0,len(temp))  #this generates a random number between 0 and len(temp)\nfiles = temp[start:start+25]         #in files we store the first 25 continuous images between the random number generated \n                                     #since temp is basically images of dogs from step 2\nfor index,file in enumerate(files):\n  plt.subplot(5,5,index+1)\n  img = load_img(path+file)\n  img = np.array(img)\n  plt.imshow(img)\n  plt.imshow(img)\n  plt.title('Casava Leaf')\n  plt.axis()","metadata":{"execution":{"iopub.status.busy":"2022-02-19T04:58:33.396814Z","iopub.execute_input":"2022-02-19T04:58:33.397257Z","iopub.status.idle":"2022-02-19T04:58:39.675663Z","shell.execute_reply.started":"2022-02-19T04:58:33.397221Z","shell.execute_reply":"2022-02-19T04:58:39.674866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"WORK_DIR = '../input/cassava-leaf-disease-classification'","metadata":{"execution":{"iopub.status.busy":"2022-02-19T04:58:39.676739Z","iopub.execute_input":"2022-02-19T04:58:39.676992Z","iopub.status.idle":"2022-02-19T04:58:39.681556Z","shell.execute_reply.started":"2022-02-19T04:58:39.676954Z","shell.execute_reply":"2022-02-19T04:58:39.680694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras import models, layers\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.optimizers import Adam\nimport cv2","metadata":{"execution":{"iopub.status.busy":"2022-02-19T04:58:39.683029Z","iopub.execute_input":"2022-02-19T04:58:39.683407Z","iopub.status.idle":"2022-02-19T04:58:40.107245Z","shell.execute_reply.started":"2022-02-19T04:58:39.683375Z","shell.execute_reply":"2022-02-19T04:58:40.106506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Total number of images in train_images folder","metadata":{}},{"cell_type":"code","source":"import os\nprint(len(os.listdir('../input/cassava-leaf-disease-classification/train_images')))","metadata":{"execution":{"iopub.status.busy":"2022-02-19T04:58:40.108494Z","iopub.execute_input":"2022-02-19T04:58:40.108732Z","iopub.status.idle":"2022-02-19T04:58:40.124775Z","shell.execute_reply.started":"2022-02-19T04:58:40.108699Z","shell.execute_reply":"2022-02-19T04:58:40.123928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.label = df.label.astype('str')\n\ntrain_datagen = ImageDataGenerator(validation_split = 0.2,\n                                     preprocessing_function = None,\n                                     rotation_range = 45,\n                                     zoom_range = 0.2,\n                                     horizontal_flip = True,\n                                     vertical_flip = True,\n                                     fill_mode = 'nearest',\n                                     shear_range = 0.1,\n                                     height_shift_range = 0.1,\n                                     width_shift_range = 0.1)\n\ntrain_generator = train_datagen.flow_from_dataframe(df,\n                         directory = os.path.join(WORK_DIR, \"train_images\"),\n                         subset = \"training\",\n                         x_col = \"image_id\",\n                         y_col = \"label\",\n                         target_size = (224,224),\n                         batch_size = 32,\n                         class_mode = \"categorical\")\n\n\nvalidation_datagen = ImageDataGenerator(validation_split = 0.2)\n\nvalidation_generator = validation_datagen.flow_from_dataframe(df,\n                         directory = os.path.join(WORK_DIR, \"train_images\"),\n                         subset = \"validation\",\n                         x_col = \"image_id\",\n                         y_col = \"label\",\n                         target_size = (224, 224),\n                         batch_size = 32,\n                         class_mode = \"categorical\")","metadata":{"execution":{"iopub.status.busy":"2022-02-19T04:58:40.127937Z","iopub.execute_input":"2022-02-19T04:58:40.128683Z","iopub.status.idle":"2022-02-19T04:58:50.599180Z","shell.execute_reply.started":"2022-02-19T04:58:40.128648Z","shell.execute_reply":"2022-02-19T04:58:50.597678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vgg = VGG16(input_shape=[224,224,3],weights='imagenet',include_top=False)","metadata":{"execution":{"iopub.status.busy":"2022-02-19T04:58:50.600441Z","iopub.execute_input":"2022-02-19T04:58:50.600686Z","iopub.status.idle":"2022-02-19T04:58:53.978094Z","shell.execute_reply.started":"2022-02-19T04:58:50.600652Z","shell.execute_reply":"2022-02-19T04:58:53.977359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for layer in vgg.layers:\n    layer.trainable = False","metadata":{"execution":{"iopub.status.busy":"2022-02-19T04:58:53.979438Z","iopub.execute_input":"2022-02-19T04:58:53.979766Z","iopub.status.idle":"2022-02-19T04:58:53.985065Z","shell.execute_reply.started":"2022-02-19T04:58:53.979728Z","shell.execute_reply":"2022-02-19T04:58:53.983946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = Flatten()(vgg.output)","metadata":{"execution":{"iopub.status.busy":"2022-02-19T04:58:53.986483Z","iopub.execute_input":"2022-02-19T04:58:53.986981Z","iopub.status.idle":"2022-02-19T04:58:53.998116Z","shell.execute_reply.started":"2022-02-19T04:58:53.986946Z","shell.execute_reply":"2022-02-19T04:58:53.997371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vgg_output = Dense(5,activation='softmax')(x)\n#This is the last layer cnsisting of the various classes we are replacing the include_top = False layer with this one","metadata":{"execution":{"iopub.status.busy":"2022-02-19T04:58:54.000827Z","iopub.execute_input":"2022-02-19T04:58:54.001008Z","iopub.status.idle":"2022-02-19T04:58:54.014283Z","shell.execute_reply.started":"2022-02-19T04:58:54.000986Z","shell.execute_reply":"2022-02-19T04:58:54.013581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(df)","metadata":{"execution":{"iopub.status.busy":"2022-02-19T04:58:54.015514Z","iopub.execute_input":"2022-02-19T04:58:54.015938Z","iopub.status.idle":"2022-02-19T04:58:54.022033Z","shell.execute_reply.started":"2022-02-19T04:58:54.015898Z","shell.execute_reply":"2022-02-19T04:58:54.021240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Model(inputs=vgg.input, outputs=vgg_output)","metadata":{"execution":{"iopub.status.busy":"2022-02-19T04:58:54.023295Z","iopub.execute_input":"2022-02-19T04:58:54.023753Z","iopub.status.idle":"2022-02-19T04:58:54.034141Z","shell.execute_reply.started":"2022-02-19T04:58:54.023717Z","shell.execute_reply":"2022-02-19T04:58:54.033431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-02-19T04:58:54.035894Z","iopub.execute_input":"2022-02-19T04:58:54.036226Z","iopub.status.idle":"2022-02-19T04:58:54.051569Z","shell.execute_reply.started":"2022-02-19T04:58:54.036190Z","shell.execute_reply":"2022-02-19T04:58:54.050785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer=Adam(lr=1e-4),\n              loss='categorical_crossentropy',\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-02-19T05:26:13.519973Z","iopub.execute_input":"2022-02-19T05:26:13.520661Z","iopub.status.idle":"2022-02-19T05:26:13.531225Z","shell.execute_reply.started":"2022-02-19T05:26:13.520623Z","shell.execute_reply":"2022-02-19T05:26:13.530476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n  train_generator,\n  validation_data=validation_generator,\n  epochs=10,\n  steps_per_epoch=len(train_generator),\n  validation_steps=len(validation_generator)\n)","metadata":{"execution":{"iopub.status.busy":"2022-02-19T05:26:19.799250Z","iopub.execute_input":"2022-02-19T05:26:19.801327Z","iopub.status.idle":"2022-02-19T06:23:02.503691Z","shell.execute_reply.started":"2022-02-19T05:26:19.801291Z","shell.execute_reply":"2022-02-19T06:23:02.502989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#loss\nplt.plot(history.history['loss'],label='train loss')\nplt.plot(history.history['val_loss'],label = 'val loss')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-19T06:23:17.833584Z","iopub.execute_input":"2022-02-19T06:23:17.834367Z","iopub.status.idle":"2022-02-19T06:23:18.008516Z","shell.execute_reply.started":"2022-02-19T06:23:17.834328Z","shell.execute_reply":"2022-02-19T06:23:18.007838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#accuracy\nplt.plot(history.history['accuracy'],label='train acc')\nplt.plot(history.history['val_accuracy'],label='val acc')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-19T06:23:58.388624Z","iopub.execute_input":"2022-02-19T06:23:58.389004Z","iopub.status.idle":"2022-02-19T06:23:58.573121Z","shell.execute_reply.started":"2022-02-19T06:23:58.388947Z","shell.execute_reply":"2022-02-19T06:23:58.572422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss = pd.read_csv(os.path.join(WORK_DIR, \"sample_submission.csv\"))\nss","metadata":{"execution":{"iopub.status.busy":"2022-02-19T06:27:41.259794Z","iopub.execute_input":"2022-02-19T06:27:41.260061Z","iopub.status.idle":"2022-02-19T06:27:41.274281Z","shell.execute_reply.started":"2022-02-19T06:27:41.260032Z","shell.execute_reply":"2022-02-19T06:27:41.273643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = []\n\nfor image_id in ss.image_id:\n    image = Image.open(os.path.join(WORK_DIR,  \"test_images\", image_id))\n    image = image.resize((224,224))\n    image = np.expand_dims(image, axis = 0)\n    preds.append(np.argmax(model.predict(image)))\n\nss['label'] = preds\nss","metadata":{"execution":{"iopub.status.busy":"2022-02-19T06:28:43.237583Z","iopub.execute_input":"2022-02-19T06:28:43.238362Z","iopub.status.idle":"2022-02-19T06:28:43.651899Z","shell.execute_reply.started":"2022-02-19T06:28:43.238317Z","shell.execute_reply":"2022-02-19T06:28:43.651255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss.to_csv('submission.csv', index = False)\n","metadata":{"execution":{"iopub.status.busy":"2022-02-19T06:28:59.935600Z","iopub.execute_input":"2022-02-19T06:28:59.935867Z","iopub.status.idle":"2022-02-19T06:28:59.943609Z","shell.execute_reply.started":"2022-02-19T06:28:59.935835Z","shell.execute_reply":"2022-02-19T06:28:59.942821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.models import Model\nmodel.save('cassava.h5')","metadata":{"execution":{"iopub.status.busy":"2022-02-19T06:29:33.563505Z","iopub.execute_input":"2022-02-19T06:29:33.563763Z","iopub.status.idle":"2022-02-19T06:29:33.688046Z","shell.execute_reply.started":"2022-02-19T06:29:33.563733Z","shell.execute_reply":"2022-02-19T06:29:33.687245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}