{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport tensorflow as tf\n\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom tensorflow.keras.optimizers import RMSprop\nfrom matplotlib import pyplot as plt\nfrom tensorflow.keras.preprocessing.image import img_to_array\nfrom tensorflow.keras.preprocessing.image import load_img\nfrom tensorflow.keras.preprocessing import image_dataset_from_directory\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        os.path.join(dirname, filename)\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-05-20T13:54:24.194164Z","iopub.execute_input":"2021-05-20T13:54:24.194544Z","iopub.status.idle":"2021-05-20T13:54:25.932129Z","shell.execute_reply.started":"2021-05-20T13:54:24.194455Z","shell.execute_reply":"2021-05-20T13:54:25.931245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ndf = pd.read_csv('/kaggle/input/plant-pathology-2021-fgvc8/train.csv')\nsubmission = pd.read_csv('../input/plant-pathology-2021-fgvc8/sample_submission.csv')\ntrain, val = train_test_split(df, test_size=0.2)","metadata":{"execution":{"iopub.status.busy":"2021-05-20T13:54:25.934701Z","iopub.execute_input":"2021-05-20T13:54:25.935092Z","iopub.status.idle":"2021-05-20T13:54:26.167129Z","shell.execute_reply.started":"2021-05-20T13:54:25.935053Z","shell.execute_reply":"2021-05-20T13:54:26.166225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Reads in the the csv to pandas dataframe. 80% of the data is used for training and 20% used for validation.","metadata":{}},{"cell_type":"code","source":"plt.imshow(np.array(img_to_array(load_img('../input/plant-pathology-2021-fgvc8/train_images/800113bb65efe69e.jpg')).astype(\"uint8\")))","metadata":{"execution":{"iopub.status.busy":"2021-05-20T13:54:26.168462Z","iopub.execute_input":"2021-05-20T13:54:26.168788Z","iopub.status.idle":"2021-05-20T13:54:27.476990Z","shell.execute_reply.started":"2021-05-20T13:54:26.168753Z","shell.execute_reply":"2021-05-20T13:54:27.476072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The first image from the training set","metadata":{}},{"cell_type":"code","source":"plt.imshow(np.array(img_to_array(load_img('../input/plant-pathology-2021-fgvc8/train_images/800113bb65efe69e.jpg',target_size=(128,128))).astype(\"uint8\")))","metadata":{"execution":{"iopub.status.busy":"2021-05-20T13:54:27.478510Z","iopub.execute_input":"2021-05-20T13:54:27.479125Z","iopub.status.idle":"2021-05-20T13:54:27.932042Z","shell.execute_reply.started":"2021-05-20T13:54:27.479079Z","shell.execute_reply":"2021-05-20T13:54:27.931012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The first image of the training set resiszed to 128x128, because the raw images are very large and take a long time to load and process.","metadata":{}},{"cell_type":"markdown","source":"The image data generator will resize the images and load them as needed for the model to train.","metadata":{}},{"cell_type":"code","source":"image_datagen = ImageDataGenerator()\n\ntrain_ds = image_datagen.flow_from_dataframe(\n    dataframe = train,\n    directory = '../input/plant-pathology-2021-fgvc8/train_images',\n    x_col = \"image\",\n    y_col = \"labels\",\n    target_size = (128,128),\n    class_mode='categorical',\n    batch_size = 64,\n    shuffle = True,\n    seed = 9,\n    validate_filenames = False,\n)\n\nval_ds = image_datagen.flow_from_dataframe(\n    dataframe = val,\n    directory = '../input/plant-pathology-2021-fgvc8/train_images',\n    x_col = \"image\",\n    y_col = \"labels\",\n    target_size = (128,128),\n    class_mode='categorical',\n    batch_size = 64,\n    shuffle = True,\n    seed = 9,\n    validate_filenames = False\n)\ntest_ds = image_datagen.flow_from_dataframe(\n    dataframe=submission,\n    directory='../input/plant-pathology-2021-fgvc8/test_images',\n    x_col = \"image\",\n    y_col = \"labels\",\n    target_size = (128,128),\n    class_mode='categorical',\n    validate_filenames=False)","metadata":{"execution":{"iopub.status.busy":"2021-05-20T13:54:27.933413Z","iopub.execute_input":"2021-05-20T13:54:27.933763Z","iopub.status.idle":"2021-05-20T13:54:28.023813Z","shell.execute_reply.started":"2021-05-20T13:54:27.933727Z","shell.execute_reply":"2021-05-20T13:54:28.022935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Convolutional networks are the most used machine leanring type for image classification","metadata":{}},{"cell_type":"code","source":"model = Sequential([\n  layers.experimental.preprocessing.Rescaling(1./255, input_shape=(128, 128, 3)),\n  layers.Conv2D(64, (3,3), padding='same', activation='relu'),\n  layers.MaxPooling2D((2,2)),\n  layers.Conv2D(64, (3,3), padding='same', activation='relu'),\n  layers.MaxPooling2D((2,2)),\n  layers.Conv2D(64, (3,3), padding='same', activation='relu'),\n  layers.MaxPooling2D((2,2)),\n  layers.Flatten(),\n  layers.Dense(12, activation='softmax')\n])","metadata":{"execution":{"iopub.status.busy":"2021-05-20T13:54:28.027292Z","iopub.execute_input":"2021-05-20T13:54:28.027547Z","iopub.status.idle":"2021-05-20T13:54:30.429739Z","shell.execute_reply.started":"2021-05-20T13:54:28.027521Z","shell.execute_reply":"2021-05-20T13:54:30.428874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The optimizer adam is also the most used type of optimizer for image classifcation. The loss function categorical crossentropy is a used for multi-class classification problems.","metadata":{}},{"cell_type":"code","source":"model.compile(optimizer='adam',\n              loss='categorical_crossentropy',\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2021-05-20T13:54:30.431576Z","iopub.execute_input":"2021-05-20T13:54:30.431909Z","iopub.status.idle":"2021-05-20T13:54:30.447994Z","shell.execute_reply.started":"2021-05-20T13:54:30.431861Z","shell.execute_reply":"2021-05-20T13:54:30.447198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Early stopping stops the model from training if the loss function on the validation data is not improving. Early stopping can help prevent over fitting and waisted time training.","metadata":{}},{"cell_type":"code","source":"early_stopping = EarlyStopping(monitor = 'val_loss',min_delta=.01,patience=3,restore_best_weights=True)","metadata":{"execution":{"iopub.status.busy":"2021-05-20T13:54:30.451225Z","iopub.execute_input":"2021-05-20T13:54:30.451612Z","iopub.status.idle":"2021-05-20T13:54:30.457470Z","shell.execute_reply.started":"2021-05-20T13:54:30.451584Z","shell.execute_reply":"2021-05-20T13:54:30.456722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n  train_ds,\n  validation_data=val_ds,\n  epochs=10,\n  callbacks=early_stopping,\n  steps_per_epoch=100,\n  validation_steps=20,\n  max_queue_size=1000,\n  use_multiprocessing=True,\n  workers=4\n)","metadata":{"execution":{"iopub.status.busy":"2021-05-20T13:54:30.458895Z","iopub.execute_input":"2021-05-20T13:54:30.459406Z","iopub.status.idle":"2021-05-20T18:57:31.266048Z","shell.execute_reply.started":"2021-05-20T13:54:30.459368Z","shell.execute_reply":"2021-05-20T18:57:31.263131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Shows the accuracy on the training data and validation data over each epoch","metadata":{}},{"cell_type":"code","source":"plt.plot(history.history['accuracy'])\nplt.plot(history.history['val_accuracy'])\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-05-20T18:57:31.277772Z","iopub.execute_input":"2021-05-20T18:57:31.289084Z","iopub.status.idle":"2021-05-20T18:57:32.949943Z","shell.execute_reply.started":"2021-05-20T18:57:31.289031Z","shell.execute_reply":"2021-05-20T18:57:32.948983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_key(val_array, my_dict):\n    keys = []\n    for val in val_array:\n        for key, value in my_dict.items():\n             if val == value:\n                keys.append(key)\n    return keys","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['labels'] = get_key(np.argmax(model.predict(test_ds), axis=1), train_ds.class_indices)","metadata":{"execution":{"iopub.status.busy":"2021-05-20T19:37:43.180575Z","iopub.execute_input":"2021-05-20T19:37:43.180926Z","iopub.status.idle":"2021-05-20T19:37:44.138111Z","shell.execute_reply.started":"2021-05-20T19:37:43.180875Z","shell.execute_reply":"2021-05-20T19:37:44.137304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('./plant_pathology_cnn')","metadata":{"execution":{"iopub.status.busy":"2021-05-20T19:38:41.182745Z","iopub.execute_input":"2021-05-20T19:38:41.183129Z","iopub.status.idle":"2021-05-20T19:38:42.476461Z","shell.execute_reply.started":"2021-05-20T19:38:41.183087Z","shell.execute_reply":"2021-05-20T19:38:42.475678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd /kaggle/working\nfrom IPython.display import FileLink as FileLink(r'*plant_pathology_cnn*')","metadata":{"execution":{"iopub.status.busy":"2021-05-20T19:42:51.521044Z","iopub.execute_input":"2021-05-20T19:42:51.521379Z","iopub.status.idle":"2021-05-20T19:42:51.527145Z","shell.execute_reply.started":"2021-05-20T19:42:51.521349Z","shell.execute_reply":"2021-05-20T19:42:51.525912Z"},"trusted":true},"execution_count":null,"outputs":[]}]}