{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport os\nimport numpy as np\nfrom pathlib import Path\nimport cv2\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nfrom skimage.transform import resize, rescale\nimport seaborn as sns\nimport random\nimport warnings\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nwarnings.filterwarnings('ignore')\nimport gc\ngc.enable()\nimport wandb\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '2'\n\nimport tensorflow as tf\nfrom tensorflow import keras\nimport keras_tuner as kt\nfrom tensorflow.keras import layers\nimport tensorflow.keras as keras\nfrom tensorflow.keras.layers.experimental import preprocessing\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.preprocessing import image_dataset_from_directory\nfrom tensorflow.keras.models import load_model\n\nfrom kaggle_secrets import UserSecretsClient\nuser_secrets = UserSecretsClient()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-13T16:10:08.701757Z","iopub.execute_input":"2022-08-13T16:10:08.702407Z","iopub.status.idle":"2022-08-13T16:10:16.933381Z","shell.execute_reply.started":"2022-08-13T16:10:08.702298Z","shell.execute_reply":"2022-08-13T16:10:16.932262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Config and EDA","metadata":{}},{"cell_type":"code","source":" cfg = {\n    'train_path': '/kaggle/input/paddy-disease-classification/train_images/',\n    'train_df_path': Path('/kaggle/input/paddy-disease-classification/train.csv'),\n    'img_size': (480,480),\n    'epochs': 100,\n    'num_workers': 16,\n    'lr':1e-03,\n    'batch_size': 16,\n    'device': 'cuda',\n    'lr_values': [1e-2, 2e-2, 1e-3, 2e-3, 1e-4, 2e-4]\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:10:16.935438Z","iopub.execute_input":"2022-08-13T16:10:16.936367Z","iopub.status.idle":"2022-08-13T16:10:16.946111Z","shell.execute_reply.started":"2022-08-13T16:10:16.936308Z","shell.execute_reply":"2022-08-13T16:10:16.944392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(cfg['train_df_path'])\nprint(df)\nprint(df.info())","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:10:16.948820Z","iopub.execute_input":"2022-08-13T16:10:16.949505Z","iopub.status.idle":"2022-08-13T16:10:17.008489Z","shell.execute_reply.started":"2022-08-13T16:10:16.949468Z","shell.execute_reply":"2022-08-13T16:10:17.007011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = ['label', 'variety', 'age']\nfor idx, col in enumerate(cols):\n    plt.figure(figsize=(22,10))\n    sns.countplot(x=col, data=df)\n    plt.title(f\"Count of images by {col} type\")\n    plt.ylabel(\"Number of images\")\n    plt.grid()\n    plt.show()\n    print('\\n')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:10:17.011173Z","iopub.execute_input":"2022-08-13T16:10:17.011626Z","iopub.status.idle":"2022-08-13T16:10:17.880433Z","shell.execute_reply.started":"2022-08-13T16:10:17.011587Z","shell.execute_reply":"2022-08-13T16:10:17.878981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = df.label\nimgs = df.image_id\ndf['paths'] = None\nfor idx in tqdm(range(len(df))):\n    df.paths[idx] = str(labels[idx])+ '/' + str(imgs[idx])\ndf.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:10:17.881932Z","iopub.execute_input":"2022-08-13T16:10:17.882404Z","iopub.status.idle":"2022-08-13T16:10:22.465343Z","shell.execute_reply.started":"2022-08-13T16:10:17.882366Z","shell.execute_reply":"2022-08-13T16:10:22.464143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = cv2.imread(os.path.join(cfg['train_path'], df.paths[1]))\nimg = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n#img = cv2.resize(img, cfg['img_size'])\nprint(df.label[1])\nplt.imshow(img)\nprint(img.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:10:22.467052Z","iopub.execute_input":"2022-08-13T16:10:22.469589Z","iopub.status.idle":"2022-08-13T16:10:22.761954Z","shell.execute_reply.started":"2022-08-13T16:10:22.469543Z","shell.execute_reply":"2022-08-13T16:10:22.761042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating dataset","metadata":{}},{"cell_type":"code","source":"encodings = {val: idx for idx, val in enumerate(df.label.unique())}\nprint(encodings)\n\ndf['label_encoded'] = [encodings[df.label[i]] for i in tqdm(range(len(df)))]\ndf","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:10:22.762988Z","iopub.execute_input":"2022-08-13T16:10:22.763665Z","iopub.status.idle":"2022-08-13T16:10:22.885099Z","shell.execute_reply.started":"2022-08-13T16:10:22.763627Z","shell.execute_reply":"2022-08-13T16:10:22.883972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = cv2.imread(os.path.join(cfg['train_path'], df.paths[1]))\nimg = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\nimg = cv2.resize(img, cfg['img_size'], interpolation = cv2.INTER_AREA)\nimg.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:10:22.886661Z","iopub.execute_input":"2022-08-13T16:10:22.887110Z","iopub.status.idle":"2022-08-13T16:10:22.907699Z","shell.execute_reply.started":"2022-08-13T16:10:22.887073Z","shell.execute_reply":"2022-08-13T16:10:22.906656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential([\n    \n#Data Augmentation layers\npreprocessing.RandomFlip('horizontal'), #Flip left to right\npreprocessing.RandomContrast(0.1, seed=0), #contrast randomly by 10%\npreprocessing.RandomRotation(factor=0.25, seed=0), #random rotation, rotates the image by\npreprocessing.RandomZoom(height_factor=0.2, width_factor=0.1, seed=0),\npreprocessing.RandomFlip('vertical'),\n    \n\n#ResNet layer\nkeras.applications.InceptionResNetV2(\n    include_top = False,\n    weights = 'imagenet',\n    input_shape = (cfg['img_size'][0], cfg['img_size'][1], 3)\n),\n    \nkeras.layers.GlobalAveragePooling2D(),\nkeras.layers.Flatten(),\nkeras.layers.Dense(256, activation = 'relu', bias_regularizer = keras.regularizers.L1L2(l1=0.01, l2=0.00)),\nkeras.layers.Dropout(0.2),\nkeras.layers.Dense(10, activation = 'softmax')\n])\n\nmodel.build(input_shape=(None, cfg['img_size'][0], cfg['img_size'][1], 3))\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:13:19.292102Z","iopub.execute_input":"2022-08-13T16:13:19.292785Z","iopub.status.idle":"2022-08-13T16:13:30.969966Z","shell.execute_reply.started":"2022-08-13T16:13:19.292746Z","shell.execute_reply":"2022-08-13T16:13:30.968775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_ds = tf.data.Dataset.list_files('../input/paddy-disease-classification/train_images/*/*', shuffle=True, seed=64)\n# print([file for file in img_ds.take(2)])\n\ntrain_size = int(len(img_ds) * 0.8) #taking 80% of data for training\ntrain_ds = img_ds.take(train_size) #this takes first 90% of data for training\nvalid_ds = img_ds.skip(train_size) #this skips firat 90% of data and takes next 10% for validation","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:13:30.972075Z","iopub.execute_input":"2022-08-13T16:13:30.972448Z","iopub.status.idle":"2022-08-13T16:13:32.055605Z","shell.execute_reply.started":"2022-08-13T16:13:30.972411Z","shell.execute_reply":"2022-08-13T16:13:32.054552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def parse_label(label):\n    idx = None\n    if label == 'bacterial_leaf_blight':\n        idx = 0\n    elif label == 'bacterial_leaf_streak':\n        idx = 1\n    elif label == 'bacterial_panicle_blight':\n        idx = 2\n    elif label == 'blast':\n        idx = 3\n    elif label == 'brown_spot':\n        idx = 4\n    elif label == 'dead_heart':\n        idx = 5\n    elif label == 'downy_mildew':\n        idx = 6\n    elif label == 'hispa':\n        idx = 7\n    elif label == 'normal':\n        idx = 8\n    else: #label == 'tungro'\n        idx = 9\n    return tf.one_hot(idx, depth=10)\n\n# def parse_label(label):\n#     idx = encodings[label]\n#     return tf.one_hot(idx, depth=10)\n\ndef create_dataset(file_path):\n    label = tf.strings.split(file_path, os.path.sep)[-2]\n    label = parse_label(label)\n    img = tf.io.read_file(file_path)\n    img = tf.image.decode_jpeg(img, channels=3)\n    img = tf.image.resize(img, [cfg['img_size'][0], cfg['img_size'][1]])\n    return img, label","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:13:32.057164Z","iopub.execute_input":"2022-08-13T16:13:32.057780Z","iopub.status.idle":"2022-08-13T16:13:32.067619Z","shell.execute_reply.started":"2022-08-13T16:13:32.057738Z","shell.execute_reply":"2022-08-13T16:13:32.066642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds = train_ds.map(create_dataset)\nvalid_ds = valid_ds.map(create_dataset)\ndel img_ds\ngc.collect()\n#print(train_ds)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:13:32.070454Z","iopub.execute_input":"2022-08-13T16:13:32.070937Z","iopub.status.idle":"2022-08-13T16:13:33.003296Z","shell.execute_reply.started":"2022-08-13T16:13:32.070895Z","shell.execute_reply":"2022-08-13T16:13:33.002296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTOTUNE = tf.data.AUTOTUNE\ntrain_ds = (train_ds.batch(cfg['batch_size']).prefetch(AUTOTUNE))#use .cache() only if there is enough RAM, else the training loop will stop in between and return OOM error.\nvalid_ds = (valid_ds.batch(cfg['batch_size']).prefetch(AUTOTUNE))#use .cache() only if there is enough RAM, else the training loop will stop in between and return OOM error.","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:13:33.008409Z","iopub.execute_input":"2022-08-13T16:13:33.008872Z","iopub.status.idle":"2022-08-13T16:13:33.021584Z","shell.execute_reply.started":"2022-08-13T16:13:33.008828Z","shell.execute_reply":"2022-08-13T16:13:33.020519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(\n    optimizer=tf.keras.optimizers.Adam(lr=cfg['lr']),\n    loss= tf.keras.losses.CategoricalCrossentropy(),\n    metrics = ['Accuracy']\n)\n\ncheckpoint = tf.keras.callbacks.ModelCheckpoint(\n    filepath='/kaggle/working/best_model.h5',\n    monitor='val_loss',\n    verbose=0,\n    save_best_only=True,\n    mode='min'\n)\n\nreduce_lr = tf.keras.callbacks.ReduceLROnPlateau(\n    monitor='val_loss',\n    factor=0.25,\n    patience=3,\n    verbose=0,\n    mode='min'\n)\n\nes = tf.keras.callbacks.EarlyStopping(\n    patience=5,\n    min_delta=0,\n    monitor='val_loss',\n    restore_best_weights=True,\n    verbose=0,\n    mode='min',\n    baseline=None\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:13:33.023008Z","iopub.execute_input":"2022-08-13T16:13:33.024223Z","iopub.status.idle":"2022-08-13T16:13:33.052691Z","shell.execute_reply.started":"2022-08-13T16:13:33.024184Z","shell.execute_reply":"2022-08-13T16:13:33.051489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_ds,\n    validation_data = valid_ds,\n    batch_size=cfg['batch_size'],\n    epochs = cfg['epochs'],\n    callbacks=[es,reduce_lr,checkpoint],\n    shuffle=True,\n    verbose=1\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:13:33.055289Z","iopub.execute_input":"2022-08-13T16:13:33.055725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T11:29:49.570060Z","iopub.execute_input":"2022-07-21T11:29:49.573674Z","iopub.status.idle":"2022-07-21T11:29:49.768603Z","shell.execute_reply.started":"2022-07-21T11:29:49.573631Z","shell.execute_reply":"2022-07-21T11:29:49.767672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['Accuracy'])\nplt.plot(history.history['val_Accuracy'])\nplt.title('model accuracy')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T11:43:12.046267Z","iopub.execute_input":"2022-07-21T11:43:12.046951Z","iopub.status.idle":"2022-07-21T11:43:12.264272Z","shell.execute_reply.started":"2022-07-21T11:43:12.046917Z","shell.execute_reply":"2022-07-21T11:43:12.263325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\nfrom IPython.display import FileLink","metadata":{"execution":{"iopub.status.busy":"2022-07-21T11:48:57.529971Z","iopub.execute_input":"2022-07-21T11:48:57.530330Z","iopub.status.idle":"2022-07-21T11:48:57.534720Z","shell.execute_reply.started":"2022-07-21T11:48:57.530295Z","shell.execute_reply":"2022-07-21T11:48:57.533588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.chdir(r'/kaggle/working/')\nFileLink(r'best_model.h5')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T12:17:25.075121Z","iopub.execute_input":"2022-07-21T12:17:25.075483Z","iopub.status.idle":"2022-07-21T12:17:25.085146Z","shell.execute_reply.started":"2022-07-21T12:17:25.075451Z","shell.execute_reply":"2022-07-21T12:17:25.083682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-07-21T11:58:41.720342Z","iopub.status.idle":"2022-07-21T11:58:41.721501Z","shell.execute_reply.started":"2022-07-21T11:58:41.721218Z","shell.execute_reply":"2022-07-21T11:58:41.721242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}