{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Importing libs","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nfrom PIL import Image\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.applications import ResNet50","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:56:30.438572Z","iopub.execute_input":"2022-07-13T19:56:30.438968Z","iopub.status.idle":"2022-07-13T19:56:37.227000Z","shell.execute_reply.started":"2022-07-13T19:56:30.438934Z","shell.execute_reply":"2022-07-13T19:56:37.225948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Looking closer to the data","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(os.path.join('/kaggle', 'input', 'paddy-disease-classification', 'train.csv'))\ntrain_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:17:47.742721Z","iopub.execute_input":"2022-07-13T19:17:47.743335Z","iopub.status.idle":"2022-07-13T19:17:47.786141Z","shell.execute_reply.started":"2022-07-13T19:17:47.743298Z","shell.execute_reply":"2022-07-13T19:17:47.785199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:17:49.201398Z","iopub.execute_input":"2022-07-13T19:17:49.202493Z","iopub.status.idle":"2022-07-13T19:17:49.215939Z","shell.execute_reply.started":"2022-07-13T19:17:49.202448Z","shell.execute_reply":"2022-07-13T19:17:49.214563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## label and variety ohe (one-hot encoding)","metadata":{}},{"cell_type":"code","source":"ohe_labels = pd.get_dummies(train_df['label'])\ntrain_df_ohe = train_df.join(ohe_labels)\n#train_df_ohe.drop(columns= ['label'], inplace= True)\ntrain_df_ohe.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:17:53.770691Z","iopub.execute_input":"2022-07-13T19:17:53.771777Z","iopub.status.idle":"2022-07-13T19:17:53.802299Z","shell.execute_reply.started":"2022-07-13T19:17:53.771735Z","shell.execute_reply":"2022-07-13T19:17:53.801207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ohe_vars = pd.get_dummies(train_df['variety'])\ntrain_df_ohe = train_df_ohe.join(ohe_vars)\n#train_df_ohe.drop(columns= ['label'], inplace= True)\ntrain_df_ohe.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:17:54.695906Z","iopub.execute_input":"2022-07-13T19:17:54.696406Z","iopub.status.idle":"2022-07-13T19:17:54.726151Z","shell.execute_reply.started":"2022-07-13T19:17:54.696364Z","shell.execute_reply":"2022-07-13T19:17:54.725241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## dataviz","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize= (10,10))\nsns.countplot(data= train_df_ohe, y = 'label').set_title(\"Disease samples count\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:17:57.254676Z","iopub.execute_input":"2022-07-13T19:17:57.255292Z","iopub.status.idle":"2022-07-13T19:17:57.524259Z","shell.execute_reply.started":"2022-07-13T19:17:57.255252Z","shell.execute_reply":"2022-07-13T19:17:57.523353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize= (10,10))\nsns.histplot(data= train_df_ohe, x= 'age', hue= 'label').set_title(\"Disease across the age\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:17:57.603978Z","iopub.execute_input":"2022-07-13T19:17:57.605131Z","iopub.status.idle":"2022-07-13T19:17:58.707981Z","shell.execute_reply.started":"2022-07-13T19:17:57.605080Z","shell.execute_reply":"2022-07-13T19:17:58.707053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize= (10,10))\nsns.histplot(data= train_df_ohe, x= 'variety', hue= 'label').set_title(\"Disease across varieties\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:17:58.709702Z","iopub.execute_input":"2022-07-13T19:17:58.712502Z","iopub.status.idle":"2022-07-13T19:17:59.227422Z","shell.execute_reply.started":"2022-07-13T19:17:58.712460Z","shell.execute_reply":"2022-07-13T19:17:59.226483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# mean age accross varieties\nvar_gr = train_df_ohe.groupby('variety')\nm_age_var = var_gr['age'].mean()\nplt.figure(figsize= (7,7))\nsns.histplot(y = m_age_var.index, x= m_age_var.values).set_title(\"Mean age across varieties\")\nprint(m_age_var)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:17:59.534379Z","iopub.execute_input":"2022-07-13T19:17:59.535041Z","iopub.status.idle":"2022-07-13T19:17:59.736117Z","shell.execute_reply.started":"2022-07-13T19:17:59.535003Z","shell.execute_reply":"2022-07-13T19:17:59.735155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# heatmap\nplt.figure(figsize= (15,15))\nsns.heatmap(train_df_ohe.corr(), annot= True)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:18:02.403045Z","iopub.execute_input":"2022-07-13T19:18:02.403423Z","iopub.status.idle":"2022-07-13T19:18:04.211928Z","shell.execute_reply.started":"2022-07-13T19:18:02.403393Z","shell.execute_reply":"2022-07-13T19:18:04.208789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LABELS = train_df['label'].unique()\nLABELS","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:18:04.839139Z","iopub.execute_input":"2022-07-13T19:18:04.839733Z","iopub.status.idle":"2022-07-13T19:18:04.847324Z","shell.execute_reply.started":"2022-07-13T19:18:04.839696Z","shell.execute_reply":"2022-07-13T19:18:04.846219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating dataset","metadata":{}},{"cell_type":"code","source":"TRAIN_DATA_PATH = os.path.join('/kaggle', 'input', 'paddy-disease-classification', 'train_images')\nTEST_DATA_PATH = os.path.join('/kaggle', 'input', 'paddy-disease-classification', 'test_images')","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:18:08.693273Z","iopub.execute_input":"2022-07-13T19:18:08.693950Z","iopub.status.idle":"2022-07-13T19:18:08.699993Z","shell.execute_reply.started":"2022-07-13T19:18:08.693910Z","shell.execute_reply":"2022-07-13T19:18:08.698131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# collect mean and std from imgs\ntr_r_mean_arr, tr_g_mean_arr, tr_b_mean_arr = [], [], []\ntr_r_std_arr, tr_g_std_arr, tr_b_std_arr = [], [], []\n\n# train\nfor dis in LABELS:\n    for f_name in tqdm(os.listdir(os.path.join(TRAIN_DATA_PATH, dis))):\n        img = Image.open(os.path.join(TRAIN_DATA_PATH, dis, f_name))\n        img_np = np.asarray(img)\n        img_np = img_np/255\n        \n        r_mean, g_mean, b_mean = np.mean(img_np, axis= tuple(range(img_np.ndim - 1)))\n        r_std, g_std, b_std = np.std(img_np, axis= tuple(range(img_np.ndim - 1)))\n        \n        tr_r_mean_arr.append(r_mean)\n        tr_g_mean_arr.append(g_mean)\n        tr_b_mean_arr.append(b_mean)\n        \n        tr_r_std_arr.append(r_std)\n        tr_g_std_arr.append(g_std)\n        tr_b_std_arr.append(b_std)\n        \ntr_r_mean = np.mean(tr_r_mean_arr)\ntr_g_mean = np.mean(tr_g_mean_arr)\ntr_b_mean = np.mean(tr_b_mean_arr)\n\ntr_r_std = np.std(tr_r_std_arr)\ntr_g_std = np.std(tr_g_std_arr)\ntr_b_std = np.std(tr_b_std_arr)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T12:54:20.012081Z","iopub.execute_input":"2022-07-13T12:54:20.012665Z","iopub.status.idle":"2022-07-13T13:01:05.299332Z","shell.execute_reply.started":"2022-07-13T12:54:20.012619Z","shell.execute_reply":"2022-07-13T13:01:05.298136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# collect mean and std from imgs\ntst_r_mean_arr, tst_g_mean_arr, tst_b_mean_arr = [], [], []\ntst_r_std_arr, tst_g_std_arr, tst_b_std_arr = [], [], []\n\n# test\nfor f_name in tqdm(os.listdir(TEST_DATA_PATH)):\n        img = Image.open(os.path.join(TEST_DATA_PATH, f_name))\n        img_np = np.asarray(img)\n        img_np = img_np/255\n        \n        r_mean, g_mean, b_mean = np.mean(img_np, axis= tuple(range(img_np.ndim - 1)))\n        r_std, g_std, b_std = np.std(img_np, axis= tuple(range(img_np.ndim - 1)))\n        \n        tst_r_mean_arr.append(r_mean)\n        tst_g_mean_arr.append(g_mean)\n        tst_b_mean_arr.append(b_mean)\n        \n        tst_r_std_arr.append(r_std)\n        tst_g_std_arr.append(g_std)\n        tst_b_std_arr.append(b_std)\n        \n        \ntst_r_mean = np.mean(tst_r_mean_arr)\ntst_g_mean = np.mean(tst_g_mean_arr)\ntst_b_mean = np.mean(tst_b_mean_arr)\n\ntst_r_std = np.std(tst_r_std_arr)\ntst_g_std = np.std(tst_g_std_arr)\ntst_b_std = np.std(tst_b_std_arr)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T13:01:05.301769Z","iopub.execute_input":"2022-07-13T13:01:05.302543Z","iopub.status.idle":"2022-07-13T13:03:20.404235Z","shell.execute_reply.started":"2022-07-13T13:01:05.302501Z","shell.execute_reply":"2022-07-13T13:03:20.403282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Train mean r: {tr_r_mean}, g: {tr_g_mean}, b: {tr_b_mean}\")\nprint(f\"Train std r: {tr_r_std}, g: {tr_r_std}, b: {tr_r_std}\\n\")\n\nprint(f\"Test mean r: {tst_r_mean}, g: {tst_g_mean}, b: {tst_g_mean}\")\nprint(f\"Test std r: {tst_r_std}, g: {tst_g_std}, b: {tst_b_std}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T13:03:20.405877Z","iopub.execute_input":"2022-07-13T13:03:20.406694Z","iopub.status.idle":"2022-07-13T13:03:20.413431Z","shell.execute_reply.started":"2022-07-13T13:03:20.406654Z","shell.execute_reply":"2022-07-13T13:03:20.412382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# resize and rescale\n\nIMG_SIZE = 180\n\nRnR = tf.keras.Sequential([\n  layers.Resizing(IMG_SIZE, IMG_SIZE),\n  layers.Rescaling(1./255)\n])","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:18:13.046959Z","iopub.execute_input":"2022-07-13T19:18:13.047347Z","iopub.status.idle":"2022-07-13T19:18:16.178211Z","shell.execute_reply.started":"2022-07-13T19:18:13.047312Z","shell.execute_reply":"2022-07-13T19:18:16.177198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data augmentation\n\nTRANSFORMS = tf.keras.Sequential([\n  layers.RandomFlip(\"horizontal_and_vertical\"),\n  layers.RandomRotation(0.2),\n])","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:57:36.321538Z","iopub.execute_input":"2022-07-13T19:57:36.322662Z","iopub.status.idle":"2022-07-13T19:57:39.461383Z","shell.execute_reply.started":"2022-07-13T19:57:36.322610Z","shell.execute_reply":"2022-07-13T19:57:39.460373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 32\nIMG_SIZE = (640, 480)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:57:39.463566Z","iopub.execute_input":"2022-07-13T19:57:39.463945Z","iopub.status.idle":"2022-07-13T19:57:39.469312Z","shell.execute_reply.started":"2022-07-13T19:57:39.463908Z","shell.execute_reply":"2022-07-13T19:57:39.467992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = tf.keras.utils.image_dataset_from_directory(TRAIN_DATA_PATH,\n                                                            shuffle=True,\n                                                            validation_split=0.2,\n                                                            subset=\"training\",\n                                                            seed=113,\n                                                            batch_size=BATCH_SIZE,\n                                                            image_size=IMG_SIZE)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:57:39.471339Z","iopub.execute_input":"2022-07-13T19:57:39.472159Z","iopub.status.idle":"2022-07-13T19:57:39.501549Z","shell.execute_reply.started":"2022-07-13T19:57:39.472119Z","shell.execute_reply":"2022-07-13T19:57:39.500155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_datasset = tf.keras.utils.image_dataset_from_directory(TRAIN_DATA_PATH,\n                                                           validation_split=0.2,\n                                                           subset=\"validation\",\n                                                           seed=113,\n                                                           image_size=IMG_SIZE,\n                                                           batch_size=BATCH_SIZE)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:57:39.502801Z","iopub.status.idle":"2022-07-13T19:57:39.503567Z","shell.execute_reply.started":"2022-07-13T19:57:39.503289Z","shell.execute_reply":"2022-07-13T19:57:39.503316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_names = train_dataset.class_names\n\nplt.figure(figsize=(10, 10))\nfor images, labels in train_dataset.take(1):\n    for i in range(9):\n        ax = plt.subplot(3, 3, i + 1)\n        plt.imshow(images[i].numpy().astype(\"uint8\"))\n        plt.title(class_names[labels[i]])\n        plt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:57:44.917692Z","iopub.execute_input":"2022-07-13T19:57:44.918091Z","iopub.status.idle":"2022-07-13T19:57:44.943782Z","shell.execute_reply.started":"2022-07-13T19:57:44.918055Z","shell.execute_reply":"2022-07-13T19:57:44.942197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for image_batch, labels_batch in train_dataset:\n      print(image_batch.shape)\n      print(labels_batch.shape)\n      break","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:39:52.793738Z","iopub.execute_input":"2022-07-13T19:39:52.794082Z","iopub.status.idle":"2022-07-13T19:39:54.068998Z","shell.execute_reply.started":"2022-07-13T19:39:52.794053Z","shell.execute_reply":"2022-07-13T19:39:54.067935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# normalize train ds\nnormalized_train_ds = train_dataset.map(lambda x, y: (RnR(x), y))\nimage_batch, labels_batch = next(iter(normalized_train_ds))\nfirst_image = image_batch[0]\nprint(np.min(first_image), np.max(first_image))\nprint(f\"Shape: {first_image.shape}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:39:54.948429Z","iopub.execute_input":"2022-07-13T19:39:54.949016Z","iopub.status.idle":"2022-07-13T19:39:55.918922Z","shell.execute_reply.started":"2022-07-13T19:39:54.948977Z","shell.execute_reply":"2022-07-13T19:39:55.917914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# normalize validation ds\nnormalized_val_ds = val_datasset.map(lambda x, y: (RnR(x), y))\nimage_batch, labels_batch = next(iter(normalized_val_ds))\nfirst_image = image_batch[0]\nprint(np.min(first_image), np.max(first_image))\nprint(f\"Shape: {first_image.shape}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:39:57.096841Z","iopub.execute_input":"2022-07-13T19:39:57.097924Z","iopub.status.idle":"2022-07-13T19:39:58.256909Z","shell.execute_reply.started":"2022-07-13T19:39:57.097884Z","shell.execute_reply":"2022-07-13T19:39:58.255908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Transfer learning (ResNet50)","metadata":{}},{"cell_type":"code","source":"feat_extr = ResNet50(weights= 'imagenet',input_shape=(180, 180, 3), include_top=False)\nfeat_extr.trainable = False","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:40:01.610564Z","iopub.execute_input":"2022-07-13T19:40:01.611252Z","iopub.status.idle":"2022-07-13T19:40:02.913608Z","shell.execute_reply.started":"2022-07-13T19:40:01.611198Z","shell.execute_reply":"2022-07-13T19:40:02.912629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prepare ResNet50 for transfer learning\n\n# input layer\nnn_input = layers.Input(shape= (180,180,3))\n\n# feature axtraction layer (ResNet50)\nf_extr_layer = feat_extr(nn_input, training= False)\n\n# pooling layer\nf_extr_layer = tf.keras.layers.GlobalAveragePooling2D()(f_extr_layer)\n\n# output layer\nnn_output = tf.keras.layers.Dense(10)(f_extr_layer)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:50:15.206147Z","iopub.execute_input":"2022-07-13T19:50:15.206816Z","iopub.status.idle":"2022-07-13T19:50:15.575033Z","shell.execute_reply.started":"2022-07-13T19:50:15.206779Z","shell.execute_reply":"2022-07-13T19:50:15.574018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create model object\nwith tf.device('/GPU:0'):\n    model_trl = tf.keras.Model(nn_input, nn_output)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:53:21.272325Z","iopub.execute_input":"2022-07-13T19:53:21.272887Z","iopub.status.idle":"2022-07-13T19:53:21.288104Z","shell.execute_reply.started":"2022-07-13T19:53:21.272847Z","shell.execute_reply":"2022-07-13T19:53:21.286938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# compile with neccessary loss\nwith tf.device('/CPU:0'):\n    model_trl.compile(optimizer='adam',\n             loss='sparse_categorical_crossentropy',\n             metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:53:29.538336Z","iopub.execute_input":"2022-07-13T19:53:29.539025Z","iopub.status.idle":"2022-07-13T19:53:29.584610Z","shell.execute_reply.started":"2022-07-13T19:53:29.538985Z","shell.execute_reply":"2022-07-13T19:53:29.583656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show model summary\nmodel_trl.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:53:30.880294Z","iopub.execute_input":"2022-07-13T19:53:30.881156Z","iopub.status.idle":"2022-07-13T19:53:30.899135Z","shell.execute_reply.started":"2022-07-13T19:53:30.881109Z","shell.execute_reply":"2022-07-13T19:53:30.898268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fit the model\n\nmodel_trl.fit(normalized_train_ds, epochs= 200, validation_data= normalized_val_ds)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:57:25.595991Z","iopub.execute_input":"2022-07-13T19:57:25.596406Z","iopub.status.idle":"2022-07-13T19:57:25.621833Z","shell.execute_reply.started":"2022-07-13T19:57:25.596372Z","shell.execute_reply":"2022-07-13T19:57:25.620524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}