{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%%capture\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-18T13:20:41.022612Z","iopub.execute_input":"2022-08-18T13:20:41.023346Z","iopub.status.idle":"2022-08-18T13:20:58.929708Z","shell.execute_reply.started":"2022-08-18T13:20:41.023177Z","shell.execute_reply":"2022-08-18T13:20:58.928293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns, matplotlib.pyplot as plt\nimport random\nimport tensorflow as tf\nfrom tensorflow.keras.applications import EfficientNetB3\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.metrics import Precision, Recall\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-08-18T14:04:34.854532Z","iopub.execute_input":"2022-08-18T14:04:34.854946Z","iopub.status.idle":"2022-08-18T14:04:34.864652Z","shell.execute_reply.started":"2022-08-18T14:04:34.854916Z","shell.execute_reply":"2022-08-18T14:04:34.863066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# todo: integrate weight and biases\n# !pip install --upgrade -q wandb\n# from kaggle_secrets import UserSecretsClient\n\n# user_secrets = UserSecretsClient()\n\n# # I have saved my API token with \"wandb_api\" as Label. \n# # If you use some other Label make sure to change the same below. \n# wandb_api = user_secrets.get_secret(\"wandb_api\") \n\n# wandb.login(key=wandb_api)\n# 0eebfeee9a34735d19228b8f3257257b62ece255","metadata":{"execution":{"iopub.status.busy":"2022-08-14T00:17:22.246064Z","iopub.execute_input":"2022-08-14T00:17:22.247196Z","iopub.status.idle":"2022-08-14T00:17:22.258344Z","shell.execute_reply.started":"2022-08-14T00:17:22.247151Z","shell.execute_reply":"2022-08-14T00:17:22.257289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.chdir('/kaggle')\ntrain_csv=pd.read_csv('input/plant-pathology-2021-fgvc8/train.csv')\nos.chdir('input/plant-pathology-2021-fgvc8/train_images')\nprint(os.getcwd())","metadata":{"execution":{"iopub.status.busy":"2022-08-18T13:21:07.051507Z","iopub.execute_input":"2022-08-18T13:21:07.052122Z","iopub.status.idle":"2022-08-18T13:21:07.092433Z","shell.execute_reply.started":"2022-08-18T13:21:07.052084Z","shell.execute_reply":"2022-08-18T13:21:07.091372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-18T13:21:07.0947Z","iopub.execute_input":"2022-08-18T13:21:07.095073Z","iopub.status.idle":"2022-08-18T13:21:07.116099Z","shell.execute_reply.started":"2022-08-18T13:21:07.095027Z","shell.execute_reply":"2022-08-18T13:21:07.114738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualizations","metadata":{}},{"cell_type":"markdown","source":"- visualization needed: Contours with edges, threshold otsu, HSV, and blur","metadata":{}},{"cell_type":"markdown","source":"## Plot CSV classes","metadata":{"execution":{"iopub.status.busy":"2022-08-08T12:18:00.440046Z","iopub.execute_input":"2022-08-08T12:18:00.440787Z","iopub.status.idle":"2022-08-08T12:18:00.446695Z","shell.execute_reply.started":"2022-08-08T12:18:00.440749Z","shell.execute_reply":"2022-08-08T12:18:00.445366Z"}}},{"cell_type":"code","source":"plt.figure(figsize=(15,7))\nsns.countplot(y=train_csv['labels'], order=train_csv['labels'].value_counts().index)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-18T13:21:07.117813Z","iopub.execute_input":"2022-08-18T13:21:07.119653Z","iopub.status.idle":"2022-08-18T13:21:07.454639Z","shell.execute_reply.started":"2022-08-18T13:21:07.119614Z","shell.execute_reply":"2022-08-18T13:21:07.453492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counts=dict(train_csv['labels'].value_counts())\nprint(f\"Highest class count is: {list(counts.values())[0]}\\n\\\nLowest class count is: {list(counts.values())[-1]}\")\nprint(f\"The min count of smallest class is: {counts['powdery_mildew']}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-18T13:21:07.457091Z","iopub.execute_input":"2022-08-18T13:21:07.457428Z","iopub.status.idle":"2022-08-18T13:21:07.466823Z","shell.execute_reply.started":"2022-08-18T13:21:07.457398Z","shell.execute_reply":"2022-08-18T13:21:07.46548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualize images of classes","metadata":{}},{"cell_type":"code","source":"classes=list(counts.keys())\nnrows, ncols = len(classes),5\n# fig, ax = plt.subplots(len(classes),3,figsize=(50,50))\nfig =plt.figure(figsize=(60,6*nrows))\nfor c1, i in enumerate(classes):\n    grouped=train_csv[train_csv['labels']==i].sample(n=ncols, random_state=1)\n    for c2,image in enumerate(grouped['image']):\n            ax = fig.add_subplot(nrows, ncols, c1*ncols + (c2+1))\n            ax.imshow(plt.imread(image))\n            plt.xlabel(i)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T12:29:44.655686Z","iopub.execute_input":"2022-08-14T12:29:44.656113Z","iopub.status.idle":"2022-08-14T12:31:29.185921Z","shell.execute_reply.started":"2022-08-14T12:29:44.65608Z","shell.execute_reply":"2022-08-14T12:31:29.183535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Color Distribution","metadata":{}},{"cell_type":"code","source":"# left skwed blue channel\nclasses=list(counts.keys())\nnrows, ncols = len(classes),5\n# fig, ax = plt.subplots(len(classes),3,figsize=(50,50))\nfig =plt.figure(figsize=(40,7*nrows))\nfor c1, i in enumerate(classes):\n    grouped=train_csv[train_csv['labels']==i].sample(n=ncols, random_state=1)\n    for c2,image in enumerate(grouped['image']):\n            ax = fig.add_subplot(nrows, ncols, c1*ncols + (c2+1))\n            image=plt.imread(image)\n            ax.hist(image[:, :, 0].ravel(), bins = 256, color = 'red', alpha = 0.5)\n            ax.hist(image[:, :, 1].ravel(), bins = 256, color = 'Green', alpha = 0.5)\n            ax.hist(image[:, :, 2].ravel(), bins = 256, color = 'Blue', alpha = 0.5)\n            plt.xlabel(i)\n            plt.legend(['Total', 'Red_Channel', 'Green_Channel', 'Blue_Channel'])","metadata":{"execution":{"iopub.status.busy":"2022-08-14T12:31:29.188244Z","iopub.execute_input":"2022-08-14T12:31:29.188635Z","iopub.status.idle":"2022-08-14T12:34:12.642827Z","shell.execute_reply.started":"2022-08-14T12:31:29.188599Z","shell.execute_reply":"2022-08-14T12:34:12.641798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Image shape ","metadata":{}},{"cell_type":"code","source":"classes=list(counts.keys())\nnrows, ncols = len(classes),80\nwid, hei= [], []\n# fig, ax = plt.subplots(len(classes),3,figsize=(50,50))\nfig =plt.figure(figsize=(40,7*nrows))\nfor c1, i in enumerate(classes):\n    grouped=train_csv[train_csv['labels']==i].sample(n=ncols, random_state=1)\n    for c2,image in enumerate(grouped['image']):\n        b1, b2, _ = plt.imread(image).shape\n        wid.append(b1)\n        hei.append(b2)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:55:26.181551Z","iopub.execute_input":"2022-08-14T03:55:26.182282Z","iopub.status.idle":"2022-08-14T03:59:06.62731Z","shell.execute_reply.started":"2022-08-14T03:55:26.182234Z","shell.execute_reply":"2022-08-14T03:59:06.626118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 1,figsize =(10, 7),tight_layout = True)\nax.hist(wid, bins = 20, color = 'red', alpha = 0.5)\nax.hist(hei, bins = 20, color = 'Green', alpha = 0.5)\nplt.legend(['width','height'])\nwid,hei= np.array(wid), np.array(hei)\nprint(f'Width = {wid.mean()}, height = {hei.mean()}')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:18:42.038968Z","iopub.execute_input":"2022-08-14T04:18:42.039444Z","iopub.status.idle":"2022-08-14T04:18:42.486058Z","shell.execute_reply.started":"2022-08-14T04:18:42.039406Z","shell.execute_reply":"2022-08-14T04:18:42.484308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Labels preprocessing ","metadata":{}},{"cell_type":"code","source":"# uniqueLabels=[lb for label in list(counts.keys()) for lb in label.split()]\n# # how to do it with map\n# uniqueLabels=list(set(uniqueLabels))\nlabels = [set(inst.split()) for inst in train_csv['labels']]\nmlb = MultiLabelBinarizer()\nlabels = mlb.fit_transform(labels)\nclasses=list(mlb.classes_)\nlabels =[tuple(lb) for lb in labels]\ntrain_csv['labeled']=labels\ntrain_csv.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-18T13:21:07.468747Z","iopub.execute_input":"2022-08-18T13:21:07.469719Z","iopub.status.idle":"2022-08-18T13:21:07.557198Z","shell.execute_reply.started":"2022-08-18T13:21:07.469679Z","shell.execute_reply":"2022-08-18T13:21:07.556077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-18T13:35:26.053836Z","iopub.execute_input":"2022-08-18T13:35:26.054272Z","iopub.status.idle":"2022-08-18T13:35:26.071509Z","shell.execute_reply.started":"2022-08-18T13:35:26.054238Z","shell.execute_reply":"2022-08-18T13:35:26.070148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data-Loader","metadata":{}},{"cell_type":"markdown","source":"- Try with imagedatagenerator and flow from dataframe\n** tf.data is faster x38 time than imagedatagenerator **\n- implement the tf.data.dataset\n- Multiple idea for training:\n    - shuffle training (train all the data)\n    - Use the mixed data as validation dataset","metadata":{}},{"cell_type":"markdown","source":"## ImageDatagenerator","metadata":{}},{"cell_type":"code","source":"seed, imgsize, BATCH_SIZE = 33, 300, 32\ndatagen=ImageDataGenerator(rescale=1./255, \n                           validation_split=0.1, \n                           rotation_range =35, \n                           horizontal_flip=True, \n                           vertical_flip=True,\n                           zca_whitening=True,\n)\ntrain_generator = datagen.flow_from_dataframe(\n                                   dataframe=train_csv,\n                                   x_col='image',\n                                   y_col='strlabels',\n                                   seed=seed,\n                                   class_mode='categorical',\n                                   color_mode='rgb',\n                                   target_size=(imgsize,imgsize),\n                                   batch_size=BATCH_SIZE,\n                                   subset='training')\nval_generator = datagen.flow_from_dataframe(\n                                   dataframe=train_csv,\n                                   x_col='image',\n                                   y_col='strlabels',\n                                   seed=seed,\n                                   class_mode='categorical',\n                                   color_mode='rgb',\n                                   target_size=(imgsize,imgsize),\n                                   batch_size=BATCH_SIZE,\n                                   subset='validation')","metadata":{"execution":{"iopub.status.busy":"2022-08-18T13:59:02.741975Z","iopub.execute_input":"2022-08-18T13:59:02.742859Z","iopub.status.idle":"2022-08-18T13:59:20.97037Z","shell.execute_reply.started":"2022-08-18T13:59:02.742817Z","shell.execute_reply":"2022-08-18T13:59:20.968934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for x,y in train_generator:\n    print(y)\n    break\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-18T13:37:26.248949Z","iopub.execute_input":"2022-08-18T13:37:26.249399Z","iopub.status.idle":"2022-08-18T13:37:28.485282Z","shell.execute_reply.started":"2022-08-18T13:37:26.249362Z","shell.execute_reply":"2022-08-18T13:37:28.484065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data mapping from csv to tensors\n","metadata":{}},{"cell_type":"markdown","source":"## Data augmentation","metadata":{}},{"cell_type":"markdown","source":"## D- visualization needed: Contours with ata spliting: training and validation","metadata":{}},{"cell_type":"markdown","source":"## K-Cross validation ","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model ","metadata":{}},{"cell_type":"markdown","source":"- Loss function on a specific meteric ==> Soft Macro F1 Score (Macro vs Micro)\n- Model\n    - Global average pooling and flatten\n    - Dropout layer? \n    - Dense layer and bottleneck effect\n    - Tensorflow hub and weights fetch\n- optimizers ==> RMS is not good for transfer learning? which for what?\n- add lr schedular\n- Add callback for checkpoint and logging\n- Add callback for early stopping\n- (Search)\n   - tf.function and eager execution with determinism\n   - tf.gradienttape","metadata":{}},{"cell_type":"code","source":"def backbone(x):\n    feature_extractor = EfficientNetB3(weights= 'imagenet', include_top=False)\n    return feature_extractor(x)\ndef classifier(x):\n    x = tf.keras.layers.Flatten()(x)\n    x = tf.keras.layers.Dense(512,activation='relu')(x)\n    x = tf.keras.layers.Dropout(0.2)(x)\n    x = tf.keras.layers.Dense(len(classes),activation='sigmoid')(x)\n    return x\ndef effmodel():\n    inputs =  tf.keras.layers.Input(shape=(300, 300, 3))\n    x=backbone(inputs)\n    outputs= classifier(x)\n    model = tf.keras.Model(inputs, outputs)\n    model.compile(\n        optimizer=\"adam\", loss=\"binary_crossentropy\", metrics=[Precision(),Recall()]\n    )\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-08-18T14:04:59.441607Z","iopub.execute_input":"2022-08-18T14:04:59.442105Z","iopub.status.idle":"2022-08-18T14:04:59.452716Z","shell.execute_reply.started":"2022-08-18T14:04:59.442069Z","shell.execute_reply":"2022-08-18T14:04:59.451574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train","metadata":{}},{"cell_type":"code","source":"model = effmodel()\nmodel.summary()\nmodel.fit(x=train_generator,\n                  validation_data=val_generator,\n                    epochs=30,verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T14:05:01.958066Z","iopub.execute_input":"2022-08-18T14:05:01.95892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Metrics","metadata":{}},{"cell_type":"markdown","source":"- Tensorboard \n- plotting meterics ","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference","metadata":{}},{"cell_type":"markdown","source":"- Try infering with tensorflow serve","metadata":{}}]}