{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h1><center> 🍎 Classify foliar diseases in apple trees</center></h1>","metadata":{}},{"cell_type":"markdown","source":"# 1. Problem Statement ？\n\nApples are one of the most important temperate fruit crops in the world. Foliar (leaf) diseases pose a major threat to the overall productivity and quality of apple orchards. The current process for disease diagnosis in apple orchards is based on manual scouting by humans, which is time-consuming and expensive.\n\nThe main objective of the competition is to develop machine learning-based models to accurately classify a given leaf image from the test dataset to a particular disease category, and to identify an individual disease from multiple disease symptoms on a single leaf image.\n","metadata":{}},{"cell_type":"markdown","source":"## libraries ","metadata":{}},{"cell_type":"code","source":"!pip install opencv-python==3.4.2.17\n!pip install opencv-contrib-python==3.4.2.17","metadata":{"execution":{"iopub.status.busy":"2021-11-27T01:41:14.390701Z","iopub.execute_input":"2021-11-27T01:41:14.391295Z","iopub.status.idle":"2021-11-27T01:41:14.395877Z","shell.execute_reply.started":"2021-11-27T01:41:14.391232Z","shell.execute_reply":"2021-11-27T01:41:14.394861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n%matplotlib inline\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport cv2\nimport os\nimport warnings\nwarnings.filterwarnings('ignore')\nimport tensorflow as tf\nimport random\nimport albumentations as A\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.layers import Dense,Activation,Flatten, Conv2D, MaxPooling2D\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.callbacks import ModelCheckpoint,EarlyStopping\n\nimport seaborn as sns\nfrom tqdm import tqdm\nimport matplotlib.cm as cm\nfrom sklearn import metrics\nimport matplotlib.pyplot as plt\nfrom sklearn.utils import shuffle\nfrom sklearn.model_selection import train_test_split\n\ntqdm.pandas()\nimport plotly.express as px\nimport plotly.graph_objects as go\nimport plotly.figure_factory as ff\nfrom plotly.subplots import make_subplots\n\nnp.random.seed(0)\ntf.random.set_seed(0)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T01:41:24.278091Z","iopub.execute_input":"2021-11-27T01:41:24.278680Z","iopub.status.idle":"2021-11-27T01:41:24.296068Z","shell.execute_reply.started":"2021-11-27T01:41:24.278627Z","shell.execute_reply":"2021-11-27T01:41:24.294982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from plotly.offline import init_notebook_mode, iplot, plot\nimport plotly as py\ninit_notebook_mode(connected=True)\nimport plotly.graph_objs as go","metadata":{"execution":{"iopub.status.busy":"2021-11-27T01:44:22.912657Z","iopub.execute_input":"2021-11-27T01:44:22.913101Z","iopub.status.idle":"2021-11-27T01:44:22.919937Z","shell.execute_reply.started":"2021-11-27T01:44:22.913063Z","shell.execute_reply":"2021-11-27T01:44:22.918888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. About Dataset","metadata":{}},{"cell_type":"code","source":"train_image_path = '../input/plant-pathology-2021-fgvc8/train_images/'\ntest_image_path = '../input/plant-pathology-2021-fgvc8/test_images'\ntrain_df_path = '../input/plant-pathology-2021-fgvc8/train.csv'\ntest_df_path = '../input/plant-pathology-2021-fgvc8/sample_submission.csv'","metadata":{"execution":{"iopub.status.busy":"2021-11-27T01:41:27.859003Z","iopub.execute_input":"2021-11-27T01:41:27.859519Z","iopub.status.idle":"2021-11-27T01:41:27.863524Z","shell.execute_reply.started":"2021-11-27T01:41:27.859487Z","shell.execute_reply":"2021-11-27T01:41:27.862754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> 📌**Note**:\n* `train.csv` contains information about the image files available in `train_images`. It contains 18632 rows(images) with 2 columns i.e (image , labels )\n* `test.csv` The test set images. This competition has a hidden test set: only three images are provided here as samples while the remaining 5,000 images will be available to your notebook once it is submitted.","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(train_df_path)\ndf_test = pd.read_csv(test_df_path)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T01:41:33.039122Z","iopub.execute_input":"2021-11-27T01:41:33.039650Z","iopub.status.idle":"2021-11-27T01:41:33.071296Z","shell.execute_reply.started":"2021-11-27T01:41:33.039616Z","shell.execute_reply":"2021-11-27T01:41:33.070208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2021-11-27T01:41:35.183570Z","iopub.execute_input":"2021-11-27T01:41:35.184032Z","iopub.status.idle":"2021-11-27T01:41:35.195351Z","shell.execute_reply.started":"2021-11-27T01:41:35.183997Z","shell.execute_reply":"2021-11-27T01:41:35.194304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test","metadata":{"execution":{"iopub.status.busy":"2021-11-27T01:30:45.515004Z","iopub.execute_input":"2021-11-27T01:30:45.515515Z","iopub.status.idle":"2021-11-27T01:30:45.526143Z","shell.execute_reply.started":"2021-11-27T01:30:45.515484Z","shell.execute_reply":"2021-11-27T01:30:45.525285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.labels.value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-11-27T01:31:03.129206Z","iopub.execute_input":"2021-11-27T01:31:03.129898Z","iopub.status.idle":"2021-11-27T01:31:03.146509Z","shell.execute_reply.started":"2021-11-27T01:31:03.129855Z","shell.execute_reply":"2021-11-27T01:31:03.145574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,12))\nlabels = sns.barplot(df_train.labels.value_counts().index,df_train.labels.value_counts())\nfor item in labels.get_xticklabels():\n    item.set_rotation(45)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T01:31:05.617301Z","iopub.execute_input":"2021-11-27T01:31:05.618097Z","iopub.status.idle":"2021-11-27T01:31:05.96316Z","shell.execute_reply.started":"2021-11-27T01:31:05.618051Z","shell.execute_reply":"2021-11-27T01:31:05.962315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> 📌**Note**:\n* We have multiple labels for eg. label can be **scab** or **scab and rust**\n* Main labels are - **scab** , **healthy** , **frog_eye_leaf_spot** , **rust** , **complex** and **powdery_mildew**","metadata":{}},{"cell_type":"markdown","source":"## Batch Visualisation of Images ","metadata":{}},{"cell_type":"code","source":"def batch_visualize(df,batch_size,path):\n    sample_df = df_train.sample(9)\n    image_names = sample_df[\"image\"].values\n    labels = sample_df[\"labels\"].values\n    plt.figure(figsize=(16, 12))\n    \n    for image_ind, (image_name, label) in enumerate(zip(image_names, labels)):\n        plt.subplot(3, 3, image_ind + 1)\n        image = cv2.imread(os.path.join(path, image_name))\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        plt.imshow(image)\n        plt.title(f\"{label}\", fontsize=12)\n        plt.axis(\"off\")\n    plt.show()\n    \nbatch_visualize(df_train,9,train_image_path)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T01:31:17.224204Z","iopub.execute_input":"2021-11-27T01:31:17.22458Z","iopub.status.idle":"2021-11-27T01:31:28.511969Z","shell.execute_reply.started":"2021-11-27T01:31:17.22455Z","shell.execute_reply":"2021-11-27T01:31:28.51076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Batch visualisation with labels","metadata":{}},{"cell_type":"code","source":"def batch_visualize_with_label(df,batch_size,path,label): \n    sample_df = df_train[df_train[\"labels\"]==label].sample(9)\n    image_names = sample_df[\"image\"].values\n    labels = sample_df[\"labels\"].values\n    plt.figure(figsize=(16, 12))\n    \n    for image_ind, (image_name, label) in enumerate(zip(image_names, labels)):\n        plt.subplot(3, 3, image_ind + 1)\n        image = cv2.imread(os.path.join(path, image_name))\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        plt.imshow(image)\n        plt.axis(\"off\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-26T17:32:33.896923Z","iopub.execute_input":"2021-11-26T17:32:33.897183Z","iopub.status.idle":"2021-11-26T17:32:33.903749Z","shell.execute_reply.started":"2021-11-26T17:32:33.897156Z","shell.execute_reply":"2021-11-26T17:32:33.902797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Visualise healthy leaves","metadata":{}},{"cell_type":"code","source":"batch_visualize_with_label(df_train,9,train_image_path,'healthy')","metadata":{"execution":{"iopub.status.busy":"2021-11-26T17:32:33.904863Z","iopub.execute_input":"2021-11-26T17:32:33.905319Z","iopub.status.idle":"2021-11-26T17:32:44.656036Z","shell.execute_reply.started":"2021-11-26T17:32:33.905208Z","shell.execute_reply":"2021-11-26T17:32:44.655323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Visualise scab leaves ","metadata":{}},{"cell_type":"code","source":"batch_visualize_with_label(df_train,9,train_image_path,'scab')","metadata":{"execution":{"iopub.status.busy":"2021-11-26T17:32:44.657011Z","iopub.execute_input":"2021-11-26T17:32:44.657421Z","iopub.status.idle":"2021-11-26T17:32:55.337619Z","shell.execute_reply.started":"2021-11-26T17:32:44.657386Z","shell.execute_reply":"2021-11-26T17:32:55.336567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Visualise frog_eye_leaf_spot  leaves","metadata":{}},{"cell_type":"code","source":"batch_visualize_with_label(df_train,9,train_image_path,'frog_eye_leaf_spot')","metadata":{"execution":{"iopub.status.busy":"2021-11-26T17:32:55.33887Z","iopub.execute_input":"2021-11-26T17:32:55.339163Z","iopub.status.idle":"2021-11-26T17:33:06.026219Z","shell.execute_reply.started":"2021-11-26T17:32:55.339132Z","shell.execute_reply":"2021-11-26T17:33:06.025185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Visualise rust leaves ","metadata":{}},{"cell_type":"code","source":"batch_visualize_with_label(df_train,9,train_image_path,'rust')","metadata":{"execution":{"iopub.status.busy":"2021-11-26T17:33:06.02746Z","iopub.execute_input":"2021-11-26T17:33:06.027748Z","iopub.status.idle":"2021-11-26T17:33:14.556255Z","shell.execute_reply.started":"2021-11-26T17:33:06.027719Z","shell.execute_reply":"2021-11-26T17:33:14.555425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Visualise complex leaves","metadata":{}},{"cell_type":"code","source":"batch_visualize_with_label(df_train,9,train_image_path,'complex')","metadata":{"execution":{"iopub.status.busy":"2021-11-26T17:33:14.557252Z","iopub.execute_input":"2021-11-26T17:33:14.557644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Visualise powdery_mildew leaves","metadata":{}},{"cell_type":"code","source":"batch_visualize_with_label(df_train,9,train_image_path,'powdery_mildew')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualize with color histogram","metadata":{}},{"cell_type":"code","source":"SAMPLE_LEN = 100\n\ndef load_image(file_path):\n    image = cv2.imread(train_image_path + file_path)\n    return cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\ntrain_images = df_train[\"image\"][:SAMPLE_LEN].progress_apply(load_image)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T01:31:35.438862Z","iopub.execute_input":"2021-11-27T01:31:35.439462Z","iopub.status.idle":"2021-11-27T01:31:51.697608Z","shell.execute_reply.started":"2021-11-27T01:31:35.439425Z","shell.execute_reply":"2021-11-27T01:31:51.696235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### All channel values","metadata":{}},{"cell_type":"code","source":"red_values = [np.mean(train_images[idx][:, :, 0]) for idx in range(len(train_images))]\ngreen_values = [np.mean(train_images[idx][:, :, 1]) for idx in range(len(train_images))]\nblue_values = [np.mean(train_images[idx][:, :, 2]) for idx in range(len(train_images))]\nvalues = [np.mean(train_images[idx]) for idx in range(len(train_images))]","metadata":{"execution":{"iopub.status.busy":"2021-11-27T01:41:59.096808Z","iopub.execute_input":"2021-11-27T01:41:59.097240Z","iopub.status.idle":"2021-11-27T01:42:07.369713Z","shell.execute_reply.started":"2021-11-27T01:41:59.097203Z","shell.execute_reply":"2021-11-27T01:42:07.368671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = ff.create_distplot([values], group_labels=[\"Channels\"], colors=[\"purple\"])\nfig.update_layout(showlegend=False, template=\"simple_white\")\nfig.update_layout(title_text=\"Distribution of channel values\")\nfig.data[0].marker.line.color = 'rgb(0, 0, 0)'\nfig.data[0].marker.line.width = 0.5\nfig","metadata":{"execution":{"iopub.status.busy":"2021-11-27T01:44:59.530651Z","iopub.execute_input":"2021-11-27T01:44:59.531245Z","iopub.status.idle":"2021-11-27T01:44:59.601979Z","shell.execute_reply.started":"2021-11-27T01:44:59.531209Z","shell.execute_reply":"2021-11-27T01:44:59.601180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Red channel values","metadata":{}},{"cell_type":"code","source":"fig = ff.create_distplot([red_values], group_labels=[\"R\"], colors=[\"red\"])\nfig.update_layout(showlegend=False, template=\"simple_white\")\nfig.update_layout(title_text=\"Distribution of red channel values\")\nfig.data[0].marker.line.color = 'rgb(0, 0, 0)'\nfig.data[0].marker.line.width = 0.5\nfig","metadata":{"execution":{"iopub.status.busy":"2021-11-27T01:46:55.762056Z","iopub.execute_input":"2021-11-27T01:46:55.762636Z","iopub.status.idle":"2021-11-27T01:46:55.832478Z","shell.execute_reply.started":"2021-11-27T01:46:55.762595Z","shell.execute_reply":"2021-11-27T01:46:55.831608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Green channel values","metadata":{}},{"cell_type":"code","source":"fig = ff.create_distplot([green_values], group_labels=[\"G\"], colors=[\"green\"])\nfig.update_layout(showlegend=False, template=\"simple_white\")\nfig.update_layout(title_text=\"Distribution of green channel values\")\nfig.data[0].marker.line.color = 'rgb(0, 0, 0)'\nfig.data[0].marker.line.width = 0.5\nfig","metadata":{"execution":{"iopub.status.busy":"2021-11-27T01:48:30.674300Z","iopub.execute_input":"2021-11-27T01:48:30.674963Z","iopub.status.idle":"2021-11-27T01:48:30.745017Z","shell.execute_reply.started":"2021-11-27T01:48:30.674924Z","shell.execute_reply":"2021-11-27T01:48:30.743658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Blue channel values","metadata":{}},{"cell_type":"code","source":"fig = ff.create_distplot([blue_values], group_labels=[\"B\"], colors=[\"blue\"])\nfig.update_layout(showlegend=False, template=\"simple_white\")\nfig.update_layout(title_text=\"Distribution of blue channel values\")\nfig.data[0].marker.line.color = 'rgb(0, 0, 0)'\nfig.data[0].marker.line.width = 0.5\nfig","metadata":{"execution":{"iopub.status.busy":"2021-11-27T01:49:16.013928Z","iopub.execute_input":"2021-11-27T01:49:16.014335Z","iopub.status.idle":"2021-11-27T01:49:16.084414Z","shell.execute_reply.started":"2021-11-27T01:49:16.014297Z","shell.execute_reply":"2021-11-27T01:49:16.083460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### All channels together","metadata":{}},{"cell_type":"code","source":"fig = go.Figure()\n\nfor idx, values in enumerate([red_values, green_values, blue_values]):\n    if idx == 0:\n        color = \"Red\"\n    if idx == 1:\n        color = \"Green\"\n    if idx == 2:\n        color = \"Blue\"\n    fig.add_trace(go.Box(x=[color]*len(values), y=values, name=color, marker=dict(color=color.lower())))\n    \nfig.update_layout(yaxis_title=\"Mean value\", xaxis_title=\"Color channel\",\n                  title=\"Mean value vs. Color channel\", template=\"plotly_white\")","metadata":{"execution":{"iopub.status.busy":"2021-11-27T01:51:42.363339Z","iopub.execute_input":"2021-11-27T01:51:42.363738Z","iopub.status.idle":"2021-11-27T01:51:42.419097Z","shell.execute_reply.started":"2021-11-27T01:51:42.363706Z","shell.execute_reply":"2021-11-27T01:51:42.418241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = ff.create_distplot([red_values, green_values, blue_values],\n                         group_labels=[\"R\", \"G\", \"B\"],\n                         colors=[\"red\", \"green\", \"blue\"])\nfig.update_layout(title_text=\"Distribution of red channel values\", template=\"simple_white\")\nfig.data[0].marker.line.color = 'rgb(0, 0, 0)'\nfig.data[0].marker.line.width = 0.5\nfig.data[1].marker.line.color = 'rgb(0, 0, 0)'\nfig.data[1].marker.line.width = 0.5\nfig.data[2].marker.line.color = 'rgb(0, 0, 0)'\nfig.data[2].marker.line.width = 0.5\nfig","metadata":{"execution":{"iopub.status.busy":"2021-11-27T01:52:01.310537Z","iopub.execute_input":"2021-11-27T01:52:01.311391Z","iopub.status.idle":"2021-11-27T01:52:01.426187Z","shell.execute_reply.started":"2021-11-27T01:52:01.311343Z","shell.execute_reply":"2021-11-27T01:52:01.424992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualize targets","metadata":{}},{"cell_type":"code","source":"df_train['healthy'] = [1 if 'healthy' in x.split(' ') else 0 for x in df_train['labels']]\ndf_train['rust'] = [1 if 'rust' in x.split(' ') else 0 for x in df_train['labels']]\ndf_train['scab'] = [1 if 'scab' in x.split(' ') else 0 for x in df_train['labels']]\ndf_train['frog_eye_leaf_spot'] = [1 if 'frog_eye_leaf_spot' in x.split(' ') else 0 for x in df_train['labels']]\ndf_train['powdery_mildew'] = [1 if 'powdery_mildew' in x.split(' ') else 0 for x in df_train['labels']]\ndf_train['complex'] = [1 if 'complex' in x.split(' ') else 0 for x in df_train['labels']]","metadata":{"execution":{"iopub.status.busy":"2021-11-27T02:15:14.454303Z","iopub.execute_input":"2021-11-27T02:15:14.454921Z","iopub.status.idle":"2021-11-27T02:15:14.558384Z","shell.execute_reply.started":"2021-11-27T02:15:14.454880Z","shell.execute_reply":"2021-11-27T02:15:14.557262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_drop_labels = df_train.drop('labels', axis=1)\ndf_train_drop_labels.head()","metadata":{"execution":{"iopub.status.busy":"2021-11-27T02:17:00.104099Z","iopub.execute_input":"2021-11-27T02:17:00.104497Z","iopub.status.idle":"2021-11-27T02:17:00.119259Z","shell.execute_reply.started":"2021-11-27T02:17:00.104466Z","shell.execute_reply":"2021-11-27T02:17:00.118347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prepare data\ndf_train[\"Healthy\"] = df_train[\"healthy\"].apply(bool).apply(str)\n\ntrue = df_train[\"Healthy\"][df_train.Healthy == 'True']\nfalse = df_train[\"Healthy\"][df_train.Healthy == 'False']\n\ntrace1 = go.Histogram(\n    x=true,\n    opacity=0.75,\n    name = \"Healthy\",\n    marker=dict(color='rgba(171, 50, 96, 0.6)'))\ntrace2 = go.Histogram(\n    x=false,\n    opacity=0.75,\n    name = \"Unhealthy\",\n    marker=dict(color='rgba(12, 50, 196, 0.6)'))\n\ndata = [trace1, trace2]\nlayout = go.Layout(barmode='overlay',\n                   title='Healthy distribution',\n                   xaxis=dict(title='Healthy'),\n                   yaxis=dict( title='Count'),\n)\nfig = go.Figure(data=data, layout=layout)\niplot(fig)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T03:12:37.509579Z","iopub.execute_input":"2021-11-27T03:12:37.510006Z","iopub.status.idle":"2021-11-27T03:12:37.887000Z","shell.execute_reply.started":"2021-11-27T03:12:37.509969Z","shell.execute_reply":"2021-11-27T03:12:37.885908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prepare data\ndf_train[\"Scab\"] = df_train[\"scab\"].apply(bool).apply(str)\n\ntrue = df_train[\"Scab\"][df_train.Scab == 'True']\nfalse = df_train[\"Scab\"][df_train.Scab == 'False']\n\ntrace1 = go.Histogram(\n    x=true,\n    opacity=0.75,\n    name = \"True\",\n    marker=dict(color='rgba(171, 50, 96, 0.6)'))\ntrace2 = go.Histogram(\n    x=false,\n    opacity=0.75,\n    name = \"False\",\n    marker=dict(color='rgba(12, 50, 196, 0.6)'))\n\ndata = [trace1, trace2]\nlayout = go.Layout(barmode='overlay',\n                   title='Scab distribution',\n                   xaxis=dict(title='Scab'),\n                   yaxis=dict( title='Count'),\n)\nfig = go.Figure(data=data, layout=layout)\niplot(fig)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T03:13:47.179618Z","iopub.execute_input":"2021-11-27T03:13:47.180049Z","iopub.status.idle":"2021-11-27T03:13:47.557952Z","shell.execute_reply.started":"2021-11-27T03:13:47.180016Z","shell.execute_reply":"2021-11-27T03:13:47.556861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prepare data\ndf_train[\"Rust\"] = df_train[\"rust\"].apply(bool).apply(str)\n\ntrue = df_train[\"Rust\"][df_train.Rust == 'True']\nfalse = df_train[\"Rust\"][df_train.Rust == 'False']\n\ntrace1 = go.Histogram(\n    x=true,\n    opacity=0.75,\n    name = \"True\",\n    marker=dict(color='rgba(171, 50, 96, 0.6)'))\ntrace2 = go.Histogram(\n    x=false,\n    opacity=0.75,\n    name = \"False\",\n    marker=dict(color='rgba(12, 50, 196, 0.6)'))\n\ndata = [trace1, trace2]\nlayout = go.Layout(barmode='overlay',\n                   title='Rust distribution',\n                   xaxis=dict(title='Rust'),\n                   yaxis=dict( title='Count'),\n)\nfig = go.Figure(data=data, layout=layout)\niplot(fig)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T03:17:37.801415Z","iopub.execute_input":"2021-11-27T03:17:37.801836Z","iopub.status.idle":"2021-11-27T03:17:38.175016Z","shell.execute_reply.started":"2021-11-27T03:17:37.801804Z","shell.execute_reply":"2021-11-27T03:17:38.173949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prepare data\ndf_train[\"Frogeye\"] = df_train[\"frog_eye_leaf_spot\"].apply(bool).apply(str)\n\ntrue = df_train[\"Frogeye\"][df_train.Frogeye == 'True']\nfalse = df_train[\"Frogeye\"][df_train.Frogeye == 'False']\n\ntrace1 = go.Histogram(\n    x=true,\n    opacity=0.75,\n    name = \"True\",\n    marker=dict(color='rgba(171, 50, 96, 0.6)'))\ntrace2 = go.Histogram(\n    x=false,\n    opacity=0.75,\n    name = \"False\",\n    marker=dict(color='rgba(12, 50, 196, 0.6)'))\n\ndata = [trace1, trace2]\nlayout = go.Layout(barmode='overlay',\n                   title='Frogeye leaf spot distribution',\n                   xaxis=dict(title='Frogeye'),\n                   yaxis=dict( title='Count'),\n)\nfig = go.Figure(data=data, layout=layout)\niplot(fig)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T03:23:25.397360Z","iopub.execute_input":"2021-11-27T03:23:25.397815Z","iopub.status.idle":"2021-11-27T03:23:25.780198Z","shell.execute_reply.started":"2021-11-27T03:23:25.397780Z","shell.execute_reply":"2021-11-27T03:23:25.779051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prepare data\ndf_train[\"Powdery\"] = df_train[\"powdery_mildew\"].apply(bool).apply(str)\n\ntrue = df_train[\"Powdery\"][df_train.Powdery == 'True']\nfalse = df_train[\"Powdery\"][df_train.Powdery == 'False']\n\ntrace1 = go.Histogram(\n    x=true,\n    opacity=0.75,\n    name = \"True\",\n    marker=dict(color='rgba(171, 50, 96, 0.6)'))\ntrace2 = go.Histogram(\n    x=false,\n    opacity=0.75,\n    name = \"False\",\n    marker=dict(color='rgba(12, 50, 196, 0.6)'))\n\ndata = [trace1, trace2]\nlayout = go.Layout(barmode='overlay',\n                   title='Powdery mildew distribution',\n                   xaxis=dict(title='Powdery mildew'),\n                   yaxis=dict( title='Count'),\n)\nfig = go.Figure(data=data, layout=layout)\niplot(fig)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T03:27:27.122775Z","iopub.execute_input":"2021-11-27T03:27:27.123352Z","iopub.status.idle":"2021-11-27T03:27:27.514592Z","shell.execute_reply.started":"2021-11-27T03:27:27.123308Z","shell.execute_reply":"2021-11-27T03:27:27.513217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prepare data\ndf_train[\"Complex\"] = df_train[\"complex\"].apply(bool).apply(str)\n\ntrue = df_train[\"Complex\"][df_train.Complex == 'True']\nfalse = df_train[\"Complex\"][df_train.Complex == 'False']\n\ntrace1 = go.Histogram(\n    x=true,\n    opacity=0.75,\n    name = \"True\",\n    marker=dict(color='rgba(171, 50, 96, 0.6)'))\ntrace2 = go.Histogram(\n    x=false,\n    opacity=0.75,\n    name = \"False\",\n    marker=dict(color='rgba(12, 50, 196, 0.6)'))\n\ndata = [trace1, trace2]\nlayout = go.Layout(barmode='overlay',\n                   title='Complex distribution',\n                   xaxis=dict(title='Complex'),\n                   yaxis=dict( title='Count'),\n)\nfig = go.Figure(data=data, layout=layout)\niplot(fig)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T03:28:25.128592Z","iopub.execute_input":"2021-11-27T03:28:25.129121Z","iopub.status.idle":"2021-11-27T03:28:25.522694Z","shell.execute_reply.started":"2021-11-27T03:28:25.129080Z","shell.execute_reply":"2021-11-27T03:28:25.521606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualize frequency map","metadata":{}},{"cell_type":"markdown","source":"### Random one image","metadata":{}},{"cell_type":"code","source":"img = cv2.imread('../input/plant-pathology-2021-fgvc8/train_images/800113bb65efe69e.jpg', 0)\nplt.imshow(img, cmap='gray')\nplt.axis(\"off\")\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# convert image to floats and do dft saving as complex output\ndft = cv2.dft(np.float32(img), flags = cv2.DFT_COMPLEX_OUTPUT)\n\n# apply shift of origin from upper left corner to center of image\ndft_shift = np.fft.fftshift(dft)\n\nmagnitude_spectrum = np.log(cv2.magnitude(dft_shift[:,:,0],dft_shift[:,:,1]))\nfig = plt.figure(figsize=(8,8))\nplt.imshow(magnitude_spectrum, cmap='gray')\nplt.axis(\"off\")\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Batch label visualize","metadata":{}},{"cell_type":"code","source":"def frequency_visualize_by_label(df,batch_size,path,label): \n    sample_df = df_train[df_train[\"labels\"]==label].sample(9)\n    image_names = sample_df[\"image\"].values\n    labels = sample_df[\"labels\"].values\n    plt.figure(figsize=(16, 12))\n    \n    for image_ind, (image_name, label) in enumerate(zip(image_names, labels)):\n        plt.subplot(3, 3, image_ind + 1)\n        img = cv2.imread(os.path.join(path, image_name), 0)\n        # convert image to floats and do dft saving as complex output\n        dft = cv2.dft(np.float32(img), flags = cv2.DFT_COMPLEX_OUTPUT)\n\n        # apply shift of origin from upper left corner to center of image\n        dft_shift = np.fft.fftshift(dft)\n\n        magnitude_spectrum = np.log(cv2.magnitude(dft_shift[:,:,0],dft_shift[:,:,1]))\n        plt.imshow(magnitude_spectrum, cmap = 'gray')\n        plt.axis(\"off\")\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"frequency_visualize_by_label(df_train,9,train_image_path,'healthy')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"frequency_visualize_by_label(df_train,9,train_image_path,'rust')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"frequency_visualize_by_label(df_train,9,train_image_path,'powdery_mildew')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"frequency_visualize_by_label(df_train,9,train_image_path,'scab')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"frequency_visualize_by_label(df_train,9,train_image_path,'frog_eye_leaf_spot')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"frequency_visualize_by_label(df_train,9,train_image_path,'complex')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Local Features","metadata":{}},{"cell_type":"markdown","source":"### Histogram of Gradient (HoG)","metadata":{}},{"cell_type":"markdown","source":"#### Random an image","metadata":{}},{"cell_type":"code","source":"import skimage\nimport copy\n\nimg = cv2.imread('../input/plant-pathology-2021-fgvc8/test_images/ad8770db05586b59.jpg')\n\nscale_percent = 40 # percent of original size\nwidth = int(img.shape[1] * scale_percent / 100)\nheight = int(img.shape[0] * scale_percent / 100)\ndim = (width, height)\n \n# resize image\nresized = cv2.resize(img, dim, interpolation=cv2.INTER_AREA)\n\nfd, hog_image = skimage.feature.hog(resized, orientations=8, pixels_per_cell=(16, 16),\n                                    cells_per_block=(1, 1), visualize=True, multichannel=True)\n\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 6), sharex=True, sharey=True)\n\nax1.axis('off')\nax1.imshow(resized, cmap=plt.cm.gray)\nax1.set_title('Input image')\n\n# Rescale histogram for better display\nhog_image_rescaled = skimage.exposure.rescale_intensity(hog_image, in_range=(0, 5))\n\nax2.axis('off')\nax2.imshow(hog_image_rescaled, cmap=plt.cm.gray)\nax2.set_title('Histogram of Oriented Gradients')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### HoG by label","metadata":{}},{"cell_type":"code","source":"def hog_visualize_by_label(df,batch_size,path,label): \n    sample_df = df_train[df_train[\"labels\"]==label].sample(9)\n    image_names = sample_df[\"image\"].values\n    labels = sample_df[\"labels\"].values\n    plt.figure(figsize=(16, 12))\n    \n    for image_ind, (image_name, label) in enumerate(zip(image_names, labels)):\n        plt.subplot(3, 3, image_ind + 1)\n        img = cv2.imread(os.path.join(path, image_name))\n        \n        scale_percent = 40 # percent of original size\n        width = int(img.shape[1] * scale_percent / 100)\n        height = int(img.shape[0] * scale_percent / 100)\n        dim = (width, height)\n\n        # resize image\n        resized = cv2.resize(img, dim, interpolation=cv2.INTER_AREA)\n        \n        fd, hog_image = skimage.feature.hog(resized, orientations=8, pixels_per_cell=(16, 16), \n                                            cells_per_block=(1, 1), visualize=True, multichannel=True)\n        \n        # Rescale histogram for better display\n        hog_image_rescaled = skimage.exposure.rescale_intensity(hog_image, in_range=(0, 10))\n        \n        plt.imshow(hog_image_rescaled, cmap=plt.cm.gray)\n        plt.axis(\"off\")\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hog_visualize_by_label(df_train,9,train_image_path,'healthy')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hog_visualize_by_label(df_train,9,train_image_path,'rust')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hog_visualize_by_label(df_train,9,train_image_path,'powdery_mildew')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hog_visualize_by_label(df_train,9,train_image_path,'scab')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hog_visualize_by_label(df_train,9,train_image_path,'frog_eye_leaf_spot')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hog_visualize_by_label(df_train,9,train_image_path,'complex')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Scale-Invariant Feature Transform (SIFT)","metadata":{}},{"cell_type":"markdown","source":"#### Random an image","metadata":{}},{"cell_type":"code","source":"import cv2\nfrom matplotlib import pyplot as plt\n\nimg = cv2.imread('../input/plant-pathology-2021-fgvc8/train_images/8002cb321f8bfcdf.jpg')\n\nscale_percent = 20 # percent of original size\nwidth = int(img.shape[1] * scale_percent / 100)\nheight = int(img.shape[0] * scale_percent / 100)\ndim = (width, height)\n \n# resize image\nresized = cv2.resize(img, dim, interpolation = cv2.INTER_AREA)\n\ngray = cv2.cvtColor(resized, cv2.COLOR_BGR2GRAY)\n\nsift = cv2.xfeatures2d.SIFT_create()\n\nkp, des = sift.detectAndCompute(gray,None)\n\n#img=cv2.drawKeypoints(gray,kp,img)\nimg=cv2.drawKeypoints(gray,kp,img, flags=cv2.DRAW_MATCHES_FLAGS_DRAW_RICH_KEYPOINTS)\n\n#img_final = cv2.drawKeypoints(img, keypoint, None, flags=cv2.DRAW_MATCHES_FLAGS_DRAW_RICH_KEYPOINTS)\n\nplt.figure(figsize=(8, 4))\nplt.imshow(img)\nplt.axis('off')\nplt.show","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Visualize by label","metadata":{}},{"cell_type":"code","source":"def sift_visualize_by_label(df,batch_size,path,label): \n    sample_df = df_train[df_train[\"labels\"]==label].sample(9)\n    image_names = sample_df[\"image\"].values\n    labels = sample_df[\"labels\"].values\n    plt.figure(figsize=(16, 12))\n    \n    for image_ind, (image_name, label) in enumerate(zip(image_names, labels)):\n        plt.subplot(3, 3, image_ind + 1)\n        img = cv2.imread(os.path.join(path, image_name))\n        \n        scale_percent = 40 # percent of original size\n        width = int(img.shape[1] * scale_percent / 100)\n        height = int(img.shape[0] * scale_percent / 100)\n        dim = (width, height)\n\n        # resize image\n        resized = cv2.resize(img, dim, interpolation=cv2.INTER_AREA)\n        \n        gray = cv2.cvtColor(resized, cv2.COLOR_BGR2GRAY)\n        \n        sift = cv2.xfeatures2d.SIFT_create()\n\n        kp, des = sift.detectAndCompute(gray,None)\n\n        img = cv2.drawKeypoints(gray, kp, img, flags=cv2.DRAW_MATCHES_FLAGS_DRAW_RICH_KEYPOINTS)\n        \n        plt.imshow(img, cmap=plt.cm.gray)\n        plt.axis(\"off\")\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sift_visualize_by_label(df_train,9,train_image_path,'healthy')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sift_visualize_by_label(df_train,9,train_image_path,'rust')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sift_visualize_by_label(df_train,9,train_image_path,'powdery_mildew')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sift_visualize_by_label(df_train,9,train_image_path,'scab')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sift_visualize_by_label(df_train,9,train_image_path,'frog_eye_leaf_spot')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sift_visualize_by_label(df_train,9,train_image_path,'complex')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}