{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Loading libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport json\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n%matplotlib inline","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading data","metadata":{}},{"cell_type":"code","source":"df_data = pd.read_csv('../input/plant-pathology-2021-fgvc8/train.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data.head(1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Cleaning data\n\nAs we can see there is two columns in the dataset. One named images and other named labels. Images give the image id and labels give the diseases in the leaf. Lets make a new colum for image id and path of image.","metadata":{}},{"cell_type":"code","source":"# removing .jpg from images and storing id in seperate column\ndf_data['image_id'] = df_data['image'].map(lambda x: x.rstrip('.jpg'))\ndf_data.head(1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking for missing values in new column\ndf_data['image_id'].isnull().unique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creating a column of file location\ndf_data['file_path'] = \"../input/plant-pathology-2021-fgvc8/train_images/\"+df_data['image']\ndf_data.head(1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# removing unwanted columns\ndf_data = df_data.drop(['image'], axis=1)\n# changing order of column\ndf_data = df_data[['image_id','labels','file_path']]\ndf_data.head(1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#removing null values\ndf_data = df_data.dropna(how = 'all')\n# cheking for missing data \ndf_data.isnull().sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### EDA","metadata":{}},{"cell_type":"code","source":"df_data.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(df_data['labels'].unique())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data['labels'].unique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are total of 18632 images given and a total of 12 diseases labels given. On further inspection we see that there are five individual diseases namely \n- frog_eye_leaf_spot \n- complex\n- rust\n- scab\n- powdery_mildew\n\nOther six labels are the combination of two or more of these diseases. The remining one label is healthy leaf without any diseases","metadata":{}},{"cell_type":"code","source":"#Number of images in each label\nplt.figure(figsize=(15, 10))\ndf_data['labels'].value_counts().plot.bar()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15, 15))\ndf_data['labels'].value_counts().plot.pie(autopct='%.2f')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the plot we can see that most of the images are concentrated in the five individual diseases labels. Labels with multiple diseases have asmall dataset. This might skew the detection matrix.","metadata":{}},{"cell_type":"code","source":"# list of labels\nn = df_data['labels'].unique().tolist()\n# finding number of images in each label\nnumber=[]\nfor i in range(len(n)):\n    number.append(len(df_data.loc[df_data['labels']==n[i]]))\n    \n# creating a dataset with labels and number of images in them    \ntable =pd.DataFrame(df_data['labels'].unique(),columns =['labels'])\ntable['Number_of_images_in_label'] = number\ntable = table.sort_values(by=['Number_of_images_in_label'], ascending=False)\nprint(table.to_markdown())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualize","metadata":{}},{"cell_type":"code","source":"# plotting image by image id for single image\ndef  visualize(image_id):\n    \n    path = df_data.loc[df_data['image_id'] == image_id, 'file_path'].iloc[0]\n    label = df_data.loc[df_data['image_id'] == image_id, 'labels'].iloc[0]\n    \n    plt.figure(figsize=(10, 10))\n    image = cv2.imread(path)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    plt.imshow(image)\n    plt.title(f\"Label: {label}\", fontsize=10,)\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize('800113bb65efe69e')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot image by label . plot 15 random images from label\ndef visualize_label(label):\n    \n     \n    df = df_data.loc[df_data['labels'] == label]\n    \n    x = np.random.choice(df['image_id'], 15, replace=False).tolist()\n    plt.figure(figsize=(18, 18))\n    for i, j in zip(x, range(15)):       \n            plt.subplot(3, 5, j + 1)\n            path = df.loc[df['image_id'] == i, 'file_path'].iloc[0]\n            image = cv2.imread(path)\n            image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n            plt.imshow(image)\n           \n            labels = df.loc[df['image_id'] == i, 'labels'].iloc[0]\n            imageid= df.loc[df['image_id'] == i, 'image_id'].iloc[0]\n            plt.title(f\" Label: {labels}\\n Image_id:{imageid}\", fontsize=9,)\n            plt.axis(\"off\")\n            #plt.savefig('saved_figure.png')\n    \n    plt.show()                \n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Healthy leaf","metadata":{}},{"cell_type":"code","source":"visualize_label('healthy')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Scab","metadata":{}},{"cell_type":"code","source":"visualize_label('scab')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Complex","metadata":{}},{"cell_type":"code","source":"visualize_label('complex')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Rust","metadata":{}},{"cell_type":"code","source":"visualize_label('rust')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Frog_eye_leaf_spot","metadata":{}},{"cell_type":"code","source":"visualize_label('frog_eye_leaf_spot')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Powdery_mildew","metadata":{}},{"cell_type":"code","source":"visualize_label('powdery_mildew')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Creating test dataset","metadata":{}}]}