{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### **EDA**","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport tensorflow as tf\nimport cv2\nfrom tqdm import tqdm\nimport os\n\nimport plotly.express as px\nimport imagehash","metadata":{"execution":{"iopub.status.busy":"2021-06-11T14:44:58.221544Z","iopub.execute_input":"2021-06-11T14:44:58.222254Z","iopub.status.idle":"2021-06-11T14:45:06.899714Z","shell.execute_reply.started":"2021-06-11T14:44:58.222117Z","shell.execute_reply":"2021-06-11T14:45:06.898503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.getcwd()","metadata":{"execution":{"iopub.status.busy":"2021-06-11T14:45:06.901416Z","iopub.execute_input":"2021-06-11T14:45:06.901744Z","iopub.status.idle":"2021-06-11T14:45:06.910156Z","shell.execute_reply.started":"2021-06-11T14:45:06.901714Z","shell.execute_reply":"2021-06-11T14:45:06.909105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1. Preparing the ground","metadata":{}},{"cell_type":"code","source":"path_dict = {'test_images' : '../input/plant-pathology-2021-fgvc8/test_images',\n            'train_images': '../input/plant-pathology-2021-fgvc8/train_images',\n            'train_csv'   : '../input/plant-pathology-2021-fgvc8/train.csv'\n            }\n\n\ntrain = pd.read_csv(path_dict['train_csv'])\nstr_label = train['labels']\n\nlabel_dict = {}\nfor i, label in enumerate(str_label.unique()):\n    label_dict[label] = i\n\n    \n'''label_dict = {'healthy': 0,\n 'scab frog_eye_leaf_spot complex': 4,\n 'scab': 1,\n 'complex': 4,\n 'rust': 2,\n 'frog_eye_leaf_spot': 3,\n 'powdery_mildew': 4,\n 'scab frog_eye_leaf_spot': 4,\n 'frog_eye_leaf_spot complex': 4,\n 'rust frog_eye_leaf_spot': 4,\n 'powdery_mildew complex': 4,\n 'rust complex': 4}\n '''\n    \n# Integer Coding    \ntrain['int_encoder'] = train['labels']\ntrain.replace({'int_encoder': label_dict}, inplace = True)\ntrain_y = tf.keras.utils.to_categorical(train['int_encoder'].to_numpy(), num_classes = 12)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T14:45:06.912402Z","iopub.execute_input":"2021-06-11T14:45:06.912849Z","iopub.status.idle":"2021-06-11T14:45:06.991280Z","shell.execute_reply.started":"2021-06-11T14:45:06.912774Z","shell.execute_reply":"2021-06-11T14:45:06.990292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.concat([train,pd.DataFrame(train_y)], axis = 1)\n","metadata":{"execution":{"iopub.status.busy":"2021-06-11T14:45:06.992966Z","iopub.execute_input":"2021-06-11T14:45:06.993298Z","iopub.status.idle":"2021-06-11T14:45:07.037654Z","shell.execute_reply.started":"2021-06-11T14:45:06.993267Z","shell.execute_reply":"2021-06-11T14:45:07.036480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Classifying 12 disease in in 4 classes in final_label_dict and Finding the Percentage of disese in complete data","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\n\ndf = train.copy()\ndf = df.groupby('labels').count()\ndf = df.reset_index()\ndf = df[['labels','image']]\n\ndf['percentage'] = df['image'] / len(train) * 100\n\nfig = plt.figure(figsize = (25,10))\nax = sns.barplot(x = 'labels' , y = 'percentage' , data = df)\nax.set_xticklabels(ax.get_xticklabels(), rotation=30, ha=\"right\")\nplt.title('Percentage of Disease in data')\nplt.xlabel('Types of Disease')\n","metadata":{"execution":{"iopub.status.busy":"2021-06-11T14:45:07.038947Z","iopub.execute_input":"2021-06-11T14:45:07.039245Z","iopub.status.idle":"2021-06-11T14:45:07.399064Z","shell.execute_reply.started":"2021-06-11T14:45:07.039216Z","shell.execute_reply":"2021-06-11T14:45:07.398018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Seaborn Introduction** \n\n### Related to Catagorical (categorical vs integer variable)\n1. BoxPlot\n2. Barplot\n3. Voilin Plot\n4. count PLot\n5. stripplot and swarmplot\n6. factorplot - This is in general plot having kind parameter which canbe set as bar, box, violin.\n","metadata":{}},{"cell_type":"markdown","source":"### Related to Distribution plot (for examining univariate and bivariate distributions.)\n1. Distplot  -  univariant set of observations and visualizes it through a histogram.\n2. Joinplot  -  draw a plot of two variables with bivariate and univariate graphs.\n3. Pair Plot -  plot between each pair varibales.\n4. RugbPlot  -  dashes plot for a single column \n","metadata":{}},{"cell_type":"markdown","source":"### Related to Regression Plots\n1. Simple linear plot","metadata":{}},{"cell_type":"markdown","source":"### Visulaisation in pi chart","metadata":{}},{"cell_type":"code","source":"%config Completer.use_jedi = False","metadata":{"execution":{"iopub.status.busy":"2021-06-11T14:45:07.400428Z","iopub.execute_input":"2021-06-11T14:45:07.400742Z","iopub.status.idle":"2021-06-11T14:45:07.415363Z","shell.execute_reply.started":"2021-06-11T14:45:07.400711Z","shell.execute_reply":"2021-06-11T14:45:07.414163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (10,10))\nplt.pie(x = df['percentage'],labels = df['labels'] ,autopct='%1.1f%%',labeldistance=1.2,radius=0.9,pctdistance= 0.7)\nplt.legend(loc='upper right')","metadata":{"execution":{"iopub.status.busy":"2021-06-11T14:45:07.417832Z","iopub.execute_input":"2021-06-11T14:45:07.418204Z","iopub.status.idle":"2021-06-11T14:45:07.736643Z","shell.execute_reply.started":"2021-06-11T14:45:07.418162Z","shell.execute_reply":"2021-06-11T14:45:07.735496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### **Conclusion**\n\n1. Data set is imbalanced","metadata":{}},{"cell_type":"markdown","source":"### Plotting Sample Images","metadata":{}},{"cell_type":"code","source":"def visualize_batch(path,image_id, labels):\n    plt.figure(figsize=(16, 12))\n    \n    for ind, (image_id, label) in enumerate(zip(image_id, labels)):\n        plt.subplot(5,4, ind + 1)\n        image = cv2.imread(os.path.join(path, image_id))\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\n        plt.imshow(image)\n        plt.title(f\"Class: {label}\", fontsize=12)\n        plt.axis(\"off\")\n    plt.show()\n    \nsample_df = train.sample(20)\nimage_id = sample_df[\"image\"].values\nlabels = sample_df[\"labels\"].values\nvisualize_batch(path_dict['train_images'],image_id,labels)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T14:45:07.739338Z","iopub.execute_input":"2021-06-11T14:45:07.739782Z","iopub.status.idle":"2021-06-11T14:45:25.151886Z","shell.execute_reply.started":"2021-06-11T14:45:07.739734Z","shell.execute_reply":"2021-06-11T14:45:25.149271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"arr = cv2.imread(path_dict['train_images'] + '/'+ 'acc99659863d9f0a.jpg')\narr = cv2.cvtColor(arr, cv2.COLOR_BGR2RGB)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T14:45:25.153619Z","iopub.execute_input":"2021-06-11T14:45:25.153961Z","iopub.status.idle":"2021-06-11T14:45:25.366182Z","shell.execute_reply.started":"2021-06-11T14:45:25.153931Z","shell.execute_reply":"2021-06-11T14:45:25.365077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Distinct List of labels\n\n* healthy\n* complex\n* rust\n* frog_eye_leaf_spot\n* powdery_mildew\n* scab","metadata":{}},{"cell_type":"markdown","source":"### Training Data ['labels'] splitting by ''","metadata":{}},{"cell_type":"code","source":"distinct_labels = ['healthy','complex','rust','frog_eye_leaf_spot','powdery_mildew','scab']\ntrain['label_list'] = train['labels'].str.split()\nfor x in distinct_labels:\n    train[x] = 0\n\n\ndef overlapping_category(label_list, coln):\n    if coln in label_list :\n        return 1\n    else:\n        return 0\n\nfor x in distinct_labels:\n    myfunc = np.vectorize(overlapping_category)\n    train[x] = myfunc(train['label_list'], x)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T14:45:25.367542Z","iopub.execute_input":"2021-06-11T14:45:25.367840Z","iopub.status.idle":"2021-06-11T14:45:25.427084Z","shell.execute_reply.started":"2021-06-11T14:45:25.367811Z","shell.execute_reply":"2021-06-11T14:45:25.425953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"px.parallel_categories(train , dimensions= distinct_labels, color= 'healthy', color_continuous_scale=\"sunset\")","metadata":{"execution":{"iopub.status.busy":"2021-06-11T14:45:25.428382Z","iopub.execute_input":"2021-06-11T14:45:25.428722Z","iopub.status.idle":"2021-06-11T14:45:26.833742Z","shell.execute_reply.started":"2021-06-11T14:45:25.428692Z","shell.execute_reply":"2021-06-11T14:45:26.833054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Removing Duplicate Images\n\nwe are going to use image hash library to delete the duplicate images.","metadata":{}},{"cell_type":"code","source":"from PIL import Image\n\nthreshold = .9\nimg_height = 8\nimg_width = 9\nseed = 42\n\nroot = path_dict['train_images']\nlist_of_images = os.listdir(root)\ndf = pd.read_csv(path_dict['train_csv'], index_col='image')\n\nfor i in tqdm(list_of_images, total=len(list_of_images)):\n    image = os.path.join(root,i)\n    tens = tf.io.read_file(image)\n    image = tf.image.decode_png(tens, channels=3)\n    image = tf.image.resize(image,[img_height, img_width])\n    image = tf.cast(image, tf.uint8).numpy()\n    plt.imsave(i, image)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T14:45:26.834791Z","iopub.execute_input":"2021-06-11T14:45:26.835203Z","iopub.status.idle":"2021-06-11T15:05:47.752997Z","shell.execute_reply.started":"2021-06-11T14:45:26.835162Z","shell.execute_reply":"2021-06-11T15:05:47.751845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import PIL\nhash_functions = [\n    imagehash.average_hash,\n    imagehash.phash,\n    imagehash.dhash,\n    imagehash.whash]\n\nimage_ids = []\nhashes = []\n\nlist_of_images_2 = tf.io.gfile.glob('./*.jpg')\n\nfor i in tqdm(list_of_images_2, total=len(list_of_images_2)):\n\n    image = PIL.Image.open(i)\n\n    hashes.append(np.array([x(image).hash for x in hash_functions]).reshape(-1,))\n    image_ids.append(i.split('/')[-1])\n    \nhashes = np.array(hashes)\nimage_ids = np.array(image_ids)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T15:05:47.754660Z","iopub.execute_input":"2021-06-11T15:05:47.755202Z","iopub.status.idle":"2021-06-11T15:06:15.288730Z","shell.execute_reply.started":"2021-06-11T15:05:47.755153Z","shell.execute_reply":"2021-06-11T15:06:15.287475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"duplicate_ids = []   # To store all duplicate ids\nid_similar_id = {}      # to store which image_id is simalilar to which image_ids\n\nfor id, hash in tqdm(zip(image_ids,hashes), total=len(hashes)):\n    if id not in duplicate_ids:\n        similarity = (hash == hashes).mean(axis=1)\n        similar_to = list(image_ids[similarity > threshold])\n        similar_to.remove(id)\n        if(len(similar_to)>0):\n            id_similar_id[id] = similar_to\n        \n        for i in similar_to:\n            duplicate_ids.append(i)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T15:13:28.481653Z","iopub.execute_input":"2021-06-11T15:13:28.482254Z","iopub.status.idle":"2021-06-11T15:16:35.817178Z","shell.execute_reply.started":"2021-06-11T15:13:28.482201Z","shell.execute_reply":"2021-06-11T15:16:35.816094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,100))\n\n\nfor i ,(key, value) in enumerate(id_similar_id.items()):\n    plt.subplot(27,2,2*i+1)\n    plt.imshow(cv2.imread(path_dict['train_images'] +'/'+key))\n    plt.title(key)\n    plt.axis('off')\n    \n    plt.subplot(27,2,2*i+2)\n    plt.imshow(cv2.imread(path_dict['train_images'] +'/'+value[0]))\n    plt.title(value[0])\n    plt.axis('off')\n\n    \n    \n    \n","metadata":{"execution":{"iopub.status.busy":"2021-06-11T16:07:04.832218Z","iopub.execute_input":"2021-06-11T16:07:04.832629Z","iopub.status.idle":"2021-06-11T16:07:36.045389Z","shell.execute_reply.started":"2021-06-11T16:07:04.832591Z","shell.execute_reply":"2021-06-11T16:07:36.044471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}