{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import thư viện + đọc dữ liệu","metadata":{}},{"cell_type":"code","source":"!pip install opencv-python==3.4.2.17 -q\n!pip install opencv-contrib-python==3.4.2.17 -q","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:42:49.535263Z","iopub.execute_input":"2021-11-29T12:42:49.535924Z","iopub.status.idle":"2021-11-29T12:43:17.171228Z","shell.execute_reply.started":"2021-11-29T12:42:49.535821Z","shell.execute_reply":"2021-11-29T12:43:17.170309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\nimport plotly.express as px\nimport plotly.graph_objects as go\nimport plotly.figure_factory as ff\n\nprint(os.listdir('../'))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-11-29T12:40:49.97623Z","iopub.execute_input":"2021-11-29T12:40:49.976488Z","iopub.status.idle":"2021-11-29T12:40:53.358141Z","shell.execute_reply.started":"2021-11-29T12:40:49.976457Z","shell.execute_reply":"2021-11-29T12:40:53.357558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_image_path = '../input/plant-pathology-2021-fgvc8/train_images/'\ntest_image_path = '../input/plant-pathology-2021-fgvc8/test_images'\ntrain_df_path = '../input/plant-pathology-2021-fgvc8/train.csv'\ntest_df_path = '../input/plant-pathology-2021-fgvc8/sample_submission.csv'\n\ndf_train = pd.read_csv(train_df_path)","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:40:53.359258Z","iopub.execute_input":"2021-11-29T12:40:53.359592Z","iopub.status.idle":"2021-11-29T12:40:53.401178Z","shell.execute_reply.started":"2021-11-29T12:40:53.359563Z","shell.execute_reply":"2021-11-29T12:40:53.400549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Train: {}'.format(len(os.listdir(train_image_path))))\nprint('Test : {}'.format(len(os.listdir(test_image_path))))\nprint('Data set: ')\nprint(df_train.head(3))\n\n\nprint('Label: {}\\n{}'.format(len(df_train.labels.unique()), df_train.labels.unique()))","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:40:53.402273Z","iopub.execute_input":"2021-11-29T12:40:53.402639Z","iopub.status.idle":"2021-11-29T12:40:53.712752Z","shell.execute_reply.started":"2021-11-29T12:40:53.402608Z","shell.execute_reply":"2021-11-29T12:40:53.712044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA cơ bản","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nlabels = sns.barplot(df_train.labels.value_counts().index, df_train.labels.value_counts())\nfor item in labels.get_xticklabels():\n    item.set_rotation(45)","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:40:53.714587Z","iopub.execute_input":"2021-11-29T12:40:53.714943Z","iopub.status.idle":"2021-11-29T12:40:54.077896Z","shell.execute_reply.started":"2021-11-29T12:40:53.714911Z","shell.execute_reply":"2021-11-29T12:40:54.077326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"source = df_train['labels'].value_counts()\nfig = go.Figure(data=[go.Pie(labels=source.index,values=source.values)])\nfig.update_layout(title='Label distribution')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:40:54.078817Z","iopub.execute_input":"2021-11-29T12:40:54.079352Z","iopub.status.idle":"2021-11-29T12:40:54.160436Z","shell.execute_reply.started":"2021-11-29T12:40:54.079321Z","shell.execute_reply":"2021-11-29T12:40:54.159868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['healthy'] = [1 if 'healthy' in x.split(' ') else 0 for x in df_train['labels']]\ndf_train['rust'] = [1 if 'rust' in x.split(' ') else 0 for x in df_train['labels']]\ndf_train['scab'] = [1 if 'scab' in x.split(' ') else 0 for x in df_train['labels']]\ndf_train['frog_eye_leaf_spot'] = [1 if 'frog_eye_leaf_spot' in x.split(' ') else 0 for x in df_train['labels']]\ndf_train['powdery_mildew'] = [1 if 'powdery_mildew' in x.split(' ') else 0 for x in df_train['labels']]\ndf_train['complex'] = [1 if 'complex' in x.split(' ') else 0 for x in df_train['labels']]","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:40:54.161377Z","iopub.execute_input":"2021-11-29T12:40:54.162022Z","iopub.status.idle":"2021-11-29T12:40:54.261775Z","shell.execute_reply.started":"2021-11-29T12:40:54.161985Z","shell.execute_reply":"2021-11-29T12:40:54.26087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(35,20))\nfig = px.parallel_categories(df_train[['healthy','complex','rust','frog_eye_leaf_spot','powdery_mildew','scab']], color=\"healthy\", color_continuous_scale=\"sunset\",\\\n                             title=\"Parallel categories plot of targets\")\nfig","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:40:54.263237Z","iopub.execute_input":"2021-11-29T12:40:54.263641Z","iopub.status.idle":"2021-11-29T12:40:55.305651Z","shell.execute_reply.started":"2021-11-29T12:40:54.263595Z","shell.execute_reply":"2021-11-29T12:40:55.304861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA từng loại bệnh","metadata":{}},{"cell_type":"code","source":"def batch_visualize_with_label(df, path, label, batch_size=3): \n    sample_df = df[df[\"labels\"]==label].sample(batch_size)\n    image_names = sample_df[\"image\"].values\n    labels = sample_df[\"labels\"].values\n    plt.figure(figsize=(12, 12))\n    \n    for image_ind, (image_name, label) in enumerate(zip(image_names, labels)):\n        plt.subplot(1, batch_size, image_ind + 1)\n        image = cv2.imread(os.path.join(path, image_name))\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        plt.imshow(image)\n        plt.axis(\"off\")\n    present_disease = ''\n    if label == 'scab':\n        present_disease = 'bệnh ghẻ, đốm trên lá màu xám xanh'\n    elif label == 'frog_eye_leaf_spot':\n        present_disease = 'đốm nâu, sẫm màu giống mắt ếch'\n    elif label == 'rust':\n        present_disease = 'bệnh gỉ sắt, giống vết gỉ sắt màu vàng cam'\n    elif label == 'complex':\n        present_disease = '...'\n    elif label == 'powdery_mildew':\n        present_disease = 'bệnh phấn trắng, nên lá có bụi phủ giống phấn'\n    plt.title(label + ': ' + present_disease, loc='right')","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:40:55.306954Z","iopub.execute_input":"2021-11-29T12:40:55.307199Z","iopub.status.idle":"2021-11-29T12:40:55.317173Z","shell.execute_reply.started":"2021-11-29T12:40:55.307169Z","shell.execute_reply":"2021-11-29T12:40:55.316312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in df_train.labels.unique():\n    batch_visualize_with_label(df_train, train_image_path, i, batch_size=5)","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:40:55.318381Z","iopub.execute_input":"2021-11-29T12:40:55.318738Z","iopub.status.idle":"2021-11-29T12:41:58.499153Z","shell.execute_reply.started":"2021-11-29T12:40:55.318702Z","shell.execute_reply":"2021-11-29T12:41:58.498407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EDA_IMG_SHAPE = (512, 256)\n\ndef getImage(image_id, SHAPE=EDA_IMG_SHAPE):\n    img = cv2.imread(os.path.join(train_image_path, image_id))\n    img = cv2.resize(img, SHAPE)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    \n    return img\n\ncolors = ['rgb(200, 0, 0)', 'rgb(0, 200, 0)', 'rgb(0,0,200)']\n\ndef plotChannelDistribution(df, condition):\n    \n    distributions = []\n        \n    for channel in range(3):\n        distributions.append([np.mean(getImage(img)[:,:,channel]) for img in df[df[\"labels\"]==condition].sample(50).image.values])\n    \n    fig = ff.create_distplot(distributions,\n                            group_labels=['red','green','blue'],\n                            colors=colors)\n    \n    fig.update_layout(title=f'{condition.capitalize()} leaves channel distribution')\n    \n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:41:58.500547Z","iopub.execute_input":"2021-11-29T12:41:58.501281Z","iopub.status.idle":"2021-11-29T12:41:58.512379Z","shell.execute_reply.started":"2021-11-29T12:41:58.501239Z","shell.execute_reply":"2021-11-29T12:41:58.511372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in df_train.labels.unique():\n    plotChannelDistribution(df_train, i)","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:41:58.513952Z","iopub.execute_input":"2021-11-29T12:41:58.514439Z"},"trusted":true},"execution_count":null,"outputs":[]}]}