{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import torch\nimport cv2\nimport os\nimport torch.nn as tnn\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport plotly.express as px\nimport seaborn as sns\n\nfrom PIL import Image\nfrom skimage import io, transform\nfrom torchvision.transforms import transforms\nfrom torchvision import utils\nfrom torchvision import datasets\nfrom torch.utils.data import DataLoader, Dataset\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom collections import Counter","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_PATH = \"../input/plant-pathology-2021-fgvc8/train_images/\"\nTEST_IMG_PATH = \"../input/plant-pathology-2021-fgvc8/test_images/\"\nTRAIN_PATH = \"../input/plant-pathology-2021-fgvc8/train.csv\"\nSUB_PATH = \"../input/plant-pathology-2021-fgvc8/sample_submission.csv\"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = pd.read_csv(TRAIN_PATH)\ntrain_labels","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels['labels'].unique()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(18,12))\nplt.title(\"Phân phối số lượng ảnh trong các nhãn\",size= 25)\nplt.ylabel(\"Số lượng ảnh\", size=20);\nplt.xlabel(\"Nhãn\", size=20);\nlabels = sns.barplot(train_labels.labels.value_counts().index,train_labels.labels.value_counts())\nfor item in labels.get_xticklabels():\n    item.set_rotation(45)\nplt.savefig('plot.png')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mlb = MultiLabelBinarizer().fit(train_labels.labels.apply(lambda x : x.split()))\nlabels = pd.DataFrame(mlb.transform(train_labels.labels.apply(lambda x : x.split())), columns = mlb.classes_)\n\nlabels = pd.concat([train_labels['image'], labels], axis=1)\nlabels.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = ['1','2','3']\nvalue = labels.iloc[:,1:].sum(axis=1).value_counts().values\ncolors = ['mediumturquoise', 'burlywood','sandybrown']\nplt.figure(figsize=(8, 8))\nplt.bar(data, value, color = colors)\nplt.title('Ảnh có nhiều nhãn',fontsize = 14)\nplt.xlabel('Số nhãn',fontsize = 12)\nplt.ylabel('Số lượng ảnh',fontsize = 12)\nplt.savefig('plot2.png')\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_name = labels.iloc[:,0].tolist()\nhs = []\nws = []\nfor i in range(len(img_name)):\n        img = Image.open(IMAGE_PATH+(img_name[i]))\n        h, w = img.size\n        hs.append(h)\n        ws.append(w)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels, values = zip(*Counter(hs).items())\n\nindexes = np.arange(len(labels))\nwidth = 1\n\nplt.bar(indexes, values, width)\nplt.xticks(indexes + width * 0.5, labels)\nplt.savefig('plot4.png')\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels, values = zip(*Counter(ws).items())\n\nindexes = np.arange(len(labels))\nwidth = 1\n\nplt.bar(indexes, values, width)\nplt.xticks(indexes + width * 0.5, labels)\nplt.savefig('plot5.png')\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = pd.DataFrame(mlb.transform(train_labels.labels.apply(lambda x : x.split())), columns = mlb.classes_)\n\nlabels = pd.concat([train_labels['image'], labels], axis=1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def visualize_batch(path,image_ids, labels):\n    plt.figure(figsize=(16, 12))\n    \n    for ind, (image_id, label) in enumerate(zip(image_ids, labels)):\n        plt.subplot(3, 3, ind + 1)\n        image = cv2.imread(os.path.join(path, image_id))\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\n        plt.imshow(image)\n        plt.title(f\"Class: {label}\", fontsize=12)\n        plt.axis(\"off\")\n        plt.savefig('plot3.png')\n    plt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_s = train_labels.sample(9)\nimage_ids = img_s[\"image\"].values\nlabels_s = img_s[\"labels\"].values\nvisualize_batch(IMAGE_PATH,image_ids,labels_s)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"l_complex = labels.loc[labels['complex'] == 1].iloc[:,0].tolist()\nfrog_eye_leaf_spot = labels.loc[labels['frog_eye_leaf_spot'] == 1].iloc[:,0].tolist()\nhealthy = labels.loc[labels['healthy'] == 1].iloc[:,0].tolist()\npowdery_mildew = labels.loc[labels['powdery_mildew'] == 1].iloc[:,0].tolist()\nrust = labels.loc[labels['rust'] == 1].iloc[:,0].tolist()\nscab = labels.loc[labels['scab'] == 1].iloc[:,0].tolist()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(12, 12))\nfor i in range(0,9):\n        img_array = np.array(Image.open(IMAGE_PATH +healthy[i]))\n        fig.add_subplot(3, 3, i+1) \n        plt.imshow(img_array)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(12, 12))\nfor i in range(0,9):\n        img_array = np.array(Image.open(IMAGE_PATH +l_complex[i]))\n        fig.add_subplot(3, 3, i+1) \n        plt.imshow(img_array)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(12, 12))\nfor i in range(0,9):\n        img_array = np.array(Image.open(IMAGE_PATH +frog_eye_leaf_spot[i]))\n        fig.add_subplot(3, 3, i+1) \n        plt.imshow(img_array)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(12, 12))\nfor i in range(0,9):\n        img_array = np.array(Image.open(IMAGE_PATH +scab[i]))\n        fig.add_subplot(3, 3, i+1) \n        plt.imshow(img_array)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(12, 12))\nfor i in range(0,9):\n        img_array = np.array(Image.open(IMAGE_PATH +powdery_mildew[i]))\n        fig.add_subplot(3, 3, i+1) \n        plt.imshow(img_array)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(12, 12))\nfor i in range(0,9):\n        img_array = np.array(Image.open(IMAGE_PATH +rust[i]))\n        fig.add_subplot(3, 3, i+1) \n        plt.imshow(img_array)","metadata":{},"execution_count":null,"outputs":[]}]}