{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import glob\n\nimport matplotlib.pyplot as plt\nfrom matplotlib.image import imread\n\nimport pandas as pd\n# do not truncate dataframe data when displaying\npd.set_option('display.max_colwidth', None)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-08-12T18:54:46.686013Z","iopub.execute_input":"2021-08-12T18:54:46.686478Z","iopub.status.idle":"2021-08-12T18:54:46.699866Z","shell.execute_reply.started":"2021-08-12T18:54:46.686383Z","shell.execute_reply":"2021-08-12T18:54:46.698691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# define constants and see what file structure kaggle defines\nbase_dir = \"/kaggle/input/plant-pathology-2021-fgvc8/\"\ntrain_dir = base_dir + \"train_images/\"\nglob.glob(base_dir + \"*\")","metadata":{"execution":{"iopub.status.busy":"2021-08-12T18:54:46.701727Z","iopub.execute_input":"2021-08-12T18:54:46.702210Z","iopub.status.idle":"2021-08-12T18:54:46.726745Z","shell.execute_reply.started":"2021-08-12T18:54:46.702166Z","shell.execute_reply":"2021-08-12T18:54:46.725468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# read the train.csv file into a dataframe\ninput_df = pd.read_csv(base_dir + \"train.csv\")\ninput_df['image'] = input_df['image'].apply(lambda x: train_dir + x)\ninput_df","metadata":{"execution":{"iopub.status.busy":"2021-08-12T18:54:46.728945Z","iopub.execute_input":"2021-08-12T18:54:46.729304Z","iopub.status.idle":"2021-08-12T18:54:46.934744Z","shell.execute_reply.started":"2021-08-12T18:54:46.729270Z","shell.execute_reply":"2021-08-12T18:54:46.934055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get a series from labels and count how many we have of each type\n# https://pandas.pydata.org/docs/reference/api/pandas.Series.plot.html\nsplit_labels_df = input_df.copy()\nsplit_labels_df['labels'] = split_labels_df['labels'].apply(lambda x: x.split(' '))\nsplit_labels_df['labels'].explode('labels').value_counts().sort_values(ascending=True).plot(kind='barh')","metadata":{"execution":{"iopub.status.busy":"2021-08-12T18:55:34.561856Z","iopub.execute_input":"2021-08-12T18:55:34.562313Z","iopub.status.idle":"2021-08-12T18:55:34.925096Z","shell.execute_reply.started":"2021-08-12T18:55:34.562272Z","shell.execute_reply":"2021-08-12T18:55:34.923892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get a preview_df with a sample of sample_count images for each label\nsample_count = 3\npreview_df = input_df.groupby('labels').head(sample_count).sort_values(by=['labels'])\n\n# for each label display a subplot with sample_count images\nfor group_name, df_group in preview_df.groupby('labels'):\n    figure, axes = plt.subplots(nrows=1, ncols=sample_count, figsize=[20, 20])\n    for row_index, row in df_group.reset_index().iterrows():\n        image = plt.imread(row['image'])\n        axes[row_index].imshow(image)\n        axes[row_index].set_title(group_name + str(row_index))\n        axes[row_index].axis('off')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-12T18:54:46.936106Z","iopub.execute_input":"2021-08-12T18:54:46.936618Z","iopub.status.idle":"2021-08-12T18:55:34.560670Z","shell.execute_reply.started":"2021-08-12T18:54:46.936585Z","shell.execute_reply":"2021-08-12T18:55:34.559863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}