{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-23T18:07:06.745761Z","iopub.execute_input":"2023-08-23T18:07:06.747159Z","iopub.status.idle":"2023-08-23T18:07:06.752559Z","shell.execute_reply.started":"2023-08-23T18:07:06.747114Z","shell.execute_reply":"2023-08-23T18:07:06.751689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfor dirname, dirs, _ in os.walk('/kaggle/input'):\n    for dir in dirs:\n        print(os.path.join(dirname, dir))","metadata":{"execution":{"iopub.status.busy":"2023-08-28T12:04:28.691407Z","iopub.execute_input":"2023-08-28T12:04:28.692208Z","iopub.status.idle":"2023-08-28T12:04:53.681370Z","shell.execute_reply.started":"2023-08-28T12:04:28.692165Z","shell.execute_reply":"2023-08-28T12:04:53.679994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-08-28T12:05:11.848602Z","iopub.execute_input":"2023-08-28T12:05:11.849093Z","iopub.status.idle":"2023-08-28T12:05:11.854920Z","shell.execute_reply.started":"2023-08-28T12:05:11.849057Z","shell.execute_reply":"2023-08-28T12:05:11.853317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_path = '/kaggle/input/plant-pathology-2021-fgvc8/'","metadata":{"execution":{"iopub.status.busy":"2023-08-28T12:05:12.237474Z","iopub.execute_input":"2023-08-28T12:05:12.238077Z","iopub.status.idle":"2023-08-28T12:05:12.245922Z","shell.execute_reply.started":"2023-08-28T12:05:12.238033Z","shell.execute_reply":"2023-08-28T12:05:12.244050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(img_path+'train.csv')\nsubmission = pd.read_csv(img_path+'sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-08-28T12:05:12.623346Z","iopub.execute_input":"2023-08-28T12:05:12.624059Z","iopub.status.idle":"2023-08-28T12:05:12.706867Z","shell.execute_reply.started":"2023-08-28T12:05:12.624024Z","shell.execute_reply":"2023-08-28T12:05:12.705208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-28T12:05:13.026879Z","iopub.execute_input":"2023-08-28T12:05:13.027375Z","iopub.status.idle":"2023-08-28T12:05:13.054405Z","shell.execute_reply.started":"2023-08-28T12:05:13.027337Z","shell.execute_reply":"2023-08-28T12:05:13.052914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_list = train['labels']","metadata":{"execution":{"iopub.status.busy":"2023-08-28T12:06:34.680544Z","iopub.execute_input":"2023-08-28T12:06:34.681014Z","iopub.status.idle":"2023-08-28T12:06:34.688627Z","shell.execute_reply.started":"2023-08-28T12:06:34.680981Z","shell.execute_reply":"2023-08-28T12:06:34.686860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# labels 종류가 뭐가 있는지 알고싶어\nlabel_list = list(set(label_list.tolist()))\nprint(f'label list:{label_list}')","metadata":{"execution":{"iopub.status.busy":"2023-08-28T12:06:35.825827Z","iopub.execute_input":"2023-08-28T12:06:35.826277Z","iopub.status.idle":"2023-08-28T12:06:35.835356Z","shell.execute_reply.started":"2023-08-28T12:06:35.826245Z","shell.execute_reply":"2023-08-28T12:06:35.833415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-28T12:06:37.050591Z","iopub.execute_input":"2023-08-28T12:06:37.051059Z","iopub.status.idle":"2023-08-28T12:06:37.063592Z","shell.execute_reply.started":"2023-08-28T12:06:37.051025Z","shell.execute_reply":"2023-08-28T12:06:37.062065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[train['labels'] == 'healthy']","metadata":{"execution":{"iopub.status.busy":"2023-08-28T12:06:39.839788Z","iopub.execute_input":"2023-08-28T12:06:39.840323Z","iopub.status.idle":"2023-08-28T12:06:39.863049Z","shell.execute_reply.started":"2023-08-28T12:06:39.840290Z","shell.execute_reply":"2023-08-28T12:06:39.861319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nums = []\nfor label in label_list:\n    nums.append(train[train['labels']==label])","metadata":{"execution":{"iopub.status.busy":"2023-08-28T12:06:59.845948Z","iopub.execute_input":"2023-08-28T12:06:59.846417Z","iopub.status.idle":"2023-08-28T12:06:59.902627Z","shell.execute_reply.started":"2023-08-28T12:06:59.846383Z","shell.execute_reply":"2023-08-28T12:06:59.901280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib as mlt\nimport matplotlib.pyplot as plt\n\nmlt.rc('font', size=6)\nplt.figure(figsize=(10,10))\n\nsizes = [len(num) for num in nums]\nplt.pie(sizes, labels=label_list, autopct='%1.1f%%')\nplt.plot() # need stratify ","metadata":{"execution":{"iopub.status.busy":"2023-08-28T12:07:04.191515Z","iopub.execute_input":"2023-08-28T12:07:04.192448Z","iopub.status.idle":"2023-08-28T12:07:04.581392Z","shell.execute_reply.started":"2023-08-28T12:07:04.192403Z","shell.execute_reply":"2023-08-28T12:07:04.580156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 왜 plt가 아니라 cv2를 쓰게 된거지??\n# - plt: 이미지를 출력하는것에 강점이 있으며, jupyter notebook환경을 가장 선호함. 처리능력은 떨어짐. (단순 출력용)\n# - cv2: 이미지 처리에 강점이 있다. 출력은 그다지 중요한 기능을 취급하지 않는다. (처리용)\n\n# 단순 이미지를 출력하는 것이므로, plt로 만들어보도록 하자.\ndef show_image(imgs, rows=2, cols=3):\n    assert len(imgs) <= rows*cols\n    \n    plt.figure(figsize=(15,8))\n    grid = gridspec.GridSpec(rows, cols)\n    \n    for idx, img in enumerate(imgs):\n        img_path = f'{data_path}/train_images/{img}'\n        ax = plt.subplot(grid[idx])\n        ax.imshow(image)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T12:16:34.196205Z","iopub.execute_input":"2023-08-28T12:16:34.196735Z","iopub.status.idle":"2023-08-28T12:16:34.207961Z","shell.execute_reply.started":"2023-08-28T12:16:34.196696Z","shell.execute_reply":"2023-08-28T12:16:34.204958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_of_imgs = 6\nlast_","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1. EDA(데이터를 불러와서 haed를 통해서 불러보기, 비율이 어떻게 되는지 알아보기, 6개정도 이미지 출력해보기)\n\n2. 데이터셋을 만들고 모델을 불러와서 학습시키고 결과보기\n3. 제출해보기","metadata":{}},{"cell_type":"markdown","source":"1. test 이미지가 3개밖에 없어. -> 어떻게 해결?","metadata":{}}]}