{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"> notebook Sample code learning"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# import define\n# os:운영체제에서 제공되는 여러 기능을 파이썬에서 수행할 수있게 함\n# json: Json 데이터를 처리하기 위해서 사용되는 파이썬 내장 모듈\n# numpy(np): 고성능의 수치계산. 대규모 다차원 배열과 행렬 연산에 필요한 다양한 함수 제공\n# pandas(pd): 데이터분석 라이브러리로 행과열로 이루어진 데이터 객체를 만들어 다룸.\n# seaborn(sn): 데이터프레임으로 다양한 통계 지표를 낼수 있는 시각화 차트를 제공\n# matplotlib.pyolot(plt): 데이터를 차트나 플롯(Plot)으로 그려주는 데이터 시각화\n#                       라인플롯, 바 차트, 파이차트, 히스토그램, BoxPlot 등\n# cv2: openCV 컴퓨터 비전 라이브러리로 객체 얼굴 행동 모션 추적 등의 응용에서 사용\n# albumentations:  이미지 형태 변환\n# \nimport os\nimport json\n\nimport numpy as np\nimport pandas as pd\nimport seaborn as sn\nimport matplotlib.pyplot as plt\nimport cv2\nimport albumentations as A\nfrom sklearn import metrics as sk_metrics","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Kaggle base diretory def\n\nBASE_DIR = \"../input/cassava-leaf-disease-classification/\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#os.path.join: 경로를 병합하여 새 경로를 생성\nwith open(os.path.join(BASE_DIR, \"label_num_to_disease_map.json\")) as file:\n    \n    #file 데이터를 json.load()를 사용하여 객체로 읽어오기 -> map_classes\n    map_classes = json.loads(file.read())\n    \n    # for var in 문자열(튜플, 문자열)\n    map_classes = {int(k) : v for k, v in map_classes.items()}\n\n    #json.dumpsL 객체를 json 데이터로 쓰기, 직렬화 ,인코딩\nprint(json.dumps(map_classes, indent=4))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#os.listdir 해당 디렉토리에 있는 파일들의 리스트를 구함\ninput_files = os.listdir(os.path.join(BASE_DIR, \"train_images\"))\n\n#len() 리스트에 들어 있는 원소의 갯수\nprint(f\"Number of train images: {len(input_files)}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img_shapes = {}\nfor image_name in os.listdir(os.path.join(BASE_DIR, \"train_images\"))[:300]:\n    \n    #cv2.imread: 함수를 이용하여 이미지 파일을 읽음\n    image = cv2.imread(os.path.join(BASE_DIR, \"train_images\", image_name))\n    #shape image 변수의 이미지 shape를 확인(height, width, channl)\n    img_shapes[image.shape] = img_shapes.get(image.shape, 0) + 1\n\nprint(img_shapes)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#pandas library의 read_csv() 함수를 이용하여 train.csv 함수를 읽음\ndf_train = pd.read_csv(os.path.join(BASE_DIR, \"train.csv\"))\n\n#map은 리스트의 요소를 지정된 함수로 처리해주는 함수\n# list(map(함수, 리스트))\n# https://dojang.io/pluginfile.php/13699/mod_page/content/3/022019.png\ndf_train[\"class_name\"] = df_train[\"label\"].map(map_classes)\n\ndf_train","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":""},{"metadata":{"trusted":true},"cell_type":"code","source":"#figure() 함수는 matplotlib에서 figure를 만들고 편집할 수 있게 만들어주는 함수\n# ex1) fig=figure(num=figure number, figsize = (x,y))\n# ex2) fig=figure(figure number, (x,y))\n# ex3) fig=girure()\n plt.figure(figsize=(8, 4))\n\n#countplot 카테고리별 데이터 양 확인 \nsn.countplot(y=\"class_name\", data=df_train);","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(4, 10))\n\n#countplot 카테고리별 데이터 양 확인 \nsn.countplot(x=\"class_name\", data=df_train);","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# define visualize_batch \ndef visualize_batch(image_ids, labels):\n    plt.figure(figsize=(16, 12))\n    \n    #enumerate :반복문 사용시 몇번째 반복문인지 확인이 필요할 수 있음\n    #인덱스 번호와 컬렉션의 원소를 tuple 형태로 반환\n    \n    for ind, (image_id, label) in enumerate(zip(image_ids, labels)):\n        \n        #subplot 한 화면에 여러 그래프를 나눠서 그려주는 기능\n        plt.subplot(3, 3, ind + 1)\n        image = cv2.imread(os.path.join(BASE_DIR, \"train_images\", image_id))\n        \n        #open cv에는 color-space를 변환방법이 약 150가지 있음\n        #cv2.cvtColor( image_src , color_code )\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\n        #imshow() 이미지를 모니터에 출력\n        plt.imshow(image)\n        plt.title(f\"Class: {label}\", fontsize=12)\n        # x,y축의 범위를 설정할수 있게 하는것과 동시에 여러 옵션을 설정할 수 있는 함수\n        plt.axis(\"off\")\n    \n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#smaple(컬랙션, 샘플수) 지정된 컬렉션으로 부터 샘플수 만큼 추출\ntmp_df = df_train.sample(9)\n\nimage_ids = tmp_df[\"image_id\"].values\nlabels = tmp_df[\"class_name\"].values\n\nvisualize_batch(image_ids, labels)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#smaple(컬랙션, 샘플수) 지정된 컬렉션으로 부터 샘플수 만큼 추출\ntmp_df = df_train.sample(6)\n\nimage_ids = tmp_df[\"image_id\"].values\nlabels = tmp_df[\"class_name\"].values\n\nvisualize_batch(image_ids, labels)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* 1. 0 - CBB - Cassava Bacterial Blight(카사바 박테리아 병균)\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"tmp_df = df_train[df_train[\"label\"] == 0]\nprint(f\"Total train images for class 0: {tmp_df.shape[0]}\")\n\ntmp_df = tmp_df.sample(9)\nimage_ids = tmp_df[\"image_id\"].values\nlabels = tmp_df[\"label\"].values\n\nvisualize_batch(image_ids, labels)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* 1 - CBSD - Cassava Brown Streak Disease\n* "},{"metadata":{"trusted":true},"cell_type":"code","source":"tmp_df = df_train[df_train[\"label\"] == 1]\nprint(f\"Total train images for class 1: {tmp_df.shape[0]}\")\n\ntmp_df = tmp_df.sample(9)\nimage_ids = tmp_df[\"image_id\"].values\nlabels = tmp_df[\"label\"].values\n\nvisualize_batch(image_ids, labels)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* 2 - CGM - Cassava Green Mottle\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"tmp_df = df_train[df_train[\"label\"] == 2]\nprint(f\"Total train images for class 2: {tmp_df.shape[0]}\")\n\ntmp_df = tmp_df.sample(9)\nimage_ids = tmp_df[\"image_id\"].values\nlabels = tmp_df[\"label\"].values\n\nvisualize_batch(image_ids, labels)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* 3 - CMD - Cassava Mosaic Disease\n* "},{"metadata":{"trusted":true},"cell_type":"code","source":"tmp_df = df_train[df_train[\"label\"] == 3]\nprint(f\"Total train images for class 3: {tmp_df.shape[0]}\")\n\ntmp_df = tmp_df.sample(9)\nimage_ids = tmp_df[\"image_id\"].values\nlabels = tmp_df[\"label\"].values\n\nvisualize_batch(image_ids, labels)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* 4 - Healthy\n* "},{"metadata":{"trusted":true},"cell_type":"code","source":"tmp_df = df_train[df_train[\"label\"] == 4]\nprint(f\"Total train images for class 4: {tmp_df.shape[0]}\")\n\ntmp_df = tmp_df.sample(9)\nimage_ids = tmp_df[\"image_id\"].values\nlabels = tmp_df[\"label\"].values\n\nvisualize_batch(image_ids, labels)\n","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}