{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nfrom sklearn.model_selection import StratifiedKFold\n\nimport cassava_utils as utils","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Exploratory Data Analyses (EDA)"},{"metadata":{"trusted":true},"cell_type":"code","source":"folderpath = \"../input/cassava-leaf-disease-classification\"\nlabel2name_json = \"../input/cassava-leaf-disease-classification/label_num_to_disease_map.json\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# check how many files are in the folders\ntrain_files = os.listdir(os.path.join(folderpath, \"train_images\"))\ntest_files = os.listdir(os.path.join(folderpath, \"test_images\"))\nprint(\"images in training set:\", str(len(train_files)))\nprint(\"images in test set:\", str(len(test_files)))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"There is only one sample in the test set. There will be added more test images in submitting process."},{"metadata":{"trusted":true},"cell_type":"code","source":"# check if all files end with \".jpeg\"\nprint(\"all training files end with '.jpg':\", \n      str(sum([f.endswith('.jpg') for f in train_files]) == len(train_files)))\nprint(\"all test files end with '.jpg':\", \n      str(sum([f.endswith('.jpg') for f in test_files]) == len(test_files)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# check for duplicates\nprint(\"no duplicates in train_files:\",len(set(train_files))==len(train_files))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.read_csv(os.path.join(folderpath, \"train.csv\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# check if there is a label for all files in train_images folder\nprint(\"All files are in df:\", sorted(train_files) == sorted(list(df.image_id)))\nprint(df.label.unique())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"All files are jpgs, unique and labeled with 0 to 4"},{"metadata":{"trusted":true},"cell_type":"code","source":"# check class distribution in train files\nclass_dist = df.label.value_counts(normalize=True)\nprint(class_dist)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"base_color = sns.color_palette()[0]\nsns.countplot(data=df, x=\"label\", color=base_color)\nplt.ylabel(\"samples\")\nplt.xlabel(\"\")\nplt.xticks([0, 1, 2, 3, 4], [\"CBB\", \"CBSD\", \"CGM\", \"CMD\", \"healthy\"])\n# get counts\nn_points = df.shape[0]\ncounts = df.label.value_counts()\n#loop through locations and label pairs of diagramm (xticks)\nlocs,_ = plt.xticks()\nfor loc, label_ in enumerate(locs):\n    count = counts[label_]\n    percentage = \"{:0.1f}%\".format(100*count/n_points)\n    plt.text(loc, count-800, percentage, ha=\"center\", color=\"w\")\n    ","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* label 3 (disease CMD) is overrepresented (61%)\n* only 12% of the files show healthy plants (label 4)"},{"metadata":{"trusted":true},"cell_type":"code","source":"# # check image sizes\n# image_sizes = []\n\n# for img_id in train_files:\n#     img = Image.open(os.path.join(folderpath, \"train_images\", img_id))\n#     image_sizes.append(img.size)\n    \n# print(\"Image sizes of train_files:\", set(image_sizes))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"All image files are of pixel size 800x600 (width x height)."},{"metadata":{},"cell_type":"markdown","source":"### Display Images"},{"metadata":{"trusted":true},"cell_type":"code","source":"# split image_ids according to their label\nfiles_label_0 = df.loc[df.label==0].image_id.values\nfiles_label_1 = df.loc[df.label==1].image_id.values\nfiles_label_2 = df.loc[df.label==2].image_id.values\nfiles_label_3 = df.loc[df.label==3].image_id.values\nfiles_label_4 = df.loc[df.label==4].image_id.values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# display images of label 0\nutils.display_images(files_label_0, df, os.path.join(folderpath, \"train_images\"), label2name_json)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# display images of label 1\nutils.display_images(files_label_1, df, os.path.join(folderpath, \"train_images\"), label2name_json)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# display images of label 2\nutils.display_images(files_label_2, df, os.path.join(folderpath, \"train_images\"), label2name_json)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# display images of label 3\nutils.display_images(files_label_3, df, os.path.join(folderpath, \"train_images\"), label2name_json)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# display images of label 4\nutils.display_images(files_label_4, df, os.path.join(folderpath, \"train_images\"), label2name_json)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Data Preprocessing"},{"metadata":{},"cell_type":"markdown","source":"## Train/Val Split"},{"metadata":{"trusted":true},"cell_type":"code","source":"### Version 1: using just one train/val split with sklearn.train_test_split\n\n# train_df, val_df = train_test_split(df, test_size=0.2, \n#                                     random_state=seed)\n# #save dataframes as csv\n# train_df.to_csv(\"train_df.csv\", index=False)\n# val_df.to_csv(\"val_df.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"### Version 2: using k-Fold cross validation with stratified k-folds\n# from \"Approaching almost all machine learning problems\" by Abhishek Thaku\n\n# we create a new column called kfold and fill it with -1\ndf[\"kfold\"] = -1\n# the next step is to randomize the rows of the data\ndf = df.sample(frac=1, random_state=seed).reset_index(drop=True)\n # fetch targets\ny = df.label.values\n# initiate the kfold class from model_selection module\nkf = StratifiedKFold(n_splits=5)\n# fill the new kfold column\nfor f, (t_, v_) in enumerate(kf.split(X=df, y=y)):\n    df.loc[v_, 'kfold'] = f\n# save the new csv with kfold column\ndf.to_csv(\"train_folds.csv\", index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}