{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h1 style='background:#2cab6c; border:0; color:white'><center>Importing Libraries</center></h1>","metadata":{}},{"cell_type":"code","source":"import json\n\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nfrom pathlib import Path\nfrom sklearn.model_selection import StratifiedKFold","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:32:42.299558Z","iopub.execute_input":"2022-01-12T10:32:42.300234Z","iopub.status.idle":"2022-01-12T10:32:43.397163Z","shell.execute_reply.started":"2022-01-12T10:32:42.300098Z","shell.execute_reply":"2022-01-12T10:32:43.396178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style='background:#2cab6c; border:0; color:white'><center>Paths, files</center></h1>","metadata":{}},{"cell_type":"code","source":"# Paths to the base directories/files of the dataset\nbase_dir = Path('/kaggle/input/cassava-leaf-disease-classification')\ntrain_df = pd.read_csv(f'{base_dir}/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:32:43.769714Z","iopub.execute_input":"2022-01-12T10:32:43.769990Z","iopub.status.idle":"2022-01-12T10:32:43.804408Z","shell.execute_reply.started":"2022-01-12T10:32:43.769957Z","shell.execute_reply":"2022-01-12T10:32:43.803692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read the JSON file and write its contents to a variable\nwith open(f'{base_dir}/label_num_to_disease_map.json') as f:\n    class_names = json.loads(f.read())\nf.close()","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:32:44.146430Z","iopub.execute_input":"2022-01-12T10:32:44.147069Z","iopub.status.idle":"2022-01-12T10:32:44.157081Z","shell.execute_reply.started":"2022-01-12T10:32:44.147035Z","shell.execute_reply":"2022-01-12T10:32:44.156260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style='background:#2cab6c; border:0; color:white'><center>Data Preprocessing</center></h1>","metadata":{}},{"cell_type":"code","source":"# Let's show the names of the classes \nclass_names","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:32:45.330189Z","iopub.execute_input":"2022-01-12T10:32:45.330471Z","iopub.status.idle":"2022-01-12T10:32:45.339221Z","shell.execute_reply.started":"2022-01-12T10:32:45.330438Z","shell.execute_reply":"2022-01-12T10:32:45.338620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's check our DataFrame with training data\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:32:46.182565Z","iopub.execute_input":"2022-01-12T10:32:46.183066Z","iopub.status.idle":"2022-01-12T10:32:46.200690Z","shell.execute_reply.started":"2022-01-12T10:32:46.183016Z","shell.execute_reply":"2022-01-12T10:32:46.200072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add a new column with appropriate class names for labels\ntrain_df['label_name'] = train_df['label'].apply(lambda x: class_names[str(x)])","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:32:46.958754Z","iopub.execute_input":"2022-01-12T10:32:46.959332Z","iopub.status.idle":"2022-01-12T10:32:46.979314Z","shell.execute_reply.started":"2022-01-12T10:32:46.959282Z","shell.execute_reply":"2022-01-12T10:32:46.978454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check the result\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:32:47.306505Z","iopub.execute_input":"2022-01-12T10:32:47.306805Z","iopub.status.idle":"2022-01-12T10:32:47.318275Z","shell.execute_reply.started":"2022-01-12T10:32:47.306772Z","shell.execute_reply":"2022-01-12T10:32:47.317607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's look at the distribution of classes\ntrain_df.groupby('label')['image_id'].count().plot(kind='bar', title='Class distribution');","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:32:48.164035Z","iopub.execute_input":"2022-01-12T10:32:48.164630Z","iopub.status.idle":"2022-01-12T10:32:48.422600Z","shell.execute_reply.started":"2022-01-12T10:32:48.164590Z","shell.execute_reply":"2022-01-12T10:32:48.421737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style='background:#2cab6c; border:0; color:white'><center>Stratified K-Folds</center></h1>","metadata":{}},{"cell_type":"code","source":"# Let's use StratifiedKFold to split the dataset into 4 parts\nsk = StratifiedKFold(n_splits=4, random_state=42, shuffle=True)\n\nfor fold, (train, val) in enumerate(sk.split(train_df, train_df.label)):\n    train_df.loc[val, 'fold'] = fold","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:32:53.174220Z","iopub.execute_input":"2022-01-12T10:32:53.175122Z","iopub.status.idle":"2022-01-12T10:32:53.195151Z","shell.execute_reply.started":"2022-01-12T10:32:53.175060Z","shell.execute_reply":"2022-01-12T10:32:53.194136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Converting from float type to int\ntrain_df.fold = train_df.fold.astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:32:53.553926Z","iopub.execute_input":"2022-01-12T10:32:53.554226Z","iopub.status.idle":"2022-01-12T10:32:53.560913Z","shell.execute_reply.started":"2022-01-12T10:32:53.554193Z","shell.execute_reply":"2022-01-12T10:32:53.559992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check the result\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:32:55.918215Z","iopub.execute_input":"2022-01-12T10:32:55.918561Z","iopub.status.idle":"2022-01-12T10:32:55.936641Z","shell.execute_reply.started":"2022-01-12T10:32:55.918525Z","shell.execute_reply":"2022-01-12T10:32:55.935595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have successfully applied the dataset splitting into 4 parts using sklearn <a href='https://scikit-learn.org/stable/modules/generated/sklearn.model_selection.StratifiedKFold.html'>sklearn StratifiedKFold</a>, let's save our updated dataset for later training the model.","metadata":{}},{"cell_type":"markdown","source":"<h1 style='background:#2cab6c; border:0; color:white'><center>Save Dataset</center></h1>","metadata":{}},{"cell_type":"code","source":"# Save updated dataset\ntrain_df.to_csv('/kaggle/working/train_splitted.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:32:59.226904Z","iopub.execute_input":"2022-01-12T10:32:59.227235Z","iopub.status.idle":"2022-01-12T10:32:59.314795Z","shell.execute_reply.started":"2022-01-12T10:32:59.227199Z","shell.execute_reply":"2022-01-12T10:32:59.313827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}