{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Disclaimer\nThis is supposed to be a baseline notebook for those starting with image classification. Please don't expect SoTA results :D"},{"metadata":{},"cell_type":"markdown","source":"# Creating Training DF\n\nWe will create a dataframe containing the training image ID, label, and path"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train = pd.read_csv(r'/kaggle/input/cassava-leaf-disease-classification/train.csv')\ndf_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import json \n\nwith open(r'/kaggle/input/cassava-leaf-disease-classification/label_num_to_disease_map.json') as json_file: \n    label_map = json.load(json_file) \n\nlabel_map = {int(k):v for k,v in label_map.items()}\nlabel_map","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train['disease'] = df_train['label'].map(label_map)\ndf_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import glob\ntrain_path = glob.glob(r'/kaggle/input/cassava-leaf-disease-classification/train_images/*.jpg')\ntrain_path.sort()\nprint(len(train_path))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train['path'] = train_path\ndf_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train.groupby(['disease']).size().plot(kind='bar')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"This is a very imbalanced dataset"},{"metadata":{},"cell_type":"markdown","source":"# First look at the Images"},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom PIL import Image","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img = Image.open(df_train.path[0])\nimg","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img.size","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Each image is of size 800x600 pixels"},{"metadata":{},"cell_type":"markdown","source":"# Taking a subsample of the set to prevent RAM overflow"},{"metadata":{"trusted":true},"cell_type":"code","source":"from tqdm.notebook import tqdm #to monitor progress\nnp.random.seed(42) #to get reproducible results","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_samp = pd.DataFrame()\n\ndf_samp = df_samp.append(df_train.sample(2000), ignore_index=True)\n\ndf_samp.groupby(by='disease').count()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"#### To take equal number of samples from each category\ndf_samp = pd.DataFrame()\n\nfor label in tqdm(df_train.label.unique()):\n    df_samp = df_samp.append(df_train[df_train.label==label].sample(100), \n                           ignore_index=True)\n\ndf_samp.groupby(by='disease').count()"},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.utils import shuffle\n\ndf_samp = shuffle(df_samp).reset_index(drop=True) #shuffling the dataframe","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Train-Test split"},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX = df_samp.drop(columns=['label'])\ny = df_samp.label\n\nX_train, X_valid, y_train, y_valid = train_test_split(X, y, test_size=0.3, stratify=y)\n\nprint(X_train.shape)\nprint(len(y_train))\nprint(X_valid.shape)\nprint(len(y_valid))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Reducing image size and saving images as arrays"},{"metadata":{"trusted":true},"cell_type":"code","source":"compressed_size = (200,150)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_array = np.array([np.asarray(Image.open(path).resize(compressed_size, Image.ANTIALIAS)) for path in tqdm(X_train.path)])\nvalid_array = np.array([np.asarray(Image.open(path).resize(compressed_size, Image.ANTIALIAS)) for path in tqdm(X_valid.path)])\n\nprint(train_array.shape)\nprint(valid_array.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(20,12))\n\nfor i, img in tqdm(enumerate(train_array[:5])):\n    plt.subplot(1, 5, i+1)\n    plt.xticks([])\n    plt.yticks([])\n    plt.grid(False)\n    plt.imshow(img)\n    plt.title(X_train.disease.iloc[i])\n    plt.xlabel(X_train.image_id.iloc[i])\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(20,12))\n\nfor i, img in tqdm(enumerate(valid_array[:5])):\n    plt.subplot(1, 5, i+1)\n    plt.xticks([])\n    plt.yticks([])\n    plt.grid(False)\n    plt.imshow(img)\n    plt.title(X_valid.disease.iloc[i])\n    plt.xlabel(X_valid.image_id.iloc[i])\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f'Length of the training array is {len(train_array)}')\nprint(f'Shape of the training array is {train_array.shape}')\nprint(f'Shape of each training image array is {train_array[0].shape}')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"The training input array currently has 700 rows, with each row containing the image array.\n\nSince a LogReg model accepts each input as a row array, we need to reshape our image arrays to a single row having `150`x`200`x`3` values.\n\nThe shape of the input array will then become `(700, 150*200*3)`"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_array.resize(len(train_array), train_array.shape[1]*train_array.shape[2]*train_array.shape[3])\n\nprint(f'New length of the training array is {len(train_array)}')\nprint(f'New shape of the training array is {train_array.shape}')\nprint(f'New shape of each training image array is {train_array[0].shape}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"valid_array.resize(len(valid_array), valid_array.shape[1]*valid_array.shape[2]*valid_array.shape[3])\n\nprint(f'New length of the validation array is {len(valid_array)}')\nprint(f'New shape of the validation array is {valid_array.shape}')\nprint(f'New shape of each validation image array is {valid_array[0].shape}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"# Model Instantiation and Training"},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nlr = LogisticRegression(class_weight='balanced', verbose=5, n_jobs=-1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lr.fit(train_array, y_train)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Prediction and Evaluation"},{"metadata":{"trusted":true},"cell_type":"code","source":"preds = lr.predict(valid_array)\npreds","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, classification_report, f1_score\nimport seaborn as sns\n\nlabel = sorted(y_valid.unique())\nsns.heatmap(confusion_matrix(y_valid, preds), annot=True, square=True, fmt='g', \n            xticklabels=label, yticklabels=label, cbar=False)\n\nplt.title('Confusion matrix')\nplt.xlabel('Predicted')\nplt.ylabel('Actual')\nplt.show()\n\nprint(classification_report(y_valid, preds))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"An accuracy of 54% is definitely not bad for a simple logreg model trained on just 4.3% of the training set, and with the training images compressed by a factor of 8 (random guessing would give an accuracy of 20%).\n\nLets see how we can improve this accuracy in further versions of this notebook.\nWould love you have your suggestions too!"},{"metadata":{},"cell_type":"markdown","source":"# Submission"},{"metadata":{"trusted":true},"cell_type":"code","source":"test_path = glob.glob(r'/kaggle/input/cassava-leaf-disease-classification/test_images/*.jpg')\ntest_path.sort()\nprint(len(test_path))\ntest_path","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"Image.open(test_path[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_array = np.array([np.asarray(Image.open(path).resize(compressed_size, Image.ANTIALIAS)) for path in tqdm(test_path)])\ntest_array.resize(len(test_array), test_array.shape[1]*test_array.shape[2]*test_array.shape[3])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission = lr.predict(test_array)\nsubmission","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_df = pd.DataFrame({'image_id':[path.split('/')[-1] for path in test_path], \n                              'label':submission})\nsubmission_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_df.to_csv('/kaggle/working/submission.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}