{"cells":[{"metadata":{"_uuid":"6ebfc8c5e505e245fc0e38d01ff56402edfe44d0"},"cell_type":"markdown","source":"### Resize iWildCam 2019 and iNat Idaho data\n\nTrain models on complete image data (iWildCam 2019 + supplemental iNat Idaho)\n\n* iWildCam2019: https://www.kaggle.com/c/iwildcam-2019-fgvc6/data\n* Supplemental iNat Idaho data: https://github.com/visipedia/iwildcam_comp\n* Original idea of resizing imgs from https://www.kaggle.com/xhlulu/reducing-image-sizes-to-32x32\n"},{"metadata":{},"cell_type":"markdown","source":"**How to use this kernel**\n* Fork kernel\n* Modify *USR_IMG_SIZE* to the desired image dimensions \n* Commit kernel\n* In a new kernel, click *+ add dataset* -> *Kernel output files* -> *Your Work* -> *iWildCam 2019 + iNat Idaho*\n\nYou will now have access to x_train, y_train and y_test in /kaggle/input to start building models!"},{"metadata":{"trusted":true},"cell_type":"code","source":"# modify as needed\nUSR_IMG_SIZE = 32 # random sample images will be saved as 32x32","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Implementation\n\nBelow implementation can remain untouched unless you want to add extra preprocessing to images.\n\n*USR_IMG_SIZE* is the only main user modifications"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import cv2\nimport numpy as np \nimport pandas as pd\nimport json\nfrom sklearn.model_selection import StratifiedShuffleSplit","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fcec4faf374aa01d5a92ab3d465daf1c8e396cb8"},"cell_type":"markdown","source":"**iWildCam 2019 (Kaggle input data)**"},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/sample_submission.csv')\ntest_df['file_path'] = test_df['Id'].apply(lambda x: f'/kaggle/input/test_images/{x}.jpg')\ntest_df.drop(columns=['Predicted','Id'],inplace=True)\ntest_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"42d2ea950ed12f6f896c20c8b23435b36a67f757"},"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/train.csv')\ntrain_df['file_path'] = train_df['id'].apply(lambda x: f'/kaggle/input/train_images/{x}.jpg')\ntrain_df = train_df[['file_path','category_id']]\ntrain_df['is_supp'] = False\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Supplemental iNat Idaho**"},{"metadata":{},"cell_type":"markdown","source":"Extract"},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\n!wget https://wildcamdrop.blob.core.windows.net/wildcamdropcontainer/iWildCam_2019_iNat_Idaho.tar.gz -P /kaggle/supplemental","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\n!tar xvf /kaggle/supplemental/iWildCam_2019_iNat_Idaho.tar.gz -C /kaggle/supplemental/ >> /kaggle/working/log.txt\n\n# images now in /kaggle/supplemental/iWildCam_2019_iNat_Idaho/train_val2017/ and /kaggle/supplemental/iWildCam_2019_iNat_Idaho/train_val2018/\n\n!rm /kaggle/supplemental/iWildCam_2019_iNat_Idaho.tar.gz\n!rm /kaggle/working/log.txt","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Get a dataframe"},{"metadata":{"trusted":true},"cell_type":"code","source":"with open('/kaggle/supplemental/iWildCam_2019_iNat_Idaho/iWildCam_2019_iNat_Idaho.json') as json_data:\n    supp_data = json.load(json_data)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_df = pd.DataFrame.from_dict(supp_data['images'])\nimage_df = image_df[['file_name','id']]\nimage_df.rename(columns={'id':'image_id'}, inplace=True)\nimage_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"annotation_df = pd.DataFrame.from_dict(supp_data['annotations'])\nannotation_df = annotation_df[['category_id','image_id']]\nannotation_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"supp_df = pd.merge(image_df, annotation_df, on='image_id', how='inner')\n\nsupp_df['is_supp'] = True\nsupp_df['file_path'] = supp_df['file_name'].apply(lambda x: f'/kaggle/supplemental/iWildCam_2019_iNat_Idaho/{x}')\nsupp_df.drop(columns=['image_id','file_name'],inplace=True)\n\nsupp_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"** Combine kaggle data with supplemental **"},{"metadata":{"trusted":true},"cell_type":"code","source":"complete_df = pd.concat([train_df,supp_df], ignore_index = True)\ncomplete_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Resize all images**"},{"metadata":{"trusted":true,"_uuid":"b8ca9a0a8e8ec91f3855cb7d15bc8859a1b70c8e"},"cell_type":"code","source":"# Image transformations and more data prep can be done in a child kernel\ndef resize(image_path, desired_size):\n    img = cv2.imread(image_path)\n    return cv2.resize(img, (desired_size,)*2).astype('uint8')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fb1d093a03ac5af72b9eeaa696bd877e73e89481"},"cell_type":"code","source":"%%time\ntrain_resized_imgs = [resize(file_path,USR_IMG_SIZE) for file_path in complete_df['file_path']]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\ntest_resized_imgs = [resize(file_path,USR_IMG_SIZE) for file_path in test_df['file_path']]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e61354b00dd73b27ea20d04d1e0d039c08cb169b"},"cell_type":"code","source":"X_train = np.stack(train_resized_imgs)\nX_test = np.stack(test_resized_imgs)\ny_train = pd.get_dummies(complete_df['category_id']).values\n\nprint(X_train.shape)\nprint(X_test.shape)\nprint(y_train.shape)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5bc69bfab2ae1440089e93491a06721c77e7db17"},"cell_type":"markdown","source":"**Saving**"},{"metadata":{"trusted":true,"_uuid":"ef26b6ee0a6ab8b3ef6c1c64cdef4b31dd657426"},"cell_type":"code","source":"# No need to save the IDs of X_test, since they are in the same order as the ID column in sample_submission.csv\nnp.save('X_train.npy', X_train)\nnp.save('X_test.npy', X_test)\nnp.save('y_train.npy', y_train)\n\n# save the complete_df for future reference\ncomplete_df.to_pickle(\"/kaggle/working/complete_df.pkl\")","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}