{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport fastai\nfrom sklearn.model_selection import StratifiedKFold\nimport torchvision.models as torch_models\nimport dill\nfrom fastai.vision.all import *\n","metadata":{"execution":{"iopub.status.busy":"2021-05-22T19:48:48.890103Z","iopub.execute_input":"2021-05-22T19:48:48.890439Z","iopub.status.idle":"2021-05-22T19:48:51.611401Z","shell.execute_reply.started":"2021-05-22T19:48:48.890409Z","shell.execute_reply":"2021-05-22T19:48:51.610525Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fastai.__version__","metadata":{"execution":{"iopub.status.busy":"2021-05-22T19:48:51.613018Z","iopub.execute_input":"2021-05-22T19:48:51.613327Z","iopub.status.idle":"2021-05-22T19:48:51.625866Z","shell.execute_reply.started":"2021-05-22T19:48:51.613293Z","shell.execute_reply":"2021-05-22T19:48:51.625057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# About\nThis notebook trains a simple model for the [Hotel-ID to Combat Human Trafficking 2021 - FGVC8](https://www.kaggle.com/c/hotel-id-2021-fgvc8/overview) competition.\nThe notebook uses the downsized version of the competiton train images. The max height or width is 512 pix. The score can be improved by running more epochs.","metadata":{}},{"cell_type":"code","source":"kaggle_path = '../input/hotel-id-2021-fgvc8/'\ntrain_img_path = '../input/fgvc8hoteltrain512/'","metadata":{"execution":{"iopub.status.busy":"2021-05-22T19:56:42.009697Z","iopub.execute_input":"2021-05-22T19:56:42.010044Z","iopub.status.idle":"2021-05-22T19:56:42.016669Z","shell.execute_reply.started":"2021-05-22T19:56:42.010013Z","shell.execute_reply":"2021-05-22T19:56:42.015884Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Set the DEBUG flag = `True` to construct a validation set and to restrict the number of train data. Set the flag = `False` to train on the entire set without validation set. ","metadata":{}},{"cell_type":"code","source":"DEBUG = False","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load train info and define validation set\nLoad train image descriptions and add path information of downsized images.","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(kaggle_path+'train.csv').sample(frac=1, random_state = 2021).reset_index(drop = True)\n\n# add image path\ntrain['image'] = train_img_path + train.chain.astype('str') + '/' + train.image\n","metadata":{"execution":{"iopub.status.busy":"2021-05-22T19:56:42.509806Z","iopub.execute_input":"2021-05-22T19:56:42.510136Z","iopub.status.idle":"2021-05-22T19:56:42.759062Z","shell.execute_reply.started":"2021-05-22T19:56:42.510106Z","shell.execute_reply":"2021-05-22T19:56:42.758119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"If in DEBUG mode limit data and construct validation set.","metadata":{}},{"cell_type":"code","source":"if DEBUG:\n    ## Map chains to 6 buckets.\n    train['h_c'] = train['hotel_id'].astype('str')+' c'+train['chain'].astype('str')\n    train['chain_bucket']=train['chain'].map({0:1,\n                                               6:2,\n                                               5:3,\n                                               90:4,\n                                               3:4,\n                                               89:4,\n                                               87:5,\n                                               4:5,\n                                               2:5,\n                                               88:5,\n                                               9:6,\n                                               82:6,\n                                               78:6}).fillna(0)\n    \n    # Do a stratified KFold to construct a proper validation set.\n    skf = StratifiedKFold(n_splits = 5, random_state = None, shuffle = False)\n\n    train.kfold = -1\n\n    for f, (t,v) in enumerate(skf.split(X = train, y = train.hotel_id.values)):\n        train.loc[v, 'kfold'] = f\n\n    train.groupby('kfold')['hotel_id'].count()","metadata":{"execution":{"iopub.status.busy":"2021-05-22T19:56:43.488122Z","iopub.execute_input":"2021-05-22T19:56:43.48844Z","iopub.status.idle":"2021-05-22T19:56:43.494667Z","shell.execute_reply.started":"2021-05-22T19:56:43.488406Z","shell.execute_reply":"2021-05-22T19:56:43.493754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['is_demo_valid'] = False\n\nif DEBUG:\n    # limit data to one bucket\n    train=train[train.chain_bucket==6]\n    # set fold 0 as validation set\n    train.loc[train['kfold'] == 0, 'is_demo_valid'] = True\n\ntrain[(train.is_demo_valid)]","metadata":{"execution":{"iopub.status.busy":"2021-05-22T20:12:23.561493Z","iopub.execute_input":"2021-05-22T20:12:23.56191Z","iopub.status.idle":"2021-05-22T20:12:23.576298Z","shell.execute_reply.started":"2021-05-22T20:12:23.561873Z","shell.execute_reply":"2021-05-22T20:12:23.575218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataloader and augmentation\nSetting pad_mode to `reflection` gave a significant boost.","metadata":{}},{"cell_type":"code","source":"dls = ImageDataLoaders.from_df(df = train[['image', 'hotel_id', 'is_demo_valid']], path = '.',folder = '.', valid_col = 'is_demo_valid',\n                                item_tfms=Resize(448, method='pad', pad_mode='reflection')\n                               ,batch_tfms=aug_transforms(size=224)\n                               ,bs=32)","metadata":{"execution":{"iopub.status.busy":"2021-05-22T20:15:31.231525Z","iopub.execute_input":"2021-05-22T20:15:31.23189Z","iopub.status.idle":"2021-05-22T20:15:42.692236Z","shell.execute_reply.started":"2021-05-22T20:15:31.231852Z","shell.execute_reply":"2021-05-22T20:15:42.691451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check some augmented sample images.","metadata":{}},{"cell_type":"code","source":"dls.show_batch() ","metadata":{"execution":{"iopub.status.busy":"2021-05-22T20:15:42.69389Z","iopub.execute_input":"2021-05-22T20:15:42.694206Z","iopub.status.idle":"2021-05-22T20:15:43.975887Z","shell.execute_reply.started":"2021-05-22T20:15:42.69417Z","shell.execute_reply":"2021-05-22T20:15:43.974471Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training\nDensenet161 worked best in my experiment but training is slow. Resnet101 was a good faster alternative but results were slightly worse. CrossEntropy-Loss (CE) worked better for me than label smoothing and a bit better than FocalLoss (FL).\n\nQHAdam gave a significant boost.","metadata":{}},{"cell_type":"code","source":"learn = cnn_learner(dls, densenet161, metrics=[accuracy, top_k_accuracy], opt_func=QHAdam).to_fp16()","metadata":{"execution":{"iopub.status.busy":"2021-05-22T20:21:50.727554Z","iopub.execute_input":"2021-05-22T20:21:50.727944Z","iopub.status.idle":"2021-05-22T20:21:51.754171Z","shell.execute_reply.started":"2021-05-22T20:21:50.72791Z","shell.execute_reply":"2021-05-22T20:21:51.752125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%time\n\nlearn.fine_tune(12, 0.005, freeze_epochs=3)","metadata":{"execution":{"iopub.status.busy":"2021-05-22T19:57:02.820818Z","iopub.execute_input":"2021-05-22T19:57:02.821147Z","iopub.status.idle":"2021-05-22T19:57:22.134772Z","shell.execute_reply.started":"2021-05-22T19:57:02.821114Z","shell.execute_reply":"2021-05-22T19:57:22.131216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.export(f'export_dn161_kaggle_notebook.pkl', pickle_module=dill)","metadata":{"execution":{"iopub.status.busy":"2021-05-22T19:55:45.859194Z","iopub.execute_input":"2021-05-22T19:55:45.85953Z","iopub.status.idle":"2021-05-22T19:55:46.521438Z","shell.execute_reply.started":"2021-05-22T19:55:45.859499Z","shell.execute_reply":"2021-05-22T19:55:46.520647Z"},"trusted":true},"execution_count":null,"outputs":[]}]}