{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This is required due to this error: https://www.kaggle.com/product-feedback/279990\n!pip install --user torch==1.9.0 > /dev/null 2>&1","metadata":{"execution":{"iopub.status.busy":"2022-02-15T11:47:01.705554Z","iopub.execute_input":"2022-02-15T11:47:01.706296Z","iopub.status.idle":"2022-02-15T11:48:08.269119Z","shell.execute_reply.started":"2022-02-15T11:47:01.706199Z","shell.execute_reply":"2022-02-15T11:48:08.267808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from fastai.vision.all import *","metadata":{"execution":{"iopub.status.busy":"2022-02-15T11:48:08.273880Z","iopub.execute_input":"2022-02-15T11:48:08.274181Z","iopub.status.idle":"2022-02-15T11:48:10.867932Z","shell.execute_reply.started":"2022-02-15T11:48:08.274144Z","shell.execute_reply":"2022-02-15T11:48:10.866918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Intro\n\nI'm going to train a model to classify whether the example came from the train or test set. If the distribution in the train and test set is exactly the same, we expect an ROC of about 0.5. Any higher than that suggests that there is something quite different about the test set.\n\nI'm using the processed dataset [here](https://www.kaggle.com/c/happy-whale-and-dolphin/discussion/307287) to speed up training.","metadata":{}},{"cell_type":"markdown","source":"# Params and Dataset","metadata":{}},{"cell_type":"code","source":"SEED = 420\nIMG_PATH_BASE = '../input/happy-whale-512'\nIMG_SIZE = 224\nBS = 64\nARCH = resnet18","metadata":{"execution":{"iopub.status.busy":"2022-02-15T11:48:10.869636Z","iopub.execute_input":"2022-02-15T11:48:10.869980Z","iopub.status.idle":"2022-02-15T11:48:10.879528Z","shell.execute_reply.started":"2022-02-15T11:48:10.869931Z","shell.execute_reply":"2022-02-15T11:48:10.878526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(f'{IMG_PATH_BASE}/train.csv')\ntest_df = pd.read_csv(f'{IMG_PATH_BASE}/sample_submission.csv')\n\ntrain_df['image_path'] = f'{IMG_PATH_BASE}/train_images/' + train_df.image\ntrain_df['is_test'] = False\n\ntest_df['image_path'] = f'{IMG_PATH_BASE}/test_images/' + test_df.image\ntest_df['is_test'] = True","metadata":{"execution":{"iopub.status.busy":"2022-02-15T11:48:10.882846Z","iopub.execute_input":"2022-02-15T11:48:10.883202Z","iopub.status.idle":"2022-02-15T11:48:14.170526Z","shell.execute_reply.started":"2022-02-15T11:48:10.883155Z","shell.execute_reply":"2022-02-15T11:48:14.169486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Remove corrupt examples","metadata":{}},{"cell_type":"code","source":"from tqdm.notebook import tqdm\n\ndef remove_corrupt_examples(df):\n    valid_rows = []\n    num = 0\n    for idx, row in tqdm(df.iterrows(), total=len(df)):\n        try:\n            Image.open(row.image_path)\n            valid_rows.append(row)\n        except Exception:\n            num += 1\n            continue\n\n    print(f'Found {num} corrupt examples')\n    \n    return pd.DataFrame(valid_rows)","metadata":{"execution":{"iopub.status.busy":"2022-02-15T11:48:14.172145Z","iopub.execute_input":"2022-02-15T11:48:14.172470Z","iopub.status.idle":"2022-02-15T11:48:14.181067Z","shell.execute_reply.started":"2022-02-15T11:48:14.172424Z","shell.execute_reply":"2022-02-15T11:48:14.180042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = remove_corrupt_examples(train_df)\ntest_df = remove_corrupt_examples(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-02-15T11:48:14.182912Z","iopub.execute_input":"2022-02-15T11:48:14.183653Z","iopub.status.idle":"2022-02-15T11:59:52.466048Z","shell.execute_reply.started":"2022-02-15T11:48:14.183608Z","shell.execute_reply":"2022-02-15T11:59:52.465022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_df = pd.concat([\n    train_df[['image_path', 'is_test']], test_df[['image_path', 'is_test']]]\n).reset_index(drop=True).sample(frac=1., random_state=SEED)","metadata":{"execution":{"iopub.status.busy":"2022-02-15T11:59:52.467872Z","iopub.execute_input":"2022-02-15T11:59:52.468266Z","iopub.status.idle":"2022-02-15T11:59:52.493241Z","shell.execute_reply.started":"2022-02-15T11:59:52.468209Z","shell.execute_reply":"2022-02-15T11:59:52.492281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_df.is_test.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-02-15T11:59:52.494617Z","iopub.execute_input":"2022-02-15T11:59:52.494861Z","iopub.status.idle":"2022-02-15T11:59:52.510121Z","shell.execute_reply.started":"2022-02-15T11:59:52.494830Z","shell.execute_reply":"2022-02-15T11:59:52.509260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I'm using fastai library for training as we can do so much with very little code.","metadata":{}},{"cell_type":"markdown","source":"# Setup Data","metadata":{}},{"cell_type":"markdown","source":"[Datablock](https://docs.fast.ai/data.block.html) is a tool that let's you create a Dataset from configuration.","metadata":{}},{"cell_type":"code","source":"datablock = DataBlock(\n    blocks=(ImageBlock, CategoryBlock),\n    getters=[\n        ColReader('image_path'), ColReader('is_test')\n    ],\n    splitter=RandomSplitter(seed=SEED),\n    item_tfms=Resize(IMG_SIZE),\n    batch_tfms=aug_transforms(size=IMG_SIZE, max_rotate=30., min_scale=0.75, flip_vert=True, do_flip=True)\n)","metadata":{"execution":{"iopub.status.busy":"2022-02-15T11:59:52.511355Z","iopub.execute_input":"2022-02-15T11:59:52.512081Z","iopub.status.idle":"2022-02-15T11:59:52.522205Z","shell.execute_reply.started":"2022-02-15T11:59:52.512035Z","shell.execute_reply":"2022-02-15T11:59:52.521011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls = datablock.dataloaders(source=all_df, bs=BS)","metadata":{"execution":{"iopub.status.busy":"2022-02-15T11:59:52.526415Z","iopub.execute_input":"2022-02-15T11:59:52.526955Z","iopub.status.idle":"2022-02-15T12:00:09.049767Z","shell.execute_reply.started":"2022-02-15T11:59:52.526907Z","shell.execute_reply":"2022-02-15T12:00:09.048717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls.show_batch()","metadata":{"execution":{"iopub.status.busy":"2022-02-15T12:00:09.051580Z","iopub.execute_input":"2022-02-15T12:00:09.052153Z","iopub.status.idle":"2022-02-15T12:00:10.427578Z","shell.execute_reply.started":"2022-02-15T12:00:09.052104Z","shell.execute_reply":"2022-02-15T12:00:10.425097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model and training","metadata":{}},{"cell_type":"code","source":"def get_learner(dls, lr=1e-3):\n    opt_func = partial(Adam, lr=lr, wd=0.01, eps=1e-8)\n\n    learn = cnn_learner(\n        dls, ARCH, opt_func=opt_func,\n        metrics=[RocAucBinary()]).to_fp16()\n\n    return learn","metadata":{"execution":{"iopub.status.busy":"2022-02-15T12:00:10.428767Z","iopub.execute_input":"2022-02-15T12:00:10.429428Z","iopub.status.idle":"2022-02-15T12:00:10.436903Z","shell.execute_reply.started":"2022-02-15T12:00:10.429380Z","shell.execute_reply":"2022-02-15T12:00:10.435732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn = get_learner(dls)","metadata":{"execution":{"iopub.status.busy":"2022-02-15T12:00:10.438910Z","iopub.execute_input":"2022-02-15T12:00:10.439989Z","iopub.status.idle":"2022-02-15T12:00:13.201207Z","shell.execute_reply.started":"2022-02-15T12:00:10.439940Z","shell.execute_reply":"2022-02-15T12:00:13.200025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I previously ran lr_find to get the learning rate.","metadata":{}},{"cell_type":"code","source":"# learn.lr_find()","metadata":{"execution":{"iopub.status.busy":"2022-02-15T12:00:13.203085Z","iopub.execute_input":"2022-02-15T12:00:13.203498Z","iopub.status.idle":"2022-02-15T12:00:13.209692Z","shell.execute_reply.started":"2022-02-15T12:00:13.203446Z","shell.execute_reply":"2022-02-15T12:00:13.208535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.fit_one_cycle(1)\nlearn.unfreeze()\nlearn.fit_one_cycle(4, slice(1e-4, 1e-3))","metadata":{"execution":{"iopub.status.busy":"2022-02-15T12:00:13.211507Z","iopub.execute_input":"2022-02-15T12:00:13.212760Z","iopub.status.idle":"2022-02-15T12:08:33.732165Z","shell.execute_reply.started":"2022-02-15T12:00:13.212711Z","shell.execute_reply":"2022-02-15T12:08:33.726971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Result","metadata":{}},{"cell_type":"code","source":"loss, metric = learn.validate()","metadata":{"execution":{"iopub.status.busy":"2022-02-15T12:08:38.085163Z","iopub.execute_input":"2022-02-15T12:08:38.085480Z","iopub.status.idle":"2022-02-15T12:10:10.862194Z","shell.execute_reply.started":"2022-02-15T12:08:38.085445Z","shell.execute_reply":"2022-02-15T12:10:10.861131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metric","metadata":{"execution":{"iopub.status.busy":"2022-02-15T12:10:10.864880Z","iopub.execute_input":"2022-02-15T12:10:10.865235Z","iopub.status.idle":"2022-02-15T12:10:10.874255Z","shell.execute_reply.started":"2022-02-15T12:10:10.865185Z","shell.execute_reply":"2022-02-15T12:10:10.873119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Compare Result to Random Baseline","metadata":{}},{"cell_type":"markdown","source":"Since the AUC is > 0.5, it indicates that there may be some signal allowing a model to differentiate the sets.\n\nHowever, we should first get the AUC if we train a model with a \"fake\" test set. We make believe 20% of examples from train are test set, train a model on that and get a metric reading. If it's roughly the same as what we saw before, we know that the test set is from the same distribution as train.","metadata":{"execution":{"iopub.status.busy":"2022-02-15T09:22:00.287887Z","iopub.execute_input":"2022-02-15T09:22:00.288202Z","iopub.status.idle":"2022-02-15T09:22:00.312448Z","shell.execute_reply.started":"2022-02-15T09:22:00.288123Z","shell.execute_reply":"2022-02-15T09:22:00.311423Z"}}},{"cell_type":"code","source":"fake_train = train_df.copy()\nfake_train['is_test'] = False\nfake_train_samp = fake_train.sample(frac=0.2)\nfake_train_samp['is_test'] = True\nfake_train.update(fake_train_samp)","metadata":{"execution":{"iopub.status.busy":"2022-02-15T12:10:10.876217Z","iopub.execute_input":"2022-02-15T12:10:10.877141Z","iopub.status.idle":"2022-02-15T12:10:10.924619Z","shell.execute_reply.started":"2022-02-15T12:10:10.877090Z","shell.execute_reply":"2022-02-15T12:10:10.923599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fake_train.is_test.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-02-15T12:10:10.926990Z","iopub.execute_input":"2022-02-15T12:10:10.927324Z","iopub.status.idle":"2022-02-15T12:10:10.941368Z","shell.execute_reply.started":"2022-02-15T12:10:10.927279Z","shell.execute_reply":"2022-02-15T12:10:10.940236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datablock = DataBlock(\n    blocks=(ImageBlock, CategoryBlock),\n    getters=[\n        ColReader('image_path'), ColReader('is_test')\n    ],\n    splitter=RandomSplitter(seed=SEED),\n    item_tfms=Resize(IMG_SIZE),\n    batch_tfms=aug_transforms(size=IMG_SIZE, max_rotate=30., min_scale=0.75, flip_vert=True, do_flip=True)\n)\ndls = datablock.dataloaders(source=fake_train, bs=BS)\nlearn = get_learner(dls)\nlearn.fit_one_cycle(1)\nlearn.unfreeze()\nlearn.fit_one_cycle(4, slice(1e-4, 1e-3))","metadata":{"execution":{"iopub.status.busy":"2022-02-15T12:10:10.943399Z","iopub.execute_input":"2022-02-15T12:10:10.944131Z","iopub.status.idle":"2022-02-15T12:10:50.361994Z","shell.execute_reply.started":"2022-02-15T12:10:10.944082Z","shell.execute_reply":"2022-02-15T12:10:50.359982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss, baseline_metric = learn.validate()","metadata":{"execution":{"iopub.status.busy":"2022-02-15T12:10:51.580313Z","iopub.execute_input":"2022-02-15T12:10:51.581276Z","iopub.status.idle":"2022-02-15T12:11:50.020820Z","shell.execute_reply.started":"2022-02-15T12:10:51.581237Z","shell.execute_reply":"2022-02-15T12:11:50.019801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Random Baseline Result","metadata":{}},{"cell_type":"code","source":"baseline_metric","metadata":{"execution":{"iopub.status.busy":"2022-02-15T12:11:50.023663Z","iopub.execute_input":"2022-02-15T12:11:50.024039Z","iopub.status.idle":"2022-02-15T12:11:50.031045Z","shell.execute_reply.started":"2022-02-15T12:11:50.023970Z","shell.execute_reply":"2022-02-15T12:11:50.030034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Conclusion","metadata":{}},{"cell_type":"markdown","source":"So train and test set do have some minor differences. An ROC-AUC of > 0.5 is indicative of some signal to distinguish between the 2 sets, but not a lot.","metadata":{}}]}