{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Setup","metadata":{}},{"cell_type":"code","source":"from fastai.vision.all import *","metadata":{"execution":{"iopub.status.busy":"2021-11-13T15:42:54.560098Z","iopub.execute_input":"2021-11-13T15:42:54.560691Z","iopub.status.idle":"2021-11-13T15:42:57.240047Z","shell.execute_reply.started":"2021-11-13T15:42:54.560589Z","shell.execute_reply":"2021-11-13T15:42:57.239055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading the DataFrame","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('../input/sartorius-cell-instance-segmentation/train.csv')\ndf.tail(1)","metadata":{"execution":{"iopub.status.busy":"2021-11-13T15:42:57.243746Z","iopub.execute_input":"2021-11-13T15:42:57.243959Z","iopub.status.idle":"2021-11-13T15:42:57.765031Z","shell.execute_reply.started":"2021-11-13T15:42:57.243934Z","shell.execute_reply":"2021-11-13T15:42:57.764370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Splitting in 5 folds based on image and cell type","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nimg_df = df[['id', 'cell_type']].drop_duplicates().reset_index(drop = True)\nimg_df['fold'] = -1\nskf = StratifiedKFold(n_splits = 5, random_state = 42, shuffle = True)\nfor fold, (train_index, test_index) in enumerate(skf.split(img_df['id'], img_df['cell_type'])):\n    img_df.loc[test_index, 'fold'] = fold\nimg_df.to_csv('train_fold.csv', index = False)\nimg_df.tail()","metadata":{"execution":{"iopub.status.busy":"2021-11-13T15:42:57.766519Z","iopub.execute_input":"2021-11-13T15:42:57.766915Z","iopub.status.idle":"2021-11-13T15:42:57.825913Z","shell.execute_reply.started":"2021-11-13T15:42:57.766880Z","shell.execute_reply":"2021-11-13T15:42:57.825116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_df.groupby('fold')['cell_type'].value_counts().to_frame().T","metadata":{"execution":{"iopub.status.busy":"2021-11-13T15:42:57.828171Z","iopub.execute_input":"2021-11-13T15:42:57.828428Z","iopub.status.idle":"2021-11-13T15:42:57.850640Z","shell.execute_reply.started":"2021-11-13T15:42:57.828393Z","shell.execute_reply":"2021-11-13T15:42:57.850017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Building final DataFrame","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('./train_fold.csv')\ndf = pd.concat([df, pd.get_dummies(df['fold'], prefix = 'fold', dtype = bool)], axis = 1).drop('fold', axis = 1)\ndf.tail(1)","metadata":{"execution":{"iopub.status.busy":"2021-11-13T15:42:57.852528Z","iopub.execute_input":"2021-11-13T15:42:57.852967Z","iopub.status.idle":"2021-11-13T15:42:57.872296Z","shell.execute_reply.started":"2021-11-13T15:42:57.852932Z","shell.execute_reply":"2021-11-13T15:42:57.871722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dataloaders","metadata":{}},{"cell_type":"code","source":"## Global variables\nBS = 32\nWORKERS = 4\nBASE_DIR = '../input/sartorius-cell-instance-segmentation/train/'\nFILE_EXT = '.png'\nFOLD = 'fold_0'","metadata":{"execution":{"iopub.status.busy":"2021-11-13T15:49:25.464762Z","iopub.execute_input":"2021-11-13T15:49:25.465289Z","iopub.status.idle":"2021-11-13T15:49:25.469204Z","shell.execute_reply.started":"2021-11-13T15:49:25.465252Z","shell.execute_reply":"2021-11-13T15:49:25.468509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dblock = DataBlock(\n    blocks = (ImageBlock, CategoryBlock),\n    get_x = ColReader('id', pref = BASE_DIR, suff = FILE_EXT),\n    get_y = ColReader('cell_type'),\n    splitter = ColSplitter(FOLD)\n)\ndls = dblock.dataloaders(df, bs = BS, num_workers = WORKERS)\ndls.show_batch(figsize = (30, 22))","metadata":{"execution":{"iopub.status.busy":"2021-11-13T15:50:05.426327Z","iopub.execute_input":"2021-11-13T15:50:05.426610Z","iopub.status.idle":"2021-11-13T15:50:07.764650Z","shell.execute_reply.started":"2021-11-13T15:50:05.426561Z","shell.execute_reply":"2021-11-13T15:50:07.763851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Basic CNN Learner with resnet18","metadata":{}},{"cell_type":"code","source":"learn = cnn_learner(dls, resnet18, metrics = accuracy)","metadata":{"execution":{"iopub.status.busy":"2021-11-13T15:43:02.814149Z","iopub.execute_input":"2021-11-13T15:43:02.814384Z","iopub.status.idle":"2021-11-13T15:43:06.363504Z","shell.execute_reply.started":"2021-11-13T15:43:02.814355Z","shell.execute_reply":"2021-11-13T15:43:06.362803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Finding optimal LR and training for 5 epochs","metadata":{}},{"cell_type":"code","source":"learn.lr_find()","metadata":{"execution":{"iopub.status.busy":"2021-11-13T15:43:06.364704Z","iopub.execute_input":"2021-11-13T15:43:06.365347Z","iopub.status.idle":"2021-11-13T15:43:51.691934Z","shell.execute_reply.started":"2021-11-13T15:43:06.365312Z","shell.execute_reply":"2021-11-13T15:43:51.691113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.fine_tune(5, 1e-3)","metadata":{"execution":{"iopub.status.busy":"2021-11-13T15:43:51.694861Z","iopub.execute_input":"2021-11-13T15:43:51.695351Z","iopub.status.idle":"2021-11-13T15:44:46.705281Z","shell.execute_reply.started":"2021-11-13T15:43:51.695317Z","shell.execute_reply":"2021-11-13T15:44:46.704511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualizing the results","metadata":{}},{"cell_type":"code","source":"learn.show_results(figsize = (30, 22))","metadata":{"execution":{"iopub.status.busy":"2021-11-13T15:50:15.326105Z","iopub.execute_input":"2021-11-13T15:50:15.326605Z","iopub.status.idle":"2021-11-13T15:50:18.106110Z","shell.execute_reply.started":"2021-11-13T15:50:15.326551Z","shell.execute_reply":"2021-11-13T15:50:18.105468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"interp = ClassificationInterpretation.from_learner(learn)","metadata":{"execution":{"iopub.status.busy":"2021-11-13T15:44:48.858336Z","iopub.execute_input":"2021-11-13T15:44:48.859099Z","iopub.status.idle":"2021-11-13T15:44:51.523981Z","shell.execute_reply.started":"2021-11-13T15:44:48.859064Z","shell.execute_reply":"2021-11-13T15:44:51.523142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"interp.plot_confusion_matrix()","metadata":{"execution":{"iopub.status.busy":"2021-11-13T15:44:51.525491Z","iopub.execute_input":"2021-11-13T15:44:51.525782Z","iopub.status.idle":"2021-11-13T15:44:51.711869Z","shell.execute_reply.started":"2021-11-13T15:44:51.525744Z","shell.execute_reply":"2021-11-13T15:44:51.711097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Temporary fix for the broken version of plot_top_losses (untill they update the kaggle repo)\ndef plot_top_losses_fix(interp, k, largest=True, **kwargs):\n        losses,idx = interp.top_losses(k, largest)\n        if not isinstance(interp.inputs, tuple): interp.inputs = (interp.inputs,)\n        if isinstance(interp.inputs[0], Tensor): inps = tuple(o[idx] for o in interp.inputs)\n        else: inps = interp.dl.create_batch(interp.dl.before_batch([tuple(o[i] for o in interp.inputs) for i in idx]))\n        b = inps + tuple(o[idx] for o in (interp.targs if is_listy(interp.targs) else (interp.targs,)))\n        x,y,its = interp.dl._pre_show_batch(b, max_n=k)\n        b_out = inps + tuple(o[idx] for o in (interp.decoded if is_listy(interp.decoded) else (interp.decoded,)))\n        x1,y1,outs = interp.dl._pre_show_batch(b_out, max_n=k)\n        if its is not None:\n            plot_top_losses(x, y, its, outs.itemgot(slice(len(inps), None)), interp.preds[idx], losses,  **kwargs)","metadata":{"execution":{"iopub.status.busy":"2021-11-13T15:48:03.297390Z","iopub.execute_input":"2021-11-13T15:48:03.297660Z","iopub.status.idle":"2021-11-13T15:48:03.306927Z","shell.execute_reply.started":"2021-11-13T15:48:03.297629Z","shell.execute_reply":"2021-11-13T15:48:03.306250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_top_losses_fix(interp, 3, figsize = (30, 8))","metadata":{"execution":{"iopub.status.busy":"2021-11-13T15:49:47.166469Z","iopub.execute_input":"2021-11-13T15:49:47.167149Z","iopub.status.idle":"2021-11-13T15:49:47.814923Z","shell.execute_reply.started":"2021-11-13T15:49:47.167111Z","shell.execute_reply":"2021-11-13T15:49:47.813406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}