{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom pathlib import Path\nimport fastai\nfrom fastai.vision import *\nimport cv2\nimport torch\nfrom torch.nn import Conv2d\nfrom sklearn.model_selection import train_test_split\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"path = Path('../input/recursion-cellular-image-classification')","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"tr_df = pd.read_csv(path/'train.csv')\ntr_df = tr_df.append(pd.read_csv(path/'train_controls.csv'), ignore_index=True, sort=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tr_df['site'] = 1\ntr_df_copy = tr_df.copy()\ntr_df_copy['site'] = 2\ntr_df = tr_df.append(tr_df_copy, ignore_index=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tr_df.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_img_path(experiment, plate, well, site=1):\n    return experiment + '/' + 'Plate' + str(plate) + '/' + well + '_s{}'.format(site)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tr_df['path'] = tr_df.apply(lambda row : get_img_path(row.experiment, row.plate, row.well, row.site), axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tr_df['cellline'] = tr_df.apply(lambda row : row.experiment[:-3], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cell_lines = list(tr_df.cellline.unique())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cell_lines","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for cell_line in cell_lines:\n    !mkdir {cell_line}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!ls","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tr_dfs = [tr_df.loc[lambda df: df.cellline == cell_line] for cell_line in cell_lines]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"list(map(len, tr_dfs))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class RecursionImageList(ImageList):\n    def open(self, fn:PathOrStr) -> Image:\n        fn = str(fn)\n        imgs = []\n        for channel in [1, 2, 3, 4, 5, 6]:\n            img_path = fn + '_w{}'.format(channel) + '.png'\n            imgs.append(cv2.imread(img_path, cv2.IMREAD_GRAYSCALE))\n        img = cv2.merge(imgs)\n        img = img / 255.0\n        return Image(px=pil2tensor(img, np.float32))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"num_classes = len(tr_df.sirna.unique())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"num_classes","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"srcs = [\n    RecursionImageList.from_df(tr_cell_line_df, path=path/'train', cols=['path'])\n    for tr_cell_line_df in tr_dfs\n]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def split_ds(src, tr_df):\n    df = tr_df\n    test_size = max(num_classes, int(len(df) * 0.05))\n    tr_df, va_df = train_test_split(df, test_size=test_size, random_state=42, stratify = df.sirna)\n    src = src.split_by_list(RecursionImageList.from_df(tr_df, path=path/'train', cols=['path']), \n                            RecursionImageList.from_df(va_df, path=path/'train', cols=['path']))\n    return src","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"srcs = list(map(split_ds, srcs, tr_dfs))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"srcs = [src.label_from_df(cols=['sirna']) for src in srcs]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data = [src.transform(size=512).databunch(bs=32, pin_memory=True) for src in srcs]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"precision = Precision(average=\"macro\")\nrecall = Recall(average=\"macro\")\nloss_func = CrossEntropyFlat()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learners = [cnn_learner(data[i], models.resnet18, metrics=[accuracy, error_rate],\n                    loss_func=loss_func,\n                    model_dir='/kaggle/working/' + cell_lines[i], pretrained=True)\n            for i in range(len(data))\n           ]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learners[0].model[0][0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# let's make our model work with single channel\nfor learn in learners:\n    trained_kernel = learn.model[0][0].weight\n    new_conv = Conv2d(6, 64, kernel_size=(7, 7), stride=(2, 2), padding=(3, 3), bias=False)\n    with torch.no_grad():\n        new_conv.weight[:,:] = torch.stack([torch.mean(trained_kernel, 1)] * 6, dim=1)\n    learn.model[0][0] = new_conv\n    learn.model.cuda();\n#     learn.load('../../input/recursion-fast-ai-channel-2-4-resnet50/stage-04');","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# validation_results = []\n# for learn in learners:\n#     validation_results.append(learn.validate())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# validation_results","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"i = 0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# # Find a good learning rate\n# learners[i].lr_find(start_lr=1.0e-5, end_lr=1e-1, num_it=256)\n# learners[i].recorder.plot(suggestion=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# !nvidia-smi","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learners[i].fit_one_cycle(10, 2e-3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learners[i].recorder.plot_losses()\nlearners[i].recorder.plot_metrics()\nlearners[i].recorder.plot_lr(show_moms=True)\nlearners[i].save('stage-01')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"i = 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# # Find a good learning rate\n# learners[i].lr_find(start_lr=1.0e-5, end_lr=1e-1, num_it=256)\n# learners[i].recorder.plot(suggestion=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learners[i].fit_one_cycle(10, 5e-3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learners[i].recorder.plot_losses()\nlearners[i].recorder.plot_metrics()\nlearners[i].recorder.plot_lr(show_moms=True)\nlearners[i].save('stage-01')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"i = 2","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# # Find a good learning rate\n# learners[i].lr_find(start_lr=1.0e-4, end_lr=1e-1, num_it=256)\n# learners[i].recorder.plot(suggestion=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learners[i].fit_one_cycle(10, 5e-3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learners[i].recorder.plot_losses()\nlearners[i].recorder.plot_metrics()\nlearners[i].recorder.plot_lr(show_moms=True)\nlearners[i].save('stage-01')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"i = 3","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# # Find a good learning rate\n# learners[i].lr_find(start_lr=1.0e-4, end_lr=1e-1, num_it=256)\n# learners[i].recorder.plot(suggestion=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learners[i].fit_one_cycle(10, 4e-3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learners[i].recorder.plot_losses()\nlearners[i].recorder.plot_metrics()\nlearners[i].recorder.plot_lr(show_moms=True)\nlearners[i].save('stage-01')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Prediction"},{"metadata":{"trusted":true},"cell_type":"code","source":"te_df = pd.read_csv(path/'test.csv')\nte_df['path1'] = te_df.apply(lambda row : get_img_path(row.experiment, row.plate, row.well, site=1), axis=1)\nte_df['path2'] = te_df.apply(lambda row : get_img_path(row.experiment, row.plate, row.well, site=2), axis=1)\nte_df['cellline'] = te_df.apply(lambda row : row.experiment[:-3], axis=1)\nte_dfs = [te_df.loc[lambda df: df.cellline == cell_line] for cell_line in cell_lines]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"te_df['cellline'].unique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"list(map(len, te_dfs))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_srcs1 = [\n    RecursionImageList.from_df(te_cell_line_df, path=path/'test', cols=['path1'])\n    for te_cell_line_df in te_dfs\n]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_srcs2 = [\n    RecursionImageList.from_df(te_cell_line_df, path=path/'test', cols=['path2'])\n    for te_cell_line_df in te_dfs\n]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for cell_line, learn, test_src1, test_src2, te_df in zip(cell_lines, learners, test_srcs1, test_srcs2, te_dfs):\n    print(cell_line)\n    learn.data.add_test(test_src1)\n    preds1, y = learn.get_preds(DatasetType.Test)\n    learn.data.add_test(test_src2)\n    preds2,y = learn.get_preds(DatasetType.Test)\n    preds = 0.5 * (preds1 + preds2)\n    te_df['sirna'] = preds.argmax(1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"te_df = pd.concat(te_dfs, ignore_index=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"te_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"te_df.to_csv('submission.csv', index=False, columns=['id_code','sirna'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}