{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom pathlib import Path\nimport fastai\nfrom fastai.vision import *\nimport cv2\nimport torch\nfrom torch.nn import Conv2d\nimport torch.nn.functional as F\nfrom sklearn.model_selection import train_test_split\nfrom fastai.callbacks import *\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\n# print(os.listdir(\"../input\"))\n# !ls ../input\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"path = Path('../input/recursion-cellular-image-classification')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pixel_stats = pd.read_csv(path/'pixel_stats.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pixel_stats.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# pixel_stats['exp_plate'] = pixel_stats['experiment'] + '' + pixel_stats['plate'].astype(str)\npixel_stats['exp_plate_channel'] = pixel_stats['experiment'] + '' + pixel_stats['plate'].astype(str) + '' + pixel_stats['channel'].astype(str)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pixel_stats_channel_mean = pixel_stats.groupby('exp_plate_channel').agg(np.mean).reset_index()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pixel_stats_channel_mean.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pixel_stats_channel_mean['exp_plate'] = pixel_stats_channel_mean.exp_plate_channel.apply(lambda s:s[:-1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"exp_plate_mean = dict()\nfor group in pixel_stats_channel_mean.groupby('exp_plate'):\n    exp_plate_mean[group[0]] = torch.from_numpy(group[1]['mean'].to_numpy(dtype=np.float32).reshape(6, 1, 1))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"exp_plate_std = dict()\nfor group in pixel_stats_channel_mean.groupby('exp_plate'):\n    exp_plate_std[group[0]] = torch.from_numpy(group[1]['std'].to_numpy(dtype=np.float32).reshape(6, 1, 1))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tr_df = pd.read_csv(path/'train.csv')\ntr_df['site'] = 1\ntr_df_copy = tr_df.copy()\ntr_df_copy['site'] = 2\ntr_df = tr_df.append(tr_df_copy, ignore_index=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tr_df['cellline'] = tr_df.apply(lambda row : row.experiment[:-3], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_img_path(suffix, experiment, plate, well, site=1):\n    return suffix + experiment.lower() + '/' + 'plate' + str(plate) + '/' + well + '_s{}.npy'.format(site)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tr_df['path'] = tr_df.apply(lambda row : get_img_path('recursion-npy-train-', row.experiment, row.plate, row.well, row.site), axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tr_df.experiment.unique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tr_exps = ['HUVEC-02', 'HUVEC-03', 'HUVEC-04', 'HUVEC-05',\n           'HUVEC-06', 'HUVEC-07', 'HUVEC-08', 'HUVEC-09',\n           'HUVEC-10', 'HUVEC-11', 'HUVEC-12', 'HUVEC-13']\nva_exps = ['HUVEC-01', 'HUVEC-14', 'HUVEC-15', 'HUVEC-16']#, 'HEPG2-01', 'RPE-01', 'U2OS-01']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# tr_exps = ['U2OS-02', 'U2OS-03']\n# va_exps = ['U2OS-01']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# tr_exps = ['RPE-02', 'RPE-03']\n# va_exps = ['RPE-01']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# tr_exps = ['HEPG2-02', 'U2OS-02', 'RPE-02', 'HUVEC-02']\n# va_exps = ['HEPG2-01', 'U2OS-01', 'RPE-01', 'HUVEC-01']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tr_df = tr_df[tr_df.experiment.str.contains('|'.join(tr_exps + va_exps))]\ntr_df['is_valid'] = tr_df.experiment.str.contains('|'.join(va_exps))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class RecursionImageList(ImageList):\n    def open(self, fn:PathOrStr) -> Image:\n        plate = fn[-12]\n        exp = fn[29:-18]\n        exp_plate = exp.upper() + plate\n        img = np.load(fn)\n#         img = img / 255.0\n        img = pil2tensor(img, dtype=np.float32)\n#         img.sub_(exp_plate_mean[exp_plate])\n#         img.div_(exp_plate_std[exp_plate])\n#         img = F.avg_pool2d(img, kernel_size=2)\n        return Image(img)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"src = (\n    RecursionImageList.from_df(tr_df, path=Path('../input'), cols=['path'])\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"src = src.split_from_df(col='is_valid')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"src = src.label_from_df(cols=['sirna'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img_size = 256\nr = img_size / (512 * 2)\n# r = 0.5","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"transforms = ([\n    crop(size=img_size, row_pct=(r, 1 - r), col_pct=(r, 1 - r)),\n    dihedral(),\n], [\n    crop(size=img_size, row_pct=0.5, col_pct=0.5),\n])\n# transforms = ([], [])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data = src.transform(size=img_size, tfms=transforms).databunch(bs=16, pin_memory=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"precision = Precision(average=\"macro\")\nrecall = Recall(average=\"macro\")\nloss_func = CrossEntropyFlat()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn = cnn_learner(data, models.densenet201, metrics=[accuracy, error_rate],\n                    loss_func=loss_func,\n                    model_dir='/kaggle/working', pretrained=True, \n                    callback_fns=[CSVLogger])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"trained_kernel = learn.model[0][0][0].weight\nnew_conv = Conv2d(6, 64, kernel_size=(7, 7), stride=(2, 2), padding=(3, 3), bias=False)\nwith torch.no_grad():\n    new_conv.weight[:,:] = torch.stack([torch.mean(trained_kernel, 1)] * 6, dim=1)\nlearn.model[0][0][0] = new_conv\nlearn.model[0][0][0].requires_grad = True","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# learn.load('../input/recursion-fast-ai-channel-all/huvec-25', with_opt=True);","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.model.cuda();","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# learn.unfreeze()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.lr_find()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.recorder.plot(skip_start=35, skip_end=18, suggestion=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.fit_one_cycle(25, 3e-3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.csv_logger.read_logged_file()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# learn.fit_one_cycle(20, 3e-3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# learn.fit_one_cycle(20, 3e-3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# learn.fit_one_cycle(20, 3e-3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# learn.fit_one_cycle(20, 3e-3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# learn.fit_one_cycle(20, 3e-3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# learn.fit_one_cycle(20, 3e-3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.recorder.plot_losses()\nlearn.recorder.plot_metrics()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.save('huvec-25')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Prediction"},{"metadata":{"trusted":true},"cell_type":"code","source":"te_df = pd.read_csv(path/'test.csv')\nte_df['path1'] = te_df.apply(lambda row : get_img_path('recursion-npy-test-', row.experiment, row.plate, row.well, site=1), axis=1)\nte_df['path2'] = te_df.apply(lambda row : get_img_path('recursion-npy-test-', row.experiment, row.plate, row.well, site=2), axis=1)\ntest_src1 = RecursionImageList.from_df(te_df, path=Path('../input'), cols=['path1'])\ntest_src2 = RecursionImageList.from_df(te_df, path=Path('../input'), cols=['path2'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.data.add_test(test_src1)\npreds1, y = learn.get_preds(DatasetType.Test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.data.add_test(test_src2)\npreds2,y = learn.get_preds(DatasetType.Test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds = 0.5 * (preds1 + preds2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"te_df['sirna'] = preds.argmax(1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"te_df.to_csv('submission.csv', index=False, columns=['id_code','sirna'])","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}