{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# fastai quick submission template\n\nSolution overview: https://www.kaggle.com/c/hpa-single-cell-image-classification/discussion/221550\n\nI want to experiment quickly and can't wait for the lb score, especially that with weak labels I haven't found a reasonable way to do CV, and depend on the public lb score. I have pre-processed the public test images in the same way as my prototyping dataset and submit my preds only for this piece. These submissions will get zero score on private, but there is still lots of time in the competition and better approaches will be developed .","metadata":{}},{"cell_type":"code","source":"# pip install fastai==1.0.61","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! pip install gdown","metadata":{"execution":{"iopub.status.busy":"2021-06-11T06:44:05.658407Z","iopub.execute_input":"2021-06-11T06:44:05.658767Z","iopub.status.idle":"2021-06-11T06:44:26.840604Z","shell.execute_reply.started":"2021-06-11T06:44:05.658737Z","shell.execute_reply":"2021-06-11T06:44:26.839335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!gdown --id 1eOc0pgbjGA-JVIx4jdA3m1xeYaf0xsx2","metadata":{"execution":{"iopub.status.busy":"2021-06-11T06:44:51.256295Z","iopub.execute_input":"2021-06-11T06:44:51.256625Z","iopub.status.idle":"2021-06-11T06:44:55.764591Z","shell.execute_reply.started":"2021-06-11T06:44:51.256597Z","shell.execute_reply":"2021-06-11T06:44:55.762951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# path = Path('../input/hpa-cell-tiles-sample-balanced-dataset')\n# df = pd.read_csv(path/'cell_df.csv')\n# df.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport pickle\nimport torch\nfrom matplotlib import pyplot as plt\n%matplotlib inline","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Name Label Dictionary\nname_label_dict = {\n0:  \"Nucleoplasm\", \n1:  \"Nuclear membrane\",   \n2:  \"Nucleoli\",   \n3:  \"Nucleoli fibrillar center\" ,  \n4:  \"Nuclear speckles\"   ,\n5:  \"Nuclear bodies\"   ,\n6:  \"Endoplasmic reticulum\",   \n7:  \"Golgi apparatus\"   ,\n8:  \"Intermediate filaments\",   \n9:  \"Actin filaments\"   ,\n10:  \"Microtubules\"   ,\n11:  \"Mitotic spindle\"   ,\n12:  \"Centrosome\" , \n13:  \"Plasma membrane\",   \n14:  \"Mitochondria\"   ,\n15:  \"Aggresome\"   ,\n16:  \"Cytosol\",  \n17:  \"Vesicles and punctate cytosolic patterns\",   \n18:  \"Negative\" \n}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Load Train Dataset \npath_to_train = '../input/hpa-single-cell-image-classification/train/'\ndata = pd.read_csv('../input/hpa-single-cell-image-classification/train.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head(), data.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_labels(df):\n    # dataframe with column for each class\n    labels = list()\n    \n    for index, sample in df.iterrows():\n        # zero out class array\n        label = [0] * 19\n        \n        # for each class found in training sample, flip lablel value to one\n        for cls in sample['Label'].split('|'):\n            label[int(cls)] = 1\n\n        # Append label to list\n        labels.append( np.array(label) )\n\n    return np.vstack(labels)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_labels = build_labels(data)\ndf_train_labels.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_fs = 'spanwar'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add full path to images files\ndata['image_path']=path_fs + '/' + data['ID']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Join train_labels \n## Just Testing Green \"Cells\"\ndf_train_green = pd.DataFrame(data['image_path']).join(pd.DataFrame(df_train_labels))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_green.head()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_green['image_path'] = df_train_green['image_path']+'_green.png'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_colwidth', None)\ndf_train_green.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"colours = ['_red.png', '_blue.png', '_yellow.png', '_green.png']\nTRAIN_PATHS = '../input/hpa-single-cell-image-classification/train'\npaths = [[os.path.join(TRAIN_PATHS, data.iloc[idx,0])+ colour for colour in colours] for idx in range(len(data))]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths[0]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titles = ['microtubules', 'nuclei', 'endoplasmic reticulum', 'protein of interest']\nfig, axs = plt.subplots(3, 4, figsize =(16,8))\nfor entry in range(3):\n    for channel in range(4):\n        img = plt.imread(paths[entry][channel])\n        axs[entry, channel].imshow(img)        \n        if entry == 0:\n            axs[0, channel].set_title(titles[channel])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def to_rgb(idx,paths = paths, gbr = False, blue_only = False):\n    red = cv2.imread(paths[idx][0],0)\n    yellow = cv2.imread(paths[idx][2],0)\n    blue = cv2.imread(paths[idx][1],0)\n    green = cv2.imread(paths[idx][3],0)\n    \n    if blue_only: \n        return cv2.resize(blue, (144,144))\n    else:\n        return np.dstack((cv2.resize(red,(144,144)),\n                          cv2.resize(yellow,(144,144)) ,cv2.resize(blue,(144, 144))))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.Label = data.Label.apply(lambda x: list(map(int,x.split('|'))))\ndata.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#to visualise classes, will take images with single class labels. \n\nsingle_label = []\nfor label in range(19):\n    for idx in range(len(data)):\n        if data.loc[idx, 'Label'] == [label]:\n            single_label.append(idx)\n            break\ntitles_2 = [\n'0-Nucleoplasm',\n'1-Nuclear membrane',\n'2-Nucleoli',\n'3-Nucleoli fibrillar center',\n'4-Nuclear speckles',\n'5-Nuclear bodies',\n'6-Endoplasmic reticulum',\n'7-Golgi apparatus',\n'8-Intermediate filaments',\n'9-Actin filaments',\n'10-Microtubules',\n'11-Mitotic spindle',\n'12-Centrosome',\n'13-Plasma membrane',\n'14-Mitochondria',\n'15-Aggresome',\n'16-Cytosol',\n'17-Vesicles and punctate cytosolic patterns',\n'18-Negative'\n]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"single_label,data.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(19,1, figsize = (10, 100))\n\nfor label, idx in enumerate(single_label):\n    axs[label].imshow(to_rgb(idx))\n    axs[label].set_title(titles_2[label])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(data['ID'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Data Prepairing\n# img_set = []\n# for i in range(2000):\n#     img_set.append(to_rgb(i))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# len(img_set)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !mkdir spanwar","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for i in range(2000):\n#     name = './spanwar/' + data.loc[i,'ID'] + '.jpg'\n#     cv2.imwrite(name,img_set[i])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/hpa-cell-tiles-sample-balanced-dataset/cell_df.csv')\ndf.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfs = df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = [str(i) for i in range(19)]\nfor x in labels: df[x] = df['image_labels'].apply(lambda r: int(x in r.split('|')))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def get_labels(x):\n#     row = data[data['ID']==x]\n#     label = row['Label']\n#     return label\n# get_labels('5c27f04c-bb99-11e8-b2b9-ac1f6b6435d0')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !mkdir \"a\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install iterative-stratification","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from fastai import *\nfrom fastai.vision import *\nfrom fastai.vision.all import *","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fs_data = pd.DataFrame()\n# fs_data['image_path'] = data['image_path']+'.jpg'\n# fs_data['Label'] = data['Label']\n# fs_data = fs_data[0:2000]\n# fs_data.to_csv('a/train.csv', index= False)\n\nnfold = 5\nseed = 42\n\ny = dfs[labels].values\nX = dfs[['image_id', 'cell_id']].values\n\ndfs['fold'] = np.nan\n\nfrom iterstrat.ml_stratifiers import MultilabelStratifiedKFold\nmskf = MultilabelStratifiedKFold(n_splits=nfold, random_state=seed)\nfor i, (_, test_index) in enumerate(mskf.split(X, y)):\n    dfs.iloc[test_index, -1] = i\n    \ndfs['fold'] = dfs['fold'].astype('int')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfs['is_valid'] = False\ndfs['is_valid'][dfs['fold'] == 4] = True","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfs","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# path_fs = \"a\"\n# folder = \"spanwar\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pd.read_csv('a/train.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pip install fastai","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def get_x(r): return r['image_path']+'.jpg'\n# def get_y(r): return r['Label']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = Path('../input/hpa-cell-tiles-sample-balanced-dataset')\ndef get_x(r): return path/'cells'/(r['image_id']+'_'+str(r['cell_id'])+'.jpg')\ndef get_y(r): return r['image_labels'].split('|')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pip install fastai==1.0.61","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_stats = ([0.07237246, 0.04476176, 0.07661699], [0.17179589, 0.10284516, 0.14199627])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"item_tfms = RandomResizedCrop(224, min_scale=0.75, ratio=(1.,1.))\nbatch_tfms = [*aug_transforms(flip_vert=True, size=128, max_warp=0), Normalize.from_stats(*sample_stats)]\nbs=256","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dblock = DataBlock(blocks=(ImageBlock, MultiCategoryBlock(vocab=titles_2)),\n                splitter=ColSplitter(col='is_valid'),\n                get_x=get_x,\n                get_y=get_y,\n                item_tfms=item_tfms,\n                batch_tfms=batch_tfms\n                )\ndls = dblock.dataloaders(dfs, bs=bs)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls=ImageDataLoaders.from_df(fs_data, path='.', valid_pct=0.2,\n                         fn_col=0, folder=None, label_col=1, label_delim=None, bs=64)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nsize = 64\nbatchsize = 32\ntfms = get_transforms(do_flip = True)\nimage_path_1 = './spanwar'\nsrc = (ImageList.from_csv(path_fs,'train.csv',folder)\n       .split_by_rand_pct(0.2).label_from_df(sep='|'))\n\ndata_1 = (src.transform(tfms, size=size, tfm_y=False)\n .databunch(bs=batchsize)\n .normalize(imagenet_stats, do_y = False))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"src","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_1.show_batch(rows=2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preds","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class_means = preds.mean(dim=0).numpy()\n# labels = range(19)\n# plt.bar(labels, class_means)\n# plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# row_max = preds.max(dim=-1).values.numpy()\n# plt.hist(row_max, bins=100)\n# plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fig = plt.figure(figsize=(16, 12))\n\n# for i in labels:\n#     ax = fig.add_subplot(5,4,i+1)\n#     ax.hist(preds[:,i].numpy(), bins=100)\n#     ax.set_title(i)\n\n# plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cell_df = pd.read_csv('../input/fastai-cell-tile-prototyping-3/cell_df.csv')\n# cell_df.head()\n# cell_df['cls'] = ''","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# threshold = 0.0\n\n# for i in range(preds.shape[0]): \n#     p = torch.nonzero(preds[i] > threshold).squeeze().numpy().tolist()\n#     if type(p) != list: p = [p]\n#     if len(p) == 0: cls = [(preds[i].argmax().item(), preds[i].max().item())]\n#     else: cls = [(x, preds[i][x].item()) for x in p]\n#     cell_df['cls'].loc[i] = cls","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def combine(r):\n#     cls = r[0]\n#     enc = r[1]\n#     classes = [str(c[0]) + ' ' + str(c[1]) + ' ' + enc for c in cls]\n#     return ' '.join(classes)\n\n# combine(cell_df[['cls', 'enc']].loc[24])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cell_df['pred'] = cell_df[['cls', 'enc']].apply(combine, axis=1)\n# cell_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# subm = cell_df.groupby(['image_id'])['pred'].apply(lambda x: ' '.join(x)).reset_index()\n# # subm = subm.loc[3:]\n# subm.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sample_submission = pd.read_csv('../input/hpa-single-cell-image-classification/sample_submission.csv')\n# sample_submission.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sub = pd.merge(\n#     sample_submission,\n#     subm,\n#     how=\"left\",\n#     left_on='ID',\n#     right_on='image_id',\n# )\n# sub.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def isNaN(num):\n#     return num != num\n\n# for i, row in sub.iterrows():\n#     if isNaN(row['pred']): continue\n#     sub.PredictionString.loc[i] = row['pred']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sub = sub[sample_submission.columns]\n# sub.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sub.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}