{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":23823,"databundleVersionId":1920183,"sourceType":"competition"},{"sourceId":1988985,"sourceType":"datasetVersion","datasetId":1189227},{"sourceId":2091935,"sourceType":"datasetVersion","datasetId":1254377},{"sourceId":2112880,"sourceType":"datasetVersion","datasetId":1267656},{"sourceId":2144980,"sourceType":"datasetVersion","datasetId":1287115},{"sourceId":2173797,"sourceType":"datasetVersion","datasetId":1304958},{"sourceId":2203982,"sourceType":"datasetVersion","datasetId":1323533},{"sourceId":2214479,"sourceType":"datasetVersion","datasetId":1329486},{"sourceId":2326972,"sourceType":"datasetVersion","datasetId":1364159}],"dockerImageVersionId":30097,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Description :**\n\nThis work was inspired by theses great kernels :\n- **tito** which gave me the opportunity to learn more about MMDetection\n    - https://www.kaggle.com/its7171/mmdetection-for-segmentation-training\n    - https://www.kaggle.com/its7171/mmdetection-for-segmentation-inference\n- **Rai** which illustrate perceptual hashing algorithm in order to find duplicate/similar images\n    - https://www.kaggle.com/rai555/hpa-duplicate-images-in-train","metadata":{}},{"cell_type":"markdown","source":"## Table of contents :\n* [I - Load dataset](#first)\n    * [1 - Global variables](#first_1)\n    * [2 - Read train dataset](#first_2)\n    * [3 - Duplicate labels](#first_3)\n    * [4 - Feature engineering](#first_4)\n* [II - EDA](#second)\n    * [1 - Labels](#second_1)\n        * [1.1 - Total labels](#second_1_1)\n        * [1.2 - Label count per cell](#second_1_2)\n        * [1.3 - Label occurrence](#second_1_3)\n        * [1.4 - Label combinations](#second_1_4)\n        * [1.5 - Co-occurrence matrix](#second_1_5)\n        * [1.6 - Display single label filters](#second_1_6)\n    * [2 - Single cells](#second_2)\n        * [2.1 - Distribution](#second_2_1)\n        * [2.2 - Heterogeneity](#second_2_2)\n    * [3 - Masks](#second_3)\n        * [3.1 - Generate masks with HPA Cell segmentator (example with random label)](#second_3_1)\n        * [3.2 - Generate masks with HPA Cell segmentator (all data)](#second_3_2)\n        * [3.3 - Evaluate segmentation quality](#second_3_3)\n    * [4 - Perceptual hashing](#second_4)\n* [III - Modeling](#third)\n    * [1 - Preprocessing (prepare training data) ](#third_1)\n    * [2 - HTC+](#third_2)\n        \n* [IV - Helpers](#fourth)\n* [V - Drafts](#sixth)","metadata":{}},{"cell_type":"code","source":"# # Install MMDetection environment\n!pip install mmdet\nimport torch\ntorch_version = f'torch{torch.__version__[:3]}'\ncuda_version = 'cu{}'.format(torch.version.cuda.replace('.', '')[:3])\nprint(f'PyTorch version : {torch_version}\\nCUDA version : {cuda_version}')\n!pip install mmcv-full -f https://download.openmmlab.com/mmcv/dist/{cuda_version}/{torch_version}/index.html\n# # Install Apex\n!git clone https://github.com/NVIDIA/apex\n!mv ./apex/ ./apex1\n!cp -r ./apex1/* ./ && rm -rf ./apex1\n!pip install -v --no-cache-dir --global-option=\"--cpp_ext\" --global-option=\"--cuda_ext\" ./\n!find ./ -mindepth 1 ! -regex '^./apex\\(/.*\\)?' -delete\n# Install MMDetection Swin Transformer version\n!git clone https://github.com/SwinTransformer/Swin-Transformer-Object-Detection.git\n!pip install -r ./Swin-Transformer-Object-Detection/requirements.txt\n!mv ./Swin-Transformer-Object-Detection/* ./ && rm -rf ./Swin-Transformer-Object-Detection/\n!pip install -v -e ./\n# Get pretrained Swin Transformer backbones\n# !cp ../input/swin-weights/* ./\n\n# # Install custom package (+ additional packages)\n\n!mkdir src train_cell_masks train_nuclei_masks logs hpa_train_jpg hpa_pix2pix_cell_mask\n!cp -r ../input/hpa-src/src/* ../working/src/\n!pip install --upgrade pip\n!pip install -r /kaggle/working/src/requirements.txt\n!pip install pandarallel scikit-multilearn torchsummary # imagededup instaboostfast \n# !pip install https://github.com/CellProfiling/HPA-Cell-Segmentation/archive/master.zip","metadata":{"execution":{"iopub.status.busy":"2021-06-12T23:40:42.403472Z","iopub.execute_input":"2021-06-12T23:40:42.403866Z","iopub.status.idle":"2021-06-12T23:57:17.778798Z","shell.execute_reply.started":"2021-06-12T23:40:42.403785Z","shell.execute_reply":"2021-06-12T23:57:17.777804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport json\nimport tempfile\nimport zipfile\nimport multiprocessing as mp\nimport random\nimport functools\nimport scipy\nimport PIL\nimport cv2\nimport skimage\nimport sklearn\nimport warnings\n\n# Find similar/duplicate images with Perceptual hashing\n# https://en.wikipedia.org/wiki/Perceptual_hashing\n# from imagededup.methods import PHash\n\n# Pandas multiprocessing\nfrom pandarallel import pandarallel\n\n# PyTorch ecosystem\nimport torch\nimport torchvision\nimport torchvision.transforms as T\nimport torchvision.transforms.functional as F\nfrom torchsummary import summary\n\n# MMdetection\nimport mmcv\nfrom mmcv import Config\nfrom mmdet.apis import *\nfrom mmdet.datasets import build_dataset, PIPELINES \nfrom mmdet.datasets.pipelines.auto_augment import AutoAugment\nfrom mmdet.models import build_detector, LOSSES\n\n# # COCO\nfrom pycocotools import mask as coco_mask\nfrom pycocotools.coco import COCO\n\nfrom tqdm.notebook import tqdm\n\n# Stratify multiclass & multi-label train & test sets\nfrom sklearn.model_selection import train_test_split\nfrom skmultilearn.model_selection import iterative_train_test_split\n\n# Custom EDA modules \nfrom src.datacleaner import *\nfrom src.analyzer.univariate import *\nfrom src.analyzer.multivariate import *\nfrom src.analyzer.univariate import *\nfrom src.evaluator import pickle_data\n\n# HPA Cell segmentation\n# from hpacellseg import cellsegmentator\n# from hpacellseg.utils import label_cell, label_nuclei\n\nwarnings.filterwarnings('ignore')\n\npandarallel.initialize(progress_bar=True)","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2021-06-13T00:10:45.867971Z","iopub.execute_input":"2021-06-13T00:10:45.868339Z","iopub.status.idle":"2021-06-13T00:11:05.589844Z","shell.execute_reply.started":"2021-06-13T00:10:45.868303Z","shell.execute_reply":"2021-06-13T00:11:05.589024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# I - Load dataset <a class=\"anchor\" id=\"first\"></a>","metadata":{}},{"cell_type":"markdown","source":"### 1 - Global variables <a class=\"anchor\" id=\"first_1\"></a>","metadata":{}},{"cell_type":"code","source":"# Main variables\n\nTRAIN_IMG_PATH = '../input/hpa-single-cell-image-classification/train/'\nTRAIN_JPG_IMG_PATH = '../input/hpa-jpg/hpa_train/hpa_train_jpg/'\n\nTRAIN_CELL_MASK_PATH = './train_cell_masks/' # '../input/hpa-mask/hpa_cell_mask'\n# './train_cell_masks/'\nTRAIN_NUCLEI_MASK_PATH = './train_nuclei_masks/' # '../input/hpa-mask/hpa_nuclei_mask'\n# './train_nuclei_masks/'\n\nIMG_FILTERS = ('_green.png', '_blue.png', '_red.png', '_yellow.png')\n\nIMG_FILTER_LABELS = ('protein_of_interest', 'nucleus', 'microtubules', 'endoplasmic_reticulum')\n\n# Labels mapper\nRAW_LABELS_DICT = {0: 'Nucleoplasm',\n                   1: 'Nuclear membrane',\n                   2: 'Nucleoli',\n                   3: 'Nucleoli fibrillar center',\n                   4: 'Nuclear speckles',\n                   5: 'Nuclear bodies',\n                   6: 'Endoplasmic reticulum',\n                   7: 'Golgi apparatus',\n                   8: 'Intermediate filaments',\n                   9: 'Actin filaments',\n                   10: 'Microtubules',\n                   11: 'Mitotic spindle',\n                   12: 'Centrosome',\n                   13: 'Plasma membrane',\n                   14: 'Mitochondria',\n                   15: 'Aggresome',\n                   16: 'Cytosol',\n                   17: 'Vesicles and punctate cytosolic patterns',\n                   18: 'Negative'\n                  }\n\n# Assign index label\nLABELS_DICT = {n: f'({n}) {label}' for n, label in RAW_LABELS_DICT.items()}","metadata":{"execution":{"iopub.status.busy":"2021-06-13T00:11:05.592869Z","iopub.execute_input":"2021-06-13T00:11:05.593127Z","iopub.status.idle":"2021-06-13T00:11:05.601391Z","shell.execute_reply.started":"2021-06-13T00:11:05.593096Z","shell.execute_reply":"2021-06-13T00:11:05.60066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2 - Read train dataset <a class=\"anchor\" id=\"first_2\"></a>","metadata":{}},{"cell_type":"code","source":"original_train_dataset_path = '../input/hpa-single-cell-image-classification/train.csv'\n# Load train dataset from EDA backup\neda_backup_train_dataset_path  = '../input/hpa-train-csv/hpa_train.csv' \ndf = pd.read_csv(eda_backup_train_dataset_path)\n# Extract labels (str -> list)\ndf['LabelList'] = df['Label'].map(lambda x: list(map(int, x.split('|'))))\ndf['JPGImagePath'] = df['ImagePath'].map(lambda path: TRAIN_JPG_IMG_PATH \\\n                                         + path.split('/')[-1].replace('png', 'jpg'))\n# Get total images \ntotal_images = df.shape[0]\nprint(f'Total images : {read_int_cleaner(total_images)}')\ndf.head(2)","metadata":{"execution":{"iopub.status.busy":"2021-06-06T16:08:02.516512Z","iopub.execute_input":"2021-06-06T16:08:02.51702Z","iopub.status.idle":"2021-06-06T16:08:02.766886Z","shell.execute_reply.started":"2021-06-06T16:08:02.516969Z","shell.execute_reply":"2021-06-06T16:08:02.766182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3 - Duplicate labels <a class=\"anchor\" id=\"first_3\"></a>","metadata":{}},{"cell_type":"code","source":"# Check duplicate labels\ndf['DuplicateLabel'] = df.LabelList.map(lambda l: (len(set(l)) != len(l)) if len(l) > 1 else False)\n# Get all images with duplicate labels\ndf_duplicate_labels = df[df['DuplicateLabel']]\n# Total images with duplicate labels\ntotal_duplicate_labels = df_duplicate_labels.shape[0]\n# Get duplicate labels\nduplicate_labels = set([duplicate_label for labels in df_duplicate_labels['LabelList'] \n                        for duplicate_label in set([label for label in labels if labels.count(label) > 1])])\n# Get duplicate labels as string\nduplicate_labels_str = ', '.join([LABELS_DICT[n] for n in sorted(duplicate_labels)])\nprint('{} duplicate labels\\nConcerns {} labels'.format(total_duplicate_labels,\n                                                        duplicate_labels_str))\nprint(df[df['DuplicateLabel']].head(2))\n# Remove duplicate classes\ndf['LabelList'] = df['LabelList'].map(lambda l: list(set(l)))\ndf['Label'] = df['LabelList'].map(lambda l: '|'.join(map(str, l)) if len(l) > 1 else ''.join(map(str, l)))\ndf.drop('DuplicateLabel',  axis=1, inplace=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4 - Feature engineering <a class=\"anchor\" id=\"first_4\"></a>","metadata":{}},{"cell_type":"code","source":"%%time\n\n# Get image filepath\ndf['ImagePath'] = TRAIN_IMG_PATH + df['ID'] + '_green.png'\n# Get mask filepath\ndf['MaskPath'] = df['ID'].map(lambda img_id: f'{TRAIN_CELL_MASK_PATH}/{img_id}.npz')\n# Map organelle names from int labels\ndf['LabelStr'] = df['LabelList'].map(lambda x: '\\n'.join([LABELS_DICT[n] for n in x]))\n# Get total labels per image\ndf['TotalLabels'] = df['LabelList'].map(len)\n# Get total cells count per mask\ndf['TotalCellsPerMask'] = df['MaskPath'].parallel_apply(get_total_cells_from_mask) # Expensive (16 min)\n# Re-order columns\ndf = df[['ID',\n         'ImagePath',\n         'MaskPath',\n         'Label',\n         'LabelList',\n         'LabelStr',\n         'TotalLabels',\n         'TotalCellsPerMask'\n        ]]\ndf.head(2)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_cols = ['ImagePath', 'MaskPath']","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****Save image files as jpg****","metadata":{}},{"cell_type":"code","source":"%%time\n\n\nhpa_train = df.ImagePath.tolist()\n\n\ndef convert_png_to_jpg(filepath, ext_path=None, quality_thr=95):\n    file = cv2.imread(filepath)\n    if ext_path is None:\n        filename = filepath.replace('png', 'jpg')\n    else:\n        filename = filepath.split('/')[-1].replace('png', 'jpg')\n        filepath = ext_path + filename\n    file = cv2.resize(file, (512, 512), interpolation=cv2.INTER_NEAREST)\n    cv2.imwrite(filepath, file, [int(cv2.IMWRITE_JPEG_QUALITY), quality_thr])\n    return filepath\n\n\nwith zipfile.ZipFile('hpa_train.zip', 'w') as zip_: \n    for filepath in hpa_train:\n        jpg_filepath = convert_png_to_jpg(filepath, './hpa_train_jpg/')\n        zip_.write(jpg_filepath, compress_type=zipfile.ZIP_DEFLATED)\n        os.remove(jpg_filepath)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# II - EDA <a class=\"anchor\" id=\"second\"></a>","metadata":{}},{"cell_type":"markdown","source":"## 1 - Labels <a class=\"anchor\" id=\"second_1\"></a>","metadata":{}},{"cell_type":"markdown","source":"### 1.1 - Total labels <a class=\"anchor\" id=\"second_1_1\"></a>","metadata":{}},{"cell_type":"code","source":"# Total labels (18 + negative class)\ntotal_labels = set([label for label_list in df['LabelList'] for label in label_list])\nprint(f'Total labels : {len(total_labels)}')","metadata":{"execution":{"iopub.status.busy":"2021-06-06T15:46:47.000065Z","iopub.execute_input":"2021-06-06T15:46:47.000456Z","iopub.status.idle":"2021-06-06T15:46:47.020776Z","shell.execute_reply.started":"2021-06-06T15:46:47.000423Z","shell.execute_reply":"2021-06-06T15:46:47.019625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.2 - Label count per cell <a class=\"anchor\" id=\"second_1_2\"></a>","metadata":{}},{"cell_type":"code","source":"# Label frequency\ncount_plot(df=df,\n           variable='TotalLabels',\n           xlabel='Effectif',\n           ylabel='Label Frequency',\n           sort=True,\n           precision=2,\n           bar_type='h',\n           plotsize=(8, 4))\n\n# 88.69 % of image cells have at most 2 organelles","metadata":{"execution":{"iopub.status.busy":"2021-06-06T15:46:49.943518Z","iopub.execute_input":"2021-06-06T15:46:49.944034Z","iopub.status.idle":"2021-06-06T15:46:50.194744Z","shell.execute_reply.started":"2021-06-06T15:46:49.943983Z","shell.execute_reply":"2021-06-06T15:46:50.194073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Outliers \nlabel_freq_outliers = {n: df[df['TotalLabels'] == n] for n in range(3, 6)}\nlabel_freq_outliers[5]","metadata":{"execution":{"iopub.status.busy":"2021-06-06T15:47:17.906794Z","iopub.execute_input":"2021-06-06T15:47:17.907312Z","iopub.status.idle":"2021-06-06T15:47:17.937158Z","shell.execute_reply.started":"2021-06-06T15:47:17.907278Z","shell.execute_reply":"2021-06-06T15:47:17.935759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_label_mode(label_freq_outliers[4])","metadata":{"execution":{"iopub.execute_input":"2021-06-01T14:31:45.477075Z","iopub.status.busy":"2021-06-01T14:31:45.476723Z","iopub.status.idle":"2021-06-01T14:31:45.485641Z","shell.execute_reply":"2021-06-01T14:31:45.484522Z","shell.execute_reply.started":"2021-06-01T14:31:45.477044Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.3 - Label occurrence <a class=\"anchor\" id=\"second_1_3\"></a>","metadata":{}},{"cell_type":"code","source":"# Get occurrence matrix\nmulti_label_bin_matrix = binary_matrix(df, 'LabelList')\nlabel_occurrence = pd.DataFrame(multi_label_bin_matrix.sum(axis=0), columns=['Occurrence'])\n# Sort labels by occurrence\n# label_occurrence.sort_values(by='Occurrence', axis=0, ascending=False, inplace=True)\n# Rename labels (int to str) \nlabel_occurrence.index = label_occurrence.index.map(LABELS_DICT)\n# Plot labels occurrence\nplot_label_occurrence(label_occurrence, total_images, precision=1) # ERROR WITH GPU\n\n# Sort label occurrence by label number\n# label_occurrence = label_occurrence.reindex(sorted(label_occurrence.index,\n#                                                    key=lambda x: {v: k for k, v in LABELS_DICT.items()}[x]))\n# label_occurrence.plot(kind='barh');plt.gca().invert_yaxis()","metadata":{"execution":{"iopub.status.busy":"2021-06-06T15:52:13.574682Z","iopub.execute_input":"2021-06-06T15:52:13.575038Z","iopub.status.idle":"2021-06-06T15:52:18.251097Z","shell.execute_reply.started":"2021-06-06T15:52:13.574987Z","shell.execute_reply":"2021-06-06T15:52:18.250102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Skewness (γ1) = \", label_occurrence['Occurrence'].skew())\nprint(\"Kurtosis (γ2) = \", label_occurrence['Occurrence'].kurtosis())","metadata":{"execution":{"iopub.status.busy":"2021-06-06T15:53:26.386835Z","iopub.execute_input":"2021-06-06T15:53:26.387289Z","iopub.status.idle":"2021-06-06T15:53:26.399341Z","shell.execute_reply.started":"2021-06-06T15:53:26.38725Z","shell.execute_reply":"2021-06-06T15:53:26.397927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Nucleoplasm (0) -> most common label\n# Mitotic spindle (11) and Aggresome (15) -> underpopulated","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.4 - Label combinations <a class=\"anchor\" id=\"second_1_4\"></a>","metadata":{}},{"cell_type":"code","source":"# Many different proteins come together at a specific location to perform a task,\n# proteins combinations are important because the exact outcome of this task \n# depends on which proteins are present. \n\ntotal_organelles_combinations = len(df.Label.unique().tolist())\nprint(f'Total organelles combinations : {total_organelles_combinations}')\nplot_n_top_values(df, 'Label', n=25, plot_size=(12, 8), precision=1)","metadata":{"execution":{"iopub.status.busy":"2021-06-06T15:53:30.625747Z","iopub.execute_input":"2021-06-06T15:53:30.626228Z","iopub.status.idle":"2021-06-06T15:53:31.113994Z","shell.execute_reply.started":"2021-06-06T15:53:30.626189Z","shell.execute_reply":"2021-06-06T15:53:31.11293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.5 - Co-occurrence matrix <a class=\"anchor\" id=\"second_1_5\"></a>","metadata":{}},{"cell_type":"code","source":"# Build & plot co-occurrence matrix\nmulti_label_bin_matrix_df = pd.DataFrame(multi_label_bin_matrix).rename(LABELS_DICT, axis=1)\ncorrelation_matrix(multi_label_bin_matrix_df, size=(12, 12))","metadata":{"execution":{"iopub.status.busy":"2021-06-06T15:53:38.600171Z","iopub.execute_input":"2021-06-06T15:53:38.600586Z","iopub.status.idle":"2021-06-06T15:53:39.949651Z","shell.execute_reply.started":"2021-06-06T15:53:38.60055Z","shell.execute_reply":"2021-06-06T15:53:39.948968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get top 5 co-occurrences\nget_top_n_correlations(multi_label_bin_matrix_df, abs_corr=False, n=5)","metadata":{"execution":{"iopub.execute_input":"2021-06-01T14:32:37.487973Z","iopub.status.busy":"2021-06-01T14:32:37.48763Z","iopub.status.idle":"2021-06-01T14:32:37.529907Z","shell.execute_reply":"2021-06-01T14:32:37.528843Z","shell.execute_reply.started":"2021-06-01T14:32:37.487943Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.6 - Display single label filters <a class=\"anchor\" id=\"second_1_6\"></a>","metadata":{}},{"cell_type":"code","source":"# all_single_labels = df_train[df_train['Label'].isin(set(map(str, total_labels)))].index.tolist()\n\nrandom_single_label_ids = [get_random_single_label_id(df, str(int_label)) for int_label in total_labels]\n\nfor random_single_label_id in random_single_label_ids:\n    display_image_sample(random_single_label_id, data=df, plot_size=(12, 4))","metadata":{"execution":{"iopub.execute_input":"2021-06-01T14:32:45.618576Z","iopub.status.busy":"2021-06-01T14:32:45.618164Z","iopub.status.idle":"2021-06-01T14:33:40.219399Z","shell.execute_reply":"2021-06-01T14:33:40.218492Z","shell.execute_reply.started":"2021-06-01T14:32:45.618536Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Nuclear membrane (1)\ndisplay_image_filters(random_single_label_ids[1], image_filter='stacked_filters', size=(14, 14))","metadata":{"execution":{"iopub.execute_input":"2021-06-01T14:36:24.023268Z","iopub.status.busy":"2021-06-01T14:36:24.022877Z","iopub.status.idle":"2021-06-01T14:36:25.341531Z","shell.execute_reply":"2021-06-01T14:36:25.340239Z","shell.execute_reply.started":"2021-06-01T14:36:24.023235Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2 - Single Cells <a class=\"anchor\" id=\"second_2\"></a>","metadata":{}},{"cell_type":"markdown","source":"### 2.1 - Distribution <a class=\"anchor\" id=\"second_2_1\"></a>","metadata":{}},{"cell_type":"code","source":"df['TotalCellsPerMask'].hist()\nmeasures_table(df, 'TotalCellsPerMask').T.astype(int)","metadata":{"execution":{"iopub.execute_input":"2021-06-01T16:03:54.049928Z","iopub.status.busy":"2021-06-01T16:03:54.04945Z","iopub.status.idle":"2021-06-01T16:03:54.271599Z","shell.execute_reply":"2021-06-01T16:03:54.270318Z","shell.execute_reply.started":"2021-06-01T16:03:54.049894Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check outliers\ndf[df.TotalCellsPerMask == 0]","metadata":{"execution":{"iopub.execute_input":"2021-06-01T14:36:41.665664Z","iopub.status.busy":"2021-06-01T14:36:41.665265Z","iopub.status.idle":"2021-06-01T14:36:41.682686Z","shell.execute_reply":"2021-06-01T14:36:41.681622Z","shell.execute_reply.started":"2021-06-01T14:36:41.665622Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_image_by_id('5fb643ee-bb99-11e8-b2b9-ac1f6b6435d0',\n                     image_filter='protein_of_interest',\n                     plot_mask=True,\n                     size=(12, 12))\n\n# Remove outliers\ndf = df[df.TotalCellsPerMask != 0]\ndf.reset_index(inplace=True, drop=True)\n# N.B : it seems some organelles are hard to mask (e.g : mitochondria)\n# we need to evaluate mask precision and potentially improve mask generation method","metadata":{"execution":{"iopub.execute_input":"2021-06-01T14:36:48.099319Z","iopub.status.busy":"2021-06-01T14:36:48.098775Z","iopub.status.idle":"2021-06-01T14:36:49.825134Z","shell.execute_reply":"2021-06-01T14:36:49.823981Z","shell.execute_reply.started":"2021-06-01T14:36:48.099286Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2.2 - Heterogeneity <a class=\"anchor\" id=\"second_2_2\"></a>","metadata":{}},{"cell_type":"markdown","source":"#### 2.2.1 - Single label example","metadata":{}},{"cell_type":"code","source":"single_cell_ids_with_identical_single_labels = df['ID'][df.Label == '0'].head(8).tolist()\n\ndisplay_image_by_id(single_cell_ids_with_identical_single_labels,\n                    image_filter='protein_of_interest',\n                    plot_mask=False,\n                    shape=(2, 4),\n                    adjust=True,\n                    adjust_h=-0.5,\n                    adjust_w=0.5,\n                    size=(16, 16))","metadata":{"execution":{"iopub.execute_input":"2021-06-01T14:38:22.216122Z","iopub.status.busy":"2021-06-01T14:38:22.215677Z","iopub.status.idle":"2021-06-01T14:38:30.608418Z","shell.execute_reply":"2021-06-01T14:38:30.60761Z","shell.execute_reply.started":"2021-06-01T14:38:22.21607Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 2.2.2 - Multi-label example","metadata":{}},{"cell_type":"code","source":"single_cell_ids_with_identical_multiple_labels = df['ID'][df.Label == '0|14'].head(8).tolist()\n\ndisplay_image_by_id(single_cell_ids_with_identical_multiple_labels,\n                    image_filter='protein_of_interest',\n                    plot_mask=False,\n                    shape=(2, 4),\n                    adjust=True,\n                    adjust_h=-0.5,\n                    adjust_w=0.5,\n                    size=(16, 16))","metadata":{"execution":{"iopub.execute_input":"2021-06-01T14:38:38.292777Z","iopub.status.busy":"2021-06-01T14:38:38.292256Z","iopub.status.idle":"2021-06-01T14:38:45.397281Z","shell.execute_reply":"2021-06-01T14:38:45.396243Z","shell.execute_reply.started":"2021-06-01T14:38:38.292743Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3 - Masks  <a class=\"anchor\" id=\"second_3\"></a>","metadata":{}},{"cell_type":"markdown","source":"#### 3.1 - Generate masks with HPA Cell segmentator (example with random label) <a class=\"anchor\" id=\"second_3_1\"></a>","metadata":{}},{"cell_type":"code","source":"# Get a sample of training data\ndf_label_sample = df.sample(1)\n\nprint(\"Protein of interest : \\n{}\".format(df_label_sample['LabelStr'].iloc[0]))\n\nrandom_label_ids = df_label_sample['ID'].tolist()\n\ncell_segmentator_data = generate_cell_segmentator_data(random_label_ids)\nrandom_label_image_data = generate_cell_masks(cell_segmentator_data, plot=True, verbose=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 3.2 - Generate masks with HPA Cell segmentator (all data) <a class=\"anchor\" id=\"second_3_2\"></a>","metadata":{}},{"cell_type":"code","source":"# %%time\n\n# Get train image labels\nlabel_ids = df['ID'].tolist()\nTOTAL_CELLS = len(label_ids)\n\n# Build cell segmentation model with HPA package\ncell_segmentator_model = cellsegmentator.CellSegmentator(nuclei_model='.nuclei_model.pth',\n                                                         cell_model='.cell_model.pth',\n                                                         scale_factor=0.25,\n                                                         device='cuda',\n                                                         padding=False,\n                                                         multi_channel_model=True)\n\n\ndef generate_cell_masks_multiprocess(label_id):\n    cell_segmentator_data = generate_cell_segmentator_data([label_id])\n    # Generate image mask\n    generate_cell_masks(cell_segmentator_data,\n                        cell_segmentator_model,\n                        total_cells=TOTAL_CELLS,\n                        save=True,\n                        verbose=False)\n    return \n    \nwith mp.Pool(processes=mp.cpu_count()) as p:\n    res = list(tqdm(p.imap(generate_cell_masks_multiprocess, label_ids), total=TOTAL_CELLS))\n    \n# Alternative : Add dataset from https://www.kaggle.com/its7171/hpa-mask","metadata":{"execution":{"iopub.status.busy":"2021-06-06T16:15:41.437572Z","iopub.execute_input":"2021-06-06T16:15:41.438167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# TO DO : Nb of boxes vs images\n# https://www.kaggle.com/gocoding/torchvision-faster-r-cnn-finetuning","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 3.3 - Evaluate segmentation quality <a class=\"anchor\" id=\"second_3_3\"></a>","metadata":{}},{"cell_type":"code","source":"df[df['TotalCellsPerMask'] == 1]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_image_by_id(df.ID.iloc[12583],\n                    image_filter='protein_of_interest',\n                    plot_mask=True,\n                    shape=(2, 4),\n                    adjust=True,\n                    adjust_h=-0.5,\n                    adjust_w=0.5,\n                    size=(16, 16))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ndf['MaskQuality'] = df[path_cols].parallel_apply(lambda x: compute_segmentation_quality(x[0],\n                                                                                        x[1]), axis=1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask_quality_table = measures_table(df, 'MaskQuality').T\nmask_quality_table","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[df['MaskQuality'] <= 0.05]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_image_by_id(df.ID.iloc[20649], # 20649\n                    image_filter='stacked_filters',\n                    plot_mask=True,\n                    shape=(2, 4),\n                    adjust=True,\n                    adjust_h=-0.5,\n                    adjust_w=0.5,\n                    size=(16, 16))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['TotalBBoxes'] = df['TotalLabels'] * df['TotalCellsPerMask']\n\ndf['TotalBBoxes'].hist()\nmeasures_table(df, 'TotalBBoxes').T.astype(int)\nprint('Total Single Cell Bounding Boxes : ', read_int_cleaner(np.sum(df['TotalBBoxes'])))","metadata":{"execution":{"iopub.execute_input":"2021-06-01T14:35:46.546518Z","iopub.status.busy":"2021-06-01T14:35:46.546103Z","iopub.status.idle":"2021-06-01T14:35:46.567681Z","shell.execute_reply":"2021-06-01T14:35:46.566418Z","shell.execute_reply.started":"2021-06-01T14:35:46.546482Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# EDA dataset backup\n# df.to_csv('hpa_train.csv', index=False)","metadata":{"execution":{"iopub.execute_input":"2021-06-01T14:50:02.917889Z","iopub.status.busy":"2021-06-01T14:50:02.917509Z","iopub.status.idle":"2021-06-01T14:50:03.13995Z","shell.execute_reply":"2021-06-01T14:50:03.138566Z","shell.execute_reply.started":"2021-06-01T14:50:02.917858Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4 - Perceptual hashing <a class=\"anchor\" id=\"second_4\"></a>","metadata":{}},{"cell_type":"code","source":"# Find similar/duplicate images with perceptual hashing\n# see : https://www.kaggle.com/rai555/hpa-duplicate-images-in-train\n\n\ndef find_similar_images(build_encodings=False,\n                        img_dir='../input/hpa-single-cell-image-classification/train/',\n                        encodings_filename='train_similar_image_encodings',\n                        distance_thr=10,\n                        to_df=True):\n    phash = PHash()\n    if build_encodings:\n        # Encode images\n        encodings = phash.encode_images(image_dir=img_dir)\n        # Only keep green filter\n        encodings = {img_file: hash_str for img_file, hash_str in encodings.items() if '_green' in img_file}\n        pickle_data(filename=encodings_filename, data=encodings, method='w')\n    else:\n        encodings = pickle_data(filename=encodings_filename, method='r')\n    # Find similar/duplicate images based on maximum distance threshold\n    similar_images = phash.find_duplicates(encoding_map=encodings,\n                                           scores=True,\n                                           max_distance_threshold=distance_thr)\n    # Filter null results\n    similar_images = {img1: results for img1, results in similar_images.items() if len(results) > 0}\n    # Convert results as dataframe\n    if to_df:\n        duplicates_list = []\n        scores = []\n        for image1, results in similar_images.items():\n            for candidate_and_score in results:\n                image_n, score = candidate_and_score\n                if image1 != image_n and ((image1, image_n) not in duplicates_list) \\\n                                     and ((image_n, image1) not in duplicates_list):\n                    duplicates_list.append((image1, image_n))\n                    scores.append(score)\n        similar_images = pd.DataFrame(duplicates_list, columns=['img1', 'img2'])\n        similar_images['score'] = scores\n    return similar_images\n\ntrain_encodings_filename = '../input/train-img-phash-encodings/train_similar_image_encodings'\n\nsimilar_images_df = find_similar_images(encodings_filename=train_encodings_filename,\n                                        distance_thr=10)\nsimilar_images_df.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get similar image files in order to use custom data augmentation during training process\nSIMILAR_IMG_FILES = similar_images_df['img1'].tolist()\n# pickle_data(filename='similar_train_img_files', data=SIMILAR_IMG_FILES, method='w')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# III - Modeling <a class=\"anchor\" id=\"third\"></a>","metadata":{}},{"cell_type":"markdown","source":"Strategy :\n\n1. Build **Mask R-CNN baseline**\n    - find best backbone between **ResNet50** and **Swin Transformer**\n2. Build best model (called **HTC+** which is inspired from **HTC++**)","metadata":{}},{"cell_type":"markdown","source":"## 1 - Preprocessing (prepare training data) <a class=\"anchor\" id=\"third_1\"></a>","metadata":{}},{"cell_type":"markdown","source":"****Custom dataset****","metadata":{}},{"cell_type":"code","source":"###########################################################################################\n#                                    CUSTOM DATASET                                       #\n###########################################################################################\n\n\nclass SingleCellDataset(torch.utils.data.Dataset):\n    \n    def __init__(self,\n                 imgs,\n                 masks,\n                 labels,\n                 img_size,\n                 transforms=None,\n                 iou_threshold=IOU_THRESHOLD,\n                 background_label=False):\n        # Image filepaths list\n        self.imgs = imgs\n        # Mask filepaths list\n        self.masks = masks\n        # Labels filepaths list (multilabel -> list of list)\n        self.labels = labels\n        # Images and masks size (tuple -> h x w)\n        self.img_size = img_size\n        # Image data-augmentation/transformation function\n        self.transforms = transforms\n        # IoU threshold between single cell from original image and mask\n        self.iou_threshold = iou_threshold\n        # Background label boolean value (if true increment negative class)\n        self.background_label = background_label\n        \n    def __getitem__(self, idx):\n        # Load image and mask filepaths\n        img_path = os.path.join(self.imgs[idx])\n        mask_path = os.path.join(self.masks[idx])\n        # Read image and mask as numpy arrays\n        img = cv2.imread(img_path)\n        mask = np.load(mask_path)['arr_0']\n        # Resize \n        if self.img_size is not None:\n            img, mask = [cv2.resize(i, self.img_size, interpolation=cv2.INTER_NEAREST) for i in [img, mask]]\n        # Load labels (single list or list of list if multiple labels)\n        labels = [self.labels[idx]] if type(self.labels[idx]) in (int, np.int64) else self.labels[idx]\n        # Get single cell ids from mask (first id is the background, so remove it)\n        total_single_cells = np.unique(mask)[1:]\n        boolean_masks = mask == total_single_cells[:, None, None]\n        # Target data (single cell masks, boxes, labels)\n        single_cell_masks = []\n        boxes = []\n        single_cell_labels = []\n        # Extract single cell data (mask, bounding box, label(s))\n        for cell_number in range(len(total_single_cells)):\n            #############################\n            #  EXTRACT SINGLE CELL MASK #\n            #############################\n            single_cell_mask = boolean_masks[cell_number]\n            #########################################\n            # EXTRACT SINGLE CELL MASK BOUNDING BOX #\n            #########################################\n            pos = np.where(single_cell_mask)\n            xmin, xmax = np.min(pos[1]), np.max(pos[1])\n            ymin, ymax = np.min(pos[0]), np.max(pos[0])\n            ###########################################\n            # EXTRACT SINGLE CELL FROM ORIGINAL IMAGE #\n            ###########################################\n            # Single cell binary mask\n            binary_mask = single_cell_mask.astype(int)[..., None]\n            single_cell_image = img * binary_mask\n            #################\n            # ASSIGN LABELS #\n            #################\n            # Compute IoU between single cell from original image and mask\n            iou_score = compute_IoU(single_cell_mask, single_cell_image[:, :, 0])\n            # Check if protein of interest image overlap single cell mask\n            if iou_score >= self.iou_threshold:\n                # Duplicate mask and bounding box for each labels (multilabel instance segmentation)\n                for label in labels:\n                    single_cell_labels.append(label)\n                    single_cell_masks.append(single_cell_mask)\n                    boxes.append([xmin, ymin, xmax, ymax])\n            # (if not, assign negative label)\n            else:\n                # print(iou_score)\n                single_cell_labels.append(19 if self.background_label else 18)\n                single_cell_masks.append(single_cell_mask)\n                boxes.append([xmin, ymin, xmax, ymax])\n        # Build target dictionary (convert everything into a torch.Tensor)\n        boxes = torch.as_tensor(boxes, dtype=torch.float32)\n        target = {}\n        target[\"image_id\"] = torch.tensor([idx])\n        target[\"masks\"] = torch.as_tensor(single_cell_masks, dtype=torch.uint8)\n        target[\"boxes\"] = boxes\n        target[\"labels\"] = torch.as_tensor(single_cell_labels, dtype=torch.int64) # int64\n        try:\n            target[\"area\"] = (boxes[:, 3] - boxes[:, 1]) * (boxes[:, 2] - boxes[:, 0])\n        except:\n            print(target)\n        # suppose all instances are not crowd\n        target[\"iscrowd\"] = torch.zeros((len(single_cell_masks),), dtype=torch.int64) # int64\n        # Apply image data augmentation/transformations\n        if self.transforms is not None:\n            img, target = self.transforms(img, target)\n        return img, target\n\n    def __len__(self):\n        return len(self.imgs)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****Build JSON annotations from custom dataset****","metadata":{}},{"cell_type":"code","source":"%%time\n\n# Build sample dataset for Mask R-CNN model\npath_cols = ['JPGImagePath', 'MaskPath', 'LabelList']\ndf_to_sample = df[path_cols].join(pd.DataFrame(binary_matrix(df, 'LabelList'))).reset_index()\nX = df_to_sample['index']\ny = df_to_sample.iloc[:, ~df_to_sample.columns.isin(path_cols + ['index'])]\n# Multilabel stratification\n# resource : http://scikit.ml/api/skmultilearn.model_selection.iterative_stratification.html\nX_train, y_train, X_test, y_test = iterative_train_test_split(X.values, y.values, 0.2)\n\ntrain_idx = [x[0] for x in X_train.tolist()]\ntest_idx = [x[0] for x in X_test.tolist()]\n\n\nds = SingleCellDataset(imgs=df['JPGImagePath'],\n                       masks=df['MaskPath'],\n                       labels=df['LabelList'],\n                       img_size=(512, 512))\n\n\ndef get_filename_by_idx(idx, filename_col='JPGImagePath'):\n    return df[filename_col].iloc[idx].split('/')[-1]\n\n\ndef convert_to_coco_api(idx, ds):\n    dataset = {'images': [], 'annotations': [], 'categories': []}\n    categories = set()\n    img, targets = ds[idx]\n    img = torchvision.transforms.functional.to_tensor(img)\n    image_id = targets[\"image_id\"].item()\n    img_dict = {}\n    img_dict['file_name'] = get_filename_by_idx(idx)\n    img_dict['id'] = image_id\n    img_dict['height'] = img.shape[-2]\n    img_dict['width'] = img.shape[-1]\n    dataset['images'].append(img_dict)\n    bboxes = targets[\"boxes\"]\n    bboxes[:, 2:] -= bboxes[:, :2]\n    bboxes = bboxes.tolist()\n    labels = targets['labels'].tolist()\n    areas = targets['area'].tolist()\n    iscrowd = targets['iscrowd'].tolist()\n    if 'masks' in targets:\n        masks = targets['masks']\n        masks = masks.permute(0, 2, 1).contiguous().permute(0, 2, 1)\n    num_objs = len(bboxes)\n    for i in range(num_objs):\n        ann = {}\n        ann['image_id'] = image_id\n        ann['bbox'] = bboxes[i]\n        ann['category_id'] = labels[i]\n        categories.add(labels[i])\n        ann['area'] = areas[i]\n        ann['iscrowd'] = iscrowd[i]\n        if 'masks' in targets:\n            rle = coco_mask.encode(masks[i].numpy())\n            rle['counts'] = rle['counts'].decode('utf-8')\n            ann[\"segmentation\"] = rle\n        dataset['annotations'].append(ann)\n    return dataset\n\n# annotation IDs need to start at 1, not 0, see torchvision issue #1530\ndef build_annotations(results, ann_id=1, filename='train_annotations.json', bg_label=False):\n    dataset = {'images': [], 'annotations': [], 'categories': []}\n    for k in list(dataset.keys())[:2]:\n        for img_dict in results:\n            if k == 'annotations':\n                for i in range(len(img_dict[k])):\n                    img_dict[k][i]['id'] = ann_id\n                    ann_id += 1\n            dataset[k].extend(img_dict[k])\n    labels = RAW_LABELS_DICT.items()\n    dataset['categories'] = [{'id': n+1 if bg_label else n, 'name': label} for n, label in labels]\n    mmcv.dump(dataset, filename)\n    return ann_id\n    \n\nwith mp.Pool(processes=mp.cpu_count()) as p:\n    test_results = list(tqdm(p.imap(functools.partial(convert_to_coco_api, ds=ds), test_idx),\n                                    total=len(test_idx)))\n    \nwith mp.Pool(processes=mp.cpu_count()) as p:\n    train_results = list(tqdm(p.imap(functools.partial(convert_to_coco_api, ds=ds), train_idx),\n                                    total=len(train_idx)))\n\nann_id = build_annotations(test_results, ann_id=1, filename='test_annotations.json')\nann_id = build_annotations(train_results, ann_id, filename='train_annotations.json')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2 - HTC+  <a class=\"anchor\" id=\"third_2\"></a>","metadata":{}},{"cell_type":"markdown","source":"****Custom Data Augmentation****","metadata":{}},{"cell_type":"code","source":"# https://mmdetection.readthedocs.io/en/latest/tutorials/data_pipeline.html#extend-and-use-custom-pipelines\n\n\nSIMILAR_IMG_FILES = pickle_data(filename='../input/train-img-phash-encodings/similar_train_img_files',\n                                method='r')\n\n\n@PIPELINES.register_module()\nclass CustomDataAugmentation:\n    \n    def __init__(self, ext_policies=None):\n        self.policies = [\n            [\n                dict(type='Rotate', prob=1.0), # old value -> 0.5\n                dict(type='RandomShift', prob=1.0),\n                dict(type='RandomCrop', prob=1.0),\n            ],\n        ]\n        self.ext_policies = ext_policies\n        if self.ext_policies is not None:\n            self.policies.extend(self.ext_policies)\n            \n    def __call__(self, results):\n        # Handle duplicates images\n        if results['filename'] in SIMILAR_IMG_FILES:\n            augmentation = AutoAugment(self.policies)\n            results = augmentation(results)\n        return results","metadata":{"execution":{"iopub.status.busy":"2021-06-13T00:11:05.603023Z","iopub.execute_input":"2021-06-13T00:11:05.603292Z","iopub.status.idle":"2021-06-13T00:11:05.6238Z","shell.execute_reply.started":"2021-06-13T00:11:05.603263Z","shell.execute_reply":"2021-06-13T00:11:05.622979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create config\n\npng_img_path = '../input/hpa-single-cell-image-classification/train'\n\njpg_img_path = 'hpa-jpg/hpa_train/hpa_train_jpg'\n\ntraining_size = len(json.load(open('../input/hpa-annotations/train_annotations.json', 'r'))['images'])\n\n# Model config files\n\nSWIN_TYPE = 'tiny' # tiny, small, base\n\n# Mask R-CNN\nmask_rcnn_resnet50_conf_file = 'mask_rcnn/mask_rcnn_r50_fpn_1x_coco.py'\nmask_rcnn_swin_conf_file = f'swin/mask_rcnn_swin_{SWIN_TYPE}_patch4_window7_mstrain_480-800_adamw_3x_coco.py'\n# Cascade Mask R-CNN\ncascade_mask_rcnn_swin_conf_file = f'swin/cascade_mask_rcnn_swin_{SWIN_TYPE}_patch4_window7_mstrain_480-800_giou_4conv1f_adamw_3x_coco.py'\n# HTC\nhtc_conf_file =  'htc/htc_without_semantic_r50_fpn_1x_coco.py'\n\n\n\ninst_segm_params = {'CLASSES': list(RAW_LABELS_DICT.values()),\n                    'NUM_CLASSES': len(list(RAW_LABELS_DICT.values())),\n                    'IMAGE_SIZE': (224, 224), # (384, 384)\n                    'IOU_THR': 0.6,\n                    'DATA_ROOT': '../input/',\n                    'WORK_DIR': './mmdetection_working_dir',\n                    'MODEL': './configs/' + mask_rcnn_swin_conf_file, # htc_conf_file,\n                    'ALT_BACKBONE': None, # './configs/' + cascade_mask_rcnn_swin_conf_file,\n                    'BACKBONE_WEIGHTS': f'../input/swin-weights/swin_{SWIN_TYPE}_patch4_window7_224.pth',\n                    # f'./swin_{SWIN_TYPE}_patch4_window7_224.pth'\n                    'TRAIN_SIZE': training_size,\n                    'SAMPLES_PER_GPU': 16,  # Mask R-CNN = 8, default value -> 2\n                    'WORKERS_PER_GPU': 2,\n                    'TRAIN_ANNOTATIONS_PATH': 'hpa-annotations/train_annotations.json',\n                    'VAL_ANNOTATIONS_PATH': 'hpa-annotations/test_annotations.json',\n                    'TEST_ANNOTATIONS_PATH': 'hpa-annotations/test_annotations.json',\n                    'TRAIN_IMAGES_PATH': jpg_img_path,\n                    'VAL_IMAGES_PATH': jpg_img_path,\n                    'TEST_IMAGES_PATH': jpg_img_path,\n                    'IMAGE_NORM': dict(mean=[123.675, 116.28, 103.53], std=[58.395, 57.12, 57.375], to_rgb=True),\n                    'NUM_GPU': 1,\n                    'NUM_EPOCHS': 6\n                    }\n\n# Training pipeline\ninst_segm_params['TRAIN_DA'] = [\n    dict(type='LoadImageFromFile'),\n    # dict(type='InstaBoost'), # (used by HTC++, section A2.2. from https://arxiv.org/pdf/2103.14030v1.pdf)\n    dict(type='LoadAnnotations', with_bbox=True, with_mask=True),\n    dict(type='Resize',\n         img_scale=inst_segm_params['IMAGE_SIZE'],\n         # TODO : [(400, 1600), (1400, 1600)],  # Multi-scale training\n         keep_ratio=True),                      # (used by HTC++, section A2.2. from https://arxiv.org/pdf/2103.14030v1.pdf)\n    dict(type='RandomFlip', flip_ratio=0.5),\n    dict(type='CustomDataAugmentation'),\n    dict(type='Normalize', **inst_segm_params['IMAGE_NORM']),\n    dict(type='Pad', size_divisor=32),\n    dict(type='DefaultFormatBundle'),\n    dict(type='Collect', keys=['img', 'gt_bboxes', 'gt_labels', 'gt_masks']),\n]\n\n# Testing pipeline\ninst_segm_params['TEST_DA'] = [\n    dict(type='LoadImageFromFile'),\n    dict(type='MultiScaleFlipAug',\n         img_scale=inst_segm_params['IMAGE_SIZE'],\n         flip=False,\n         transforms=[dict(type='Resize', keep_ratio=True),\n                     dict(type='RandomFlip'),\n                     dict(type='Normalize', **inst_segm_params['IMAGE_NORM']),\n                     dict(type='Pad', size_divisor=32),\n                     dict(type='ImageToTensor', keys=['img']),\n                     dict(type='Collect', keys=['img'])\n                    ])\n]\n\n\nclass InstanceSegmentationConfig:\n\n    def __init__(self, parameters, metrics=('bbox', 'segm')):\n        self.parameters = parameters\n        self.metrics = metrics\n\n    def load_config(self, verbose=False):\n        cfg = Config.fromfile(self.parameters['MODEL'])\n        # Input config\n        cfg.data_root = self.parameters['DATA_ROOT']\n        cfg.classes = self.parameters['CLASSES']\n        # Model config\n        cfg.model.pretrained = self.parameters['BACKBONE_WEIGHTS']\n        if self.parameters['ALT_BACKBONE'] is not None:\n            alt_backbone_config = Config.fromfile(self.parameters['ALT_BACKBONE'])\n            cfg.model.backbone = alt_backbone_config.model.backbone\n            cfg.model.backbone.window_size = int(self.parameters['BACKBONE_WEIGHTS'].split('window')[-1].split('_')[0])\n            cfg.model.neck.in_channels = alt_backbone_config.model.neck.in_channels # [128, 256, 512, 1024]\n            # Soft NMS HTC++\n            cfg.model.test_cfg.rcnn.nms = dict(type='soft_nms', iou_thr=0.5)\n            cfg.optimizer = alt_backbone_config.optimizer\n            cfg.lr_config.step = [27, 33]\n            cfg.runner = alt_backbone_config.runner\n            cfg.fp16 = None\n            cfg.optimizer_config = alt_backbone_config.optimizer_config\n        # Handle Bbox head \n        if type(cfg.model.roi_head.bbox_head) is list:\n            for i in range(len(cfg.model.roi_head.bbox_head)):\n                cfg.model.roi_head.bbox_head[i].num_classes = self.parameters['NUM_CLASSES']\n                if ('norm_cfg' in list(cfg.model.roi_head.bbox_head[i].keys())) and (self.parameters['NUM_GPU'] == 1):\n                    cfg.model.roi_head.bbox_head[i].norm_cfg.type = 'BN'\n        else:\n            cfg.model.roi_head.bbox_head.num_classes = self.parameters['NUM_CLASSES']\n        # Handle Mask head \n        if type(cfg.model.roi_head.mask_head) is list:\n            for i in range(len(cfg.model.roi_head.mask_head)):\n                cfg.model.roi_head.mask_head[i].num_classes = self.parameters['NUM_CLASSES']\n        else:\n            cfg.model.roi_head.mask_head.num_classes = self.parameters['NUM_CLASSES']\n        # Data preprocessing config\n        cfg.img_norm_cfg = self.parameters['IMAGE_NORM']\n        cfg.train_pipeline = self.parameters['TRAIN_DA']\n        cfg.test_pipeline = self.parameters['TEST_DA']\n        # Data config\n        cfg.data.samples_per_gpu = self.parameters['SAMPLES_PER_GPU']\n        cfg.workers_per_gpu = self.parameters['WORKERS_PER_GPU']\n        # Training data config\n        cfg.data.train.classes = cfg.classes\n        cfg.data.train.ann_file = cfg.data_root + self.parameters['TRAIN_ANNOTATIONS_PATH']\n        cfg.data.train.img_prefix = cfg.data_root + self.parameters['TRAIN_IMAGES_PATH']\n        cfg.data.train.pipeline = cfg.train_pipeline\n        # Validation data config\n        cfg.data.val.classes = cfg.classes\n        cfg.data.val.ann_file = cfg.data_root + self.parameters['VAL_ANNOTATIONS_PATH']\n        cfg.data.val.img_prefix = cfg.data_root + self.parameters['VAL_IMAGES_PATH']\n        cfg.data.val.pipeline = cfg.test_pipeline\n        # Testing data config\n        cfg.data.test.classes = cfg.classes\n        cfg.data.test.ann_file = cfg.data_root + self.parameters['TEST_ANNOTATIONS_PATH']\n        cfg.data.test.img_prefix = cfg.data_root + self.parameters['TEST_IMAGES_PATH']\n        cfg.data.test.pipeline = cfg.test_pipeline\n        # Evaluation\n        cfg.evaluation = dict(metric=self.metrics) # default -> ['bbox', 'segm']\n        # Runner config\n        cfg.work_dir = self.parameters['WORK_DIR']\n        # The original learning rate (LR) is set for 8-GPU training. We divide it by 8 since we only use one GPU.\n        # cfg.optimizer.lr = 0.02 / 8\n        cfg.log_config.interval = 50 # round(self.parameters['TRAIN_SIZE'] / cfg.data.samples_per_gpu * .5) # default -> 50\n        cfg.runner.max_epochs = self.parameters['NUM_EPOCHS']\n        # Set seed thus the results are more reproducible\n        cfg.seed = 0\n        set_random_seed(0, deterministic=False)\n        cfg.gpu_ids = range(self.parameters['NUM_GPU'])\n        # We can initialize the logger for training and have a look at the final config used for training\n        if verbose:\n            print(f'Config:\\n{cfg.pretty_text}')\n        return cfg\n    \n    def save_config_file(self, cfg, filepath='./configs/htc/htc+_swin_small_1x.py'):\n        with open(filepath, 'w') as config_file:\n            config_file.write(cfg.pretty_text)\n\n            \nisc = InstanceSegmentationConfig(inst_segm_params, metrics=['segm'])    \ncfg = isc.load_config(verbose=False)\n\n# Tune config file\ncfg.model.test_cfg.rcnn.nms = dict(type='soft_nms', iou_thr=0.5)\n\ncfg.model.rpn_head.loss_cls.type = 'FocalLoss'\ncfg.model.rpn_head.loss_cls.gamma = 2.0\ncfg.model.rpn_head.loss_cls.alpha = 0.25\n\n# cfg.resume_from = '../input/hpa-model-checkpoint/epoch_9.pth'\n\nprint(f'Config:\\n{cfg.pretty_text}')","metadata":{"execution":{"iopub.status.busy":"2021-06-13T00:11:24.988924Z","iopub.execute_input":"2021-06-13T00:11:24.989249Z","iopub.status.idle":"2021-06-13T00:11:33.567809Z","shell.execute_reply.started":"2021-06-13T00:11:24.98922Z","shell.execute_reply":"2021-06-13T00:11:33.566076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Build dataset\ndatasets = [build_dataset(cfg.data.train)]\n\n# Build the detector\nmodel = build_detector(cfg.model, train_cfg=cfg.get('train_cfg'), test_cfg=cfg.get('test_cfg'))\n# Add an attribute for visualization convenience\nmodel.CLASSES = datasets[0].CLASSES\n\n# Create work_dir\nmmcv.mkdir_or_exist(os.path.abspath(cfg.work_dir))\n# Train detector\ntrain_detector(model, datasets, cfg, distributed=False, validate=True)","metadata":{"execution":{"iopub.status.busy":"2021-06-13T00:11:43.367354Z","iopub.execute_input":"2021-06-13T00:11:43.367692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!zip -r file.zip ./mmdetection_working_dir/epoch_4.pth ./mmdetection_working_dir/None.log.json ","metadata":{"execution":{"iopub.status.busy":"2021-06-12T22:56:42.59804Z","iopub.execute_input":"2021-06-12T22:56:42.598388Z","iopub.status.idle":"2021-06-12T22:57:23.119357Z","shell.execute_reply.started":"2021-06-12T22:56:42.598354Z","shell.execute_reply":"2021-06-12T22:57:23.118371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# V - Helpers <a class=\"anchor\" id=\"fifth\"></a>","metadata":{}},{"cell_type":"markdown","source":"****All functions built for this project****","metadata":{}},{"cell_type":"code","source":"# Helpers\n\n\ndef get_label_by_id(image_id, label_format='LabelStr'):\n    return df[df['ID'] == image_id][label_format].iloc[0]\n\n\ndef get_label_mode(df, col='Label', precision=1):\n    label_mode = df[col].mode().iloc[0]\n    total_rows = df.shape[0]\n    total_rows_from_label_mode = df[df[col] == label_mode].shape[0]\n    ratio = round(total_rows_from_label_mode / total_rows * 100, precision)\n    print(f'Label mode : {label_mode} (≃ {ratio} %)')\n    return\n\n\ndef binary_matrix(df, col, decrement=False):\n    total_rows = df.shape[0]\n    null_matrix = np.zeros((total_rows, len(set(np.sum(df[col])))), int)\n    for row_idx in range(total_rows):\n        for label_idx in df[col][row_idx]:\n            null_matrix[row_idx, label_idx if decrement is False else label_idx-1] = 1\n    return null_matrix\n\n\ndef multi_label_binarizer(df, multi_label_bin_matrix, labels=LABELS_DICT.values()):\n    for label, col_value in zip(labels, multi_label_bin_matrix[df.index.tolist()].T):\n        cleaned_label = label.split('(')[0].strip()\n        df[cleaned_label] = col_value\n    return \n\n\ndef plot_label_occurrence(df,\n                          effectif,\n                          precision=1,\n                          x_label='Labels',\n                          y_label='Occurrence',\n                          title='Label occurrence',\n                          plot_type='barh',\n                          plot_size=(12, 8),\n                          display_legend=0):\n    ax = df.plot(kind=plot_type,\n                 figsize=plot_size,\n                 xlabel=x_label,\n                 ylabel=y_label,\n                 title=title,\n                 legend=display_legend)\n    try:\n        for i, v in df.iterrows():\n            label = f' {v[0]:,} ({round(v[0]/effectif*100, precision)}%)'\n            ax.text(v[0], i, label, color='r', va='bottom', rotation=0)\n    except:\n        pass\n    plt.gca().invert_yaxis()\n    return\n\n\n\ndef get_image_sample(image_id,\n                     image_dir_path=TRAIN_IMG_PATH,\n                     image_filters=IMG_FILTERS,\n                     image_filter_labels=IMG_FILTER_LABELS,\n                     image_size=None,\n                     stacked_filters=False,\n                     channel_filters=('microtubules', 'protein_of_interest', 'nucleus')):\n    # Get image filters filenames\n    image_filters_filenames = [f'{image_dir_path}{image_id}{color}' for color in image_filters]\n    # Read image filters\n    image_filters = list(map(cv2.imread, image_filters_filenames))\n    # Resize image filters\n    if image_size is not None:\n        image_filters = list(map(lambda x: cv2.resize(x, image_size, interpolation=cv2.INTER_NEAREST),\n                                 image_filters))\n    # Check dtype\n    image_filters = [img if img.dtype == 'uint8' else (img/255).astype('uint8') for img in image_filters]\n    # Store image sample results\n    image_sample = {filter_label: img for filter_label, img in zip(image_filter_labels, image_filters)}\n    if stacked_filters:\n        image_sample['stacked_filters'] = stack_image_filters(image_sample, channel_filters)\n    return image_sample\n\n\ndef stack_image_filters(image_sample,\n                        channel_filters=('microtubules', 'protein_of_interest', 'nucleus')): # RGB\n    filters_to_stack = tuple([image_sample[image_filter][:, :, 0] for image_filter in channel_filters])\n    stacked_filters = np.dstack(filters_to_stack)\n    return stacked_filters\n\n\ndef display_image_filters(image, by_id=True, image_filter='stacked_filters', size=(12, 8)):\n    if by_id:\n        # Get image sample dictionary filters\n        image_sample = get_image_sample(image, stacked_filters=True)\n        # Image sample as array\n        image = image_sample[image_filter]\n    plt.figure(figsize=size)\n    return plt.imshow(image)\n\n\ndef display_channel_filter(image, channel):\n    if channel is 'microtubules': # Red\n        image[..., 1] = 0\n        image[..., 2] = 0\n    elif channel is 'protein_of_interest': # Green\n        image[..., 0] = 0\n        image[..., 2] = 0\n    elif channel is 'nucleus': # Blue\n        image[..., 0] = 0\n        image[..., 1] = 0\n    elif channel is 'endoplasmic_reticulum': # Yellow\n        image[..., 2] = 0\n    return\n\n\ndef map_organelle_labels(df, image_id, label_col, id_col='ID', label_mapper=LABELS_DICT):\n    int_labels = sorted(df.set_index(id_col).loc[image_id, label_col])\n    labels = [label_mapper[int_label] for int_label in int_labels]\n    return labels \n\n\ndef display_image_sample(image_sample_id=None,\n                         image_sample=None,\n                         data=None,\n                         label_col='LabelList',\n                         image_size=None,\n                         stacked_filters=True,\n                         channel_filters=('microtubules', 'protein_of_interest', 'nucleus'),\n                         plot_size=(12, 4)):\n    if image_sample is None:\n        image_sample = get_image_sample(image_sample_id, image_size=image_size)\n    if stacked_filters:\n        image_sample['stacked_filters'] = stack_image_filters(image_sample, channel_filters)\n    total_filters = len(image_sample)\n    fig = plt.figure(figsize=plot_size)\n    for filter_idx, (filter_label, image) in enumerate(image_sample.items()):\n        plt.subplot(1, total_filters, filter_idx + 1)\n        title = ' '.join(filter_label.split('_')).title()\n        display_channel_filter(image, filter_label)\n        plt.imshow(image)\n        plt.title(title)\n    if data is not None:\n        # Display image labels subplot title\n        labels = map_organelle_labels(data, image_sample_id, label_col)\n        fig.suptitle('\\n'.join(labels))\n    fig.tight_layout()\n    return\n\n\ndef display_image_by_id(image_ids,\n                        image_filter='stacked_filters',\n                        size=(12, 8),\n                        plot_mask=False,\n                        shape=None,\n                        adjust=False,\n                        adjust_h=None,\n                        adjust_w=None,\n                        mask_path=TRAIN_CELL_MASK_PATH):\n\n    images = []\n    masks = []\n    if type(image_ids) != list:\n        image_ids = [image_ids]\n    for image_id in image_ids:\n        # Get image sample dictionary filters\n        image_sample = get_image_sample(image_id, stacked_filters=True)\n        # Image sample as array\n        image = image_sample[image_filter]\n        if image_filter is not 'stacked_filters':\n            display_channel_filter(image, image_filter)\n        images.append(image)\n        if plot_mask:\n            # Get image mask sample\n            mask = np.load(f'{mask_path}/{image_id}.npz')['arr_0']\n            masks.append(mask)\n    # Build subplot\n    data_to_plot = images if plot_mask is False else [x for z in zip(images, masks) for x in z]\n    subplot_shape = len(images) if len(images) > 1 else len(images) + 1\n    subplot = Subplot((subplot_shape, subplot_shape) if shape is None else shape,\n                      size,\n                      data_to_plot)\n    # Clean image filter type\n    img_filter_cleaned = image_filter.replace('_', ' ').title()\n    # Dynamic title function\n    def dynamic_title(img):\n        if len(img.shape) < 3:\n            mask_id = image_ids[list(map(id, masks)).index(id(img))]\n            return f'Label(s) :\\n' + get_label_by_id(mask_id)\n        else:\n            img_id = image_ids[list(map(id, images)).index(id(img))]\n            return f'Label(s) :\\n' + get_label_by_id(img_id)\n    # Plot image and mask\n    subplot.plot_data(lambda axe, img: axe.imshow(img),\n                      title=dynamic_title,\n                      x_label='Width',\n                      y_label='Height')\n    # Add space between subplots\n    if adjust:\n        subplot.fig.subplots_adjust(hspace=adjust_h, wspace=adjust_w)\n    return\n\n\ndef get_random_single_label_id(df, single_label, id_col='ID', label_col='Label', sample_size=1):\n    return df[id_col][df[label_col].eq(single_label)].sample(sample_size).iloc[0]\n\n\ndef generate_cell_segmentator_data(image_ids, image_size=None, cell_filters=('microtubules',\n                                                                             'endoplasmic_reticulum',\n                                                                             'nucleus',\n                                                                             'stacked_filters')):\n    # Build cell & nuclei image level\n    cells_list = []\n    nuclei_list = []\n    stacked_filters_list = []\n    # Get cell filters from training sample indexes\n    cell_dicts = [get_image_sample(label_id,\n                                   image_size=image_size,\n                                   stacked_filters=True) for label_id in image_ids]\n    # Get image data for each cell filter in order to train model based on training sample\n    # Cell filters order image channels as RGB channels (+ stacked filters)\n    for cell_filter in cell_filters:\n        cell_list = [cell_dict[cell_filter][:, :, 0] for cell_dict in cell_dicts]\n        if cell_filter == 'nucleus':\n            nuclei_list.extend(cell_list)\n            cells_list.append(cell_list)\n        elif cell_filter == 'stacked_filters':\n            stacked_filters_list.extend(cell_list)\n        else:\n            cells_list.append(cell_list)\n            \n    cell_segmentator_data = {'filters': {'cell_filters': cells_list,\n                                         'nuclei_filters': nuclei_list,\n                                         'stacked_filters': stacked_filters_list},\n                             'ids': image_ids}\n    return cell_segmentator_data\n\n\ndef generate_cell_masks(cell_segmentator_data,\n                        cell_segmentator_model=None,\n                        total_cells=None,\n                        save=False,\n                        plot=False,\n                        plot_data=None,\n                        verbose=True,\n                        external_mask_counter=None):\n    image_data = []\n    image_filters = cell_segmentator_data['filters']\n    image_ids = cell_segmentator_data['ids']\n    if total_cells is None:\n        total_cells = len(image_ids)\n    \n    if cell_segmentator_model is None:\n        # Build cell segmentation model with HPA package\n        cell_segmentator_model = cellsegmentator.CellSegmentator(nuclei_model='.nuclei_model.pth',\n                                                                 cell_model='.cell_model.pth',\n                                                                 scale_factor=0.25,\n                                                                 device='cuda',\n                                                                 padding=False,\n                                                                 multi_channel_model=True)\n    # Cell level prediction\n    segmented_cells = cell_segmentator_model.pred_cells(image_filters['cell_filters'])\n    # Nuclei level prediction\n    segmented_nuclei = cell_segmentator_model.pred_nuclei(image_filters['nuclei_filters'])\n    # Generate label masks\n    for i in range(len(image_ids)):\n        mask_count = i+1 if external_mask_counter is None else i + external_mask_counter\n        if verbose:\n            print('Mask n°{} / {} ({}%)'.format(mask_count,\n                                                total_cells,\n                                                round(mask_count/total_cells*100, 1)))\n        # Generate nuclei and cell masks\n        nuclei_mask, cell_mask = label_cell(segmented_nuclei[i], segmented_cells[i])\n        # Save masks\n        if save:\n            np.savez_compressed(f'{TRAIN_CELL_MASK_PATH}{image_ids[i]}', cell_mask)\n            np.savez_compressed(f'{TRAIN_NUCLEI_MASK_PATH}{image_ids[i]}', nuclei_mask)\n        # Visualize results\n        if plot:\n            # Build cell sample to visualize (which are stacked filters, cell and nuclei masks)\n            cell_sample = {'image': image_filters['stacked_filters'][i],\n                           'cell_mask': cell_mask,\n                           'nuclei_mask': nuclei_mask}\n            image_data.append(cell_sample)\n            # Visualize cell sample\n            display_image_sample(image_sample=cell_sample,\n                                 image_sample_id=image_ids[i],\n                                 data=plot_data,\n                                 stacked_filters=False,\n                                 plot_size=(12, 4))\n    if len(image_data) > 0:\n        return image_data\n\n    \ndef get_total_cells_from_mask(mask_file):\n    mask_array = np.load(mask_file)['arr_0']\n    return len(np.unique(mask_array)) - 1\n\n\ndef check_data_coherence(df1, df2):\n    df1 = df1[df1.index.isin(df2.index.tolist())].sort_index()\n    df2 = df2.sort_index()\n    return df1.equals(df2)\n\n\ndef compute_IoU(img_array1, img_array2):\n    overlap = img_array1 * img_array2\n    union = img_array1 + img_array2\n    IoU = overlap.sum() / float(union.sum())\n    return IoU\n\n\ndef compute_image_similarity(img1, img2):\n    # https://stackoverflow.com/questions/11541154/checking-images-for-similarity-with-opencv\n    normalized_img1 = img1 / np.sqrt(np.sum(img1**2))\n    normalized_img2 = img2 / np.sqrt(np.sum(img2**2))\n    similarity_score = np.sum(normalized_img1 * normalized_img2)\n    return similarity_score\n\n\ndef compute_segmentation_quality(img_path, mask_path):\n    img = cv2.imread(img_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY).astype(float)\n    mask = np.load(mask_path)['arr_0'].astype(float)\n    similarity_score = compute_image_similarity(img, mask)\n    return similarity_score\n\n\ndef normalize_class_weights(y_train):\n    total_classes = np.unique(y_train)\n    class_weights = sklearn.utils.class_weight.compute_class_weight('balanced', total_classes, y_train)\n    # check also : sklearn.utils.class_weight.compute_sample_weight\n    class_weights = {label: weight for label, weight in zip(total_classes, class_weights)}\n    return class_weights\n\n\ndef apply_mask_color(mask, mask_color):\n    return np.concatenate(([mask[ ... , np.newaxis] * color for color in mask_color]), axis=2)\n\n\ndef morphological_transform(img_array, op, kernel_size=(5, 5), n_iter=1):\n    kernel = np.ones(kernel_size, np.uint8)\n    if op is 'erosion':\n        img_array = cv2.erode(img_array, kernel, iterations=n_iter)\n    elif op is 'dilation':\n        img_array = cv2.dilate(img_array, kernel, iterations=n_iter)\n    elif op is 'opening':  # Erosion followed by Dilation\n        img_array = cv2.morphologyEx(img_array, cv2.MORPH_OPEN, kernel)\n    elif op is 'closing':  # Dilation followed by Erosion\n        img_array = cv2.morphologyEx(img_array, cv2.MORPH_CLOSE, kernel)\n    elif op is 'gradient': # Difference between Dilation and Erosion of an image\n        img_array = cv2.morphologyEx(img_array, cv2.MORPH_GRADIENT, kernel)\n    elif op is 'tophat':  # Difference between input image and Opening of the image\n        img_array = cv2.morphologyEx(img_array, cv2.MORPH_TOPHAT, kernel)\n    elif op is 'backhat': # Difference between the Closing of the input image and input image\n        img_array = cv2.morphologyEx(img_array, cv2.MORPH_BLACKHAT, kernel)\n    else:\n        raise Exception(f'{op} not implemented')\n    return img_array ","metadata":{"execution":{"iopub.status.busy":"2021-06-06T16:12:19.89231Z","iopub.execute_input":"2021-06-06T16:12:19.89269Z","iopub.status.idle":"2021-06-06T16:12:19.956661Z","shell.execute_reply.started":"2021-06-06T16:12:19.892661Z","shell.execute_reply":"2021-06-06T16:12:19.955688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Annex (load extra train images)","metadata":{}},{"cell_type":"code","source":"trainset_extra_path = '../input/hpa-challenge-2021-extra-train-images/HPA-Challenge-2021-trainset-extra/'\n\nextra_data_filepaths = glob(trainset_extra_path + '*_green.png')\n\ndf2 = pd.read_csv('../input/publichpa-withcellline/kaggle_2021.tsv')\ndf2['ID'] = df2['Image'].map(lambda p: p.split('/')[-1])\ndf2['ImagePath'] = df2['Image'].map(lambda p: trainset_extra_path + p.split('/')[-1] + '_green.png')\ndf2.rename(columns={'Label': 'LabelStr', 'Label_idx': 'Label'}, inplace=True)\n\ndf_extra_data = df2[df2.ImagePath.isin(extra_data_filepaths)][['ID', 'ImagePath', 'Label', 'LabelStr']]\ndf_extra_data['LabelList'] = df_extra_data['Label'].map(lambda x: list(map(int, x.split('|'))))\ndf_extra_data['LabelStr'] = df_extra_data['LabelList'].map(lambda x: '\\n'.join([LABELS_DICT[n] for n in x]))\ndf_extra_data['TotalLabels'] = df_extra_data['LabelList'].map(len)\n\ndf_copy = df[df_extra_data.columns].copy()\ndf_with_extra_data = pd.concat([df_copy, df_extra_data], ignore_index=True)\ndf_with_extra_data","metadata":{"execution":{"iopub.status.busy":"2021-06-06T16:08:07.829928Z","iopub.execute_input":"2021-06-06T16:08:07.830279Z","iopub.status.idle":"2021-06-06T16:08:09.787878Z","shell.execute_reply.started":"2021-06-06T16:08:07.830245Z","shell.execute_reply":"2021-06-06T16:08:09.786838Z"},"trusted":true},"execution_count":null,"outputs":[]}]}