{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"conda install cudatoolkit=10.2","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rsync -a ../input/mmdetection-v280/mmdetection ../\n!pip install ../input/mmdetection-v280/src/mmdet-2.8.0/mmdet-2.8.0/\n!pip install ../input/mmdetection-v280/src/mmpycocotools-12.0.3/mmpycocotools-12.0.3/\n!pip install ../input/mmdetection-v280/src/addict-2.4.0-py3-none-any.whl\n!pip install ../input/mmdetection-v280/src/yapf-0.30.0-py2.py3-none-any.whl\n!pip install ../input/mmdetection-v280/src/mmcv_full-1.2.6-cp37-cp37m-manylinux1_x86_64.whl","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -r ../input/mmdetection-v280/mmdetection/requirements.txt","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from itertools import groupby\nfrom pycocotools import mask as mutils\nimport numpy as np\nfrom tqdm import tqdm\nimport pandas as pd\nimport os\nimport pickle\nimport cv2\nfrom multiprocessing import Pool\nimport matplotlib.pyplot as plt","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# exp_name = \"v1\"\ncell_mask_dir = '../input/hpa-mask/hpa_cell_mask'    \nROOT = '../input/hpa-single-cell-image-classification/'\nMAX_THRE = 4\ntrain_or_test = 'train'\ndebug = False\n\nimg_dir = f'../work/mmdet_unique_{train_or_test}'\n!mkdir -p {img_dir}\ndf = pd.read_csv(os.path.join(ROOT, 'train.csv'))\nif debug:\n    df = df[:100]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def coco_rle_encode(mask):\n    '''\n    implement storing binary masks by rles format\n    input:\n        mask: mask of image\n    output:\n        rle encode format\n    '''\n    rle = {'counts': [], 'size': list(mask.shape)}\n    counts = rle.get('counts')\n    for i, (value, elements) in enumerate(groupby(mask.ravel(order='F'))):\n        if i == 0 and value == 1:\n            counts.append(0)\n        counts.append(len(list(elements)))\n    return rle\n\ndef get_rles_from_mask(image_id):\n    '''\n    implement get rle of all masks in image\n    input:\n        image_id : id of image\n    return:\n        list of rle\n        height of image\n        width of image\n    '''\n    img = np.load(f'{cell_mask_dir}/{image_id}.npz')['arr_0']\n    rle_list = []\n    for val in np.unique(img):\n        if val == 0:\n            continue\n        binary_mask = np.where(img == val, val, 0).astype(bool)\n        counts = []\n        rle = coco_rle_encode(binary_mask)\n        rle_list.append(rle)\n    return rle_list, img.shape[0], img.shape[1]\n\ndef mk_mmdet_custom_data(image_id):\n    '''\n    convert data to coco format\n    input:\n        image_id: id of image\n    return:\n        annotation json files in coco format\n    '''\n    rles, height, width = get_rles_from_mask(image_id)\n    if len(rles) == 0:\n        return {\n            'filename': image_id+'.jpg',\n            'width': width,\n            'height': height,\n            'ann': {}\n        }\n    rles = mutils.frPyObjects(rles, height, width)\n    masks = mutils.decode(rles)\n    bboxes = mutils.toBbox(mutils.encode(np.asfortranarray(masks.astype(np.uint8))))\n    bboxes[:, 2] += bboxes[:, 0]\n    bboxes[:, 3] += bboxes[:, 1]\n    return {\n        'filename': image_id+'.jpg',\n        'width': width,\n        'height': height,\n        'ann':\n            {\n                'bboxes': np.array(bboxes, dtype=np.float32),\n                'labels': np.zeros(len(bboxes)),\n                'masks': rles\n            }\n    }\n\ndef print_masked_img(image_id, mask):\n    '''\n    visualize image\n    input:\n        image_id: id of image\n        mask: mask of image with above image_id\n    '''\n    img = load_RGB_image(image_id, train_or_test)\n    \n    plt.figure(figsize=(15, 15))\n    plt.subplot(1, 3, 1)\n    plt.imshow(img)\n    plt.title('Image')\n    plt.axis('off')\n    \n    plt.subplot(1, 3, 2)\n    plt.imshow(mask)\n    plt.title('Mask')\n    plt.axis('off')\n    \n    plt.subplot(1, 3, 3)\n    plt.imshow(img)\n    plt.imshow(mask, alpha=0.6)\n    plt.title('Image + Mask')\n    plt.axis('off')\n    plt.show()\n    \ndef read_img(image_id, color, train_or_test='train', image_size=None):\n    filename = f'{ROOT}/{train_or_test}/{image_id}_{color}.png'\n    assert os.path.exists(filename), f'not found {filename}'\n    img = cv2.imread(filename, cv2.IMREAD_UNCHANGED)\n    if image_size is not None:\n        img = cv2.resize(img, (image_size, image_size))\n    if img.max() > 255:\n        img_max = img.max()\n        img = (img/255).astype('uint8')\n    return img\n\ndef load_RGB_image(image_id, train_or_test='train', image_size=None):\n    '''\n    load image with each channels follow by stack them\n    '''\n    red = read_img(image_id, \"red\", train_or_test, image_size)\n    green = read_img(image_id, \"green\", train_or_test, image_size)\n    blue = read_img(image_id, \"blue\", train_or_test, image_size)\n    # using rgb only here\n    #yellow = read_img(image_id, \"yellow\", train_or_test, image_size)\n    stacked_images = np.transpose(np.array([red, green, blue]), (1,2,0))\n    return stacked_images\n\ndef mk_ann(idx):\n    '''\n    get the annotation json files in coco format for each image\n    input:\n        idx\n    output:\n        anno: the annotation json\n        image_id : id of image\n    '''\n    image_id = df.iloc[idx].ID\n    anno = mk_mmdet_custom_data(image_id)\n    img = load_RGB_image(image_id, train_or_test)\n    cv2.imwrite(f'{img_dir}/{image_id}.jpg', img)\n    return anno, idx, image_id","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_mask_dir = '../input/hpa-mask/hpa_cell_mask'    \nfor idx in range(3):\n    image_id = df.iloc[idx].ID\n    cell_mask = np.load(f'{cell_mask_dir}/{image_id}.npz')['arr_0']\n    print_masked_img(image_id, cell_mask)\n    load_RGB_image(image_id)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len([idxx for idxx in range(len(df)) if '|' not in df['Label'].iloc[idxx]])[:5000]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"p = Pool(processes=MAX_THRE)\nannos = []\nfor i, (anno, idx, image_id) in enumerate(p.imap(mk_ann, range(len([idxx for idxx in range(len(df)) if '|' not in df['Label'].iloc[idxx]])))):\n    if len(anno['ann']) > 0:\n        annos.append(anno)\n    if i % 100 == 0:\n        print (idx, image_id)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lbl_cnt_dict = df.set_index('ID').to_dict()['Label']\ntrn_annos = []\nval_annos = []\nval_len = int(len(annos)*0.5)\nfor idx in range(len(annos)):\n    ann = annos[idx]\n    filename = ann['filename'].replace('.jpg','').replace('.png','')\n    label_id = lbl_cnt_dict[filename]\n    # only images which are composed of only one class of cells\n    if '|' not in label_id:\n        label_id = int(label_id)\n        ann['ann']['labels'] = np.full(len(ann['ann']['bboxes']), label_id)\n        if idx < val_len:\n            val_annos.append(ann)\n        else:\n            trn_annos.append(ann)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print (len(trn_annos))\nprint (len(val_annos))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(f'../work/mmdet_unique_full.pkl', 'wb') as f:\n    pickle.dump(annos, f)\nwith open(f'../work/mmdet_unique_trn.pkl', 'wb') as f:\n    pickle.dump(trn_annos, f)\nwith open(f'../work/mmdet_unique_val.pkl', 'wb') as f:\n    pickle.dump(val_annos, f)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# config = f'configs/hpa_{exp_name}/mask_rcnn_r50_fpn_1x_coco.py'\nconfig = f'configs/mask_rcnn_unique/mask_rcnn_r50_fpn_1x_coco.py'\n\n# using --no-validate to avoid some errors for custom dataset metrics\n# additional_conf = '--no-validate '\nadditional_conf = ' --cfg-options workflow=\"[(train,1),(val,1)]\"'\nadditional_conf += f' --cfg-options optimizer.lr=0.0025'\nadditional_conf += f' --cfg-options work_dir=../working/work_dir'\nadditional_conf += f' --cfg-options load_from=../input/mmdetection-v280/pretrained/mask_rcnn_r50_fpn_2x_coco_bbox_mAP-0.392__segm_mAP-0.354_20200505_003907-3e542a40.pth'\ncmd = f'bash -x tools/dist_train.sh {config} 1 {additional_conf}'\n!cd ../mmdetection; {cmd}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls ../mmdetection/work_dirs/mask_rcnn_r50_fpn_1x_coco","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cmd = f'python ../input/mmdetection-v280/mmdetection/tools/analysis_tools/analyze_logs.py plot_curve ../mmdetection/work_dirs/mask_rcnn_r50_fpn_1x_coco/20210519_103834.log.json --keys loss_cls loss_bbox --legend loss_cls loss_bbox --out /kaggle/working/plot.pdf'\n!cd ../mmdetection; {cmd}","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}