{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!nvidia-smi","metadata":{"papermill":{"duration":0.705422,"end_time":"2021-07-08T22:53:11.391414","exception":false,"start_time":"2021-07-08T22:53:10.685992","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-23T10:23:01.220093Z","iopub.execute_input":"2021-08-23T10:23:01.220401Z","iopub.status.idle":"2021-08-23T10:23:02.013968Z","shell.execute_reply.started":"2021-08-23T10:23:01.220371Z","shell.execute_reply":"2021-08-23T10:23:02.012823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!nvcc --version","metadata":{"papermill":{"duration":0.666457,"end_time":"2021-07-08T22:53:12.083346","exception":false,"start_time":"2021-07-08T22:53:11.416889","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-23T10:23:02.015987Z","iopub.execute_input":"2021-08-23T10:23:02.016654Z","iopub.status.idle":"2021-08-23T10:23:02.824575Z","shell.execute_reply.started":"2021-08-23T10:23:02.016598Z","shell.execute_reply":"2021-08-23T10:23:02.82356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch, torchvision\nprint(torch.__version__, torch.cuda.is_available())","metadata":{"papermill":{"duration":1.348408,"end_time":"2021-07-08T22:53:13.457144","exception":false,"start_time":"2021-07-08T22:53:12.108736","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-23T10:23:02.827029Z","iopub.execute_input":"2021-08-23T10:23:02.827417Z","iopub.status.idle":"2021-08-23T10:23:04.193348Z","shell.execute_reply.started":"2021-08-23T10:23:02.827373Z","shell.execute_reply":"2021-08-23T10:23:04.192407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* It seems CUDA=11.0 and torch==1.7.0 is used in this kaggle docker image.\n* See installation for details. https://detectron2.readthedocs.io/en/latest/tutorials/install.html","metadata":{"papermill":{"duration":0.025139,"end_time":"2021-07-08T22:53:13.508347","exception":false,"start_time":"2021-07-08T22:53:13.483208","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Install Pre-Built Detectron2","metadata":{"papermill":{"duration":0.024919,"end_time":"2021-07-08T22:53:13.558394","exception":false,"start_time":"2021-07-08T22:53:13.533475","status":"completed"},"tags":[]}},{"cell_type":"code","source":"!pip install /kaggle/input/detectron2/omegaconf-2.0.6-py3-none-any.whl\n\n!pip install /kaggle/input/detectron2/iopath-0.1.8-py3-none-any.whl\n\n!pip install /kaggle/input/detectron2/fvcore-0.1.3.post20210317/fvcore-0.1.3.post20210317/\n\n!pip install /kaggle/input/detectron2/pycocotools-2.0.2/dist/pycocotools-2.0.2.tar\n\n!pip install /kaggle/input/detectron2/detectron2-0.4cu110-cp37-cp37m-linux_x86_64.whl","metadata":{"_kg_hide-output":true,"papermill":{"duration":138.682502,"end_time":"2021-07-08T22:55:32.266133","exception":false,"start_time":"2021-07-08T22:53:13.583631","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-23T10:23:04.19508Z","iopub.execute_input":"2021-08-23T10:23:04.195647Z","iopub.status.idle":"2021-08-23T10:25:23.669456Z","shell.execute_reply.started":"2021-08-23T10:23:04.19559Z","shell.execute_reply":"2021-08-23T10:25:23.66852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import Libraries","metadata":{"papermill":{"duration":0.034674,"end_time":"2021-07-08T22:55:32.406413","exception":false,"start_time":"2021-07-08T22:55:32.371739","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore') #Ignore \"future\" warnings and Data-Frame-Slicing warnings.\n\nimport numpy as np \nimport pandas as pd \nfrom datetime import datetime\nimport time\nfrom tqdm import tqdm_notebook as tqdm # progress bar\nimport matplotlib.pyplot as plt\n\nfrom math import ceil\nfrom typing import Any, Dict, List\nfrom typing import List\nfrom dataclasses import dataclass, field\nfrom typing import Dict\nfrom numpy import ndarray\nfrom glob import glob\n\n\nimport os, json, cv2, random\nimport skimage.io as io\n\nimport pickle\nfrom pathlib import Path\nfrom typing import Optional\nfrom tqdm import tqdm\n\n# numba\nimport numba\nfrom numba import jit\n\n\n# import some common detectron2 utilities\nfrom detectron2 import model_zoo\nfrom detectron2.engine import DefaultPredictor\nfrom detectron2.config import get_cfg\nfrom detectron2.utils.visualizer import Visualizer\nfrom detectron2.data import DatasetCatalog, MetadataCatalog\nfrom detectron2.utils.visualizer import ColorMode","metadata":{"_kg_hide-input":true,"papermill":{"duration":1.541907,"end_time":"2021-07-08T22:55:33.982894","exception":false,"start_time":"2021-07-08T22:55:32.440987","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-23T10:25:23.671065Z","iopub.execute_input":"2021-08-23T10:25:23.671392Z","iopub.status.idle":"2021-08-23T10:25:25.243566Z","shell.execute_reply.started":"2021-08-23T10:25:23.671354Z","shell.execute_reply":"2021-08-23T10:25:25.242702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# configs","metadata":{"papermill":{"duration":0.034749,"end_time":"2021-07-08T22:55:34.053252","exception":false,"start_time":"2021-07-08T22:55:34.018503","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# --- configs ---\nthing_classes = [\n    \"atypical\",\n    \"indeterminate\",\n    \"negative\",\n    \"typical\"\n]\ndebug=False\n\ncategory_name_to_id = {class_name: index for index, class_name in enumerate(thing_classes)}\ncategory_name_to_id","metadata":{"papermill":{"duration":0.046951,"end_time":"2021-07-08T22:55:34.136147","exception":false,"start_time":"2021-07-08T22:55:34.089196","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-23T10:25:25.244912Z","iopub.execute_input":"2021-08-23T10:25:25.245258Z","iopub.status.idle":"2021-08-23T10:25:25.257012Z","shell.execute_reply.started":"2021-08-23T10:25:25.245221Z","shell.execute_reply":"2021-08-23T10:25:25.255992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Register Dataset","metadata":{"papermill":{"duration":0.035118,"end_time":"2021-07-08T22:55:34.206271","exception":false,"start_time":"2021-07-08T22:55:34.171153","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import pandas as pd \ndf_meta = pd.read_csv(\"../input/siim-covid19-resized-1024px/meta.csv\")\ntest_meta=df_meta[df_meta.split==\"test\"]\n\n\ndef get_COVID19_data_dicts_test(\n    imgdir: Path, test_meta: pd.DataFrame, use_cache: bool = True, debug: bool = False,\n):\n    debug_str = f\"_debug{int(debug)}\"\n    cache_path = Path(\".\") / f\"dataset_dicts_cache_test.pkl\"\n    if not use_cache or not cache_path.exists():\n        print(\"Creating data...\")\n        # test_meta = pd.read_csv(imgdir / \"test_meta.csv\")\n        df_meta = pd.read_csv(\"../input/siim-covid19-resized-1024px/meta.csv\")\n        test_meta=df_meta[df_meta.split==\"test\"]\n        if debug:\n            test_meta = test_meta.iloc[:10]  # For debug....\n        # Load 1 image to get image size.\n        image_id = test_meta.iloc[0,0]\n        #image_path = str(imgdir / \"test\" / f\"{image_id}.jpg\")\n        image_path = str(f'../input/siim-covid19-resized-1024px/test/{image_id}.jpg')\n        image = cv2.imread(image_path)\n        resized_height, resized_width, ch = image.shape\n        print(f\"image shape: {image.shape}\")\n\n        dataset_dicts = []\n        for index, test_meta_row in tqdm(test_meta.iterrows(), total=len(test_meta)):\n            record = {}\n\n            image_id, height, width,s = test_meta_row.values\n            #filename = str(imgdir / \"test\" / f\"{image_id}.jpg\")\n            filename = str(f'../input/siim-covid19-resized-1024px/test/{image_id}.jpg')\n            record[\"file_name\"] = filename\n            # record[\"image_id\"] = index\n            record[\"image_id\"] = image_id\n            record[\"height\"] = resized_height\n            record[\"width\"] = resized_width\n            # objs = []\n            # record[\"annotations\"] = objs\n            dataset_dicts.append(record)\n        with open(cache_path, mode=\"wb\") as f:\n            pickle.dump(dataset_dicts, f)\n\n    #print(f\"Load from cache {cache_path}\")\n    with open(cache_path, mode=\"rb\") as f:\n        dataset_dicts = pickle.load(f)\n    return dataset_dicts","metadata":{"papermill":{"duration":0.08294,"end_time":"2021-07-08T22:55:34.324475","exception":false,"start_time":"2021-07-08T22:55:34.241535","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-23T10:25:25.258528Z","iopub.execute_input":"2021-08-23T10:25:25.258954Z","iopub.status.idle":"2021-08-23T10:25:25.301365Z","shell.execute_reply.started":"2021-08-23T10:25:25.258912Z","shell.execute_reply":"2021-08-23T10:25:25.30053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgdir = \"../input/siim-covid19-resized-1024px\"\nDatasetCatalog.register(\n    \"COVID19_data_test\", lambda: get_COVID19_data_dicts_test(imgdir, test_meta, debug=debug)\n)\n","metadata":{"papermill":{"duration":0.041897,"end_time":"2021-07-08T22:55:34.401674","exception":false,"start_time":"2021-07-08T22:55:34.359777","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-23T10:25:25.304181Z","iopub.execute_input":"2021-08-23T10:25:25.304535Z","iopub.status.idle":"2021-08-23T10:25:25.308674Z","shell.execute_reply.started":"2021-08-23T10:25:25.304505Z","shell.execute_reply":"2021-08-23T10:25:25.307596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MetadataCatalog.get(\"COVID19_data_test\").set(thing_classes=thing_classes)\nmetadata = MetadataCatalog.get(\"COVID19_data_test\")\ndataset_dicts = get_COVID19_data_dicts_test(imgdir, test_meta, debug=debug)","metadata":{"papermill":{"duration":0.256786,"end_time":"2021-07-08T22:55:34.693361","exception":false,"start_time":"2021-07-08T22:55:34.436575","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-23T10:25:25.310889Z","iopub.execute_input":"2021-08-23T10:25:25.311322Z","iopub.status.idle":"2021-08-23T10:25:25.464391Z","shell.execute_reply.started":"2021-08-23T10:25:25.311287Z","shell.execute_reply":"2021-08-23T10:25:25.461986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load model","metadata":{"papermill":{"duration":0.035831,"end_time":"2021-07-08T22:55:34.765938","exception":false,"start_time":"2021-07-08T22:55:34.730107","status":"completed"},"tags":[]}},{"cell_type":"code","source":"from detectron2.config import get_cfg\ncfg = get_cfg()\nconfig_name = \"COCO-Detection/faster_rcnn_R_50_FPN_3x.yaml\" \n#config_name = \"COCO-Detection/faster_rcnn_R_101_C4_3x.yaml\"\ncfg.merge_from_file(model_zoo.get_config_file(config_name))\ncfg.MODEL.ROI_HEADS.NUM_CLASSES = 4  # \n\n\ncfg.MODEL.WEIGHTS = \"../input/d/ammarnassanalhajali/1siim-covid19-detectron2-weights/output/model_final.pth\"\ncfg.MODEL.ROI_HEADS.SCORE_THRESH_TEST = 0.180 # set the testing threshold for this model\n\npredictor = DefaultPredictor(cfg)","metadata":{"papermill":{"duration":9.058909,"end_time":"2021-07-08T22:55:43.861018","exception":false,"start_time":"2021-07-08T22:55:34.802109","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-23T10:25:25.465653Z","iopub.execute_input":"2021-08-23T10:25:25.465989Z","iopub.status.idle":"2021-08-23T10:25:33.974068Z","shell.execute_reply.started":"2021-08-23T10:25:25.465942Z","shell.execute_reply":"2021-08-23T10:25:33.973191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def format_pred(labels: ndarray, boxes: ndarray, scores: ndarray) -> str:\n    pred_strings = []\n    for label, score, bbox in zip(labels, scores, boxes):\n        xmin, ymin, xmax, ymax = bbox.astype(np.int64)\n        if label==2:\n            labelstr='none'\n        else:\n            labelstr='opacity'\n        pred_strings.append(f\"{labelstr} {score:0.3f} {xmin} {ymin} {xmax} {ymax}\") \n    return \" \".join(pred_strings)\n\ndef predict_batch(predictor: DefaultPredictor, im_list: List[ndarray]) -> List:\n    with torch.no_grad():  # https://github.com/sphinx-doc/sphinx/issues/4258\n        inputs_list = []\n        for original_image in im_list:\n            # Apply pre-processing to image.\n            if predictor.input_format == \"RGB\":\n                # whether the model expects BGR inputs or RGB\n                original_image = original_image[:, :, ::-1]\n            height, width = original_image.shape[:2]\n            # Do not apply original augmentation, which is resize.\n            # image = predictor.aug.get_transform(original_image).apply_image(original_image)\n            image = original_image\n            image = torch.as_tensor(image.astype(\"float32\").transpose(2, 0, 1))\n            inputs = {\"image\": image, \"height\": height, \"width\": width}\n            inputs_list.append(inputs)\n        predictions = predictor.model(inputs_list)\n        return predictions","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.047473,"end_time":"2021-07-08T22:55:43.945088","exception":false,"start_time":"2021-07-08T22:55:43.897615","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-23T10:25:33.975313Z","iopub.execute_input":"2021-08-23T10:25:33.975666Z","iopub.status.idle":"2021-08-23T10:25:33.988156Z","shell.execute_reply.started":"2021-08-23T10:25:33.975625Z","shell.execute_reply":"2021-08-23T10:25:33.987348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inferance","metadata":{"papermill":{"duration":0.036428,"end_time":"2021-07-08T22:55:44.017658","exception":false,"start_time":"2021-07-08T22:55:43.98123","status":"completed"},"tags":[]}},{"cell_type":"code","source":"if debug:\n    dataset_dicts = dataset_dicts[:30]\n\nresults_list = []\nindex = 0\nbatch_size = 1\n\nfig, ax = plt.subplots(2, 5, figsize =(20,8))\nindices=[ax[0][0],ax[1][0],ax[0][1],ax[1][1],ax[0][2],ax[1][2],ax[0][3],ax[1][3],ax[0][4],ax[1][4] ]\n\n\nfor i in tqdm(range(ceil(len(dataset_dicts) / batch_size))):\n    inds = list(range(batch_size * i, min(batch_size * (i + 1), len(dataset_dicts))))\n    dataset_dicts_batch = [dataset_dicts[i] for i in inds]\n    im_list = [cv2.imread(d[\"file_name\"]) for d in dataset_dicts_batch]\n    outputs_list = predict_batch(predictor, im_list)\n\n    for im, outputs, d in zip(im_list, outputs_list, dataset_dicts_batch):\n        resized_height, resized_width, ch = im.shape\n        # outputs = predictor(im)\n        #############################################################\n\n        if index < 10:\n            # format is documented at https://detectron2.readthedocs.io/tutorials/models.html#model-output-format\n            v = Visualizer(\n                im[:, :, ::-1],\n                metadata=metadata,\n                scale=0.8,\n                instance_mode=ColorMode.IMAGE_BW\n                # remove the colors of unsegmented pixels. This option is only available for segmentation models\n            )\n            out = v.draw_instance_predictions(outputs[\"instances\"].to(\"cpu\"))\n            # cv2_imshow(out.get_image()[:, :, ::-1])\n            #cv2.imwrite(str(outdir / f\"pred_{index}.jpg\"), out.get_image()[:, :, ::-1])\n            \n            #cv2.imwrite(f\"./pred_{index}.jpg\", out.get_image()[:, :, ::-1])\n            \n            indices[index].grid(False)\n            indices[index].imshow(out.get_image()[:, :, ::-1])\n            \n                        \n         ###############################################################   \n            \n            \n            \n        df_meta = pd.read_csv(\"../input/siim-covid19-resized-1024px/meta.csv\")\n        test_meta=df_meta[df_meta.split==\"test\"]\n        \n        image_id, dim0, dim1,s = test_meta.iloc[index].values\n\n        instances = outputs[\"instances\"]\n        if len(instances) == 0:\n            # No finding, let's set 2 1.0 0 0 1 1x. Negative\n            result = {\n                \"image_id\": image_id +\"_image\",\n                \"negative\": 1,\n                \"typical\": 0,\n                \"indeterminate\": 0,\n                \"atypical\": 0,\n                 \"PredictionString\": \"none 1 0 0 1 1\"}\n        else:\n            # Find some bbox...\n            # print(f\"index={index}, find {len(instances)} bbox.\")\n            fields: Dict[str, Any] = instances.get_fields()\n            pred_classes = fields[\"pred_classes\"]  # (n_boxes,)\n            pred_scores = fields[\"scores\"]\n            # shape (n_boxes, 4). (xmin, ymin, xmax, ymax)\n            pred_boxes = fields[\"pred_boxes\"].tensor\n\n            h_ratio = dim0 / resized_height\n            w_ratio = dim1 / resized_width\n            pred_boxes[:, [0, 2]] *= w_ratio\n            pred_boxes[:, [1, 3]] *= h_ratio\n\n            pred_classes_array = pred_classes.cpu().numpy()\n            pred_boxes_array = pred_boxes.cpu().numpy()\n            pred_scores_array = pred_scores.cpu().numpy()\n            pred_classes_scores_array=np.stack((pred_classes_array,pred_scores_array), axis=-1)\n            #{'atypical': 0, 'indeterminate': 1, 'negative': 2, 'typical': 3}\n            \n            \n            typical= np.sum(pred_classes_scores_array[pred_classes_scores_array[:,0]==3, 1],axis=0)\n            negative= np.sum(pred_classes_scores_array[pred_classes_scores_array[:,0]==2, 1],axis=0)\n            indeterminate= np.sum(pred_classes_scores_array[pred_classes_scores_array[:,0]==1, 1],axis=0)\n            atypical= np.sum(pred_classes_scores_array[pred_classes_scores_array[:,0]==0, 1],axis=0)\n            \n            total=typical+negative+indeterminate+atypical\n            \n            typical=typical/total\n            negative=negative/total\n            indeterminate=indeterminate/total\n            atypical=atypical/total\n            \n                \n            result = {\n                \"image_id\": image_id +\"_image\",\n                \"negative\": negative,\n                \"typical\": typical,\n                \"indeterminate\": indeterminate,\n                \"atypical\": atypical,\n                \"PredictionString\": format_pred(pred_classes_array, pred_boxes_array, pred_scores_array),\n            }\n        results_list.append(result)\n        index += 1","metadata":{"papermill":{"duration":899.79283,"end_time":"2021-07-08T23:10:43.84749","exception":false,"start_time":"2021-07-08T22:55:44.05466","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-23T10:25:33.989461Z","iopub.execute_input":"2021-08-23T10:25:33.989871Z","iopub.status.idle":"2021-08-23T10:28:00.001259Z","shell.execute_reply.started":"2021-08-23T10:25:33.989837Z","shell.execute_reply":"2021-08-23T10:28:00.000373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_img_pre=None\ndf_img_pre = pd.DataFrame(results_list)\ndf_img_pre","metadata":{"papermill":{"duration":0.388635,"end_time":"2021-07-08T23:10:44.608942","exception":false,"start_time":"2021-07-08T23:10:44.220307","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-23T10:28:00.002327Z","iopub.execute_input":"2021-08-23T10:28:00.00273Z","iopub.status.idle":"2021-08-23T10:28:00.03209Z","shell.execute_reply.started":"2021-08-23T10:28:00.002665Z","shell.execute_reply":"2021-08-23T10:28:00.031101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df=None\nfilepaths = glob('/kaggle/input/siim-covid19-detection/test/**/*dcm',recursive=True)\ntest_df = pd.DataFrame({'filepath':filepaths,})\ntest_df['image_id'] = test_df.filepath.map(lambda x: x.split('/')[-1].replace('.dcm', '')+'_image')\ntest_df['study_id'] = test_df.filepath.map(lambda x: x.split('/')[-3].replace('.dcm', '')+'_study')\ntest_df.drop(['filepath'], axis=1, inplace=True)\ntest_df","metadata":{"papermill":{"duration":4.828166,"end_time":"2021-07-08T23:10:49.805358","exception":false,"start_time":"2021-07-08T23:10:44.977192","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-23T10:28:00.033417Z","iopub.execute_input":"2021-08-23T10:28:00.033773Z","iopub.status.idle":"2021-08-23T10:28:04.652244Z","shell.execute_reply.started":"2021-08-23T10:28:00.033737Z","shell.execute_reply":"2021-08-23T10:28:04.651423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_img_pre_df=None\ntest_img_pre_df=pd.merge(test_df, df_img_pre, on = 'image_id', how = 'left')\ntest_img_pre_df.sort_values('typical')","metadata":{"papermill":{"duration":0.38799,"end_time":"2021-07-08T23:10:50.554882","exception":false,"start_time":"2021-07-08T23:10:50.166892","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-23T10:28:04.65349Z","iopub.execute_input":"2021-08-23T10:28:04.65385Z","iopub.status.idle":"2021-08-23T10:28:04.680285Z","shell.execute_reply.started":"2021-08-23T10:28:04.653814Z","shell.execute_reply":"2021-08-23T10:28:04.679531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_img_pre_df=test_img_pre_df.groupby(['study_id']).mean()\ntest_img_pre_df.reset_index(inplace = True) \ntest_img_pre_df.sort_values('typical')\ntest_img_pre_df=test_img_pre_df.fillna(0)\ntest_img_pre_df","metadata":{"papermill":{"duration":0.396574,"end_time":"2021-07-08T23:10:51.526276","exception":false,"start_time":"2021-07-08T23:10:51.129702","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-23T10:28:04.681405Z","iopub.execute_input":"2021-08-23T10:28:04.681673Z","iopub.status.idle":"2021-08-23T10:28:04.706225Z","shell.execute_reply.started":"2021-08-23T10:28:04.681646Z","shell.execute_reply":"2021-08-23T10:28:04.705533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\n\ndf = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')\nid_laststr_list  = []\nfor i in range(df.shape[0]):\n    id_laststr_list.append(df.loc[i,'id'][-1])\ndf['id_last_str'] = id_laststr_list\nstudy_len = df[df['id_last_str'] == 'y'].shape[0]\nimage_len = df[df['id_last_str'] == 'e'].shape[0]\nprint(\"study_len:\" + str(study_len) + \"   image_len:\" + str(image_len))\ndf=df[df.id_last_str == 'y']\ndf","metadata":{"papermill":{"duration":0.417057,"end_time":"2021-07-08T23:10:52.304527","exception":false,"start_time":"2021-07-08T23:10:51.88747","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-23T10:28:04.708785Z","iopub.execute_input":"2021-08-23T10:28:04.709032Z","iopub.status.idle":"2021-08-23T10:28:04.762046Z","shell.execute_reply.started":"2021-08-23T10:28:04.709007Z","shell.execute_reply":"2021-08-23T10:28:04.761287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission=None\ndf_submission=pd.merge(df,test_img_pre_df, left_on='id', right_on='study_id', how = 'left')\ndf_submission=df_submission.fillna(0)\n#df_submission.drop(['PredictionString'], axis=1, inplace=True)\ndf_submission","metadata":{"papermill":{"duration":0.385628,"end_time":"2021-07-08T23:10:53.060426","exception":false,"start_time":"2021-07-08T23:10:52.674798","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-23T10:28:04.763159Z","iopub.execute_input":"2021-08-23T10:28:04.763492Z","iopub.status.idle":"2021-08-23T10:28:04.789666Z","shell.execute_reply.started":"2021-08-23T10:28:04.76345Z","shell.execute_reply":"2021-08-23T10:28:04.788903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nid_laststr_list  = []\nfor i in range(df.shape[0]):\n    id_laststr_list.append(df.loc[i,'id'][-1])\ndf['id_last_str'] = id_laststr_list\nstudy_len = df[df['id_last_str'] == 'y'].shape[0]\n\nfor i in range(study_len):\n    negative =  df_submission.loc[i,'negative'] \n    typical = df_submission.loc[i,'typical']\n    indeterminate = df_submission.loc[i,'indeterminate']\n    atypical = df_submission.loc[i,'atypical']\n    \n    negative_st=''\n    typical_st=''\n    indeterminate_st=''\n    atypical_st=''\n\n    if negative>0:\n        negative_st =f'negative {negative:0.3f} 0 0 1 1 '\n    if typical>0:\n        typical_st =f'typical {typical:0.3f} 0 0 1 1 '\n    if indeterminate>0:\n        indeterminate_st =f'indeterminate {indeterminate:0.3f} 0 0 1 1 '\n    if atypical>0:\n        atypical_st =f'atypical {atypical:0.3f} 0 0 1 1 '\n\n    #study_str = f'{negative_st}{typical_st}{indeterminate_st}{atypical_st}'\n        \n    #study_str = 'negative 1 0 0 1 1 atypical 1 0 0 1 1 typical 1 0 0 1 1 indeterminate 1 0 0 1 1'\n    study_str = f'negative {negative} 0 0 1 1 typical {typical} 0 0 1 1 indeterminate {indeterminate} 0 0 1 1 atypical {atypical} 0 0 1 1'\n    df.loc[i, 'PredictionString'] = study_str\n    \nsubmission_file_study = df[['id', 'PredictionString']]\nsubmission_file_study","metadata":{"papermill":{"duration":0.548663,"end_time":"2021-07-08T23:10:53.97261","exception":false,"start_time":"2021-07-08T23:10:53.423947","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-23T10:28:04.790937Z","iopub.execute_input":"2021-08-23T10:28:04.7913Z","iopub.status.idle":"2021-08-23T10:28:04.978371Z","shell.execute_reply.started":"2021-08-23T10:28:04.791266Z","shell.execute_reply":"2021-08-23T10:28:04.977467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_file_image = df_img_pre[['image_id', 'PredictionString']]\nsubmission_file_image.rename(columns={'image_id': 'id'}, inplace=True)\n#submission_file_image.PredictionString=\"none 1 0 0 1 1\"\nsubmission_file_image","metadata":{"papermill":{"duration":0.379892,"end_time":"2021-07-08T23:10:54.717528","exception":false,"start_time":"2021-07-08T23:10:54.337636","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-23T10:28:04.979618Z","iopub.execute_input":"2021-08-23T10:28:04.979955Z","iopub.status.idle":"2021-08-23T10:28:04.992305Z","shell.execute_reply.started":"2021-08-23T10:28:04.97992Z","shell.execute_reply":"2021-08-23T10:28:04.991252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_file = pd.concat([submission_file_study, submission_file_image], ignore_index=[True])","metadata":{"papermill":{"duration":0.375639,"end_time":"2021-07-08T23:10:55.458716","exception":false,"start_time":"2021-07-08T23:10:55.083077","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-23T10:28:04.993757Z","iopub.execute_input":"2021-08-23T10:28:04.994189Z","iopub.status.idle":"2021-08-23T10:28:05.000807Z","shell.execute_reply.started":"2021-08-23T10:28:04.99415Z","shell.execute_reply":"2021-08-23T10:28:04.999794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#submission_file.to_csv('submission.csv', index=False)\nsubmission_file","metadata":{"papermill":{"duration":0.388578,"end_time":"2021-07-08T23:10:56.212205","exception":false,"start_time":"2021-07-08T23:10:55.823627","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-23T10:28:05.002134Z","iopub.execute_input":"2021-08-23T10:28:05.002616Z","iopub.status.idle":"2021-08-23T10:28:05.016999Z","shell.execute_reply.started":"2021-08-23T10:28:05.002511Z","shell.execute_reply":"2021-08-23T10:28:05.016067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')\ndf.drop(['PredictionString'], axis=1, inplace=True)","metadata":{"papermill":{"duration":0.376833,"end_time":"2021-07-08T23:10:56.953217","exception":false,"start_time":"2021-07-08T23:10:56.576384","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-23T10:28:05.023215Z","iopub.execute_input":"2021-08-23T10:28:05.024165Z","iopub.status.idle":"2021-08-23T10:28:05.044912Z","shell.execute_reply.started":"2021-08-23T10:28:05.024099Z","shell.execute_reply":"2021-08-23T10:28:05.043801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_file1 = df.join(submission_file.set_index('id'), on = 'id')","metadata":{"papermill":{"duration":0.376035,"end_time":"2021-07-08T23:10:57.694769","exception":false,"start_time":"2021-07-08T23:10:57.318734","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-23T10:28:05.046779Z","iopub.execute_input":"2021-08-23T10:28:05.047327Z","iopub.status.idle":"2021-08-23T10:28:05.064046Z","shell.execute_reply.started":"2021-08-23T10:28:05.047292Z","shell.execute_reply":"2021-08-23T10:28:05.06286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_file1","metadata":{"papermill":{"duration":0.380107,"end_time":"2021-07-08T23:10:58.439374","exception":false,"start_time":"2021-07-08T23:10:58.059267","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-23T10:28:05.065159Z","iopub.execute_input":"2021-08-23T10:28:05.067708Z","iopub.status.idle":"2021-08-23T10:28:05.086566Z","shell.execute_reply.started":"2021-08-23T10:28:05.067664Z","shell.execute_reply":"2021-08-23T10:28:05.085807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training","metadata":{}},{"cell_type":"code","source":"\nimport numpy as np \nimport pandas as pd \nfrom datetime import datetime\nimport time\nfrom tqdm import tqdm_notebook as tqdm # progress bar\nimport matplotlib.pyplot as plt\n\nimport os, json, cv2, random\nimport skimage.io as io\nimport copy\nimport pickle\nfrom pathlib import Path\nfrom typing import Optional\nfrom tqdm import tqdm\n\n# torch\nimport torch\n\n\n\n# Albumenatations\nimport albumentations as A\nfrom albumentations.pytorch.transforms import ToTensorV2\n\n#from pycocotools.coco import COCO\nfrom sklearn.model_selection import StratifiedKFold\n\n# glob\nfrom glob import glob\n\n# numba\nimport numba\nfrom numba import jit\n\nimport warnings\nwarnings.filterwarnings('ignore') #Ignore \"future\" warnings and Data-Frame-Slicing warnings.\n\n\n# detectron2\nfrom detectron2.structures import BoxMode\nfrom detectron2 import model_zoo\nfrom detectron2.config import get_cfg\nfrom detectron2.data import DatasetCatalog, MetadataCatalog\nfrom detectron2.engine import DefaultPredictor, DefaultTrainer, launch\nfrom detectron2.evaluation import COCOEvaluator\nfrom detectron2.structures import BoxMode\nfrom detectron2.utils.visualizer import ColorMode\nfrom detectron2.utils.logger import setup_logger\nfrom detectron2.utils.visualizer import Visualizer\n\nfrom detectron2.data import DatasetCatalog, MetadataCatalog, build_detection_test_loader, build_detection_train_loader\nfrom detectron2.data import detection_utils as utils\n\n\nfrom detectron2.data import DatasetCatalog, MetadataCatalog, build_detection_test_loader, build_detection_train_loader\nfrom detectron2.data import detection_utils as utils\nimport detectron2.data.transforms as T\nfrom detectron2.evaluation import COCOEvaluator, inference_on_dataset\n\nsetup_logger()","metadata":{"execution":{"iopub.status.busy":"2021-08-23T10:28:05.090205Z","iopub.execute_input":"2021-08-23T10:28:05.090941Z","iopub.status.idle":"2021-08-23T10:28:06.358364Z","shell.execute_reply.started":"2021-08-23T10:28:05.090904Z","shell.execute_reply":"2021-08-23T10:28:06.35757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# --- Read data ---\nimgdir = \"../input/siim-covid19-resized-384512-and-640px/SIIM-COVID19-Resized/img_sz_640\"\n# Read in the data CSV files\ntrain_df = pd.read_csv(\"../input/d/ammarnassanalhajali/siimcovid19-detection-training-label/train_image_df.csv\")\nlen(train_df)","metadata":{"execution":{"iopub.status.busy":"2021-08-23T10:28:06.359621Z","iopub.execute_input":"2021-08-23T10:28:06.359939Z","iopub.status.idle":"2021-08-23T10:28:06.454329Z","shell.execute_reply.started":"2021-08-23T10:28:06.359909Z","shell.execute_reply":"2021-08-23T10:28:06.453583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# --- configs ---\nthing_classes = [\n    \"atypical\",\n    \"indeterminate\",\n    \"negative\",\n    \"typical\"\n]\n\ndebug=False\nsplit_mode=\"valid20\" # Or  valid20 all_train\n\n\ncategory_name_to_id = {class_name: index for index, class_name in enumerate(thing_classes)}\ncategory_name_to_id","metadata":{"execution":{"iopub.status.busy":"2021-08-23T10:28:06.455659Z","iopub.execute_input":"2021-08-23T10:28:06.456016Z","iopub.status.idle":"2021-08-23T10:28:06.465523Z","shell.execute_reply.started":"2021-08-23T10:28:06.455978Z","shell.execute_reply":"2021-08-23T10:28:06.464636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from glob import glob\n\ndef get_COVID19_data_dicts(\n    imgdir: Path,\n    train_df: pd.DataFrame,\n    use_cache: bool = True,\n    target_indices: Optional[np.ndarray] = None,\n    debug: bool = False,\n    data_type:str=\"train\"\n   \n):\n\n    cache_path = Path(\".\") / f\"dataset_dicts_cache_{data_type}.pkl\"\n    if not use_cache or not cache_path.exists():\n        print(\"Creating data...\")\n        df_meta = pd.read_csv(\"../input/siim-covid19-resized-384512-and-640px/SIIM-COVID19-Resized/img_sz_640/meta_sz_640.csv\")\n        train_meta=df_meta[df_meta.split==\"train\"]\n        if debug:\n            train_meta = train_meta.iloc[:100]  # For debug....\n\n        # Load 1 image to get image size.\n        image_id = train_meta.iloc[0,0]\n        #image_path = str(imgdir / \"train\" / f\"{image_id}.jpg\")\n        image_path = str(f'../input/siim-covid19-resized-384512-and-640px/SIIM-COVID19-Resized/img_sz_640/train/{image_id}.jpg')\n        image = cv2.imread(image_path)\n        resized_height, resized_width, ch = image.shape\n        print(f\"image shape: {image.shape}\")\n\n        dataset_dicts = []\n        for index, train_meta_row in tqdm(train_meta.iterrows(), total=len(train_meta)):\n            record = {}\n            image_id, height, width,s = train_meta_row.values\n            #filename = str(imgdir / \"train\" / f\"{image_id}.jpg\")\n            filename = str(f'../input/siim-covid19-resized-384512-and-640px/SIIM-COVID19-Resized/img_sz_640/train/{image_id}.jpg')\n            record[\"file_name\"] = filename\n            record[\"image_id\"] = image_id\n            record[\"height\"] = resized_height\n            record[\"width\"] = resized_width\n            objs = []\n            for index2, row in train_df.query(\"id == @image_id\").iterrows():\n                # print(row)\n                # print(row[\"class_name\"])\n                # class_name = row[\"class_name\"]\n                class_id = row[\"integer_label\"]\n                if class_id == 2: # NO class\n                    # It is \"No finding\"\n \n                    # Use this No finding class with the bbox covering all image area.\n                    #bbox_resized = [0, 0, resized_width, resized_height]\n                    bbox_resized = [50, 50, 200, 200]\n                    obj = {\n                        \"bbox\": bbox_resized,\n                        \"bbox_mode\": BoxMode.XYXY_ABS,\n                        \"category_id\": class_id,\n                    }\n                    #objs.append(obj)\n\n                else:\n                    # bbox_original = [int(row[\"x_min\"]), int(row[\"y_min\"]), int(row[\"x_max\"]), int(row[\"y_max\"])]\n                    h_ratio = resized_height / height\n                    w_ratio = resized_width / width\n                    bbox_resized = [\n                        float(row[\"x_min\"]) * w_ratio,\n                        float(row[\"y_min\"]) * h_ratio,\n                        float(row[\"x_max\"]) * w_ratio,\n                        float(row[\"y_max\"]) * h_ratio,\n                    ]\n                    obj = {\n                        \"bbox\": bbox_resized,\n                        \"bbox_mode\": BoxMode.XYXY_ABS,\n                        \"category_id\": class_id,\n                    }\n                    objs.append(obj)\n            record[\"annotations\"] = objs\n            dataset_dicts.append(record)\n        with open(cache_path, mode=\"wb\") as f:\n            pickle.dump(dataset_dicts, f)\n\n    print(f\"Load from cache {cache_path}\")\n    with open(cache_path, mode=\"rb\") as f:\n        dataset_dicts = pickle.load(f)\n    if target_indices is not None:\n        dataset_dicts = [dataset_dicts[i] for i in target_indices]\n    return dataset_dicts\n\n\ndef get_COVID19_data_dicts_test(\n    imgdir: Path, test_meta: pd.DataFrame, use_cache: bool = True, debug: bool = False,\n):\n    debug_str = f\"_debug{int(debug)}\"\n    cache_path = Path(\".\") / f\"dataset_dicts_cache_test.pkl\"\n    if not use_cache or not cache_path.exists():\n        print(\"Creating data...\")\n        # test_meta = pd.read_csv(imgdir / \"test_meta.csv\")\n        df_meta = pd.read_csv(\"../input/siim-covid19-resized-384512-and-640px/SIIM-COVID19-Resized/img_sz_640/meta_sz_640.csv\")\n        test_meta=df_meta[df_meta.split==\"test\"]\n        if debug:\n            test_meta = test_meta.iloc[:100]  # For debug....\n        # Load 1 image to get image size.\n        image_id = test_meta.iloc[0,0]\n        #image_path = str(imgdir / \"test\" / f\"{image_id}.jpg\")\n        image_path = str(f'../input/siim-covid19-resized-384512-and-640px/SIIM-COVID19-Resized/img_sz_640/test/{image_id}.jpg')\n        image = cv2.imread(image_path)\n        resized_height, resized_width, ch = image.shape\n        #print(f\"image shape: {image.shape}\")\n\n        dataset_dicts = []\n        for index, test_meta_row in tqdm(test_meta.iterrows(), total=len(test_meta)):\n            record = {}\n\n            image_id, height, width,s = test_meta_row.values\n            #filename = str(imgdir / \"test\" / f\"{image_id}.jpg\")\n            filename = str(f'../input/siim-covid19-resized-384512-and-640px/SIIM-COVID19-Resized/img_sz_640/test/{image_id}.jpg')\n            record[\"file_name\"] = filename\n            # record[\"image_id\"] = index\n            record[\"image_id\"] = image_id\n            record[\"height\"] = resized_height\n            record[\"width\"] = resized_width\n            # objs = []\n            # record[\"annotations\"] = objs\n            dataset_dicts.append(record)\n        with open(cache_path, mode=\"wb\") as f:\n            pickle.dump(dataset_dicts, f)\n\n    #print(f\"Load from cache {cache_path}\")\n    with open(cache_path, mode=\"rb\") as f:\n        dataset_dicts = pickle.load(f)\n    return dataset_dicts","metadata":{"execution":{"iopub.status.busy":"2021-08-23T10:28:06.468571Z","iopub.execute_input":"2021-08-23T10:28:06.468839Z","iopub.status.idle":"2021-08-23T10:28:06.492013Z","shell.execute_reply.started":"2021-08-23T10:28:06.468805Z","shell.execute_reply":"2021-08-23T10:28:06.491216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if split_mode == \"all_train\":\n    DatasetCatalog.register(\n        \"COVID19_data_train\",\n        lambda: get_COVID19_data_dicts(\n            imgdir,\n            train_df,\n            debug=debug,\n            data_type=\"train\"\n        ),\n    )\n    MetadataCatalog.get(\"COVID19_data_train\").set(thing_classes=thing_classes)\n    \n    \n    dataset_dicts_train = DatasetCatalog.get(\"COVID19_data_train\")\n    metadata_dicts_train = MetadataCatalog.get(\"COVID19_data_train\")\n    \n    \nelif split_mode == \"valid20\":\n\n    n_dataset = len(\n        get_COVID19_data_dicts(\n            imgdir, train_df, debug=debug,data_type=\"All\"\n        )\n    )\n    n_train = int(n_dataset * 0.90)\n    print(\"n_dataset\", n_dataset, \"n_train\", n_train)\n    rs = np.random.RandomState(42)\n    inds = rs.permutation(n_dataset)\n    train_inds, valid_inds = inds[:n_train], inds[n_train:]\n\n    DatasetCatalog.register(\n        \"COVID19_data_train\",\n        lambda: get_COVID19_data_dicts(\n            imgdir,\n            train_df,\n            target_indices=train_inds,\n            debug=debug,\n            data_type=\"train\"\n        ),\n    )\n    MetadataCatalog.get(\"COVID19_data_train\").set(thing_classes=thing_classes)\n    \n\n    DatasetCatalog.register(\n        \"COVID19_data_valid\",\n        lambda: get_COVID19_data_dicts(\n            imgdir,\n            train_df,\n            target_indices=valid_inds,\n            debug=debug,\n            data_type=\"val\"\n            ),\n        )\n    MetadataCatalog.get(\"COVID19_data_valid\").set(thing_classes=thing_classes)\n    \n    dataset_dicts_train = DatasetCatalog.get(\"COVID19_data_train\")\n    metadata_dicts_train = MetadataCatalog.get(\"COVID19_data_train\")\n\n    dataset_dicts_valid = DatasetCatalog.get(\"COVID19_data_valid\")\n    metadata_dicts_valid = MetadataCatalog.get(\"COVID19_data_valid\")\n    \nelse:\n    raise ValueError(f\"[ERROR] Unexpected value split_mode={split_mode}\")","metadata":{"execution":{"iopub.status.busy":"2021-08-23T10:28:06.494849Z","iopub.execute_input":"2021-08-23T10:28:06.495128Z","iopub.status.idle":"2021-08-23T10:29:01.080391Z","shell.execute_reply.started":"2021-08-23T10:28:06.495103Z","shell.execute_reply":"2021-08-23T10:29:01.079518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(2, 4, figsize =(20,10))\nindices=[ax[0][0],ax[1][0],ax[0][1],ax[1][1],ax[0][2],ax[1][2],ax[0][3],ax[1][3]]\ni=-1\nfor d in random.sample(dataset_dicts_train, 8):\n    i=i+1    \n    img = cv2.imread(d[\"file_name\"])\n    v = Visualizer(img[:, :, ::-1],\n                   metadata=metadata_dicts_train, \n                   scale=0.3, \n                   instance_mode=ColorMode.IMAGE_BW   # remove the colors of unsegmented pixels. This option is only available for segmentation models\n    )\n    out = v.draw_dataset_dict(d)\n    indices[i].grid(False)\n    indices[i].axis('off')\n    indices[i].imshow(out.get_image()[:, :, ::-1])","metadata":{"execution":{"iopub.status.busy":"2021-08-23T10:29:01.082554Z","iopub.execute_input":"2021-08-23T10:29:01.082925Z","iopub.status.idle":"2021-08-23T10:29:02.249545Z","shell.execute_reply.started":"2021-08-23T10:29:01.082885Z","shell.execute_reply":"2021-08-23T10:29:02.248616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def custom_mapper(dataset_dict):\n    \n    dataset_dict = copy.deepcopy(dataset_dict)\n    image = utils.read_image(dataset_dict[\"file_name\"], format=\"BGR\")\n    transform_list = [T.Resize((640,640)),\n                      T.RandomBrightness(0.8, 1.2),\n                      T.RandomFlip(prob=0.5, horizontal=False, vertical=True),\n                      T.RandomFlip(prob=0.5, horizontal=True, vertical=False)\n                      ]\n    image, transforms = T.apply_transform_gens(transform_list, image)\n    dataset_dict[\"image\"] = torch.as_tensor(image.transpose(2, 0, 1).astype(\"float32\"))\n\n    annos = [\n        utils.transform_instance_annotations(obj, transforms, image.shape[:2])\n        for obj in dataset_dict.pop(\"annotations\")\n        if obj.get(\"iscrowd\", 0) == 0\n    ]\n    instances = utils.annotations_to_instances(annos, image.shape[:2])\n    dataset_dict[\"instances\"] = utils.filter_empty_instances(instances)\n    return dataset_dict\nclass AugTrainer(DefaultTrainer):\n    @classmethod\n    def build_train_loader(cls, cfg):\n        return build_detection_train_loader(cfg, mapper=custom_mapper)","metadata":{"execution":{"iopub.status.busy":"2021-08-23T10:29:02.251081Z","iopub.execute_input":"2021-08-23T10:29:02.251428Z","iopub.status.idle":"2021-08-23T10:29:02.263137Z","shell.execute_reply.started":"2021-08-23T10:29:02.251393Z","shell.execute_reply":"2021-08-23T10:29:02.261928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cfg = get_cfg()\nconfig_name = \"COCO-Detection/faster_rcnn_R_50_FPN_3x.yaml\" \n#config_name = \"COCO-Detection/faster_rcnn_X_101_32x8d_FPN_3x.yaml\"\n#config_name = \"COCO-Detection/faster_rcnn_R_101_C4_3x.yaml\"\n\ncfg.merge_from_file(model_zoo.get_config_file(config_name))\n\ncfg.DATASETS.TRAIN = (\"COVID19_data_train\",)\n\nif split_mode == \"all_train\":\n    cfg.DATASETS.TEST = ()\nelse:\n    cfg.DATASETS.TEST = (\"COVID19_data_valid\",)\n    cfg.TEST.EVAL_PERIOD = 1000\n# ../input/d/ammarnassanalhajali/1siim-covid19-detectron2-weights/output\ncfg.DATALOADER.NUM_WORKERS = 0\n#cfg.MODEL.WEIGHTS = model_zoo.get_checkpoint_url(config_name)\ncfg.MODEL.WEIGHTS=\"../input/d/ammarnassanalhajali/1siim-covid19-detectron2-weights/output/model_final.pth\"\n\n\ncfg.SOLVER.IMS_PER_BATCH = 2\ncfg.SOLVER.BASE_LR = 0.02\n\ncfg.SOLVER.WARMUP_ITERS = 1000\ncfg.SOLVER.MAX_ITER = 5000 #adjust up if val mAP is still rising, adjust down if overfit\n#cfg.SOLVER.STEPS = (100, 500) # must be less than  MAX_ITER \n#cfg.SOLVER.GAMMA = 0.05\n\n\ncfg.SOLVER.CHECKPOINT_PERIOD = 100000  # Small value=Frequent save need a lot of storage.\ncfg.MODEL.ROI_HEADS.BATCH_SIZE_PER_IMAGE = 128\ncfg.MODEL.ROI_HEADS.NUM_CLASSES = 4\n\n\nos.makedirs(cfg.OUTPUT_DIR, exist_ok=True)\n\n\n#Training using custom trainer defined above\n#trainer = AugTrainer(cfg) \ntrainer = DefaultTrainer(cfg) \ntrainer.resume_or_load(resume=False)\ntrainer.train()","metadata":{"execution":{"iopub.status.busy":"2021-08-23T10:29:02.264903Z","iopub.execute_input":"2021-08-23T10:29:02.265268Z","iopub.status.idle":"2021-08-23T10:29:14.470539Z","shell.execute_reply.started":"2021-08-23T10:29:02.265231Z","shell.execute_reply":"2021-08-23T10:29:14.468083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaluator = COCOEvaluator(\"COVID19_data_valid\", cfg, False, output_dir=\"./output/\")\n#cfg.MODEL.WEIGHTS=\"./output/model_final.pth\"\n#cfg.MODEL.ROI_HEADS.SCORE_THRESH_TEST = 0.001   # set a custom testing threshold\nval_loader = build_detection_test_loader(cfg, \"COVID19_data_valid\")\ninference_on_dataset(trainer.model, val_loader, evaluator)","metadata":{"execution":{"iopub.status.busy":"2021-08-23T10:29:14.4718Z","iopub.status.idle":"2021-08-23T10:29:14.47252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nmetrics_df = pd.read_json(\"./output/metrics.json\", orient=\"records\", lines=True)\nmdf = metrics_df.sort_values(\"iteration\")\nmdf.head(10).T","metadata":{"execution":{"iopub.status.busy":"2021-08-23T10:29:14.473825Z","iopub.status.idle":"2021-08-23T10:29:14.474445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 1. Loss curve\nfig, ax = plt.subplots()\n\nmdf1 = mdf[~mdf[\"total_loss\"].isna()]\nax.plot(mdf1[\"iteration\"], mdf1[\"total_loss\"], c=\"C0\", label=\"train\")\nif \"validation_loss\" in mdf.columns:\n    mdf2 = mdf[~mdf[\"validation_loss\"].isna()]\n    ax.plot(mdf2[\"iteration\"], mdf2[\"validation_loss\"], c=\"C1\", label=\"validation\")\n\n# ax.set_ylim([0, 0.5])\nax.legend()\nax.set_title(\"Loss curve\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-23T10:29:14.475697Z","iopub.status.idle":"2021-08-23T10:29:14.476251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 1. Loss curve\nfig, ax = plt.subplots()\n\nmdf1 = mdf[~mdf[\"fast_rcnn/cls_accuracy\"].isna()]\nax.plot(mdf1[\"iteration\"], mdf1[\"fast_rcnn/cls_accuracy\"], c=\"C0\", label=\"train\")\n# ax.set_ylim([0, 0.5])\nax.legend()\nax.set_title(\"Accuracy curve\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-23T10:29:14.477448Z","iopub.status.idle":"2021-08-23T10:29:14.478068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}