{"cells":[{"metadata":{"_uuid":"2ec10353def0140ae4a568d0f5e403e5d8500639"},"cell_type":"markdown","source":""},{"metadata":{"_uuid":"d05916067f6ecdf1baedf0b9fd4017675aeed0f8"},"cell_type":"markdown","source":"## Phases of post-processing for RSNA predictions"},{"metadata":{"trusted":true,"_uuid":"ff6712c45c7ddf8bbe163da61aff57019a771dd2"},"cell_type":"code","source":"on_kaggle = True","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f52d644880de098e4d6496803739caf2bdd8b738"},"cell_type":"code","source":"# Non-max suppression parameters for phase 0\nOTHRESH=0.05    # Maximum acceptable overlap\nPTHRESH=0.94    # Confidence threshold\nAVPTHRESH=0.65  # Minimum average confidence\n# Class probability threshold for phase 1\nTHRESH = .175\n# Confidence threshold for phase 2\nMIN_MAX_MRCNN_CONF = .975\n# Parameters for phase 3\nMINPROB = 0.69  # Value to set for below threshold confidence (<0.7)\nC1 = 1e6  # Logistic regularization parameters\nC2 = 0.06\n# Parameters for phase 5\nUNET_MINPROB = 0.28  # Minimum probability to add Unet cases\nUNET_MINCONF = 0.35  # Minimum confidence to add Unet cases\n# Parameters for phase 6\nYOLO_MINCONF = 0.15  # Minimum confidence for a yolo box to be considered\nMINMIN = 0.20  # Minimum minimum probability in an iteration of yolo additions before stopping","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8543a3aeea5bef2782616afaa9e5b201123ca1b1"},"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nfrom sklearn import metrics\nfrom scipy.special import logit,expit\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import mean_squared_error","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"24fda38084b964edbee31084bd7c4761cfa7a0b5"},"cell_type":"code","source":"os.listdir('../input')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fd5ccbc3239c0ba39cde3fae919c48d58e0d9662"},"cell_type":"code","source":"!ls ../input/gs-dense-chexnet-predict-test-from-all-models","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"295814121c7eb96f1f52e1075b4ed29e47690464"},"cell_type":"code","source":"RESULTS_LOC = '../input'  # Outputs from base models\nRAW_DATA_LOC = '../input/rsna-pneumonia-detection-challenge'          # Raw input data\nYOLO_SUB_STEM = 'yolov3-inference-from-multiple-saved-wts-stage-2/submission_yolo'    # YoloV3 predictions\nYOLO_SUB_STEM2 = 'yolov3-inference-from-more-saved-weights-stage-2/submission_yolo'    # YoloV3 predictions\nMORE_YOLO_EPOCHS = [5300, 1500, 6700, 2000] if on_kaggle else []\nUNET_RESULTS_PATH = '../input/resnet-unet-from-saved-weights-stage-2/submission_resnet34unet_thr0.35.csv'  # Resnet-Unet predictions\nCHEXNET_OOF_STEM = 'rsna-oof-predictions/predictions_valid_fold_'        # Giulia's classification model\nCHEXNET_TEST_STEM = 'gs-dense-chexnet-predict-stage-2-from-all-models/test_preds_pth_fold'\n# ^ but fold0 may be different\nSPECIAL_FILE0 = 'gs-dense-chexnet-predict-stage-2-from-all-models/test_preds_pth_fold0_for_combined_folds'\nCHEXNET_TEST_FILE0 = SPECIAL_FILE0 if on_kaggle else CHEXNET_TEST_STEM + '0'\nADENSE_TEST_STEM = 'andy-densenet-from-multiple-saved-weights-stage-2/submission_adense_f'                    # Andy's classification model\nADEMSE_OOF_STEM = 'rsna-oof-predictions/val_dense_v5_a1_f'\nMRCNN_TEST_STEM = 'mrcnn-inference-from-saved-wts-4dig-70-stage-2/submission_mrcnn_v'  # MRCNN predicted boxes\n# ^ need to fix v8 vs v9 (same weights, different output, but here both used v8 name, so 'v8'+'',\n#   instead of 'v'+'9')\nMRCNN95_TEST_STEM = 'mrcnn-inference-from-saved-wts-2dig-95-stage-2/submission_mrcnn_v' \\\n   if on_kaggle else MRCNN_TEST_STEM  # MRCNN predicted boxes\nMRCNN_OUT_STEM = 'temp_mrcnn_v' if on_kaggle else MRCNN_TEST_STEM\nMRCNN_OOF_STEM = 'rsna-oof-predictions/val_v'\nMRCNN_OOF_OUT_STEM = 'val_v' if on_kaggle else MRCNN_OOF_STEM\nMRCNN_TEST_SEP = ''\nMRCNN_OOF_SEP = '_'\nIN_OOF_SUFFIX = ''\nIN_SUFFIX = '.csv'\nOUT_SUFFIX = '.csv'\nCHEX_CLASS_PRED_PATH = '../input/combine-fold-class-predictions-stage-2/test_probs.csv'  # Logisitc average of Giulia's test class preds\nKERNEL_OUTPUT_PATH = '../input/filter-199-final-with-higher-thresh-stage-2/filter199.csv'      # \"Pneumonia - Segm. filtered through Class.\" kernel","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"37e1a1fb39d35b98cd13e969ec31ede153323f3d"},"cell_type":"code","source":"# For phase 3 only\nN_FOLDS = 5\nOUTPUT_LOC = '.'\n\nOOF_INFILE_STEM = MRCNN_OOF_STEM + '9' + MRCNN_OOF_SEP + 'a1' + MRCNN_OOF_SEP + 'f'\nvers = '8' if on_kaggle else '9'\nTEST_INFILE_STEM = MRCNN_TEST_STEM + vers + MRCNN_TEST_SEP + 'a1' + MRCNN_TEST_SEP + 'f'\nOOF_OUTFILE_NAME = MRCNN_OOF_OUT_STEM + '10' + MRCNN_OOF_SEP + 'a1' + \\\n                   MRCNN_OOF_SEP + 'allfolds_out.csv'\n#TEST_OUTFILE_STEM = MRCNN_TEST_STEM + '10' + MRCNN_TEST_SEP + 'a1' + MRCNN_TEST_SEP + 'f_out'\n# above won't work on kaggle\nTEST_OUTFILE_STEM = MRCNN_OUT_STEM + '10' + MRCNN_TEST_SEP + 'a1' + MRCNN_TEST_SEP + 'f_out'\nOOF_CLASS_PROBS_OUTFILE = 'phase3_oof_out.csv'\nTEST_CLASS_PROBS_OUTFILE_STEM = 'phase3_test_out'\nFULL_TEST_CLASS_PROBS_OUTFILE = 'phase3_test_avg_out.csv'","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"45d8b9382f706ede8b232db53626dcb71b070194"},"cell_type":"markdown","source":"### Phase 0:<br> Apply non-max suppression to fold predictions for 2 different 5-fold assignments"},{"metadata":{"trusted":true,"_uuid":"cd0bcaa6706f7c4d118187299bb24bf85d094582"},"cell_type":"code","source":"dfs = []\n\nfor a in range(1,3):\n    for f in range(5):\n        vers = '8' if a==2 and f==0 else '1'\n        fn = RESULTS_LOC +'/'+ MRCNN95_TEST_STEM + vers + MRCNN_TEST_SEP + \\\n             'a' + str(a) + MRCNN_TEST_SEP + 'f' + str(f) + IN_SUFFIX\n        dfs.append( pd.read_csv(fn).set_index('patientId') )\n\nfor i,f in enumerate(dfs):\n    if i:\n        df = df.join(f.rename(columns={'PredictionString':'pred'+str(i)}))\n    else:\n        df = f.rename(columns={'PredictionString':'pred'+str(i)})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"25a9cf0cf518206ada9383bb1b0506e61e0ce9ad"},"cell_type":"code","source":"# Implementation of non-max suppression from\n#   https://github.com/jrosebr1/imutils/blob/master/imutils/object_detection.py\ndef non_max_suppression(boxes, probs=None, overlapThresh=OTHRESH):\n\t# if there are no boxes, return an empty list\n\tif len(boxes) == 0:\n\t\treturn []\n\n\t# if the bounding boxes are integers, convert them to floats -- this\n\t# is important since we'll be doing a bunch of divisions\n\tif boxes.dtype.kind == \"i\":\n\t\tboxes = boxes.astype(\"float\")\n\n\t# initialize the list of picked indexes\n\tpick = []\n\n\t# grab the coordinates of the bounding boxes\n\tx1 = boxes[:, 0]\n\ty1 = boxes[:, 1]\n\tx2 = boxes[:, 2]\n\ty2 = boxes[:, 3]\n\n\t# compute the area of the bounding boxes and grab the indexes to sort\n\t# (in the case that no probabilities are provided, simply sort on the\n\t# bottom-left y-coordinate)\n\tarea = (x2 - x1 + 1) * (y2 - y1 + 1)\n\tidxs = y2\n\n\t# if probabilities are provided, sort on them instead\n\tif probs is not None:\n\t\tidxs = probs\n\n\t# sort the indexes\n\tidxs = np.argsort(idxs)\n\n\t# keep looping while some indexes still remain in the indexes list\n\twhile len(idxs) > 0:\n\t\t# grab the last index in the indexes list and add the index value\n\t\t# to the list of picked indexes\n\t\tlast = len(idxs) - 1\n\t\ti = idxs[last]\n\t\tpick.append(i)\n\n\t\t# find the largest (x, y) coordinates for the start of the bounding\n\t\t# box and the smallest (x, y) coordinates for the end of the bounding\n\t\t# box\n\t\txx1 = np.maximum(x1[i], x1[idxs[:last]])\n\t\tyy1 = np.maximum(y1[i], y1[idxs[:last]])\n\t\txx2 = np.minimum(x2[i], x2[idxs[:last]])\n\t\tyy2 = np.minimum(y2[i], y2[idxs[:last]])\n\n\t\t# compute the width and height of the bounding box\n\t\tw = np.maximum(0, xx2 - xx1 + 1)\n\t\th = np.maximum(0, yy2 - yy1 + 1)\n\n\t\t# compute the ratio of overlap\n\t\toverlap = (w * h) / area[idxs[:last]]\n\n\t\t# delete all indexes from the index list that have overlap greater\n\t\t# than the provided overlap threshold\n\t\tidxs = np.delete(idxs, np.concatenate(([last],\n\t\t\tnp.where(overlap > overlapThresh)[0])))\n\n\t# return only the bounding boxes that were picked\n\treturn boxes[pick].astype(\"int\"), list(np.array(probs)[pick])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9840ada97d90aec8877c59a5043ee0ab0581bca0"},"cell_type":"code","source":"box_dict = {}\nconf_dict = {}\nfor pat,row in df.iterrows():\n    boxes = []\n    confs = []\n    maxconfs = []\n    for pred in row:\n        maxconf = 0.\n        if isinstance(pred, str):\n            s = pred.split(' ')\n            if s[-1]=='':\n                s.pop()  # remove terminating null\n            if s[0]=='':\n                s.pop(0)  # remove initial null\n            if ( len(s)%5 ):\n                print( 'Bad prediction string.')\n            while len(s):\n                conf = float(s.pop(0))\n                x = int(round(float(s.pop(0))))\n                y = int(round(float(s.pop(0))))\n                w = int(round(float(s.pop(0))))\n                h = int(round(float(s.pop(0))))\n                if conf>maxconf:\n                    maxconf = conf\n                if conf>PTHRESH:\n                    boxes.append( [x,y,x+w,y+h] )\n                    confs.append( conf )\n        maxconfs.append(maxconf)\n    avgconf = sum(maxconfs)/len(maxconfs)\n    if len(boxes) and avgconf>AVPTHRESH:\n        box_dict[pat] = boxes\n        conf_dict[pat] = confs\nlen(box_dict), len(conf_dict)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bb0324562dd2828be44a44487b088a0729832816"},"cell_type":"code","source":"box_dict_nms = {}\nconf_dict_nms = {}\nfor p in box_dict:\n    boxes, confs = non_max_suppression(np.array(box_dict[p]), np.array(conf_dict[p]))\n    box_dict_nms[p] = boxes\n    conf_dict_nms[p] = confs","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"49897e02a2376f91e49bf49bb34081af96afd4a9"},"cell_type":"code","source":"sub_dict = {}\nfor p in df.index:\n    predictionString = ''\n    if p in box_dict_nms:\n        for box, conf in zip(box_dict_nms[p], conf_dict_nms[p]):\n            # retrieve x, y, height and width\n            x, y, x2, y2 = box\n            height = y2 - y\n            width = x2 - x\n            # add to predictionString\n            predictionString += str(conf) + ' ' + str(x) + ' ' + str(y) + ' ' + str(width) + ' ' + str(height) + ' '\n    sub_dict[p] = predictionString\n\n# save submission file\nsub = pd.DataFrame.from_dict(sub_dict,orient='index')\nsub.index.names = ['patientId']\nsub.columns = ['PredictionString']\nphase0_output = sub","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e6e711a565dfe3cf39f2af4480836814e1bcc2cd"},"cell_type":"markdown","source":"### Phase 1<br>Apply classification probability threshold to non-max suppression output"},{"metadata":{"trusted":true,"_uuid":"f088a5bcd5ff49dbef93ef1b043c92a5431895b6"},"cell_type":"code","source":"probs = pd.read_csv(CHEX_CLASS_PRED_PATH)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6a6171372c211ef73bb69548c2c3c87d01ed27be"},"cell_type":"code","source":"df = phase0_output.join(probs.set_index('patientId'))\nout = df.copy()\nout.loc[df.prob<THRESH,'PredictionString'] = np.nan\nout.loc[df.PredictionString=='','PredictionString'] = np.nan\nphase1_output = out.drop(['prob'],axis=1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0cbc7463fe31749ac42757c4a4020d377fc5feac"},"cell_type":"markdown","source":"### Phase 2<br>Add high-confidnence results from phase1 to results form Kaggle kernel"},{"metadata":{"trusted":true,"_uuid":"04040d1bf267da053a1c479dcf9da1c5379ba874"},"cell_type":"code","source":"df1 = pd.read_csv(KERNEL_OUTPUT_PATH).set_index('patientId')\ndf2 = phase1_output\ndf1_cases = df1[~df1.PredictionString.isnull()]\ndf2_cases = df2[~df2.PredictionString.isnull()]\ndf1_pos_ids = df1_cases.index.values\ndf2_pos_ids = df2_cases.index.values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fc81ee9717ba9ddbec6c2f3389c97136eb863a0d"},"cell_type":"code","source":"df1_only_dict = {}\nfor p in df1_pos_ids:\n    if not p in df2_pos_ids:\n        df1_only_dict[p] = float(df1_cases.loc[p,'PredictionString'].split(' ')[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"29fe204c93361c04c4b132d97fa85ecafecdb4d7"},"cell_type":"code","source":"df2_only_dict = {}\nfor p in df2_pos_ids:\n    if not p in df1_pos_ids:\n        df2_only_dict[p] = float(df2_cases.loc[p,'PredictionString'].split(' ')[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"be5d6275f441033af7d6922b526ec5a79c99a70d"},"cell_type":"code","source":"df_out = df1.copy()\nfor p,r in df1.iterrows():\n    if p in df2_only_dict:\n        if df2_only_dict[p] > MIN_MAX_MRCNN_CONF:\n            df_out.loc[p,'PredictionString'] = df2.loc[p,'PredictionString']\nphase2_output = df_out","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2b9996f27d3fc4f7245f97adaa78acac65e9a2c1"},"cell_type":"code","source":"phase2_output.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6cbf69bb491a6657590e066e15d19c85fb6d575c"},"cell_type":"markdown","source":"### Phase 3<br>Convert MRCNN confidence to fitted class proability"},{"metadata":{"trusted":true,"_uuid":"1e2cee5b4e867157666457c9074ffee5e0b58e20"},"cell_type":"code","source":"# Read in OOF predictions from MRCNN model\n\ninput_dir = RESULTS_LOC\ninfile_stem = OOF_INFILE_STEM\noof_preds_input = []\nfor ifold in range( N_FOLDS ):\n    fp_in = os.path.join(input_dir, infile_stem + str(ifold) + IN_OOF_SUFFIX)\n    oof_preds_input.append( pd.read_csv(fp_in) )\n    oof_preds_input[-1]['fold'] = ifold\noof_preds = pd.concat(oof_preds_input).set_index('patientId')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"01aedb64f3a7968a45d41671946686432198b4a6"},"cell_type":"code","source":"# Convert to class probabilities by taking maximum conf for each patient\nprobs_dict = {}\nfor p, r in oof_preds.iterrows():\n    probs_dict[p] = MINPROB\noof_positive = oof_preds[~oof_preds.PredictionString.isnull()]\nfor p, r in oof_positive.iterrows():\n    s = r.PredictionString.split(' ')\n    if s[-1]=='':\n        s.pop()  # remove terminating null\n    if s[0]=='':\n        s.pop(0)  # remove initial null\n    if ( len(s)%5 ):\n        print( 'Bad prediction string.')\n    while len(s):\n        conf = float(s.pop(0))\n        x = int(round(float(s.pop(0))))\n        y = int(round(float(s.pop(0))))\n        w = int(round(float(s.pop(0))))\n        h = int(round(float(s.pop(0))))\n        if conf>probs_dict[p]:\n            probs_dict[p] = conf\nprob_df = pd.DataFrame(probs_dict,index=['prob']).transpose()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cf526d2e378d27e023a78af4b7ca3e168eb40ca8"},"cell_type":"code","source":"# Read in actual classes and join with class predictions\ninput_dir = RAW_DATA_LOC\nfp = os.path.join(input_dir, 'stage_2_train_labels.csv')\nact = pd.read_csv(fp).set_index('patientId').rename(columns={'Target':'actual'})[['actual']]\nact = act[~act.index.duplicated()]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"927b75661763e4cb60d8bc707564999a4e89a112"},"cell_type":"code","source":"df = act.join(prob_df,how='right')  # Have to do right join because using stage 1 OOF data\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"faa4715e8c33e83932695f8b5132a548bb972fe0"},"cell_type":"code","source":"# Fit logistic-on-logits model\nrawprobs = df.prob.values.reshape(-1,1)\nlr = LogisticRegression(C=C1)\nlr.fit(logit(rawprobs),df.actual)\nb1 = lr.coef_[0,0]\nb0 = lr.intercept_[0]\nb0, b1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"733313ae6904e51e9558df4016d04520b85300d0"},"cell_type":"code","source":"# Patient IDs for folds\nfolds = []\nfor ifold in range(N_FOLDS):\n    folds.append( oof_preds[oof_preds.fold==ifold].index.values )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a254bc2cce2af1b2db0788d0bd1b21abafbc11f8"},"cell_type":"code","source":"def run_logistic_by_fold():\n    df = act.join(prob_df,how='right')   # Have to do right join because using stage 1 OOF data\n    logits = df.prob.apply(logit)\n    df['oofprob'] = np.nan\n    b1 = []\n    b0 = []\n    for i in range(len(folds)):\n\n        fs = folds.copy()\n        te = fs.pop(i)  # pop off current validation fold\n        tr = np.concatenate(fs)  # the rest is for training\n\n        # Divide the data\n        Xtr = logits[tr].values.reshape(-1,1)\n        Xte = logits[te].values.reshape(-1,1)\n        ytr = df.actual[tr].copy()\n        yte = df.actual[te].copy()\n\n        # Fit and predict for this fold\n        lr.fit(Xtr, ytr)\n        df.loc[te,'oofprob'] = lr.predict_proba(Xte)[:,1]\n        b1.append(lr.coef_[0,0])\n        b0.append(lr.intercept_[0])\n\n    coefs = pd.DataFrame( {'b0':b0, 'b1':b1}, index=range(len(folds)) )\n    coefs.index.name = 'fold'\n\n    return( df.sort_values('oofprob'), coefs )   ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8c7a6adc99e0ea98483573d8b233d729b8411438"},"cell_type":"code","source":"lr = LogisticRegression(C=C2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9c4e2feccefbb7afdcdae4b3781b650e52fea098"},"cell_type":"code","source":"df, coefs = run_logistic_by_fold()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bc428db44073707901edba633ee560d90e68bb4b"},"cell_type":"code","source":"# Transform confidences in OOF data to show realistic probabilities\npat_dict = {}\nfold_dict = {}\nfor pat,row in oof_preds.iterrows():\n    fold = row.fold\n    pred = row.PredictionString\n    boxes = []\n    confs = []\n    if isinstance(pred, str):\n        s = pred.split(' ')\n        if s[-1]=='':\n            s.pop()  # remove terminating null\n        if s[0]=='':\n            s.pop(0)  # remove initial null\n        if ( len(s)%5 ):\n            print( 'Bad prediction string.')\n        b0 = coefs.b0[fold]\n        b1 = coefs.b1[fold]\n        while len(s):\n            conf = float(s.pop(0))\n            x = int(round(float(s.pop(0))))\n            y = int(round(float(s.pop(0))))\n            w = int(round(float(s.pop(0))))\n            h = int(round(float(s.pop(0))))\n            boxes.append( [x,y,w,h] )\n            confs.append( expit( b0+b1*logit(conf) ) )\n    predictionString = ''\n    if len(boxes):\n        for box, conf in zip(boxes, confs):\n            x, y, w, h = box\n            # add to predictionString\n            predictionString += '{:6.4f} '.format(conf)\n            predictionString += str(x) + ' ' + str(y) + ' ' + str(w) + ' ' + str(h) + ' '\n    pat_dict[pat] = predictionString\n    fold_dict[pat] = fold\nxform_oof_preds = pd.DataFrame(pat_dict, \n                               index=['PredictionString']).transpose()\nxform_oof_preds.index.name = 'patientId'\nxform_oof_preds['fold'] = [fold_dict[x] for x in xform_oof_preds.index.values]\nxform_oof_preds.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1457f6398e687c28f0eabec201329542da4d73c7"},"cell_type":"code","source":"output_dir = OUTPUT_LOC\nxform_oof_preds.to_csv(os.path.join(output_dir, OOF_OUTFILE_NAME))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"081c6716aa862a94e8d9a902a5f145a4ef7d9d56"},"cell_type":"code","source":"input_dir = RESULTS_LOC\noutput_dir = OUTPUT_LOC\ninfile_stem = TEST_INFILE_STEM\noutfile_stem = TEST_OUTFILE_STEM","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"abfa4ab252446f0e1deeea3521c9a0ba562c5117"},"cell_type":"code","source":"for ifold in range(N_FOLDS):\n    \n    fp_in = os.path.join(input_dir, infile_stem + str(ifold) + IN_SUFFIX)\n    fp_out = os.path.join(output_dir, outfile_stem + str(ifold) + OUT_SUFFIX)\n    inpreds = pd.read_csv(fp_in).set_index('patientId')\n    pat_dict = {}\n\n    for pat,row in inpreds.iterrows():\n        pred = row.PredictionString\n        boxes = []\n        confs = []\n        if isinstance(pred, str):\n            s = pred.split(' ')\n            if s[-1]=='':\n                s.pop()  # remove terminating null\n            if s[0]=='':\n                s.pop(0)  # remove initial null\n            if ( len(s)%5 ):\n                print( 'Bad prediction string.')\n            while len(s):\n                conf = float(s.pop(0))\n                x = int(round(float(s.pop(0))))\n                y = int(round(float(s.pop(0))))\n                w = int(round(float(s.pop(0))))\n                h = int(round(float(s.pop(0))))\n                boxes.append( [x,y,w,h] )\n                confs.append( expit( b0+b1*logit(conf) ) )\n        predictionString = ''\n        if len(boxes):\n            for box, conf in zip(boxes, confs):\n                x, y, w, h = box\n                # add to predictionString\n                predictionString += '{:6.4f} '.format(conf)\n                predictionString += str(x) + ' ' + str(y) + ' ' + str(w) + ' ' + str(h) + ' '\n        pat_dict[pat] = predictionString\n    outpreds = pd.DataFrame(pat_dict, \n                            index=['PredictionString']).transpose()\n    outpreds.index.name = 'patientId'\n    outpreds.to_csv(fp_out)\nfp_out","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1d9555f8a35f55e11b050b8a80abb02b1aecfe4f"},"cell_type":"code","source":"# Generate class probability file for OOF data\noutprobs = prob_df.join(oof_preds[['fold']])\nb0_ = coefs.b0[outprobs.fold].values\nb1_ = coefs.b1[outprobs.fold].values\noutprobs['prob'] =  expit( b0_ + logit(outprobs.prob)*b1_ )\noutprobs.index.name = 'patientId'\noutput_dir = OUTPUT_LOC\noutprobs.to_csv(os.path.join(output_dir, OOF_CLASS_PROBS_OUTFILE))\noutprobs.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"285df54677fcd8fad7e3f952f0541a21fa633699"},"cell_type":"code","source":"# Generate class probability files for test data (by training fold)\noutput_dir = OUTPUT_LOC\nbox_outfile_stem = TEST_OUTFILE_STEM\nprobs_outfile_stem = TEST_CLASS_PROBS_OUTFILE_STEM\nminprob_out = expit( b0 + b1*logit(MINPROB))\n\nfor ifold in range(N_FOLDS):\n    \n    fp_in = os.path.join(output_dir, box_outfile_stem + str(ifold) + OUT_SUFFIX)\n    inpreds = pd.read_csv(fp_in).set_index('patientId')\n    fp_out = os.path.join(output_dir, probs_outfile_stem + str(ifold) + OUT_SUFFIX)\n\n    probs_dict = {}\n\n    for p, r in inpreds.iterrows():\n        probs_dict[p] = minprob_out\n    inpreds_positive = inpreds[~inpreds.PredictionString.isnull()]\n    for p, r in inpreds_positive.iterrows():\n        s = r.PredictionString.split(' ')\n        if s[-1]=='':\n            s.pop()  # remove terminating null\n        if s[0]=='':\n            s.pop(0)  # remove initial null\n        if ( len(s)%5 ):\n            print( 'Bad prediction string.')\n        while len(s):\n            conf = float(s.pop(0))\n            s.pop(0)\n            s.pop(0)\n            s.pop(0)\n            s.pop(0)\n            if conf>probs_dict[p]:\n                probs_dict[p] = conf\n    outprob_df = pd.DataFrame(probs_dict,index=['prob']).transpose()\n    outprob_df.index.name = 'patientId'\n    outprob_df.to_csv(fp_out)\n    if not ifold:\n        allprobs_df = outprob_df.rename(columns={'prob':ifold})\n    else:\n        allprobs_df = allprobs_df.join(outprob_df.rename(columns={'prob':ifold}))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"56b1defcd07b057fc8305ed977391c0c85a6f9cd"},"cell_type":"markdown","source":"### Phase 4<br>Stack class probability estimates"},{"metadata":{"trusted":true,"_uuid":"aedbcbe5543bc347c965ae0ef8bd79af861b7665"},"cell_type":"code","source":"fp = OOF_CLASS_PROBS_OUTFILE\noof_probs_mrcnn = pd.read_csv(fp).set_index('patientId').rename(columns={'prob':'ph'})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fedea86d808e5d6877f73c319f98b7991f099086"},"cell_type":"code","source":"for i in range(N_FOLDS):\n    fp = TEST_CLASS_PROBS_OUTFILE_STEM + str(i) + '.csv'\n    indf = pd.read_csv(fp).set_index('patientId').rename(columns={'prob':'ph'+str(i)})\n    if i:\n        test_prob_df = test_prob_df.join(indf)\n    else:\n        test_prob_df = indf","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"efcf9a10816583c37bfd91cdc54993f73a541c0f"},"cell_type":"code","source":"test_prob_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2d3356b0ff9bd02caafdac809749ef90783ed628"},"cell_type":"code","source":"coefs","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"00ba1754ef7ae5921d3ce159a54258e9504804f8"},"cell_type":"code","source":"df.tail()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a3ae143f8516887589de6000e8e9f8685b28fe18"},"cell_type":"code","source":"f0 = pd.read_csv(RESULTS_LOC + '/' + ADEMSE_OOF_STEM + '0.csv').set_index('patientId')\nf1 = pd.read_csv(RESULTS_LOC + '/' + ADEMSE_OOF_STEM + '1.csv').set_index('patientId')\nf2 = pd.read_csv(RESULTS_LOC + '/' + ADEMSE_OOF_STEM + '2.csv').set_index('patientId')\nf3 = pd.read_csv(RESULTS_LOC + '/' + ADEMSE_OOF_STEM + '3.csv').set_index('patientId')\nf4 = pd.read_csv(RESULTS_LOC + '/' + ADEMSE_OOF_STEM + '4.csv').set_index('patientId')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"05fd20fa71b47295e2af503f2e21e74c1953a4e8"},"cell_type":"code","source":"den = pd.concat([f0,f1,f2,f3,f4],axis=0)\nprint( den.shape )\nden.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f819535680ebb6c49968dcce9cc5bb05ee107fa7"},"cell_type":"code","source":"tf0 = pd.read_csv(RESULTS_LOC + '/' + ADENSE_TEST_STEM + '0.csv').set_index('patientId')\ntf1 = pd.read_csv(RESULTS_LOC + '/' + ADENSE_TEST_STEM + '1.csv').set_index('patientId')\ntf2 = pd.read_csv(RESULTS_LOC + '/' + ADENSE_TEST_STEM + '2.csv').set_index('patientId')\ntf3 = pd.read_csv(RESULTS_LOC + '/' + ADENSE_TEST_STEM + '3.csv').set_index('patientId')\ntf4 = pd.read_csv(RESULTS_LOC + '/' + ADENSE_TEST_STEM + '4.csv').set_index('patientId')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d07f2bb5267d07d0ce7e6807326c30e13fed298c"},"cell_type":"code","source":"test_den = tf0.rename(columns={'predicted':'pa0'})\ntest_den = test_den.join(tf1.rename(columns={'predicted':'pa1'}))\ntest_den = test_den.join(tf2.rename(columns={'predicted':'pa2'}))\ntest_den = test_den.join(tf3.rename(columns={'predicted':'pa3'}))\ntest_den = test_den.join(tf4.rename(columns={'predicted':'pa4'}))\nprint( test_den.shape )\ntest_den.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b559d4e87d1890f8d1f752193e1f5ad44008010e"},"cell_type":"code","source":"# Patient IDs for folds\np0 = f0.index.values\np1 = f1.index.values\np2 = f2.index.values\np3 = f3.index.values\np4 = f4.index.values\npt = test_den.index.values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"461efdcb58e2150bb1d176bbd46f3ebdfcdf31b1"},"cell_type":"code","source":"c0 = pd.read_csv(RESULTS_LOC + '/' + CHEXNET_OOF_STEM + '0.csv').set_index('patientId')\nc1 = pd.read_csv(RESULTS_LOC + '/' + CHEXNET_OOF_STEM + '1.csv').set_index('patientId')\nc2 = pd.read_csv(RESULTS_LOC + '/' + CHEXNET_OOF_STEM + '2.csv').set_index('patientId')\nc3 = pd.read_csv(RESULTS_LOC + '/' + CHEXNET_OOF_STEM + '3.csv').set_index('patientId')\nc4 = pd.read_csv(RESULTS_LOC + '/' + CHEXNET_OOF_STEM + '4.csv').set_index('patientId')\nchex = pd.concat([c0,c1,c2,c3,c4],axis=0)\nprint( chex.shape )\nchex.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6622cce616fd7077128d5c6b7ed6ca6729983ae5"},"cell_type":"code","source":"df = den.join(chex[['validPredProba']]).rename(columns={'predicted':'pa','validPredProba':'pg'})\ndf = df.join(oof_probs_mrcnn).drop(['fold'],axis=1)\nprint( df.shape )\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"264a2e9fc9f0f7d00ea8d04ea1dc33d4360a46f4"},"cell_type":"code","source":"tc0 = pd.read_csv(RESULTS_LOC + '/' + CHEXNET_TEST_FILE0 + '.csv').set_index('patientId')\ntc1 = pd.read_csv(RESULTS_LOC + '/' + CHEXNET_TEST_STEM + '1.csv').set_index('patientId')\ntc2 = pd.read_csv(RESULTS_LOC + '/' + CHEXNET_TEST_STEM + '2.csv').set_index('patientId')\ntc3 = pd.read_csv(RESULTS_LOC + '/' + CHEXNET_TEST_STEM + '3.csv').set_index('patientId')\ntc4 = pd.read_csv(RESULTS_LOC + '/' + CHEXNET_TEST_STEM + '4.csv').set_index('patientId')\ntc0.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"999a5886b08bd13f3a78dbe8f5b059ebc2a67746"},"cell_type":"code","source":"INFINITY = 100  # No regularization\nlr = LogisticRegression(C=INFINITY)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c39fb3fb3fca3ff5aab81c0338de0f617acd58f9"},"cell_type":"code","source":"test_chex = tc0.rename(columns={'targetPredProba':'pg0'}).drop(['targetPred'],axis=1)\ntest_chex = test_chex.join(tc1.rename(columns={'targetPredProba':'pg1'}).drop(['targetPred'],axis=1))\ntest_chex = test_chex.join(tc2.rename(columns={'targetPredProba':'pg2'}).drop(['targetPred'],axis=1))\ntest_chex = test_chex.join(tc3.rename(columns={'targetPredProba':'pg3'}).drop(['targetPred'],axis=1))\ntest_chex = test_chex.join(tc4.rename(columns={'targetPredProba':'pg4'}).drop(['targetPred'],axis=1))\nprint( test_chex.shape )\ntest_chex.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"10ffb689d314cf315cceb0aa3d24fe5d90b623e5"},"cell_type":"code","source":"all_test_probs = test_prob_df.join(test_den.join(test_chex))\nall_test_probs.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e4a23a90933ef5e7fe9ddbb586ebc95f6579dc69"},"cell_type":"code","source":"test_sets = []\ntest_set = ['pa','pg','ph']\nfor i in range(5):\n    this_set = [n + str(i) for n in test_set]\n    test_sets.append(this_set)\ntest_sets","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"be7a4b34cf6d2fd90bcc47d2129cc13e1c8024a3"},"cell_type":"code","source":"yps = []\nMAXCONF = .996\nXtr = df.drop('actual',axis=1).copy()\nytr = df.actual.copy()\nX_train_adjust = Xtr\nX_train_adjust.loc[Xtr.ph>MAXCONF,'ph'] = MAXCONF\nX_train_logit = X_train_adjust.apply(logit)\nlr.fit(X_train_logit, ytr)\nfor tset in test_sets:\n    Xte = all_test_probs[tset].copy()\n    Xte.columns = test_set\n    X_test_adjust = Xte\n    X_test_adjust.loc[Xte.ph>MAXCONF,'ph'] = MAXCONF\n    X_test_logit = X_test_adjust.apply(logit)\n    yp = lr.predict_proba(X_test_logit)[:,1]\n    yps.append(yp)\nlen(yps), [len(y) for y in yps]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5013226e0f1eb46251ece96da6b3eed852e9804f"},"cell_type":"code","source":"p = pd.DataFrame(yps,columns=Xte.index).transpose().apply(logit).mean(axis=1).apply(expit)\nclass_preds = pd.DataFrame(p,columns=['prob'])\nclass_preds.index.name = 'patientId'\nprint(class_preds.shape)\nclass_preds.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"915b833c004cc37500ed5afa7076eb0ae2699963"},"cell_type":"code","source":"phase4_test_output = class_preds","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b049feeacbdf826ba64db86972647675212c5d4b"},"cell_type":"code","source":"yps = []\nids = []\nMAXCONF = .995\nfolds = [p0,p1,p2,p3,p4]\nfor i in range(len(folds)):\n    fs = folds.copy()\n    te = fs.pop(i)\n    tr = np.concatenate(fs)\n    Xtr = df.loc[tr,:].drop('actual',axis=1).copy()\n    Xte = df.loc[te,:].drop('actual',axis=1).copy()\n    ytr = df.actual[tr].copy()\n    yte = df.actual[te].copy()\n    X_train_adjust = Xtr\n    X_test_adjust = Xte\n    X_train_adjust.loc[Xtr.ph>MAXCONF,'ph'] = MAXCONF\n    X_test_adjust.loc[Xte.ph>MAXCONF,'ph'] = MAXCONF\n    X_train_logit = X_train_adjust.apply(logit)\n    X_test_logit = X_test_adjust.apply(logit)\n    lr.fit(X_train_logit, ytr)\n    yp = lr.predict_proba(X_test_logit)[:,1]\n    yps = yps + list(yp)\n    ids = ids + list(Xte.index.values)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"706d862c87e153f27d00b59d8a97b61b49fab45f"},"cell_type":"code","source":"oof_class_preds = pd.DataFrame({'prob':yps},index=ids)\noof_class_preds.index.name = 'patientId'\nprint(oof_class_preds.shape)\noof_class_preds.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e5d0424069a0da12ed13254334539161d9994c02"},"cell_type":"code","source":"phase4_oof_output = oof_class_preds","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ce6cbf3c9624538777eb41715ad6319248ead7cb"},"cell_type":"markdown","source":"### Phase 5<br>Add high-confidence Resnet-uunet results to results from phase 2"},{"metadata":{"trusted":true,"_uuid":"e08d65bac559617000bad294c7789d12dff4b88b"},"cell_type":"code","source":"df1 = phase2_output\ndf2 = pd.read_csv(UNET_RESULTS_PATH).rename(columns={\n    'predictionString':'PredictionString'}).set_index('patientId')\n\nclass_preds = phase4_test_output\nclass_probs = class_preds.prob.to_dict()\n\ndf1_cases = df1[~df1.PredictionString.isnull()]\ndf2_cases = df2[~df2.PredictionString.isnull()]\n\ndf1_pos_ids = df1_cases.index.values\ndf2_pos_ids = df2_cases.index.values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2406a80114ca3cc9f8207d1255b7363c61f4c125"},"cell_type":"code","source":"df1_only_dict = {}\nfor p in df1_pos_ids:\n    if not p in df2_pos_ids:\n        df1_only_dict[p] = float(df1_cases.loc[p,'PredictionString'].split(' ')[0])\n\ndf2_only_dict = {}\nfor p in df2_pos_ids:\n    if not p in df1_pos_ids:\n        df2_only_dict[p] = float(df2_cases.loc[p,'PredictionString'].split(' ')[0])\n        \ncandidates = pd.DataFrame(pd.Series(df2_only_dict,name='conf')).join(class_preds,how='left')\naccepted = candidates[ (candidates.prob>UNET_MINPROB) & (candidates.conf>UNET_MINCONF)\n                     ].index.values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8ec5b1cff296145af51ccbd1b362b5eb0af2ee50"},"cell_type":"code","source":"accepted","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ec39d24f488385c291be6e5c7472bb6e199bca79"},"cell_type":"code","source":"df_out = df1.copy()\nfor p,r in df1.iterrows():\n    if p in accepted:\n        df_out.loc[p,'PredictionString'] = df2.loc[p,'PredictionString']\nphase5_output = df_out\nphase5_output.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e8c4f5c2938ce8a850a8a37d1690f69a528c5f37"},"cell_type":"markdown","source":"### Phase 6<br>Add high-confidnece yolo results to results from phase 5"},{"metadata":{"trusted":true,"_uuid":"6be55b350f7e413f33b8b62c04efd6f7765e63fd"},"cell_type":"code","source":"yolo_epochs = [1500, 2000, 3200, 4100, 5300, 6700, 10000, 13700]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3b163135a6094288aaa9a5aa563b38357c2ab664"},"cell_type":"code","source":"df1 = phase5_output\n\ndfs = []\nfor eps in yolo_epochs:\n    yolo_stem = YOLO_SUB_STEM2 if (eps in MORE_YOLO_EPOCHS) else YOLO_SUB_STEM\n    dfs.append( \n        pd.read_csv(RESULTS_LOC+'/'+yolo_stem+str(eps)+'.csv').set_index('patientId'))\n\nclass_preds = phase4_test_output\nclass_probs = class_preds.prob.to_dict()\n\ndf1_cases = df1[~df1.PredictionString.isnull()]\ndf1_pos_ids = df1_cases.index.values\nincluded = df1_pos_ids.tolist()\ndf_out = df1.copy()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7051cad5af8df558549734ae3b9f8fd6f76835d5"},"cell_type":"code","source":"while True:\n    for df2 in dfs:\n        df2_cases = df2[~df2.PredictionString.isnull()]\n        df2_pos_ids = df2_cases.index.values\n        df2_only_dict = {}\n        outstring_dict = {}\n        for p in df2_pos_ids:\n            if not p in included:\n                maxconf = 0\n                boxes = []\n                confs = []\n                s = df2.loc[p,'PredictionString'].split(' ')\n                if s[-1]=='':\n                    s.pop()  # remove terminating null\n                if s[0]=='':\n                    s.pop(0)  # remove initial null\n                if ( len(s)%5 ):\n                    print( 'Bad prediction string.')\n                while len(s):\n                    conf = float(s.pop(0))\n                    x = int(round(float(s.pop(0))))\n                    y = int(round(float(s.pop(0))))\n                    w = int(round(float(s.pop(0))))\n                    h = int(round(float(s.pop(0))))\n                    if (conf>YOLO_MINCONF):\n                        boxes.append( [x,y,w,h] )\n                        confs.append( conf )\n                        if conf>maxconf:\n                            maxconf = conf\n                predictionString = ''\n                if len(boxes):\n                    for box, conf in zip(boxes, confs):\n                        x, y, w, h = box\n                        # add to predictionString\n                        predictionString += '{:6.4f} '.format(conf)\n                        predictionString += str(x) + ' ' + str(y) + ' ' + str(w) + ' ' + str(h) + ' '\n                        df2_only_dict[p] = maxconf*class_probs[p]\n                outstring_dict[p] = predictionString\n        candidates = pd.DataFrame(pd.Series(df2_only_dict,name='prod'))\n        best = candidates.sort_values('prod').tail(1).index.values[0]\n        print(best,candidates.loc[best,'prod'])\n        df_out.loc[best,'PredictionString'] = outstring_dict[best]\n        included.append( best )\n    minprob = class_preds.loc[included[-8:],:].prob.min()\n    if minprob<MINMIN:\n        break","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f3a60d302eb247e54420a272a5567a185c07654e"},"cell_type":"code","source":"class_preds.loc[included[-32:],:]   # This was 16 in stage 1. Changed to 32 for curiosity.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ad81081ed52d80dfd9fa90884444c271948589bb"},"cell_type":"code","source":"df_out.to_csv('phase6_output.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2fa558ef57b28d9e6d445b058f0bd647ef08e766"},"cell_type":"code","source":"df_out.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4ba96c556827d6d9c488eb25d5f668fa002607d4"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}