{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n\nimport os\nimport os\nimport math\nimport time\nimport random\nimport shutil\nfrom pathlib import Path\nfrom contextlib import contextmanager\nfrom collections import defaultdict, Counter\n\nimport scipy as sp\nimport numpy as np\nimport pandas as pd\n\nfrom tqdm.auto import tqdm\nfrom functools import partial\n\nimport cv2\nfrom PIL import Image\n\nfrom matplotlib import pyplot as plt\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\nOUTPUT_DIR = './'\nif not os.path.exists(OUTPUT_DIR):\n    os.makedirs(OUTPUT_DIR)\n\nTEST_PATH = '../input/ranzcr-clip-catheter-line-classification/test'\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import datetime\nimport numpy as np \nimport pandas as pd\nfrom time import time\nfrom numba import njit\nfrom sklearn.metrics import log_loss,roc_auc_score\nfrom scipy.optimize import minimize, LinearConstraint, dual_annealing","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"folds = [0,1,2,3,4]\ntarget_size=11\ntarget_cols=['ETT - Abnormal', 'ETT - Borderline', 'ETT - Normal',\n                 'NGT - Abnormal', 'NGT - Borderline', 'NGT - Incompletely Imaged', 'NGT - Normal', \n                 'CVC - Abnormal', 'CVC - Borderline', 'CVC - Normal',\n                 'Swan Ganz Catheter Present']\ncols = [f'preds_{c}' for c in target_cols]\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from contextlib import contextmanager\n\n@contextmanager\ndef timer(name):\n    t0 = time.time()\n    LOGGER.info(f'[{name}] start')\n    yield\n    LOGGER.info(f'[{name}] done in {time.time() - t0:.0f} s.')\n\n\ndef init_logger(log_file=OUTPUT_DIR+'inference.log'):\n    from logging import getLogger, INFO, FileHandler,  Formatter,  StreamHandler\n    logger = getLogger(__name__)\n    logger.setLevel(INFO)\n    handler1 = StreamHandler()\n    handler1.setFormatter(Formatter(\"%(message)s\"))\n    handler2 = FileHandler(filename=log_file)\n    handler2.setFormatter(Formatter(\"%(message)s\"))\n    logger.addHandler(handler1)\n   # logger.addHandler(handler2)\n    return logger\n\nLOGGER = init_logger()\n\n\ndef seed_torch(seed=42):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n\nseed_torch(seed=42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def rank_data(sub2):\n    sub['target'] = sub2\n    sub['target'] = sub['target'].rank() / sub['target'].rank().max()\n    return sub2","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def obj_func(x, *oof_list):\n    oof = 0\n    x[x<0] = 0\n    x = x/np.sum(x)\n    for i in range(min(len(x), len(oof_list))):\n        oof += x[i]*oof_list[i]['predict']\n    ground_truth = oof_list[0]['target']\n    return -roc_auc_score(ground_truth, oof)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# ====================================================\n# Utils\n# ====================================================\ndef oof_score(oof_df):\n    def get_score(y_true, y_pred):\n        scores = []\n        for i in range(y_true.shape[1]):\n            score = roc_auc_score(y_true[:,i], y_pred[:,i])\n            scores.append(score)\n        avg_score = np.mean(scores)\n        return avg_score\n    def get_result(result_df):\n        preds = result_df[[f'preds_{c}' for c in target_cols]].values\n        labels = result_df[target_cols].values\n        score = get_score(labels, preds)\n        #LOGGER.info(f'Score: {score:<.4f}  Scores: {np.round(scores, decimals=4)}')\n        return score\n    for fold in folds:\n        fold_oof_df = oof_df[oof_df['fold']==fold].reset_index(drop=True)\n        #LOGGER.info(f\"========== fold: {fold} result ==========\")\n       # get_result(fold_oof_df)\n    return get_result(fold_oof_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#oof1 = pd.read_csv('../input/multihead-selfattn-inception-eb5/3rdstage_self_attn_allfold_ps8_resnet200d_320/3rdstage_self_attn_allfold_ps8_resnet200d_320/oof_df.csv')\n\n\n\n#oof2 = pd.read_csv('../input/multihead-selfattn-inception-eb5/ns3rdstagefoldallpsdo8_inception/ns3rdstagefoldallpsdo8_inception/oof_df.csv')\n\noof3 = pd.read_csv('../input/multihead-selfattn-inception-eb5/ns3rdstagefoldallpsdo8_ef7/ns3rdstagefoldallpsdo8_ef7/oof_df.csv')\n\n\noof1 = pd.read_csv('../input/resnet200d-967-oof/oof.csv')\noof2 = pd.read_csv('../input/seresnet-965pub-768-oof/oof.csv')\noof4 = pd.read_csv('../input/multiheadresnet200dpseudooof/oof.csv')\noof5 = pd.read_csv('../input/efn-mh-bestcv-963-oof/oof.csv')\noof6 = pd.read_csv('../input/resnet200d-mh-966-oof-np/oof.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"oof2 = oof2.set_index('StudyInstanceUID')\noof2 = oof2.reindex(index=oof1['StudyInstanceUID'])\noof2 = oof2.reset_index()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"oof3 = oof3.set_index('StudyInstanceUID')\noof3 = oof3.reindex(index=oof1['StudyInstanceUID'])\noof3 = oof3.reset_index()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"oof4 = oof4.set_index('StudyInstanceUID')\noof4 = oof4.reindex(index=oof1['StudyInstanceUID'])\noof4 = oof4.reset_index()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"oof5","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"oof5 = oof5.set_index('StudyInstanceUID')\noof5 = oof5.reindex(index=oof1['StudyInstanceUID'])\noof5 = oof5.reset_index()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"oof6 = oof6.set_index('StudyInstanceUID')\noof6 = oof6.reindex(index=oof1['StudyInstanceUID'])\noof6 = oof6.reset_index()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#y_true = pd.read_csv('../input/lish-moa/train_targets_scored.csv', index_col = 'sig_id').values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def oof_ranking(oof):\n    for i in cols:\n        oof[i] = oof[i].rank() / oof[i].rank().max()\n        #oof[i] = oof[i].rank() / oof[i].rank().max()\n    return oof","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"oof1.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"oof2.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"oof2.columns = oof1.columns\noof3.columns = oof1.columns\noof4.columns = oof1.columns\noof5.columns = oof1.columns\noof6.columns = oof1.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"oof3['fold']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# oof1 = oof_ranking(oof1)\n# oof2 = oof_ranking(oof2)\n# oof3 = oof_ranking(oof3)\n# oof4 = oof_ranking(oof4)\n# oof5 = oof_ranking(oof5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"weights = [0.25,0.25,0.25,0.25,0.25]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\ndef func(weights):\n    coef = 1e-6\n    oof_blend = oof1.copy()\n    cols = [f'preds_{c}' for c in target_cols]\n    \n    oof_blend[cols] = weights[0] * oof1[cols].values + weights[1] * oof2[cols].values + weights[2] * oof3[cols].values + weights[3] * oof4[cols].values+ weights[4] * oof5[cols].values\n                      \n\n#     score = log_loss_metric(oof_blend)\n    score = oof_score(oof_blend)\n    penalty = coef * (np.sum(weights) - 1) ** 2\n    return score ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"func(weights)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model_oofs = [oof1,oof2,oof3,oof4,oof5,oof6]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def objective(trial):\n        weights = []\n        blends = oof1.copy()\n        for i in range(len(model_oofs)):\n            weights.append(trial.suggest_float(f\"w{i}\", 0, 1.0))\n\n        blend = np.zeros(model_oofs[0][cols].shape)\n        for i in range(len(model_oofs)):\n            blend += weights[i] * model_oofs[i][cols]\n        blend = np.clip(blend, 0, 1.0)\n        blends[cols] = blend\n        loss = oof_score(blends)\n        return loss","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import optuna\npruner = optuna.pruners.MedianPruner(\n        n_startup_trials=5,\n        n_warmup_steps=0,\n        interval_steps=1,\n    )\nsampler = optuna.samplers.TPESampler(seed=42)\nstudy = optuna.create_study(direction=\"maximize\",\n                                pruner=pruner,\n                                sampler=sampler,\n                              #  study_name=study_name,\n                              #  storage=f'sqlite:///{study_name}.db',\n                                load_if_exists=True)\n\nstudy.optimize(objective,\n                   n_trials=2350,\n                   timeout=None,\n                   gc_after_trial=True,\n                   n_jobs=-1)\n\ntrial = study.best_trial","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"study.best_params","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"study.best_trial\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}