{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<a class=\"anchor\" id=\"0\"></a>\n# [OSIC Pulmonary Fibrosis Progression](https://www.kaggle.com/c/osic-pulmonary-fibrosis-progression)","metadata":{}},{"cell_type":"markdown","source":"## 1. Import libraries <a class=\"anchor\" id=\"1\"></a>\n\n[Back to Table of Contents](#0.1)","metadata":{}},{"cell_type":"code","source":"!pip install ../input/kerasapplications/keras-team-keras-applications-3b180cb -f ./ --no-index\n!pip install ../input/efficientnet/efficientnet-1.1.0/ -f ./ --no-index","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-24T04:05:58.218802Z","iopub.execute_input":"2023-06-24T04:05:58.219144Z","iopub.status.idle":"2023-06-24T04:06:17.183839Z","shell.execute_reply.started":"2023-06-24T04:05:58.219110Z","shell.execute_reply":"2023-06-24T04:06:17.182789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport pydicom\nimport pandas as pd\nimport numpy as np \nimport tensorflow as tf \nimport matplotlib.pyplot as plt \nimport random\nfrom tqdm.notebook import tqdm \nfrom sklearn.model_selection import train_test_split, KFold\nfrom sklearn.metrics import mean_absolute_error\nfrom tensorflow_addons.optimizers import RectifiedAdam\nfrom tensorflow.keras import Model\nimport tensorflow.keras.backend as K\nimport tensorflow.keras.layers as L\nimport tensorflow.keras.models as M\nfrom tensorflow.keras.optimizers import Nadam\nimport seaborn as sns\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom PIL import Image","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-24T04:06:17.188375Z","iopub.execute_input":"2023-06-24T04:06:17.188716Z","iopub.status.idle":"2023-06-24T04:06:23.802349Z","shell.execute_reply.started":"2023-06-24T04:06:17.188681Z","shell.execute_reply":"2023-06-24T04:06:23.801355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed=2020):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:06:23.803782Z","iopub.execute_input":"2023-06-24T04:06:23.804148Z","iopub.status.idle":"2023-06-24T04:06:23.810127Z","shell.execute_reply.started":"2023-06-24T04:06:23.804108Z","shell.execute_reply":"2023-06-24T04:06:23.809219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seed_everything(42)","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:06:23.811547Z","iopub.execute_input":"2023-06-24T04:06:23.812158Z","iopub.status.idle":"2023-06-24T04:06:23.822033Z","shell.execute_reply.started":"2023-06-24T04:06:23.812119Z","shell.execute_reply":"2023-06-24T04:06:23.820891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"config = tf.compat.v1.ConfigProto()\nconfig.gpu_options.allow_growth = True\nsession = tf.compat.v1.Session(config=config)","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-24T04:06:23.826890Z","iopub.execute_input":"2023-06-24T04:06:23.827214Z","iopub.status.idle":"2023-06-24T04:06:25.916296Z","shell.execute_reply.started":"2023-06-24T04:06:23.827182Z","shell.execute_reply":"2023-06-24T04:06:25.915352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. My upgrade <a class=\"anchor\" id=\"2\"></a>\n\n[Back to Table of Contents](#0.1)","metadata":{}},{"cell_type":"markdown","source":"## 2.1. Commit now <a class=\"anchor\" id=\"2.1\"></a>\n\n[Back to Table of Contents](#0.1)","metadata":{}},{"cell_type":"code","source":"# From the best on Private LB - commit 17\n# Dropout_model = 0.25\n# FVC_weight = 0.5\n# Confidence_weight = 0.5\n# GaussianNoise_stddev = 0.2","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:06:25.917887Z","iopub.execute_input":"2023-06-24T04:06:25.918247Z","iopub.status.idle":"2023-06-24T04:06:25.923614Z","shell.execute_reply.started":"2023-06-24T04:06:25.918206Z","shell.execute_reply":"2023-06-24T04:06:25.922694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GaussianNoise_stddev_list = [0.2, 0.21, 0.15, 0.25]\nDropout_model_list = [0.25, 0.4, 0.36, 0.35, 0.32, 0.25, 0.38, 0.39, 0.37, 0.385, 0.24]\n\nConfidence_weight_list = [0.14, 0.2, 0.21, 0.19, 0.5, 0.225, 0.175, 0.15, 0.3, 0.26, 0.36]\nFVC_weight_list = [0.5, 0.25, 0.35, 0.3, 0.2, 0.15, 0.175, 0.225, 0.5, 0.19, 0.21, 0.14]\n","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:06:25.925163Z","iopub.execute_input":"2023-06-24T04:06:25.925829Z","iopub.status.idle":"2023-06-24T04:06:25.936954Z","shell.execute_reply.started":"2023-06-24T04:06:25.925790Z","shell.execute_reply":"2023-06-24T04:06:25.936275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### seed_everything\n* Commits 1-21, 24...: seed_everything = 42\n* Commit 23: seed_everything = 0","metadata":{}},{"cell_type":"code","source":"# Seed\n# commits_df['seed'] = 42\n# commits_df.loc[commits_df['commit_num'] == 23, 'seed'] = 0","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:06:25.938249Z","iopub.execute_input":"2023-06-24T04:06:25.938900Z","iopub.status.idle":"2023-06-24T04:06:26.363182Z","shell.execute_reply.started":"2023-06-24T04:06:25.938861Z","shell.execute_reply":"2023-06-24T04:06:26.361738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/train.csv') ","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:07:45.546023Z","iopub.execute_input":"2023-06-24T04:07:45.546367Z","iopub.status.idle":"2023-06-24T04:07:45.558772Z","shell.execute_reply.started":"2023-06-24T04:07:45.546336Z","shell.execute_reply":"2023-06-24T04:07:45.557885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_tab(df):\n    vector = [(df.Age.values[0] - 30) / 30] \n    \n    if df.Sex.values[0] == 'male':\n       vector.append(0)\n    else:\n       vector.append(1)\n    \n    if df.SmokingStatus.values[0] == 'Never smoked':\n        vector.extend([0,0])\n    elif df.SmokingStatus.values[0] == 'Ex-smoker':\n        vector.extend([1,1])\n    elif df.SmokingStatus.values[0] == 'Currently smokes':\n        vector.extend([0,1])\n    else:\n        vector.extend([1,0])\n    return np.array(vector) ","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-24T04:07:47.577194Z","iopub.execute_input":"2023-06-24T04:07:47.577589Z","iopub.status.idle":"2023-06-24T04:07:47.587384Z","shell.execute_reply.started":"2023-06-24T04:07:47.577553Z","shell.execute_reply":"2023-06-24T04:07:47.586014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"A = {} \nTAB = {} \nP = [] \nfor i, p in tqdm(enumerate(train.Patient.unique())):\n    sub = train.loc[train.Patient == p, :] \n    fvc = sub.FVC.values\n    weeks = sub.Weeks.values\n    c = np.vstack([weeks, np.ones(len(weeks))]).T\n    a, b = np.linalg.lstsq(c, fvc)[0]\n    \n    A[p] = a\n    TAB[p] = get_tab(sub)\n    P.append(p)","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:07:50.011844Z","iopub.execute_input":"2023-06-24T04:07:50.012207Z","iopub.status.idle":"2023-06-24T04:07:50.401548Z","shell.execute_reply.started":"2023-06-24T04:07:50.012171Z","shell.execute_reply":"2023-06-24T04:07:50.400601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_img(path):\n    d = pydicom.dcmread(path)\n    return cv2.resize(d.pixel_array / 2**11, (512, 512))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-24T04:07:55.553131Z","iopub.execute_input":"2023-06-24T04:07:55.553496Z","iopub.status.idle":"2023-06-24T04:07:55.558696Z","shell.execute_reply.started":"2023-06-24T04:07:55.553461Z","shell.execute_reply":"2023-06-24T04:07:55.557627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.utils import Sequence\n\nclass IGenerator(Sequence):\n    BAD_ID = ['ID00011637202177653955184', 'ID00052637202186188008618']\n    def __init__(self, keys, a, tab, batch_size=32):\n        self.keys = [k for k in keys if k not in self.BAD_ID]\n        self.a = a\n        self.tab = tab\n        self.batch_size = batch_size\n        \n        self.train_data = {}\n        for p in train.Patient.values:\n            self.train_data[p] = os.listdir(f'../input/osic-pulmonary-fibrosis-progression/train/{p}/')\n    \n    def __len__(self):\n        return 1000\n    \n    def __getitem__(self, idx):\n        x = []\n        a, tab = [], [] \n        keys = np.random.choice(self.keys, size = self.batch_size)\n        for k in keys:\n            try:\n                i = np.random.choice(self.train_data[k], size=1)[0]\n                img = get_img(f'../input/osic-pulmonary-fibrosis-progression/train/{k}/{i}')\n                x.append(img)\n                a.append(self.a[k])\n                tab.append(self.tab[k])\n            except:\n                print(k, i)\n       \n        x,a,tab = np.array(x), np.array(a), np.array(tab)\n        x = np.expand_dims(x, axis=-1)\n        return [x, tab] , a","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-24T04:07:58.128192Z","iopub.execute_input":"2023-06-24T04:07:58.128633Z","iopub.status.idle":"2023-06-24T04:07:58.143351Z","shell.execute_reply.started":"2023-06-24T04:07:58.128598Z","shell.execute_reply":"2023-06-24T04:07:58.142017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.layers import (\n    Dense, Dropout, Activation, Flatten, Input, BatchNormalization, GlobalAveragePooling2D, Add, Conv2D, AveragePooling2D, \n    LeakyReLU, Concatenate \n)\nimport efficientnet.tfkeras as efn\n\ndef get_efficientnet(model, shape):\n    models_dict = {\n        'b0': efn.EfficientNetB0(input_shape=shape,weights=None,include_top=False),\n        'b1': efn.EfficientNetB1(input_shape=shape,weights=None,include_top=False),\n        'b2': efn.EfficientNetB2(input_shape=shape,weights=None,include_top=False),\n        'b3': efn.EfficientNetB3(input_shape=shape,weights=None,include_top=False),\n        'b4': efn.EfficientNetB4(input_shape=shape,weights=None,include_top=False),\n        'b5': efn.EfficientNetB5(input_shape=shape,weights=None,include_top=False),\n        'b6': efn.EfficientNetB6(input_shape=shape,weights=None,include_top=False),\n        'b7': efn.EfficientNetB7(input_shape=shape,weights=None,include_top=False)\n    }\n    return models_dict[model]\n\ndef build_model(GaussianNoise_stddev, Dropout_model, shape=(512, 512, 1), model_class=None):\n    inp = Input(shape=shape)\n    base = get_efficientnet(model_class, shape)\n    x = base(inp)\n    x = GlobalAveragePooling2D()(x)\n    inp2 = Input(shape=(4,))\n    x2 = tf.keras.layers.GaussianNoise(GaussianNoise_stddev)(inp2)\n    x = Concatenate()([x, x2]) \n    x = Dropout(Dropout_model)(x)\n    x = Dense(1)(x)\n    model = Model([inp, inp2] , x)\n    \n    weights = [w for w in os.listdir('../input/osic-model-weights') if model_class in w][0]\n    model.load_weights('../input/osic-model-weights/' + weights)\n    return model\n\nmodel_classes = ['b5'] #['b0','b1','b2','b3',b4','b5','b6','b7']\n\n# models = []\n# for GaussianNoise_stddev in GaussianNoise_stddev_list:\n#     for Dropout_model in Dropout_model_list:\n#         model = build_model(GaussianNoise_stddev, Dropout_model, shape=(512, 512, 1), model_class='b5')\n#         models.append([model, GaussianNoise_stddev, Dropout_model])\n#         print(len(models))\n# print('Number of models: ' + str(len(models)))\n\nmodels = []\nfor GaussianNoise_stddev in GaussianNoise_stddev_list:\n    Dropout_model = 0.24\n    model = build_model(GaussianNoise_stddev, Dropout_model, shape=(512, 512, 1), model_class='b5')\n    models.append([model, GaussianNoise_stddev, Dropout_model])\n    print(len(models))\nprint('Number of models: ' + str(len(models)))","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:20:33.239495Z","iopub.execute_input":"2023-06-24T04:20:33.239882Z","iopub.status.idle":"2023-06-24T04:24:49.312093Z","shell.execute_reply.started":"2023-06-24T04:20:33.239849Z","shell.execute_reply":"2023-06-24T04:24:49.309529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr_p, vl_p = train_test_split(P, shuffle=True, train_size = 0.8, random_state=42) ","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:24:55.674067Z","iopub.execute_input":"2023-06-24T04:24:55.674440Z","iopub.status.idle":"2023-06-24T04:24:55.680351Z","shell.execute_reply.started":"2023-06-24T04:24:55.674382Z","shell.execute_reply":"2023-06-24T04:24:55.679240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def score(fvc_true, fvc_pred, sigma):\n    sigma_clip = np.maximum(sigma, 70) # changed from 70, trie 66.7 too\n    delta = np.abs(fvc_true - fvc_pred)\n    delta = np.minimum(delta, 1000)\n    sq2 = np.sqrt(2)\n    metric = (delta / sigma_clip)*sq2 + np.log(sigma_clip* sq2)\n    return np.mean(metric)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-24T04:25:00.993827Z","iopub.execute_input":"2023-06-24T04:25:00.994201Z","iopub.status.idle":"2023-06-24T04:25:01.002931Z","shell.execute_reply.started":"2023-06-24T04:25:00.994167Z","shell.execute_reply":"2023-06-24T04:25:01.001524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subs = []\nfor model_item in models:\n    model = model_item[0]\n    metric = []\n    for q in tqdm(range(1, 10)):\n        m = []\n        for p in vl_p:\n            x = [] \n            tab = [] \n\n            if p in ['ID00011637202177653955184', 'ID00052637202186188008618']:\n                continue\n\n            ldir = os.listdir(f'../input/osic-pulmonary-fibrosis-progression/train/{p}/')\n            for i in ldir:\n                if int(i[:-4]) / len(ldir) < 0.8 and int(i[:-4]) / len(ldir) > 0.15:\n                    x.append(get_img(f'../input/osic-pulmonary-fibrosis-progression/train/{p}/{i}')) \n                    tab.append(get_tab(train.loc[train.Patient == p, :])) \n            if len(x) < 1:\n                continue\n            tab = np.array(tab) \n\n            x = np.expand_dims(x, axis=-1) \n            _a = model.predict([x, tab]) \n            a = np.quantile(_a, q / 10)\n\n            percent_true = train.Percent.values[train.Patient == p]\n            fvc_true = train.FVC.values[train.Patient == p]\n            weeks_true = train.Weeks.values[train.Patient == p]\n\n            fvc = a * (weeks_true - weeks_true[0]) + fvc_true[0]\n            percent = percent_true[0] - a * abs(weeks_true - weeks_true[0])\n            m.append(score(fvc_true, fvc, percent))\n        print(np.mean(m))\n        metric.append(np.mean(m))\n\n    q = (np.argmin(metric) + 1)/ 10\n\n    sub = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/sample_submission.csv') \n    test = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/test.csv') \n    A_test, B_test, P_test,W, FVC= {}, {}, {},{},{} \n    STD, WEEK = {}, {} \n    for p in test.Patient.unique():\n        x = [] \n        tab = [] \n        ldir = os.listdir(f'../input/osic-pulmonary-fibrosis-progression/test/{p}/')\n        for i in ldir:\n            if int(i[:-4]) / len(ldir) < 0.8 and int(i[:-4]) / len(ldir) > 0.15:\n                x.append(get_img(f'../input/osic-pulmonary-fibrosis-progression/test/{p}/{i}')) \n                tab.append(get_tab(test.loc[test.Patient == p, :])) \n        if len(x) <= 1:\n            continue\n        tab = np.array(tab) \n\n        x = np.expand_dims(x, axis=-1) \n        _a = model.predict([x, tab]) \n        a = np.quantile(_a, q)\n        A_test[p] = a\n        B_test[p] = test.FVC.values[test.Patient == p] - a*test.Weeks.values[test.Patient == p]\n        P_test[p] = test.Percent.values[test.Patient == p] \n        WEEK[p] = test.Weeks.values[test.Patient == p]\n\n    for k in sub.Patient_Week.values:\n        p, w = k.split('_')\n        w = int(w) \n\n        fvc = A_test[p] * w + B_test[p]\n        sub.loc[sub.Patient_Week == k, 'FVC'] = fvc\n        sub.loc[sub.Patient_Week == k, 'Confidence'] = (\n            P_test[p] - A_test[p] * abs(WEEK[p] - w) \n    ) \n\n    _sub = sub[[\"Patient_Week\",\"FVC\",\"Confidence\"]].copy()\n    subs.append([_sub, model_item[1], model_item[2]])","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:25:03.387653Z","iopub.execute_input":"2023-06-24T04:25:03.388006Z","iopub.status.idle":"2023-06-24T04:45:19.615014Z","shell.execute_reply.started":"2023-06-24T04:25:03.387975Z","shell.execute_reply":"2023-06-24T04:45:19.611083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Prediction and submission <a class=\"anchor\" id=\"4\"></a>\n\n[Back to Table of Contents](#0.1)","metadata":{}},{"cell_type":"markdown","source":"## 4.1 Average prediction <a class=\"anchor\" id=\"4.1\"></a>\n\n[Back to Table of Contents](#0.1)","metadata":{}},{"cell_type":"code","source":"# img_sub = sub[[\"Patient_Week\",\"FVC\",\"Confidence\"]].copy()","metadata":{"execution":{"iopub.status.busy":"2023-06-23T19:01:23.907583Z","iopub.execute_input":"2023-06-23T19:01:23.907949Z","iopub.status.idle":"2023-06-23T19:01:23.915584Z","shell.execute_reply.started":"2023-06-23T19:01:23.907914Z","shell.execute_reply":"2023-06-23T19:01:23.914626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4.2 Osic-Multiple-Quantile-Regression <a class=\"anchor\" id=\"4.2\"></a>\n\n[Back to Table of Contents](#0.1)","metadata":{}},{"cell_type":"code","source":"ROOT = \"../input/osic-pulmonary-fibrosis-progression\"\nBATCH_SIZE=128\n\ntr = pd.read_csv(f\"{ROOT}/train.csv\")\ntr.drop_duplicates(keep=False, inplace=True, subset=['Patient','Weeks'])\nchunk = pd.read_csv(f\"{ROOT}/test.csv\")\n\nprint(\"add infos\")\nsub = pd.read_csv(f\"{ROOT}/sample_submission.csv\")\nsub['Patient'] = sub['Patient_Week'].apply(lambda x:x.split('_')[0])\nsub['Weeks'] = sub['Patient_Week'].apply(lambda x: int(x.split('_')[-1]))\nsub =  sub[['Patient','Weeks','Confidence','Patient_Week']]\nsub = sub.merge(chunk.drop('Weeks', axis=1), on=\"Patient\")","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:46:21.156842Z","iopub.execute_input":"2023-06-24T04:46:21.157215Z","iopub.status.idle":"2023-06-24T04:46:21.196018Z","shell.execute_reply.started":"2023-06-24T04:46:21.157180Z","shell.execute_reply":"2023-06-24T04:46:21.195102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr['WHERE'] = 'train'\nchunk['WHERE'] = 'val'\nsub['WHERE'] = 'test'\ndata = tr.append([chunk, sub])","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:46:23.581288Z","iopub.execute_input":"2023-06-24T04:46:23.581719Z","iopub.status.idle":"2023-06-24T04:46:23.598211Z","shell.execute_reply.started":"2023-06-24T04:46:23.581679Z","shell.execute_reply":"2023-06-24T04:46:23.597108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(tr.shape, chunk.shape, sub.shape, data.shape)\nprint(tr.Patient.nunique(), chunk.Patient.nunique(), sub.Patient.nunique(), \n      data.Patient.nunique())","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:46:26.341847Z","iopub.execute_input":"2023-06-24T04:46:26.342213Z","iopub.status.idle":"2023-06-24T04:46:26.353592Z","shell.execute_reply.started":"2023-06-24T04:46:26.342178Z","shell.execute_reply":"2023-06-24T04:46:26.352380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['min_week'] = data['Weeks']\ndata.loc[data.WHERE=='test','min_week'] = np.nan\ndata['min_week'] = data.groupby('Patient')['min_week'].transform('min')","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:46:28.532977Z","iopub.execute_input":"2023-06-24T04:46:28.533346Z","iopub.status.idle":"2023-06-24T04:46:28.548299Z","shell.execute_reply.started":"2023-06-24T04:46:28.533310Z","shell.execute_reply":"2023-06-24T04:46:28.547250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base = data.loc[data.Weeks == data.min_week]\nbase = base[['Patient','FVC']].copy()\nbase.columns = ['Patient','min_FVC']\nbase['nb'] = 1\nbase['nb'] = base.groupby('Patient')['nb'].transform('cumsum')\nbase = base[base.nb==1]\nbase.drop('nb', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:46:30.961496Z","iopub.execute_input":"2023-06-24T04:46:30.961854Z","iopub.status.idle":"2023-06-24T04:46:30.977755Z","shell.execute_reply.started":"2023-06-24T04:46:30.961821Z","shell.execute_reply":"2023-06-24T04:46:30.976855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = data.merge(base, on='Patient', how='left')\ndata['base_week'] = data['Weeks'] - data['min_week']\ndel base","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:46:33.330566Z","iopub.execute_input":"2023-06-24T04:46:33.330977Z","iopub.status.idle":"2023-06-24T04:46:33.350767Z","shell.execute_reply.started":"2023-06-24T04:46:33.330930Z","shell.execute_reply":"2023-06-24T04:46:33.349822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"COLS = ['Sex','SmokingStatus'] #,'Age'\nFE = []\nfor col in COLS:\n    for mod in data[col].unique():\n        FE.append(mod)\n        data[mod] = (data[col] == mod).astype(int)","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:46:35.745463Z","iopub.execute_input":"2023-06-24T04:46:35.745835Z","iopub.status.idle":"2023-06-24T04:46:35.760359Z","shell.execute_reply.started":"2023-06-24T04:46:35.745801Z","shell.execute_reply":"2023-06-24T04:46:35.759469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#\ndata['age'] = (data['Age'] - data['Age'].min() ) / ( data['Age'].max() - data['Age'].min() )\ndata['BASE'] = (data['min_FVC'] - data['min_FVC'].min() ) / ( data['min_FVC'].max() - data['min_FVC'].min() )\ndata['week'] = (data['base_week'] - data['base_week'].min() ) / ( data['base_week'].max() - data['base_week'].min() )\ndata['percent'] = (data['Percent'] - data['Percent'].min() ) / ( data['Percent'].max() - data['Percent'].min() )\nFE += ['age','percent','week','BASE']","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:46:38.225192Z","iopub.execute_input":"2023-06-24T04:46:38.225567Z","iopub.status.idle":"2023-06-24T04:46:38.242703Z","shell.execute_reply.started":"2023-06-24T04:46:38.225532Z","shell.execute_reply":"2023-06-24T04:46:38.241538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr = data.loc[data.WHERE=='train']\nchunk = data.loc[data.WHERE=='val']\nsub = data.loc[data.WHERE=='test']\ndel data","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:46:42.096388Z","iopub.execute_input":"2023-06-24T04:46:42.096784Z","iopub.status.idle":"2023-06-24T04:46:42.110160Z","shell.execute_reply.started":"2023-06-24T04:46:42.096750Z","shell.execute_reply":"2023-06-24T04:46:42.109266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr.shape, chunk.shape, sub.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:46:45.125793Z","iopub.execute_input":"2023-06-24T04:46:45.126153Z","iopub.status.idle":"2023-06-24T04:46:45.141598Z","shell.execute_reply.started":"2023-06-24T04:46:45.126118Z","shell.execute_reply":"2023-06-24T04:46:45.140320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4.3 The change of mloss <a class=\"anchor\" id=\"4.3\"></a>\n\n[Back to Table of Contents](#0.1)","metadata":{}},{"cell_type":"code","source":"C1, C2 = tf.constant(70, dtype='float32'), tf.constant(1000, dtype=\"float32\")\n\ndef score(y_true, y_pred):\n    tf.dtypes.cast(y_true, tf.float32)\n    tf.dtypes.cast(y_pred, tf.float32)\n    sigma = y_pred[:, 2] - y_pred[:, 0]\n    fvc_pred = y_pred[:, 1]\n    \n    #sigma_clip = sigma + C1\n    sigma_clip = tf.maximum(sigma, C1)\n    delta = tf.abs(y_true[:, 0] - fvc_pred)\n    delta = tf.minimum(delta, C2)\n    sq2 = tf.sqrt( tf.dtypes.cast(2, dtype=tf.float32) )\n    metric = (delta / sigma_clip)*sq2 + tf.math.log(sigma_clip* sq2)\n    return K.mean(metric)\n\ndef qloss(y_true, y_pred):\n    # Pinball loss for multiple quantiles\n    qs = [0.2, 0.50, 0.8]\n    q = tf.constant(np.array([qs]), dtype=tf.float32)\n    e = y_true - y_pred\n    v = tf.maximum(q*e, (q-1)*e)\n    return K.mean(v)\n\ndef mloss(_lambda):\n    def loss(y_true, y_pred):\n        return _lambda * qloss(y_true, y_pred) + (1 - _lambda)*score(y_true, y_pred)\n    return loss\n\ndef make_model(nh):\n    z = L.Input((nh,), name=\"Patient\")\n    x = L.Dense(100, activation=\"relu\", name=\"d1\")(z)\n    x = L.Dense(100, activation=\"relu\", name=\"d2\")(x)\n    p1 = L.Dense(3, activation=\"linear\", name=\"p1\")(x)\n    p2 = L.Dense(3, activation=\"relu\", name=\"p2\")(x)\n    preds = L.Lambda(lambda x: x[0] + tf.cumsum(x[1], axis=1), \n                     name=\"preds\")([p1, p2])\n    \n    model = M.Model(z, preds, name=\"CNN\")\n    model.compile(loss=mloss(0.65), optimizer=tf.keras.optimizers.Adam(lr=0.1, beta_1=0.9, beta_2=0.999, epsilon=None, decay=0.01, amsgrad=False), metrics=[score])\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:46:47.295653Z","iopub.execute_input":"2023-06-24T04:46:47.296013Z","iopub.status.idle":"2023-06-24T04:46:47.319021Z","shell.execute_reply.started":"2023-06-24T04:46:47.295979Z","shell.execute_reply":"2023-06-24T04:46:47.318145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = tr['FVC'].values\nz = tr[FE].values\nze = sub[FE].values\nnh = z.shape[1]\npe = np.zeros((ze.shape[0], 3))\npred = np.zeros((z.shape[0], 3))","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:46:50.561374Z","iopub.execute_input":"2023-06-24T04:46:50.561829Z","iopub.status.idle":"2023-06-24T04:46:50.574277Z","shell.execute_reply.started":"2023-06-24T04:46:50.561791Z","shell.execute_reply":"2023-06-24T04:46:50.573514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"net = make_model(nh)\nprint(net.summary())\nprint(net.count_params())","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:46:52.394735Z","iopub.execute_input":"2023-06-24T04:46:52.395091Z","iopub.status.idle":"2023-06-24T04:46:52.770986Z","shell.execute_reply.started":"2023-06-24T04:46:52.395057Z","shell.execute_reply":"2023-06-24T04:46:52.770199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NFOLD = 5 # originally 5\nkf = KFold(n_splits=NFOLD)","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:46:55.710022Z","iopub.execute_input":"2023-06-24T04:46:55.710374Z","iopub.status.idle":"2023-06-24T04:46:55.715804Z","shell.execute_reply.started":"2023-06-24T04:46:55.710340Z","shell.execute_reply":"2023-06-24T04:46:55.714644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ncnt = 0\nEPOCHS = 800\nfor tr_idx, val_idx in kf.split(z):\n    cnt += 1\n    print(f\"FOLD {cnt}\")\n    net = make_model(nh)\n    net.fit(z[tr_idx], y[tr_idx], batch_size=BATCH_SIZE, epochs=EPOCHS, \n            validation_data=(z[val_idx], y[val_idx]), verbose=0) #\n    print(\"train\", net.evaluate(z[tr_idx], y[tr_idx], verbose=0, batch_size=BATCH_SIZE))\n    print(\"val\", net.evaluate(z[val_idx], y[val_idx], verbose=0, batch_size=BATCH_SIZE))\n    print(\"predict val...\")\n    pred[val_idx] = net.predict(z[val_idx], batch_size=BATCH_SIZE, verbose=0)\n    print(\"predict test...\")\n    pe += net.predict(ze, batch_size=BATCH_SIZE, verbose=0) / NFOLD","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:46:59.462355Z","iopub.execute_input":"2023-06-24T04:46:59.462785Z","iopub.status.idle":"2023-06-24T04:50:40.803778Z","shell.execute_reply.started":"2023-06-24T04:46:59.462745Z","shell.execute_reply":"2023-06-24T04:50:40.801937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sigma_opt = mean_absolute_error(y, pred[:, 1])\nunc = pred[:,2] - pred[:, 0]\nsigma_mean = np.mean(unc)\nprint(sigma_opt, sigma_mean)","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:50:47.344086Z","iopub.execute_input":"2023-06-24T04:50:47.344455Z","iopub.status.idle":"2023-06-24T04:50:47.351947Z","shell.execute_reply.started":"2023-06-24T04:50:47.344400Z","shell.execute_reply":"2023-06-24T04:50:47.350848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PREDICTION\nsub['FVC1'] = 1.*pe[:, 1]\nsub['Confidence1'] = pe[:, 2] - pe[:, 0]\nsubm = sub[['Patient_Week','FVC','Confidence','FVC1','Confidence1']].copy()\nsubm.loc[~subm.FVC1.isnull()].head(10)","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:50:58.677358Z","iopub.execute_input":"2023-06-24T04:50:58.677774Z","iopub.status.idle":"2023-06-24T04:50:58.706853Z","shell.execute_reply.started":"2023-06-24T04:50:58.677738Z","shell.execute_reply":"2023-06-24T04:50:58.705909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subm.loc[~subm.FVC1.isnull(),'FVC'] = subm.loc[~subm.FVC1.isnull(),'FVC1']\nif sigma_mean<70:\n    subm['Confidence'] = sigma_opt\nelse:\n    subm.loc[~subm.FVC1.isnull(),'Confidence'] = subm.loc[~subm.FVC1.isnull(),'Confidence1']","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:51:17.443701Z","iopub.execute_input":"2023-06-24T04:51:17.444074Z","iopub.status.idle":"2023-06-24T04:51:17.458480Z","shell.execute_reply.started":"2023-06-24T04:51:17.444040Z","shell.execute_reply":"2023-06-24T04:51:17.457521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subm.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:51:20.941988Z","iopub.execute_input":"2023-06-24T04:51:20.942338Z","iopub.status.idle":"2023-06-24T04:51:20.955982Z","shell.execute_reply.started":"2023-06-24T04:51:20.942304Z","shell.execute_reply":"2023-06-24T04:51:20.954920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subm.describe().T","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:51:30.441811Z","iopub.execute_input":"2023-06-24T04:51:30.442217Z","iopub.status.idle":"2023-06-24T04:51:30.475585Z","shell.execute_reply.started":"2023-06-24T04:51:30.442179Z","shell.execute_reply":"2023-06-24T04:51:30.474851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"otest = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/test.csv')\nfor i in range(len(otest)):\n    subm.loc[subm['Patient_Week']==otest.Patient[i]+'_'+str(otest.Weeks[i]), 'FVC'] = otest.FVC[i]\n    subm.loc[subm['Patient_Week']==otest.Patient[i]+'_'+str(otest.Weeks[i]), 'Confidence'] = 0.1","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:51:34.610297Z","iopub.execute_input":"2023-06-24T04:51:34.610710Z","iopub.status.idle":"2023-06-24T04:51:34.656467Z","shell.execute_reply.started":"2023-06-24T04:51:34.610673Z","shell.execute_reply":"2023-06-24T04:51:34.655706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subm[[\"Patient_Week\",\"FVC\",\"Confidence\"]].to_csv(\"submission_regression.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:51:42.228311Z","iopub.execute_input":"2023-06-24T04:51:42.228751Z","iopub.status.idle":"2023-06-24T04:51:42.467208Z","shell.execute_reply.started":"2023-06-24T04:51:42.228712Z","shell.execute_reply":"2023-06-24T04:51:42.466242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"reg_sub = subm[[\"Patient_Week\",\"FVC\",\"Confidence\"]].copy()","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:51:45.162673Z","iopub.execute_input":"2023-06-24T04:51:45.163082Z","iopub.status.idle":"2023-06-24T04:51:45.169945Z","shell.execute_reply.started":"2023-06-24T04:51:45.163045Z","shell.execute_reply":"2023-06-24T04:51:45.168770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4.4 Ensemble and blending <a class=\"anchor\" id=\"4.4\"></a>\n\n[Back to Table of Contents](#0.1)","metadata":{}},{"cell_type":"code","source":"def calculate_metric(fvc_true, fvc_pred, confidence):\n    sigma_clipped = np.maximum(confidence, 70)\n    delta = np.minimum(np.abs(fvc_true - fvc_pred), 1000)\n    metric = -np.sqrt(2) * delta / sigma_clipped - np.log(np.sqrt(2) * sigma_clipped)\n    return metric\nfor sub_item in subs:\n    try:\n        img_sub = sub_item[0][[\"Patient_Week\",\"FVC\",\"Confidence\"]].copy()\n        df1 = img_sub.sort_values(by=['Patient_Week'], ascending=True).reset_index(drop=True)\n        df2 = reg_sub.sort_values(by=['Patient_Week'], ascending=True).reset_index(drop=True)\n\n        df = df1[['Patient_Week']].copy()\n        for FVC_weight in FVC_weight_list:\n            for Confidence_weight in Confidence_weight_list:\n                df['FVC'] = FVC_weight*df1['FVC'] + (1-FVC_weight)*df2['FVC']\n                df['Confidence'] = Confidence_weight*df1['Confidence'] + (1-Confidence_weight)*df2['Confidence']\n\n                sub_calculate = df.copy()\n                sub_calculate[['Patient', 'Weeks']] = df['Patient_Week'].str.split('_', expand=True)\n                sub_calculate['Weeks'] = sub_calculate['Weeks'].astype(int)\n\n                merged = pd.merge(train, sub_calculate, on=['Patient', 'Weeks'])\n                merged = merged.rename(columns={'FVC_x': 'FVC', 'FVC_y': 'FVC_pred'})\n                \n                merged = merged[merged['FVC'] != merged['FVC_pred']]\n\n                # Extracting the necessary columns\n                fvc_true = merged['FVC'].values\n                fvc_pred = merged['FVC_pred'].values\n                confidence = merged['Confidence'].values\n\n                metric = calculate_metric(fvc_true, fvc_pred, confidence).mean()\n                print(FVC_weight, Confidence_weight, sub_item[1], sub_item[2], metric)\n                g = open(\"log.csv\", \"a\")\n                g.write(\"\\n{},{},{},{},{}\".format(FVC_weight, Confidence_weight, sub_item[1], sub_item[2], metric))\n                g.close()\n    except Exception as ex:\n        template = \"An exception of type {0} occurred. Arguments:\\n{1!r}\"\n        message = template.format(type(ex).__name__, ex.args)\n        f = open(\"log.txt\", \"a\")\n        f.write(\"\\n******\")\n        f.write(\"\\n FVC_weight, Confidence_weight, GaussianNoise_stddev, Dropout_model, metric: {},{},{},{},{}\".format(FVC_weight, Confidence_weight, sub_item[1], sub_item[2], metric))\n        f.write(\"\\n\"+message)\n        f.write(\"\\n******\")\n        f.close()","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:56:26.120662Z","iopub.execute_input":"2023-06-24T04:56:26.121040Z","iopub.status.idle":"2023-06-24T04:56:27.870793Z","shell.execute_reply.started":"2023-06-24T04:56:26.121004Z","shell.execute_reply":"2023-06-24T04:56:27.869821Z"},"trusted":true},"execution_count":null,"outputs":[]}]}