{"cells":[{"metadata":{"papermill":{"duration":0.01277,"end_time":"2020-08-20T06:15:47.323184","exception":false,"start_time":"2020-08-20T06:15:47.310414","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Overview & Remarks\n\n- Just some experiments I did with efficientnets b0-b7 and blending predictions\n- Best LB of efficientnets was around -0.6922\n- Tried blending efficientnets b0-b7 in a single run but due to out-of-memory errors, it was not successful\n    - you may find the code to perform the mean blend here as well\n- EfficientNets are trained for 30 or 50 epochs with modified callbacks and training parameters\n- More models are being experimented currently. Will update this notebook when I have better results!","execution_count":null},{"metadata":{"papermill":{"duration":0.010869,"end_time":"2020-08-20T06:15:47.346694","exception":false,"start_time":"2020-08-20T06:15:47.335825","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Acknowledgements\n\n- Ulrich GOUE's Osic-Multiple-Quantile-Regression-Starter\n    - Model that uses images can be found at: https://www.kaggle.com/miklgr500/linear-decay-based-on-resnet-cnn\n- Michael Kazachok's Linear Decay (based on ResNet CNN)\n    - Model that uses tabular data can be found at: https://www.kaggle.com/ulrich07/osic-multiple-quantile-regression-starter\n- Replaced Michael's model with EfficientNets B0, B2, B4\n- I only tweaked the parameters for the models ","execution_count":null},{"metadata":{"papermill":{"duration":0.010748,"end_time":"2020-08-20T06:15:47.368346","exception":false,"start_time":"2020-08-20T06:15:47.357598","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Imports","execution_count":null},{"metadata":{"_kg_hide-output":true,"execution":{"iopub.execute_input":"2020-08-20T06:15:47.399099Z","iopub.status.busy":"2020-08-20T06:15:47.398249Z","iopub.status.idle":"2020-08-20T06:16:06.698679Z","shell.execute_reply":"2020-08-20T06:16:06.697503Z"},"papermill":{"duration":19.319372,"end_time":"2020-08-20T06:16:06.698858","exception":false,"start_time":"2020-08-20T06:15:47.379486","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"!pip install ../input/kerasapplications/keras-team-keras-applications-3b180cb -f ./ --no-index\n!pip install ../input/efficientnet/efficientnet-1.1.0/ -f ./ --no-index","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.execute_input":"2020-08-20T06:16:06.737182Z","iopub.status.busy":"2020-08-20T06:16:06.734497Z","iopub.status.idle":"2020-08-20T06:16:14.406257Z","shell.execute_reply":"2020-08-20T06:16:14.404646Z"},"papermill":{"duration":7.694311,"end_time":"2020-08-20T06:16:14.406389","exception":false,"start_time":"2020-08-20T06:16:06.712078","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"import os\nimport cv2\nimport pydicom\nimport pandas as pd\nimport numpy as np \nimport tensorflow as tf \nimport matplotlib.pyplot as plt \nimport random\nfrom tqdm.notebook import tqdm \nfrom sklearn.model_selection import train_test_split, KFold\nfrom sklearn.metrics import mean_absolute_error\nfrom tensorflow_addons.optimizers import RectifiedAdam\nfrom tensorflow.keras import Model\nimport tensorflow.keras.backend as K\nimport tensorflow.keras.layers as L\nimport tensorflow.keras.models as M\nfrom tensorflow.keras.optimizers import Nadam\nimport seaborn as sns\nfrom PIL import Image\n\ndef seed_everything(seed=2020):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n    \nseed_everything(42)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:16:14.438114Z","iopub.status.busy":"2020-08-20T06:16:14.437467Z","iopub.status.idle":"2020-08-20T06:16:17.895294Z","shell.execute_reply":"2020-08-20T06:16:17.894514Z"},"papermill":{"duration":3.475604,"end_time":"2020-08-20T06:16:17.895429","exception":false,"start_time":"2020-08-20T06:16:14.419825","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"config = tf.compat.v1.ConfigProto()\nconfig.gpu_options.allow_growth = True\nsession = tf.compat.v1.Session(config=config)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:16:17.931232Z","iopub.status.busy":"2020-08-20T06:16:17.930177Z","iopub.status.idle":"2020-08-20T06:16:17.94433Z","shell.execute_reply":"2020-08-20T06:16:17.943554Z"},"papermill":{"duration":0.034706,"end_time":"2020-08-20T06:16:17.944473","exception":false,"start_time":"2020-08-20T06:16:17.909767","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"train = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/train.csv') ","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.013526,"end_time":"2020-08-20T06:16:17.972042","exception":false,"start_time":"2020-08-20T06:16:17.958516","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Linear Decay (based on EfficientNets)","execution_count":null},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:16:18.009193Z","iopub.status.busy":"2020-08-20T06:16:18.00725Z","iopub.status.idle":"2020-08-20T06:16:18.01296Z","shell.execute_reply":"2020-08-20T06:16:18.012386Z"},"papermill":{"duration":0.027282,"end_time":"2020-08-20T06:16:18.013067","exception":false,"start_time":"2020-08-20T06:16:17.985785","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"def get_tab(df):\n    vector = [(df.Age.values[0] - 30) / 30] \n    \n    if df.Sex.values[0] == 'male':\n       vector.append(0)\n    else:\n       vector.append(1)\n    \n    if df.SmokingStatus.values[0] == 'Never smoked':\n        vector.extend([0,0])\n    elif df.SmokingStatus.values[0] == 'Ex-smoker':\n        vector.extend([1,1])\n    elif df.SmokingStatus.values[0] == 'Currently smokes':\n        vector.extend([0,1])\n    else:\n        vector.extend([1,0])\n    return np.array(vector) ","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:16:18.053007Z","iopub.status.busy":"2020-08-20T06:16:18.052336Z","iopub.status.idle":"2020-08-20T06:16:18.496849Z","shell.execute_reply":"2020-08-20T06:16:18.497654Z"},"papermill":{"duration":0.471292,"end_time":"2020-08-20T06:16:18.497879","exception":false,"start_time":"2020-08-20T06:16:18.026587","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"A = {} \nTAB = {} \nP = [] \nfor i, p in tqdm(enumerate(train.Patient.unique())):\n    sub = train.loc[train.Patient == p, :] \n    fvc = sub.FVC.values\n    weeks = sub.Weeks.values\n    c = np.vstack([weeks, np.ones(len(weeks))]).T\n    a, b = np.linalg.lstsq(c, fvc)[0]\n    \n    A[p] = a\n    TAB[p] = get_tab(sub)\n    P.append(p)","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.014749,"end_time":"2020-08-20T06:16:18.52867","exception":false,"start_time":"2020-08-20T06:16:18.513921","status":"completed"},"tags":[]},"cell_type":"markdown","source":"## CNN for coeff prediction","execution_count":null},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:16:18.566528Z","iopub.status.busy":"2020-08-20T06:16:18.565626Z","iopub.status.idle":"2020-08-20T06:16:18.570297Z","shell.execute_reply":"2020-08-20T06:16:18.569659Z"},"papermill":{"duration":0.027136,"end_time":"2020-08-20T06:16:18.570454","exception":false,"start_time":"2020-08-20T06:16:18.543318","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"def get_img(path):\n    d = pydicom.dcmread(path)\n    return cv2.resize(d.pixel_array / 2**11, (512, 512))","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:16:18.625371Z","iopub.status.busy":"2020-08-20T06:16:18.624302Z","iopub.status.idle":"2020-08-20T06:16:18.632515Z","shell.execute_reply":"2020-08-20T06:16:18.633329Z"},"papermill":{"duration":0.048189,"end_time":"2020-08-20T06:16:18.633556","exception":false,"start_time":"2020-08-20T06:16:18.585367","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"from tensorflow.keras.utils import Sequence\n\nclass IGenerator(Sequence):\n    BAD_ID = ['ID00011637202177653955184', 'ID00052637202186188008618']\n    def __init__(self, keys, a, tab, batch_size=32):\n        self.keys = [k for k in keys if k not in self.BAD_ID]\n        self.a = a\n        self.tab = tab\n        self.batch_size = batch_size\n        \n        self.train_data = {}\n        for p in train.Patient.values:\n            self.train_data[p] = os.listdir(f'../input/osic-pulmonary-fibrosis-progression/train/{p}/')\n    \n    def __len__(self):\n        return 1000\n    \n    def __getitem__(self, idx):\n        x = []\n        a, tab = [], [] \n        keys = np.random.choice(self.keys, size = self.batch_size)\n        for k in keys:\n            try:\n                i = np.random.choice(self.train_data[k], size=1)[0]\n                img = get_img(f'../input/osic-pulmonary-fibrosis-progression/train/{k}/{i}')\n                x.append(img)\n                a.append(self.a[k])\n                tab.append(self.tab[k])\n            except:\n                print(k, i)\n       \n        x,a,tab = np.array(x), np.array(a), np.array(tab)\n        x = np.expand_dims(x, axis=-1)\n        return [x, tab] , a","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:16:18.695366Z","iopub.status.busy":"2020-08-20T06:16:18.688614Z","iopub.status.idle":"2020-08-20T06:17:18.857905Z","shell.execute_reply":"2020-08-20T06:17:18.858754Z"},"papermill":{"duration":60.203494,"end_time":"2020-08-20T06:17:18.858949","exception":false,"start_time":"2020-08-20T06:16:18.655455","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"from tensorflow.keras.layers import (\n    Dense, Dropout, Activation, Flatten, Input, BatchNormalization, GlobalAveragePooling2D, Add, Conv2D, AveragePooling2D, \n    LeakyReLU, Concatenate \n)\nimport efficientnet.tfkeras as efn\n\ndef get_efficientnet(model, shape):\n    models_dict = {\n        'b0': efn.EfficientNetB0(input_shape=shape,weights=None,include_top=False),\n        'b1': efn.EfficientNetB1(input_shape=shape,weights=None,include_top=False),\n        'b2': efn.EfficientNetB2(input_shape=shape,weights=None,include_top=False),\n        'b3': efn.EfficientNetB3(input_shape=shape,weights=None,include_top=False),\n        'b4': efn.EfficientNetB4(input_shape=shape,weights=None,include_top=False),\n        'b5': efn.EfficientNetB5(input_shape=shape,weights=None,include_top=False),\n        'b6': efn.EfficientNetB6(input_shape=shape,weights=None,include_top=False),\n        'b7': efn.EfficientNetB7(input_shape=shape,weights=None,include_top=False)\n    }\n    return models_dict[model]\n\ndef build_model(shape=(512, 512, 1), model_class=None):\n    inp = Input(shape=shape)\n    base = get_efficientnet(model_class, shape)\n    x = base(inp)\n    x = GlobalAveragePooling2D()(x)\n    inp2 = Input(shape=(4,))\n    x2 = tf.keras.layers.GaussianNoise(0.2)(inp2)\n    x = Concatenate()([x, x2]) \n    x = Dropout(0.4)(x) \n    x = Dense(1)(x)\n    model = Model([inp, inp2] , x)\n    \n    weights = [w for w in os.listdir('../input/osic-model-weights') if model_class in w][0]\n    model.load_weights('../input/osic-model-weights/' + weights)\n    return model\n\nmodel_classes = ['b5'] #['b0','b1','b2','b3',b4','b5','b6','b7']\nmodels = [build_model(shape=(512, 512, 1), model_class=m) for m in model_classes]\nprint('Number of models: ' + str(len(models)))","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:17:18.895261Z","iopub.status.busy":"2020-08-20T06:17:18.893357Z","iopub.status.idle":"2020-08-20T06:17:18.895989Z","shell.execute_reply":"2020-08-20T06:17:18.896494Z"},"papermill":{"duration":0.022687,"end_time":"2020-08-20T06:17:18.896639","exception":false,"start_time":"2020-08-20T06:17:18.873952","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"from sklearn.model_selection import train_test_split \n\ntr_p, vl_p = train_test_split(P, \n                              shuffle=True, \n                              train_size= 0.8) ","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:17:18.928865Z","iopub.status.busy":"2020-08-20T06:17:18.928112Z","iopub.status.idle":"2020-08-20T06:17:19.213306Z","shell.execute_reply":"2020-08-20T06:17:19.214676Z"},"papermill":{"duration":0.305095,"end_time":"2020-08-20T06:17:19.214988","exception":false,"start_time":"2020-08-20T06:17:18.909893","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"sns.distplot(list(A.values()));","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:17:19.273754Z","iopub.status.busy":"2020-08-20T06:17:19.272794Z","iopub.status.idle":"2020-08-20T06:17:19.280025Z","shell.execute_reply":"2020-08-20T06:17:19.279193Z"},"papermill":{"duration":0.039907,"end_time":"2020-08-20T06:17:19.280171","exception":false,"start_time":"2020-08-20T06:17:19.240264","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"def score(fvc_true, fvc_pred, sigma):\n    sigma_clip = np.maximum(sigma, 70) # changed from 70, trie 66.7 too\n    delta = np.abs(fvc_true - fvc_pred)\n    delta = np.minimum(delta, 1000)\n    sq2 = np.sqrt(2)\n    metric = (delta / sigma_clip)*sq2 + np.log(sigma_clip* sq2)\n    return np.mean(metric)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:17:19.347844Z","iopub.status.busy":"2020-08-20T06:17:19.337515Z","iopub.status.idle":"2020-08-20T06:33:10.042541Z","shell.execute_reply":"2020-08-20T06:33:10.041842Z"},"papermill":{"duration":950.742301,"end_time":"2020-08-20T06:33:10.042704","exception":false,"start_time":"2020-08-20T06:17:19.300403","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"subs = []\nfor model in models:\n    metric = []\n    for q in tqdm(range(1, 10)):\n        m = []\n        for p in vl_p:\n            x = [] \n            tab = [] \n\n            if p in ['ID00011637202177653955184', 'ID00052637202186188008618']:\n                continue\n\n            ldir = os.listdir(f'../input/osic-pulmonary-fibrosis-progression/train/{p}/')\n            for i in ldir:\n                if int(i[:-4]) / len(ldir) < 0.8 and int(i[:-4]) / len(ldir) > 0.15:\n                    x.append(get_img(f'../input/osic-pulmonary-fibrosis-progression/train/{p}/{i}')) \n                    tab.append(get_tab(train.loc[train.Patient == p, :])) \n            if len(x) < 1:\n                continue\n            tab = np.array(tab) \n\n            x = np.expand_dims(x, axis=-1) \n            _a = model.predict([x, tab]) \n            a = np.quantile(_a, q / 10)\n\n            percent_true = train.Percent.values[train.Patient == p]\n            fvc_true = train.FVC.values[train.Patient == p]\n            weeks_true = train.Weeks.values[train.Patient == p]\n\n            fvc = a * (weeks_true - weeks_true[0]) + fvc_true[0]\n            percent = percent_true[0] - a * abs(weeks_true - weeks_true[0])\n            m.append(score(fvc_true, fvc, percent))\n        print(np.mean(m))\n        metric.append(np.mean(m))\n\n    q = (np.argmin(metric) + 1)/ 10\n\n    sub = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/sample_submission.csv') \n    test = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/test.csv') \n    A_test, B_test, P_test,W, FVC= {}, {}, {},{},{} \n    STD, WEEK = {}, {} \n    for p in test.Patient.unique():\n        x = [] \n        tab = [] \n        ldir = os.listdir(f'../input/osic-pulmonary-fibrosis-progression/test/{p}/')\n        for i in ldir:\n            if int(i[:-4]) / len(ldir) < 0.8 and int(i[:-4]) / len(ldir) > 0.15:\n                x.append(get_img(f'../input/osic-pulmonary-fibrosis-progression/test/{p}/{i}')) \n                tab.append(get_tab(test.loc[test.Patient == p, :])) \n        if len(x) <= 1:\n            continue\n        tab = np.array(tab) \n\n        x = np.expand_dims(x, axis=-1) \n        _a = model.predict([x, tab]) \n        a = np.quantile(_a, q)\n        A_test[p] = a\n        B_test[p] = test.FVC.values[test.Patient == p] - a*test.Weeks.values[test.Patient == p]\n        P_test[p] = test.Percent.values[test.Patient == p] \n        WEEK[p] = test.Weeks.values[test.Patient == p]\n\n    for k in sub.Patient_Week.values:\n        p, w = k.split('_')\n        w = int(w) \n\n        fvc = A_test[p] * w + B_test[p]\n        sub.loc[sub.Patient_Week == k, 'FVC'] = fvc\n        sub.loc[sub.Patient_Week == k, 'Confidence'] = (\n            P_test[p] - A_test[p] * abs(WEEK[p] - w) \n    ) \n\n    _sub = sub[[\"Patient_Week\",\"FVC\",\"Confidence\"]].copy()\n    subs.append(_sub)","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.02121,"end_time":"2020-08-20T06:33:10.087604","exception":false,"start_time":"2020-08-20T06:33:10.066394","status":"completed"},"tags":[]},"cell_type":"markdown","source":"## Averaging Predictions","execution_count":null},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:33:10.138694Z","iopub.status.busy":"2020-08-20T06:33:10.137896Z","iopub.status.idle":"2020-08-20T06:33:10.170793Z","shell.execute_reply":"2020-08-20T06:33:10.171688Z"},"papermill":{"duration":0.064528,"end_time":"2020-08-20T06:33:10.17189","exception":false,"start_time":"2020-08-20T06:33:10.107362","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"N = len(subs)\nsub = subs[0].copy() # ref\nsub[\"FVC\"] = 0\nsub[\"Confidence\"] = 0\nfor i in range(N):\n    sub[\"FVC\"] += subs[0][\"FVC\"] * (1/N)\n    sub[\"Confidence\"] += subs[0][\"Confidence\"] * (1/N)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:33:10.223025Z","iopub.status.busy":"2020-08-20T06:33:10.222336Z","iopub.status.idle":"2020-08-20T06:33:10.237168Z","shell.execute_reply":"2020-08-20T06:33:10.238494Z"},"papermill":{"duration":0.045013,"end_time":"2020-08-20T06:33:10.238695","exception":false,"start_time":"2020-08-20T06:33:10.193682","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"sub.head()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:33:10.29106Z","iopub.status.busy":"2020-08-20T06:33:10.29017Z","iopub.status.idle":"2020-08-20T06:33:10.463889Z","shell.execute_reply":"2020-08-20T06:33:10.462659Z"},"papermill":{"duration":0.203338,"end_time":"2020-08-20T06:33:10.464024","exception":false,"start_time":"2020-08-20T06:33:10.260686","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"sub[[\"Patient_Week\",\"FVC\",\"Confidence\"]].to_csv(\"submission_img.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:33:10.501384Z","iopub.status.busy":"2020-08-20T06:33:10.499446Z","iopub.status.idle":"2020-08-20T06:33:10.502189Z","shell.execute_reply":"2020-08-20T06:33:10.502638Z"},"papermill":{"duration":0.023987,"end_time":"2020-08-20T06:33:10.502781","exception":false,"start_time":"2020-08-20T06:33:10.478794","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"img_sub = sub[[\"Patient_Week\",\"FVC\",\"Confidence\"]].copy()","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.01407,"end_time":"2020-08-20T06:33:10.531207","exception":false,"start_time":"2020-08-20T06:33:10.517137","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Osic-Multiple-Quantile-Regression","execution_count":null},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:33:10.572993Z","iopub.status.busy":"2020-08-20T06:33:10.57217Z","iopub.status.idle":"2020-08-20T06:33:10.620072Z","shell.execute_reply":"2020-08-20T06:33:10.620581Z"},"papermill":{"duration":0.074696,"end_time":"2020-08-20T06:33:10.620699","exception":false,"start_time":"2020-08-20T06:33:10.546003","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"ROOT = \"../input/osic-pulmonary-fibrosis-progression\"\nBATCH_SIZE=128\n\ntr = pd.read_csv(f\"{ROOT}/train.csv\")\ntr.drop_duplicates(keep=False, inplace=True, subset=['Patient','Weeks'])\nchunk = pd.read_csv(f\"{ROOT}/test.csv\")\n\nprint(\"add infos\")\nsub = pd.read_csv(f\"{ROOT}/sample_submission.csv\")\nsub['Patient'] = sub['Patient_Week'].apply(lambda x:x.split('_')[0])\nsub['Weeks'] = sub['Patient_Week'].apply(lambda x: int(x.split('_')[-1]))\nsub =  sub[['Patient','Weeks','Confidence','Patient_Week']]\nsub = sub.merge(chunk.drop('Weeks', axis=1), on=\"Patient\")","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:33:10.681234Z","iopub.status.busy":"2020-08-20T06:33:10.6795Z","iopub.status.idle":"2020-08-20T06:33:10.684677Z","shell.execute_reply":"2020-08-20T06:33:10.685536Z"},"papermill":{"duration":0.05073,"end_time":"2020-08-20T06:33:10.685823","exception":false,"start_time":"2020-08-20T06:33:10.635093","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"tr['WHERE'] = 'train'\nchunk['WHERE'] = 'val'\nsub['WHERE'] = 'test'\ndata = tr.append([chunk, sub])","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:33:10.738199Z","iopub.status.busy":"2020-08-20T06:33:10.736033Z","iopub.status.idle":"2020-08-20T06:33:10.742927Z","shell.execute_reply":"2020-08-20T06:33:10.742288Z"},"papermill":{"duration":0.036207,"end_time":"2020-08-20T06:33:10.743056","exception":false,"start_time":"2020-08-20T06:33:10.706849","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"print(tr.shape, chunk.shape, sub.shape, data.shape)\nprint(tr.Patient.nunique(), chunk.Patient.nunique(), sub.Patient.nunique(), \n      data.Patient.nunique())\n#","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:33:10.781707Z","iopub.status.busy":"2020-08-20T06:33:10.780787Z","iopub.status.idle":"2020-08-20T06:33:10.794709Z","shell.execute_reply":"2020-08-20T06:33:10.794214Z"},"papermill":{"duration":0.036575,"end_time":"2020-08-20T06:33:10.794829","exception":false,"start_time":"2020-08-20T06:33:10.758254","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"data['min_week'] = data['Weeks']\ndata.loc[data.WHERE=='test','min_week'] = np.nan\ndata['min_week'] = data.groupby('Patient')['min_week'].transform('min')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:33:10.842537Z","iopub.status.busy":"2020-08-20T06:33:10.834491Z","iopub.status.idle":"2020-08-20T06:33:10.845546Z","shell.execute_reply":"2020-08-20T06:33:10.844948Z"},"papermill":{"duration":0.036389,"end_time":"2020-08-20T06:33:10.845732","exception":false,"start_time":"2020-08-20T06:33:10.809343","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"base = data.loc[data.Weeks == data.min_week]\nbase = base[['Patient','FVC']].copy()\nbase.columns = ['Patient','min_FVC']\nbase['nb'] = 1\nbase['nb'] = base.groupby('Patient')['nb'].transform('cumsum')\nbase = base[base.nb==1]\nbase.drop('nb', axis=1, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:33:10.891576Z","iopub.status.busy":"2020-08-20T06:33:10.890753Z","iopub.status.idle":"2020-08-20T06:33:10.899461Z","shell.execute_reply":"2020-08-20T06:33:10.89894Z"},"papermill":{"duration":0.034043,"end_time":"2020-08-20T06:33:10.899552","exception":false,"start_time":"2020-08-20T06:33:10.865509","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"data = data.merge(base, on='Patient', how='left')\ndata['base_week'] = data['Weeks'] - data['min_week']\ndel base","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:33:10.944949Z","iopub.status.busy":"2020-08-20T06:33:10.935383Z","iopub.status.idle":"2020-08-20T06:33:10.952325Z","shell.execute_reply":"2020-08-20T06:33:10.953021Z"},"papermill":{"duration":0.039795,"end_time":"2020-08-20T06:33:10.953169","exception":false,"start_time":"2020-08-20T06:33:10.913374","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"COLS = ['Sex','SmokingStatus'] #,'Age'\nFE = []\nfor col in COLS:\n    for mod in data[col].unique():\n        FE.append(mod)\n        data[mod] = (data[col] == mod).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:33:10.999023Z","iopub.status.busy":"2020-08-20T06:33:10.997985Z","iopub.status.idle":"2020-08-20T06:33:11.005337Z","shell.execute_reply":"2020-08-20T06:33:11.00486Z"},"papermill":{"duration":0.036078,"end_time":"2020-08-20T06:33:11.005436","exception":false,"start_time":"2020-08-20T06:33:10.969358","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"#\ndata['age'] = (data['Age'] - data['Age'].min() ) / ( data['Age'].max() - data['Age'].min() )\ndata['BASE'] = (data['min_FVC'] - data['min_FVC'].min() ) / ( data['min_FVC'].max() - data['min_FVC'].min() )\ndata['week'] = (data['base_week'] - data['base_week'].min() ) / ( data['base_week'].max() - data['base_week'].min() )\ndata['percent'] = (data['Percent'] - data['Percent'].min() ) / ( data['Percent'].max() - data['Percent'].min() )\nFE += ['age','percent','week','BASE']","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:33:11.04079Z","iopub.status.busy":"2020-08-20T06:33:11.039869Z","iopub.status.idle":"2020-08-20T06:33:11.047077Z","shell.execute_reply":"2020-08-20T06:33:11.04641Z"},"papermill":{"duration":0.027588,"end_time":"2020-08-20T06:33:11.047187","exception":false,"start_time":"2020-08-20T06:33:11.019599","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"tr = data.loc[data.WHERE=='train']\nchunk = data.loc[data.WHERE=='val']\nsub = data.loc[data.WHERE=='test']\ndel data","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:33:11.080821Z","iopub.status.busy":"2020-08-20T06:33:11.080211Z","iopub.status.idle":"2020-08-20T06:33:11.085484Z","shell.execute_reply":"2020-08-20T06:33:11.084933Z"},"papermill":{"duration":0.023605,"end_time":"2020-08-20T06:33:11.085591","exception":false,"start_time":"2020-08-20T06:33:11.061986","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"tr.shape, chunk.shape, sub.shape","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:33:11.135439Z","iopub.status.busy":"2020-08-20T06:33:11.133105Z","iopub.status.idle":"2020-08-20T06:33:11.138116Z","shell.execute_reply":"2020-08-20T06:33:11.137528Z"},"papermill":{"duration":0.038105,"end_time":"2020-08-20T06:33:11.13825","exception":false,"start_time":"2020-08-20T06:33:11.100145","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"C1, C2 = tf.constant(70, dtype='float32'), tf.constant(1000, dtype=\"float32\")\n\ndef score(y_true, y_pred):\n    tf.dtypes.cast(y_true, tf.float32)\n    tf.dtypes.cast(y_pred, tf.float32)\n    sigma = y_pred[:, 2] - y_pred[:, 0]\n    fvc_pred = y_pred[:, 1]\n    \n    #sigma_clip = sigma + C1\n    sigma_clip = tf.maximum(sigma, C1)\n    delta = tf.abs(y_true[:, 0] - fvc_pred)\n    delta = tf.minimum(delta, C2)\n    sq2 = tf.sqrt( tf.dtypes.cast(2, dtype=tf.float32) )\n    metric = (delta / sigma_clip)*sq2 + tf.math.log(sigma_clip* sq2)\n    return K.mean(metric)\n\ndef qloss(y_true, y_pred):\n    # Pinball loss for multiple quantiles\n    qs = [0.2, 0.50, 0.8]\n    q = tf.constant(np.array([qs]), dtype=tf.float32)\n    e = y_true - y_pred\n    v = tf.maximum(q*e, (q-1)*e)\n    return K.mean(v)\n\ndef mloss(_lambda):\n    def loss(y_true, y_pred):\n        return _lambda * qloss(y_true, y_pred) + (1 - _lambda)*score(y_true, y_pred)\n    return loss\n\ndef make_model(nh):\n    z = L.Input((nh,), name=\"Patient\")\n    x = L.Dense(100, activation=\"relu\", name=\"d1\")(z)\n    x = L.Dense(100, activation=\"relu\", name=\"d2\")(x)\n    #x = L.Dense(100, activation=\"relu\", name=\"d3\")(x)\n    p1 = L.Dense(3, activation=\"linear\", name=\"p1\")(x)\n    p2 = L.Dense(3, activation=\"relu\", name=\"p2\")(x)\n    preds = L.Lambda(lambda x: x[0] + tf.cumsum(x[1], axis=1), \n                     name=\"preds\")([p1, p2])\n    \n    model = M.Model(z, preds, name=\"CNN\")\n    #model.compile(loss=qloss, optimizer=\"adam\", metrics=[score])\n    model.compile(loss=mloss(0.8), optimizer=tf.keras.optimizers.Adam(lr=0.09, beta_1=0.9, beta_2=0.9, epsilon=None, decay=0.01, amsgrad=False), metrics=[score])\n    return model","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:33:11.213394Z","iopub.status.busy":"2020-08-20T06:33:11.211297Z","iopub.status.idle":"2020-08-20T06:33:11.216942Z","shell.execute_reply":"2020-08-20T06:33:11.21646Z"},"papermill":{"duration":0.064406,"end_time":"2020-08-20T06:33:11.217039","exception":false,"start_time":"2020-08-20T06:33:11.152633","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"y = tr['FVC'].values\nz = tr[FE].values\nze = sub[FE].values\nnh = z.shape[1]\npe = np.zeros((ze.shape[0], 3))\npred = np.zeros((z.shape[0], 3))","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:33:11.252922Z","iopub.status.busy":"2020-08-20T06:33:11.252091Z","iopub.status.idle":"2020-08-20T06:33:11.331146Z","shell.execute_reply":"2020-08-20T06:33:11.330144Z"},"papermill":{"duration":0.099197,"end_time":"2020-08-20T06:33:11.331262","exception":false,"start_time":"2020-08-20T06:33:11.232065","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"net = make_model(nh)\nprint(net.summary())\nprint(net.count_params())","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:33:11.365585Z","iopub.status.busy":"2020-08-20T06:33:11.365Z","iopub.status.idle":"2020-08-20T06:33:11.368388Z","shell.execute_reply":"2020-08-20T06:33:11.36908Z"},"papermill":{"duration":0.022771,"end_time":"2020-08-20T06:33:11.36921","exception":false,"start_time":"2020-08-20T06:33:11.346439","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"NFOLD = 5 # originally 5\nkf = KFold(n_splits=NFOLD)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:33:11.406882Z","iopub.status.busy":"2020-08-20T06:33:11.40624Z","iopub.status.idle":"2020-08-20T06:36:49.11264Z","shell.execute_reply":"2020-08-20T06:36:49.113398Z"},"papermill":{"duration":217.729671,"end_time":"2020-08-20T06:36:49.11365","exception":false,"start_time":"2020-08-20T06:33:11.383979","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"%%time\ncnt = 0\nEPOCHS = 800\nfor tr_idx, val_idx in kf.split(z):\n    cnt += 1\n    print(f\"FOLD {cnt}\")\n    net = make_model(nh)\n    net.fit(z[tr_idx], y[tr_idx], batch_size=BATCH_SIZE, epochs=EPOCHS, \n            validation_data=(z[val_idx], y[val_idx]), verbose=0) #\n    print(\"train\", net.evaluate(z[tr_idx], y[tr_idx], verbose=0, batch_size=BATCH_SIZE))\n    print(\"val\", net.evaluate(z[val_idx], y[val_idx], verbose=0, batch_size=BATCH_SIZE))\n    print(\"predict val...\")\n    pred[val_idx] = net.predict(z[val_idx], batch_size=BATCH_SIZE, verbose=0)\n    print(\"predict test...\")\n    pe += net.predict(ze, batch_size=BATCH_SIZE, verbose=0) / NFOLD","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:36:49.150333Z","iopub.status.busy":"2020-08-20T06:36:49.149678Z","iopub.status.idle":"2020-08-20T06:36:49.155434Z","shell.execute_reply":"2020-08-20T06:36:49.154803Z"},"papermill":{"duration":0.02681,"end_time":"2020-08-20T06:36:49.155561","exception":false,"start_time":"2020-08-20T06:36:49.128751","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"sigma_opt = mean_absolute_error(y, pred[:, 1])\nunc = pred[:,2] - pred[:, 0]\nsigma_mean = np.mean(unc)\nprint(sigma_opt, sigma_mean)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:36:49.193548Z","iopub.status.busy":"2020-08-20T06:36:49.192631Z","iopub.status.idle":"2020-08-20T06:36:49.373982Z","shell.execute_reply":"2020-08-20T06:36:49.374504Z"},"papermill":{"duration":0.203545,"end_time":"2020-08-20T06:36:49.374635","exception":false,"start_time":"2020-08-20T06:36:49.17109","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"idxs = np.random.randint(0, y.shape[0], 100)\nplt.plot(y[idxs], label=\"ground truth\")\nplt.plot(pred[idxs, 0], label=\"q25\")\nplt.plot(pred[idxs, 1], label=\"q50\")\nplt.plot(pred[idxs, 2], label=\"q75\")\nplt.legend(loc=\"best\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:36:49.412966Z","iopub.status.busy":"2020-08-20T06:36:49.412342Z","iopub.status.idle":"2020-08-20T06:36:49.416834Z","shell.execute_reply":"2020-08-20T06:36:49.41784Z"},"papermill":{"duration":0.027484,"end_time":"2020-08-20T06:36:49.417992","exception":false,"start_time":"2020-08-20T06:36:49.390508","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"print(unc.min(), unc.mean(), unc.max(), (unc>=0).mean())","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:36:49.459403Z","iopub.status.busy":"2020-08-20T06:36:49.454325Z","iopub.status.idle":"2020-08-20T06:36:49.610781Z","shell.execute_reply":"2020-08-20T06:36:49.611279Z"},"papermill":{"duration":0.177323,"end_time":"2020-08-20T06:36:49.611409","exception":false,"start_time":"2020-08-20T06:36:49.434086","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"plt.hist(unc)\nplt.title(\"uncertainty in prediction\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:36:49.65145Z","iopub.status.busy":"2020-08-20T06:36:49.650589Z","iopub.status.idle":"2020-08-20T06:36:49.674802Z","shell.execute_reply":"2020-08-20T06:36:49.675322Z"},"papermill":{"duration":0.048156,"end_time":"2020-08-20T06:36:49.675432","exception":false,"start_time":"2020-08-20T06:36:49.627276","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"sub.head()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:36:49.716845Z","iopub.status.busy":"2020-08-20T06:36:49.716008Z","iopub.status.idle":"2020-08-20T06:36:49.755415Z","shell.execute_reply":"2020-08-20T06:36:49.756454Z"},"papermill":{"duration":0.064482,"end_time":"2020-08-20T06:36:49.756637","exception":false,"start_time":"2020-08-20T06:36:49.692155","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"# PREDICTION\nsub['FVC1'] = 1.*pe[:, 1]\nsub['Confidence1'] = pe[:, 2] - pe[:, 0]\nsubm = sub[['Patient_Week','FVC','Confidence','FVC1','Confidence1']].copy()\nsubm.loc[~subm.FVC1.isnull()].head(10)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:36:49.820627Z","iopub.status.busy":"2020-08-20T06:36:49.819777Z","iopub.status.idle":"2020-08-20T06:36:49.831547Z","shell.execute_reply":"2020-08-20T06:36:49.832397Z"},"papermill":{"duration":0.046049,"end_time":"2020-08-20T06:36:49.832552","exception":false,"start_time":"2020-08-20T06:36:49.786503","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"subm.loc[~subm.FVC1.isnull(),'FVC'] = subm.loc[~subm.FVC1.isnull(),'FVC1']\nif sigma_mean<70:\n    subm['Confidence'] = sigma_opt\nelse:\n    subm.loc[~subm.FVC1.isnull(),'Confidence'] = subm.loc[~subm.FVC1.isnull(),'Confidence1']","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:36:49.889485Z","iopub.status.busy":"2020-08-20T06:36:49.88856Z","iopub.status.idle":"2020-08-20T06:36:49.897657Z","shell.execute_reply":"2020-08-20T06:36:49.898415Z"},"papermill":{"duration":0.041018,"end_time":"2020-08-20T06:36:49.89858","exception":false,"start_time":"2020-08-20T06:36:49.857562","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"subm.head()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:36:49.963Z","iopub.status.busy":"2020-08-20T06:36:49.961824Z","iopub.status.idle":"2020-08-20T06:36:49.991863Z","shell.execute_reply":"2020-08-20T06:36:49.99262Z"},"papermill":{"duration":0.06133,"end_time":"2020-08-20T06:36:49.992837","exception":false,"start_time":"2020-08-20T06:36:49.931507","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"subm.describe().T","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:36:50.051224Z","iopub.status.busy":"2020-08-20T06:36:50.050446Z","iopub.status.idle":"2020-08-20T06:36:50.084894Z","shell.execute_reply":"2020-08-20T06:36:50.085901Z"},"papermill":{"duration":0.067525,"end_time":"2020-08-20T06:36:50.086069","exception":false,"start_time":"2020-08-20T06:36:50.018544","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"otest = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/test.csv')\nfor i in range(len(otest)):\n    subm.loc[subm['Patient_Week']==otest.Patient[i]+'_'+str(otest.Weeks[i]), 'FVC'] = otest.FVC[i]\n    subm.loc[subm['Patient_Week']==otest.Patient[i]+'_'+str(otest.Weeks[i]), 'Confidence'] = 0.1","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:36:50.145399Z","iopub.status.busy":"2020-08-20T06:36:50.139735Z","iopub.status.idle":"2020-08-20T06:36:50.148752Z","shell.execute_reply":"2020-08-20T06:36:50.14823Z"},"papermill":{"duration":0.038264,"end_time":"2020-08-20T06:36:50.148855","exception":false,"start_time":"2020-08-20T06:36:50.110591","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"subm[[\"Patient_Week\",\"FVC\",\"Confidence\"]].to_csv(\"submission_regression.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:36:50.188687Z","iopub.status.busy":"2020-08-20T06:36:50.188127Z","iopub.status.idle":"2020-08-20T06:36:50.191684Z","shell.execute_reply":"2020-08-20T06:36:50.192296Z"},"papermill":{"duration":0.026211,"end_time":"2020-08-20T06:36:50.192426","exception":false,"start_time":"2020-08-20T06:36:50.166215","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"reg_sub = subm[[\"Patient_Week\",\"FVC\",\"Confidence\"]].copy()","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.01702,"end_time":"2020-08-20T06:36:50.226675","exception":false,"start_time":"2020-08-20T06:36:50.209655","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Ensemble (Simple Blend)","execution_count":null},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:36:50.269666Z","iopub.status.busy":"2020-08-20T06:36:50.269097Z","iopub.status.idle":"2020-08-20T06:36:50.272688Z","shell.execute_reply":"2020-08-20T06:36:50.273622Z"},"papermill":{"duration":0.029615,"end_time":"2020-08-20T06:36:50.273751","exception":false,"start_time":"2020-08-20T06:36:50.244136","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"df1 = img_sub.sort_values(by=['Patient_Week'], ascending=True).reset_index(drop=True)\ndf2 = reg_sub.sort_values(by=['Patient_Week'], ascending=True).reset_index(drop=True)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:36:50.323083Z","iopub.status.busy":"2020-08-20T06:36:50.322393Z","iopub.status.idle":"2020-08-20T06:36:50.328376Z","shell.execute_reply":"2020-08-20T06:36:50.327527Z"},"papermill":{"duration":0.037476,"end_time":"2020-08-20T06:36:50.328469","exception":false,"start_time":"2020-08-20T06:36:50.290993","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"df = df1[['Patient_Week']].copy()\ndf['FVC'] = 0.25*df1['FVC'] + 0.75*df2['FVC']\ndf['Confidence'] = 0.26*df1['Confidence'] + 0.74*df2['Confidence']\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T06:36:50.368395Z","iopub.status.busy":"2020-08-20T06:36:50.367337Z","iopub.status.idle":"2020-08-20T06:36:50.37615Z","shell.execute_reply":"2020-08-20T06:36:50.375638Z"},"papermill":{"duration":0.030118,"end_time":"2020-08-20T06:36:50.376246","exception":false,"start_time":"2020-08-20T06:36:50.346128","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"df.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}