{"cells":[{"metadata":{},"cell_type":"markdown","source":"# OMQR - refactored - mean slope\n\nRefactor from: https://www.kaggle.com/htopper/omqr-kaggle-sub-01\n\n---\n\n**Base code:**\n\nOOF: -6.797\n\nLB: -6.9154\n\n* Did something go wrong here?\n\n---\n\n**Base code + mean slope**\n\nOOF: -6.7885"},{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2020-09-14T12:56:42.947659Z","iopub.status.busy":"2020-09-14T12:56:42.946702Z","iopub.status.idle":"2020-09-14T12:56:44.416019Z","shell.execute_reply":"2020-09-14T12:56:44.412303Z"},"papermill":{"duration":1.515869,"end_time":"2020-09-14T12:56:44.416444","exception":false,"start_time":"2020-09-14T12:56:42.900575","status":"completed"},"scrolled":true,"tags":[],"trusted":false},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\npd.set_option('display.max_columns', None) # show all cols\nimport pydicom\nimport os\nimport random\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nfrom PIL import Image\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.model_selection import KFold\n\nimport tensorflow as tf\nimport tensorflow.keras.backend as K\nimport tensorflow.keras.layers as L\nimport tensorflow.keras.models as M","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"#env = 'local'\nenv = 'kaggle'\n\nBATCH_SIZE=128","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"if env == 'local':\n    ROOT = \"../input\"\nelse:\n    ROOT = \"../input/osic-pulmonary-fibrosis-progression\" # kaggle","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## load data"},{"metadata":{"trusted":false},"cell_type":"code","source":"train = pd.read_csv(f\"{ROOT}/train.csv\")\ntrain.drop_duplicates(keep=False, inplace=True, subset=['Patient','Weeks'])\n\ntest = pd.read_csv(f\"{ROOT}/test.csv\")\n\nsubmission = pd.read_csv(f\"{ROOT}/sample_submission.csv\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## seeds"},{"metadata":{"trusted":false},"cell_type":"code","source":"def seed_everything(seed=2020):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    if env == 'local':\n        tf.random.set_random_seed(seed) # local\n    else:\n        tf.random.set_seed(seed) # kaggle\n    \nseed_everything(42)","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.039883,"end_time":"2020-09-14T12:56:52.130957","exception":false,"start_time":"2020-09-14T12:56:52.091074","status":"completed"},"tags":[]},"cell_type":"markdown","source":"## eval metric"},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T12:56:52.22397Z","iopub.status.busy":"2020-09-14T12:56:52.223214Z","iopub.status.idle":"2020-09-14T12:56:52.227552Z","shell.execute_reply":"2020-09-14T12:56:52.226745Z"},"papermill":{"duration":0.052544,"end_time":"2020-09-14T12:56:52.227685","exception":false,"start_time":"2020-09-14T12:56:52.175141","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"## evaluation metric function\ndef laplace_log_likelihood(actual_fvc, predicted_fvc, confidence, return_values = False):\n    \"\"\"\n    Calculates the modified Laplace Log Likelihood score for this competition.\n    \"\"\"\n    sd_clipped = np.maximum(confidence, 70)\n    delta = np.minimum(np.abs(actual_fvc - predicted_fvc), 1000)\n    metric = - np.sqrt(2) * delta / sd_clipped - np.log(np.sqrt(2) * sd_clipped)\n\n    if return_values:\n        return metric\n    else:\n        return np.mean(metric)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Get all slopes and intercepts from train"},{"metadata":{"trusted":false},"cell_type":"code","source":"all_patient_ids = train['Patient'].unique()\n\npatient_slopes_df = pd.DataFrame()\npatient_slopes_df['Patient'] = all_patient_ids\n\nfor i in ['Percent','FVC']:\n    slopes = []\n    intercepts = []\n    for patient_id in all_patient_ids:\n        patient_df = train[train['Patient']==patient_id]\n        x = patient_df['Weeks'].to_numpy()\n        y = patient_df[i].to_numpy()\n\n        ## fit with polyfit\n        m, b = np.polyfit(x, y, 1)\n        slopes.append(m)\n        intercepts.append(b)\n\n    patient_slopes_df['Slope_'+i] = slopes\n    patient_slopes_df['Intercept_'+i] = intercepts\n    \nmean_slope_percent = patient_slopes_df['Slope_Percent'].mean()\nmean_slope_fvc = patient_slopes_df['Slope_FVC'].mean()\n\nprint('mean_slope_percent:',mean_slope_percent)\nprint('mean_slope_fvc:',mean_slope_fvc)\nprint('')\nprint(patient_slopes_df.shape)\npatient_slopes_df.head(3)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## put submission data in same format as train oof data\n* add data from test\n* alter FVC and Percent over the weeks with the aid of mean_slope\n\n**Add data from test**"},{"metadata":{"trusted":false},"cell_type":"code","source":"submission['Patient'] = submission['Patient_Week'].apply(lambda x:x.split('_')[0])\nsubmission['Weeks'] = submission['Patient_Week'].apply(lambda x: int(x.split('_')[-1]))\n\nsubmission =  submission[['Patient','Weeks','Confidence','Patient_Week']]\n\ntemp_test = test.copy()\n\ntemp_test.drop(columns=['Weeks'],inplace=True)\n\nsubmission = pd.merge(submission,temp_test, how='left', on=[\"Patient\"])\n\ndel temp_test\nsubmission.head(3)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**alter FVC and Percent**"},{"metadata":{"trusted":false},"cell_type":"code","source":"submission[submission['Patient']=='ID00419637202311204720264'].head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"temp_test = test.copy()\n\n# intercept_percent based on mean_slope_percent\ntemp_test['intercept_percent'] = temp_test['Percent'] - (temp_test['Weeks'] * mean_slope_percent)\n\nsubmission = pd.merge(submission,temp_test[['Patient','intercept_percent']], how='left', on=[\"Patient\"])\ndel temp_test\n\nsubmission['Percent_mean_slope'] =  submission['intercept_percent'] + (submission['Weeks'] * mean_slope_percent)\n\nsubmission['Percent'] = submission['Percent_mean_slope']\nsubmission.drop(columns=['intercept_percent','Percent_mean_slope'],inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"submission[submission['Patient']=='ID00419637202311204720264'].head()","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.043748,"end_time":"2020-09-14T12:56:53.438676","exception":false,"start_time":"2020-09-14T12:56:53.394928","status":"completed"},"tags":[]},"cell_type":"markdown","source":"## Lorem"},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T12:56:53.545406Z","iopub.status.busy":"2020-09-14T12:56:53.536035Z","iopub.status.idle":"2020-09-14T12:56:53.549693Z","shell.execute_reply":"2020-09-14T12:56:53.549043Z"},"papermill":{"duration":0.06736,"end_time":"2020-09-14T12:56:53.549835","exception":false,"start_time":"2020-09-14T12:56:53.482475","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"train['WHERE'] = 'train'\ntest['WHERE'] = 'test'\nsubmission['WHERE'] = 'submission'\n\ndata = train.append([test, submission])","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T12:56:53.64363Z","iopub.status.busy":"2020-09-14T12:56:53.64169Z","iopub.status.idle":"2020-09-14T12:56:53.6498Z","shell.execute_reply":"2020-09-14T12:56:53.650625Z"},"papermill":{"duration":0.058959,"end_time":"2020-09-14T12:56:53.650854","exception":false,"start_time":"2020-09-14T12:56:53.591895","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"print(train.shape, test.shape, submission.shape, data.shape)\nprint(train.Patient.nunique(), test.Patient.nunique(), submission.Patient.nunique(), data.Patient.nunique())\n\ndata.head(3)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T12:56:53.746231Z","iopub.status.busy":"2020-09-14T12:56:53.745436Z","iopub.status.idle":"2020-09-14T12:56:53.757692Z","shell.execute_reply":"2020-09-14T12:56:53.758434Z"},"papermill":{"duration":0.064445,"end_time":"2020-09-14T12:56:53.758614","exception":false,"start_time":"2020-09-14T12:56:53.694169","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"data['min_week'] = data['Weeks']\ndata.loc[data.WHERE=='submission','min_week'] = np.nan\ndata['min_week'] = data.groupby('Patient')['min_week'].transform('min')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T12:56:53.855865Z","iopub.status.busy":"2020-09-14T12:56:53.854849Z","iopub.status.idle":"2020-09-14T12:56:53.896839Z","shell.execute_reply":"2020-09-14T12:56:53.895983Z"},"papermill":{"duration":0.094542,"end_time":"2020-09-14T12:56:53.896985","exception":false,"start_time":"2020-09-14T12:56:53.802443","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"base = data.loc[data.Weeks == data.min_week]\nbase = base[['Patient','FVC']].copy()\nbase.columns = ['Patient','min_FVC']\nbase['nb'] = 1\nbase['nb'] = base.groupby('Patient')['nb'].transform('cumsum')\nbase = base[base.nb==1]\nbase.drop('nb', axis=1, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T12:56:53.997146Z","iopub.status.busy":"2020-09-14T12:56:53.996171Z","iopub.status.idle":"2020-09-14T12:56:54.010757Z","shell.execute_reply":"2020-09-14T12:56:54.011527Z"},"papermill":{"duration":0.070633,"end_time":"2020-09-14T12:56:54.011777","exception":false,"start_time":"2020-09-14T12:56:53.941144","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"data = data.merge(base, on='Patient', how='left')\ndata['base_week'] = data['Weeks'] - data['min_week']\ndel base","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T12:56:54.111923Z","iopub.status.busy":"2020-09-14T12:56:54.104651Z","iopub.status.idle":"2020-09-14T12:56:54.11704Z","shell.execute_reply":"2020-09-14T12:56:54.116057Z"},"papermill":{"duration":0.063435,"end_time":"2020-09-14T12:56:54.117217","exception":false,"start_time":"2020-09-14T12:56:54.053782","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"COLS = ['Sex','SmokingStatus'] #,'Age'\nFE = []\nfor col in COLS:\n    for mod in data[col].unique():\n        FE.append(mod)\n        data[mod] = (data[col] == mod).astype(int)\n#=================","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T12:56:54.225865Z","iopub.status.busy":"2020-09-14T12:56:54.224918Z","iopub.status.idle":"2020-09-14T12:56:54.22884Z","shell.execute_reply":"2020-09-14T12:56:54.22968Z"},"papermill":{"duration":0.066958,"end_time":"2020-09-14T12:56:54.229873","exception":false,"start_time":"2020-09-14T12:56:54.162915","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"#\ndata['age'] = (data['Age'] - data['Age'].min() ) / ( data['Age'].max() - data['Age'].min() )\ndata['BASE'] = (data['min_FVC'] - data['min_FVC'].min() ) / ( data['min_FVC'].max() - data['min_FVC'].min() )\ndata['week'] = (data['base_week'] - data['base_week'].min() ) / ( data['base_week'].max() - data['base_week'].min() )\n# WRONG? - does altered submission data give unrealistic percent range?\ndata['percent'] = (data['Percent'] - data['Percent'].min() ) / ( data['Percent'].max() - data['Percent'].min() )\nFE += ['age','percent','week','BASE']\n\n## for later use in variant of \"tr\"\ndata_percent_min = data['Percent'].min()\ndata_percent_max = data['Percent'].max()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T12:56:54.330446Z","iopub.status.busy":"2020-09-14T12:56:54.329523Z","iopub.status.idle":"2020-09-14T12:56:54.338974Z","shell.execute_reply":"2020-09-14T12:56:54.338306Z"},"papermill":{"duration":0.061998,"end_time":"2020-09-14T12:56:54.339139","exception":false,"start_time":"2020-09-14T12:56:54.277141","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"train = data.loc[data.WHERE=='train']\ntest = data.loc[data.WHERE=='test']\nsubmission = data.loc[data.WHERE=='submission']\ndel data","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T12:56:54.434424Z","iopub.status.busy":"2020-09-14T12:56:54.433452Z","iopub.status.idle":"2020-09-14T12:56:54.438592Z","shell.execute_reply":"2020-09-14T12:56:54.437705Z"},"papermill":{"duration":0.053837,"end_time":"2020-09-14T12:56:54.438733","exception":false,"start_time":"2020-09-14T12:56:54.384896","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"train.shape, test.shape, submission.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"submission.head()","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.045369,"end_time":"2020-09-14T12:56:54.528174","exception":false,"start_time":"2020-09-14T12:56:54.482805","status":"completed"},"tags":[]},"cell_type":"markdown","source":"## make a variant of \"train\" for OOF prediction\n\n* this variant hasnt got the \"percent leakage\"\n* this variant is more equal to the test set"},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T12:56:54.621137Z","iopub.status.busy":"2020-09-14T12:56:54.620245Z","iopub.status.idle":"2020-09-14T12:56:54.624555Z","shell.execute_reply":"2020-09-14T12:56:54.623743Z"},"papermill":{"duration":0.05293,"end_time":"2020-09-14T12:56:54.624691","exception":false,"start_time":"2020-09-14T12:56:54.571761","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"# \"weeks_0\" : percent on patient week 0 for all weeks\n# \"avg_slope\" : use globel avg slope to calc percent for all weeks\n# \"pred_slope\" : use prediction of slope to calc percent for all weeks\n\npercent_model = \"avg_slope\"","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T12:56:54.726584Z","iopub.status.busy":"2020-09-14T12:56:54.724273Z","iopub.status.idle":"2020-09-14T12:56:54.763022Z","shell.execute_reply":"2020-09-14T12:56:54.763676Z"},"papermill":{"duration":0.095397,"end_time":"2020-09-14T12:56:54.763856","exception":false,"start_time":"2020-09-14T12:56:54.668459","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"tr_adj = train[train['Weeks']>=0].groupby(['Patient'],as_index=False)['Weeks'].min()\ntr_adj = pd.merge(tr_adj,train,how='left',on=['Patient','Weeks'])\n# if multiple time FVC measured in week for patient, take the mean (Depends on notebook if this does anything)\ntr_adj = tr_adj.groupby(['Patient','Weeks'],as_index=False)[['Percent','FVC']].mean()\n\n# intercept_percent based on mean_slope_percent\ntr_adj['intercept_percent'] = tr_adj['Percent'] - (tr_adj['Weeks'] * mean_slope_percent)\n\n# rename before merge\ntr_adj.rename(columns={'Weeks': 'Weeks_0', 'Percent': 'Percent_0','FVC':'FVC_0'}, inplace=True)\ntr_adj.head()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T12:56:54.867638Z","iopub.status.busy":"2020-09-14T12:56:54.865019Z","iopub.status.idle":"2020-09-14T12:56:54.9168Z","shell.execute_reply":"2020-09-14T12:56:54.9174Z"},"papermill":{"duration":0.108144,"end_time":"2020-09-14T12:56:54.917587","exception":false,"start_time":"2020-09-14T12:56:54.809443","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"train_variant = train.copy()\ntrain_variant = pd.merge(train_variant,tr_adj,how='left',on=['Patient'])\n\ntrain_variant['Percent_mean_slope'] =  train_variant['intercept_percent'] + (train_variant['Weeks'] * mean_slope_percent)\n\n##\nif percent_model == \"weeks_0\":\n    train_variant['percent'] = (train_variant['Percent_0'] - data_percent_min) / (data_percent_max - data_percent_min)\n##\nif percent_model == \"avg_slope\":\n    train_variant['percent'] = (train_variant['Percent_mean_slope'] - data_percent_min) / (data_percent_max - data_percent_min)\n\n    \ntrain_variant.drop(columns=['Percent_mean_slope','intercept_percent'],inplace=True)\n\nprint(train_variant.shape)\ntrain_variant.head()","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.049833,"end_time":"2020-09-14T12:56:55.015194","exception":false,"start_time":"2020-09-14T12:56:54.965361","status":"completed"},"tags":[]},"cell_type":"markdown","source":"## BASELINE NN"},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T12:56:55.139648Z","iopub.status.busy":"2020-09-14T12:56:55.118464Z","iopub.status.idle":"2020-09-14T12:56:55.18524Z","shell.execute_reply":"2020-09-14T12:56:55.184296Z"},"papermill":{"duration":0.124117,"end_time":"2020-09-14T12:56:55.185378","exception":false,"start_time":"2020-09-14T12:56:55.061261","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"C1, C2 = tf.constant(70, dtype='float32'), tf.constant(1000, dtype=\"float32\")\n#=============================#\ndef score(y_true, y_pred):\n    tf.dtypes.cast(y_true, tf.float32)\n    tf.dtypes.cast(y_pred, tf.float32)\n    sigma = y_pred[:, 2] - y_pred[:, 0]\n    fvc_pred = y_pred[:, 1]\n    \n    #sigma_clip = sigma + C1\n    sigma_clip = tf.maximum(sigma, C1)\n    delta = tf.abs(y_true[:, 0] - fvc_pred)\n    delta = tf.minimum(delta, C2)\n    sq2 = tf.sqrt( tf.dtypes.cast(2, dtype=tf.float32) )\n    metric = (delta / sigma_clip)*sq2 + tf.math.log(sigma_clip* sq2)\n    return K.mean(metric)\n#============================#\ndef qloss(y_true, y_pred):\n    # Pinball loss for multiple quantiles\n    qs = [0.2, 0.50, 0.8]\n    q = tf.constant(np.array([qs]), dtype=tf.float32)\n    e = y_true - y_pred\n    v = tf.maximum(q*e, (q-1)*e)\n    return K.mean(v)\n#=============================#\ndef mloss(_lambda):\n    def loss(y_true, y_pred):\n        return _lambda * qloss(y_true, y_pred) + (1 - _lambda)*score(y_true, y_pred)\n    return loss\n#=================\ndef make_model(nh):\n    z = L.Input((nh,), name=\"Patient\")\n    x = L.Dense(100, activation=\"relu\", name=\"d1\")(z)\n    x = L.Dense(100, activation=\"relu\", name=\"d2\")(x)\n    #x = L.Dense(100, activation=\"relu\", name=\"d3\")(x)\n    p1 = L.Dense(3, activation=\"linear\", name=\"p1\")(x)\n    p2 = L.Dense(3, activation=\"relu\", name=\"p2\")(x)\n    preds = L.Lambda(lambda x: x[0] + tf.cumsum(x[1], axis=1), \n                     name=\"preds\")([p1, p2])\n    \n    model = M.Model(z, preds, name=\"CNN\")\n    #model.compile(loss=qloss, optimizer=\"adam\", metrics=[score])\n    model.compile(loss=mloss(0.8), optimizer=tf.keras.optimizers.Adam(lr=0.1, beta_1=0.9, beta_2=0.999, epsilon=None, decay=0.01, amsgrad=False), metrics=[score])\n    return model","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T12:56:55.29335Z","iopub.status.busy":"2020-09-14T12:56:55.281617Z","iopub.status.idle":"2020-09-14T12:56:55.297274Z","shell.execute_reply":"2020-09-14T12:56:55.29822Z"},"papermill":{"duration":0.068276,"end_time":"2020-09-14T12:56:55.298459","exception":false,"start_time":"2020-09-14T12:56:55.230183","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"y_train = train['FVC'].values.astype(np.float32)\nx_train = train[FE].values.astype(np.float32)\n\nx_train_variant = train_variant[FE].values.astype(np.float32) # variant of train data, for oof\n\nx_test = test[FE].values.astype(np.float32)\nx_submission = submission[FE].values.astype(np.float32)\n\npred_test = np.zeros((x_test.shape[0], 3))\npred_train = np.zeros((x_train.shape[0], 3))\npred_submission = np.zeros((x_submission.shape[0], 3))","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T12:56:55.539368Z","iopub.status.busy":"2020-09-14T12:56:55.538408Z","iopub.status.idle":"2020-09-14T12:56:55.673732Z","shell.execute_reply":"2020-09-14T12:56:55.673047Z"},"papermill":{"duration":0.190786,"end_time":"2020-09-14T12:56:55.673887","exception":false,"start_time":"2020-09-14T12:56:55.483101","status":"completed"},"scrolled":true,"tags":[],"trusted":false},"cell_type":"code","source":"num_inputs = x_train.shape[1]\n\nnet = make_model(num_inputs)\nprint(net.summary())\nprint(net.count_params())","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T12:56:55.782832Z","iopub.status.busy":"2020-09-14T12:56:55.78158Z","iopub.status.idle":"2020-09-14T12:56:55.785043Z","shell.execute_reply":"2020-09-14T12:56:55.783932Z"},"papermill":{"duration":0.063567,"end_time":"2020-09-14T12:56:55.785247","exception":false,"start_time":"2020-09-14T12:56:55.72168","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"NFOLD = 5\nrepeats = 10","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T12:56:55.903959Z","iopub.status.busy":"2020-09-14T12:56:55.899902Z","iopub.status.idle":"2020-09-14T13:12:17.288168Z","shell.execute_reply":"2020-09-14T13:12:17.286855Z"},"papermill":{"duration":921.450518,"end_time":"2020-09-14T13:12:17.288399","exception":false,"start_time":"2020-09-14T12:56:55.837881","status":"completed"},"scrolled":true,"tags":[],"trusted":false},"cell_type":"code","source":"%%time\n\n## oof\npreds_df = pd.DataFrame()\nconf_df = pd.DataFrame()\n## test\npreds_test_df = pd.DataFrame()\nconf_test_df = pd.DataFrame()\n## submisson\npreds_submission_df = pd.DataFrame()\nconf_submission_df = pd.DataFrame()\n\nfor random_state in range(repeats):\n    print('repeat:',random_state)\n    \n#     pe = np.zeros((ze.shape[0], 3))\n#     pred = np.zeros((z.shape[0], 3))\n    \n    pred_test = np.zeros((x_test.shape[0], 3))\n    pred_train = np.zeros((x_train.shape[0], 3))\n    pred_submission = np.zeros((x_submission.shape[0], 3))\n    \n    kf = KFold(n_splits=NFOLD,shuffle=True,random_state=random_state)\n    cnt = 0\n    EPOCHS = 800\n    for tr_idx, val_idx in kf.split(x_train):\n        cnt += 1\n        print(f\"FOLD {cnt}\")\n        net = make_model(num_inputs)\n        net.fit(x_train[tr_idx], y_train[tr_idx], batch_size=BATCH_SIZE, epochs=EPOCHS, \n                validation_data=(x_train[val_idx], y_train[val_idx]), verbose=0) #\n        print(\"train\", net.evaluate(x_train[tr_idx], y_train[tr_idx], verbose=0, batch_size=BATCH_SIZE))\n        print(\"val\", net.evaluate(x_train[val_idx], y_train[val_idx], verbose=0, batch_size=BATCH_SIZE))\n        print(\"predict train-val...\")\n        pred_train[val_idx] = net.predict(x_train_variant[val_idx], batch_size=BATCH_SIZE, verbose=0) # use variant for oof\n        print(\"predict test...\")\n        pred_test += net.predict(x_test, batch_size=BATCH_SIZE, verbose=0) / NFOLD\n        print(\"predict submission...\")\n        pred_submission += net.predict(x_submission, batch_size=BATCH_SIZE, verbose=0) / NFOLD\n    #==============\n\n\n    sigma_opt = mean_absolute_error(y_train, pred_train[:, 1])\n    unc = pred_train[:,2] - pred_train[:, 0]\n    sigma_mean = np.mean(unc)\n    print('----------------------------')\n    print(sigma_opt, sigma_mean)\n    \n    preds_df[random_state] = pred_train[:, 1]\n    conf_df[random_state] = unc\n    \n    preds_test_df[random_state] = pred_test[:, 1]\n    conf_test_df[random_state] = pred_test[:,2] - pred_test[:, 0]\n    \n    preds_submission_df[random_state] = pred_submission[:, 1]\n    conf_submission_df[random_state] = pred_submission[:,2] - pred_submission[:, 0]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## eval"},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T13:12:17.436508Z","iopub.status.busy":"2020-09-14T13:12:17.435282Z","iopub.status.idle":"2020-09-14T13:12:17.454481Z","shell.execute_reply":"2020-09-14T13:12:17.453642Z"},"papermill":{"duration":0.10123,"end_time":"2020-09-14T13:12:17.454631","exception":false,"start_time":"2020-09-14T13:12:17.353401","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"## oof\npreds_df['mean'] = preds_df.iloc[:,0:repeats].mean(axis=1)\nconf_df['mean'] = conf_df.iloc[:,0:repeats].mean(axis=1)\npreds_df['median'] = preds_df.iloc[:,0:repeats].median(axis=1)\nconf_df['median'] = conf_df.iloc[:,0:repeats].median(axis=1)\n## test\npreds_test_df['mean'] = preds_test_df.iloc[:,0:repeats].mean(axis=1)\nconf_test_df['mean'] = conf_test_df.iloc[:,0:repeats].mean(axis=1)\npreds_test_df['median'] = preds_test_df.iloc[:,0:repeats].median(axis=1)\nconf_test_df['median'] = conf_test_df.iloc[:,0:repeats].median(axis=1)\n## submission\npreds_submission_df['mean'] = preds_submission_df.iloc[:,0:repeats].mean(axis=1)\nconf_submission_df['mean'] = conf_submission_df.iloc[:,0:repeats].mean(axis=1)\npreds_submission_df['median'] = preds_submission_df.iloc[:,0:repeats].median(axis=1)\nconf_submission_df['median'] = conf_submission_df.iloc[:,0:repeats].median(axis=1)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T13:12:17.59134Z","iopub.status.busy":"2020-09-14T13:12:17.58788Z","iopub.status.idle":"2020-09-14T13:12:17.60708Z","shell.execute_reply":"2020-09-14T13:12:17.60804Z"},"papermill":{"duration":0.091428,"end_time":"2020-09-14T13:12:17.60825","exception":false,"start_time":"2020-09-14T13:12:17.516822","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"scores = []\nfor i in range(repeats):\n    score = laplace_log_likelihood(y_train, preds_df[i], conf_df[i])\n    print('solution',i,':',score)\n    scores.append(score)\nprint('----------------')\nprint('mean score:',np.mean(scores))","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T13:12:17.746399Z","iopub.status.busy":"2020-09-14T13:12:17.745247Z","iopub.status.idle":"2020-09-14T13:12:17.753905Z","shell.execute_reply":"2020-09-14T13:12:17.752873Z"},"papermill":{"duration":0.081714,"end_time":"2020-09-14T13:12:17.754211","exception":false,"start_time":"2020-09-14T13:12:17.672497","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"print('mean OOF lap_log:',laplace_log_likelihood(y_train, preds_df['mean'], conf_df['mean']))\nprint('median OOF lap_log:',laplace_log_likelihood(y_train, preds_df['median'], conf_df['median']))","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.063115,"end_time":"2020-09-14T13:12:17.881901","exception":false,"start_time":"2020-09-14T13:12:17.818786","status":"completed"},"tags":[]},"cell_type":"markdown","source":"**remove patient weeks 0 and check again (given values removed)**"},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T13:12:18.019026Z","iopub.status.busy":"2020-09-14T13:12:18.018187Z","iopub.status.idle":"2020-09-14T13:12:18.032528Z","shell.execute_reply":"2020-09-14T13:12:18.03324Z"},"papermill":{"duration":0.08989,"end_time":"2020-09-14T13:12:18.033416","exception":false,"start_time":"2020-09-14T13:12:17.943526","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"temp = train_variant.copy()\n\ntemp['prediction'] = preds_df['mean']\ntemp['confidence'] = conf_df['mean']\n\n## remove given rows before calc score\n# remove rows with FVC \"given\"\ntemp['for_eval'] = temp['Weeks'] != temp['Weeks_0']\ntemp = temp[temp['for_eval']==True]\n\nprint('mean OOF lap_log:',laplace_log_likelihood(temp['FVC'], temp['prediction'], temp['confidence']))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### w0 delta adjustment on OOF"},{"metadata":{"trusted":false},"cell_type":"code","source":"p_oof = train_variant.copy()\np_oof['prediction'] = preds_df['mean']\np_oof['confidence'] = conf_df['mean']\n\np_oof.head(3)","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"trusted":false},"cell_type":"code","source":"train_adj = p_oof[p_oof['Weeks']>=0].groupby(['Patient'],as_index=False)['Weeks'].min()\ntrain_adj = pd.merge(train_adj,p_oof,how='left',on=['Patient','Weeks'])\n\n# if multiple time FVC measured in week for patient, take the mean\ntrain_adj = train_adj.groupby(['Patient','Weeks'],as_index=False)[['Percent','FVC']].mean()\n\ntrain_adj.rename(columns={'Weeks': 'First_week'}, inplace=True) \n\ntrain_adj.head(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"p_oof = pd.merge(p_oof,train_adj[['Patient','First_week']],how='left',on=['Patient'])\n\np_oof_week_0 = p_oof[p_oof['Weeks']==p_oof['First_week']][['Patient','FVC','prediction']].copy()\np_oof_week_0.reset_index(inplace=True, drop=True)\np_oof_week_0['prediction_w0_delta'] = p_oof_week_0['FVC'] - p_oof_week_0['prediction']\np_oof_week_0.head(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"## add week 0 delta data\np_oof = pd.merge(p_oof,p_oof_week_0[['Patient','prediction_w0_delta']],how='left',on=['Patient'])\n## adjust prediction\np_oof['predition_adjusted'] =  p_oof['prediction'] + p_oof['prediction_w0_delta']","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"temp = p_oof.copy()\n\n# remove rows with FVC \"given\"\ntemp['for_eval'] = (temp['Weeks'] != temp['First_week'])\ntemp = temp[temp['for_eval']==True] \n\nprint('mean OOF lap_log:',laplace_log_likelihood(temp['FVC'], temp['predition_adjusted'], temp['confidence']))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## save OOF if you want to"},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T13:12:19.115394Z","iopub.status.busy":"2020-09-14T13:12:19.114235Z","iopub.status.idle":"2020-09-14T13:12:19.117495Z","shell.execute_reply":"2020-09-14T13:12:19.118306Z"},"papermill":{"duration":0.080279,"end_time":"2020-09-14T13:12:19.1185","exception":false,"start_time":"2020-09-14T13:12:19.038221","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"save_oof = False\n\nif save_oof:\n    check_me = train.copy()\n    check_me['prediction'] = preds_df['mean']\n    check_me['confidence'] = conf_df['mean']\n\n    check_me.head(3)\n    check_me.to_csv(\"oof_name_here.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.069807,"end_time":"2020-09-14T13:12:20.150897","exception":false,"start_time":"2020-09-14T13:12:20.08109","status":"completed"},"tags":[]},"cell_type":"markdown","source":"## PREDICTION"},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T13:12:20.31949Z","iopub.status.busy":"2020-09-14T13:12:20.312234Z","iopub.status.idle":"2020-09-14T13:12:20.338248Z","shell.execute_reply":"2020-09-14T13:12:20.33891Z"},"papermill":{"duration":0.114883,"end_time":"2020-09-14T13:12:20.339107","exception":false,"start_time":"2020-09-14T13:12:20.224224","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"print(submission.shape)\nsubmission.head()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T13:12:20.503772Z","iopub.status.busy":"2020-09-14T13:12:20.50255Z","iopub.status.idle":"2020-09-14T13:12:20.508052Z","shell.execute_reply":"2020-09-14T13:12:20.507241Z"},"papermill":{"duration":0.09856,"end_time":"2020-09-14T13:12:20.508204","exception":false,"start_time":"2020-09-14T13:12:20.409644","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"my_sub = submission.copy()\nmy_sub.reset_index(inplace=True, drop=True)\nmy_sub = my_sub[['Patient','Weeks']]\nmy_sub['FVC'] = preds_submission_df['mean'].values\nmy_sub['Confidence'] = conf_submission_df['mean'].values\n\nmy_sub.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## w0 delta adjustment on submission"},{"metadata":{"trusted":false},"cell_type":"code","source":"# test_temp = test[['Patient','Weeks','FVC']].copy()\n# test_temp.reset_index(inplace=True,drop=True)\n# test_temp.rename(columns={'FVC':'W0_FVC'}, inplace=True)\n\n# print(test_temp.shape)\n# test_temp.head(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"# test_temp = pd.merge(test_temp,my_sub,on=['Patient','Weeks'],how='left')\n# test_temp['FVC_delta'] = test_temp['W0_FVC'] - test_temp['FVC']\n\n# print(test_temp.shape)\n# test_temp.head(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"# my_sub = pd.merge(my_sub,test_temp[['Patient','FVC_delta']],on=['Patient'],how='left')\n# my_sub['FVC_adjusted'] = my_sub['FVC'] + my_sub['FVC_delta']\n\n# print(my_sub.shape)\n# my_sub.head(10)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T13:12:20.67656Z","iopub.status.busy":"2020-09-14T13:12:20.675154Z","iopub.status.idle":"2020-09-14T13:12:21.73663Z","shell.execute_reply":"2020-09-14T13:12:21.735802Z"},"papermill":{"duration":1.152878,"end_time":"2020-09-14T13:12:21.736772","exception":false,"start_time":"2020-09-14T13:12:20.583894","status":"completed"},"scrolled":true,"tags":[],"trusted":false},"cell_type":"code","source":"show_test_predictions = False\n\nif show_test_predictions:\n    all_patients = my_sub['Patient'].unique()\n    print('num unique patients:',len(all_patients))\n\n    for patient in all_patients[0:5]:\n        temp = my_sub[my_sub['Patient']==patient].copy()\n        print('patient:',patient)\n        plt.plot(temp['Weeks'], temp['FVC_adjusted'], '-',label='prediction',color='purple')\n        plt.axvline(x=0,color='gray',ls='--') # visual aid\n        plt.legend()\n        plt.show();\n        print('--')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T13:12:21.923682Z","iopub.status.busy":"2020-09-14T13:12:21.922587Z","iopub.status.idle":"2020-09-14T13:12:21.931477Z","shell.execute_reply":"2020-09-14T13:12:21.930014Z"},"papermill":{"duration":0.111644,"end_time":"2020-09-14T13:12:21.931731","exception":false,"start_time":"2020-09-14T13:12:21.820087","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"make_submission = True\nif make_submission:\n    #my_sub['FVC'] = my_sub['FVC_adjusted']\n    \n    my_sub['Weeks'] = my_sub['Weeks'].astype(str)\n    my_sub['Patient_Week'] = my_sub['Patient'] + '_' + my_sub['Weeks']\n    my_sub = my_sub[['Patient_Week','FVC','Confidence']]\n    \n    print(my_sub.shape)\n    my_sub.head()\n\n    my_sub.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-09-14T13:12:22.110668Z","iopub.status.busy":"2020-09-14T13:12:22.109832Z","iopub.status.idle":"2020-09-14T13:12:22.47288Z","shell.execute_reply":"2020-09-14T13:12:22.471039Z"},"papermill":{"duration":0.454981,"end_time":"2020-09-14T13:12:22.473173","exception":false,"start_time":"2020-09-14T13:12:22.018192","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"# print(my_sub.shape)\n# my_sub.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}