{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"from sklearn.model_selection import KFold, StratifiedKFold\nfrom sklearn.ensemble import GradientBoostingRegressor\n\nimport pandas as pd\nimport numpy as np\n\nimport matplotlib.pyplot as plt\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"VERBOSE=True\nSEED=2020\nFOLDS=5\nALPHA=0.8","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def metric(preds, confidence, targets):\n    confidence[confidence < 70] = 70\n    delta = np.abs(preds - targets)\n    delta[delta > 1000] = 1000\n    return -np.sqrt(2) * delta / confidence - np.log(np.sqrt(2) * confidence)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train_df = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/train.csv')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Creating Folds","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_df = pd.DataFrame()\nfor i, patient in train_df.groupby('Patient'):\n    patient_df = pd.concat([patient_df, patient])\n\npatient_df = patient_df[['Patient', 'Age', 'Sex', 'SmokingStatus']].drop_duplicates().reset_index(drop=True)\npatient_df['Sex'] = patient_df['Sex'].factorize()[0]\npatient_df['SmokingStatus'] = patient_df['SmokingStatus'].factorize()[0]\n\npatient_df['SS'] = patient_df.apply(lambda x: str(x['Sex']) + '-' + str(x['SmokingStatus']), axis=1).astype('category')\n\nkf = StratifiedKFold(n_splits=FOLDS, shuffle=True, random_state=SEED)\npatient_df['fold'] = 0\nfold = 0\nfor train_index, test_index in kf.split(patient_df[['Age', 'Sex', 'SmokingStatus']], patient_df['SS']):\n    patient_df['fold'].iloc[test_index] = fold\n    fold += 1","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Creating Training Data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train = train_df.merge(patient_df[['Patient', 'fold']], on='Patient')\n\n# Winsorize FVC\nall_fvc = train.FVC.mean()\ntrain.FVC = train.FVC.clip(all_fvc - 2 * train.FVC.std(), all_fvc + 2 * train.FVC.std())\n\n# Winsorize Age\nall_age = train.Age.mean()\ntrain.Age = train.Age.clip(all_age - 2 * train.Age.std(), all_age + 2 * train.Age.std())\n\n\noutput = pd.DataFrame()\nfor patient_id, patient in train.groupby('Patient'):\n    \n    usr_output = pd.DataFrame()\n    for week, tmp in patient.groupby('Weeks'):\n        rename_cols = {'Weeks': 'base_Week', 'FVC': 'base_FVC', 'Percent': 'base_Percent', 'Age': 'base_Age'}\n        tmp = tmp.rename(columns=rename_cols)\n        drop_cols = ['Age', 'Sex', 'SmokingStatus', 'Percent', 'fold']\n        _usr_output = patient.drop(columns=drop_cols).rename(columns={'Weeks': 'predict_Week'}).merge(tmp, on='Patient')\n        _usr_output['Week_passed'] = _usr_output['predict_Week'] - _usr_output['base_Week']\n        _usr_output['val'] = 0\n        _usr_output.tail(n=3)['val'] = 1\n        usr_output = pd.concat([usr_output, _usr_output])\n    output = pd.concat([output, usr_output])\n    \ntrain = output[output['Week_passed']!=0].reset_index(drop=True)\n\ntrain['Sex'] = train['Sex'].factorize()[0]\ntrain['SmokingStatus'] = train['SmokingStatus'].factorize()[0]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Creating Test Data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"test = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/test.csv')\\\n        .rename(columns={'Weeks': 'base_Week', 'FVC': 'base_FVC', 'Percent': 'base_Percent', 'Age': 'base_Age'})\nsubmission = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/sample_submission.csv')\nsubmission['Patient'] = submission['Patient_Week'].apply(lambda x: x.split('_')[0])\nsubmission['predict_Week'] = submission['Patient_Week'].apply(lambda x: x.split('_')[1]).astype(int)\ntest = submission.drop(columns=['FVC', 'Confidence']).merge(test, on='Patient')\ntest['Week_passed'] = test['predict_Week'] - test['base_Week']\n\ntest['Sex'] = test['Sex'].factorize()[0]\ntest['SmokingStatus'] = test['SmokingStatus'].factorize()[0]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Training and Predicting Using GradientBoostingRegression with Quantile Loss","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\n# Feature and Target columns that are used in Test as well\nx_cols = ['base_Age', 'Sex', 'SmokingStatus', 'base_Week', 'base_FVC', 'base_Percent', 'Week_passed']\ny_cols = ['FVC']\n\nfvc_up_prediction = np.zeros(len(train))\nfvc_md_prediction = np.zeros(len(train))\nfvc_lw_prediction = np.zeros(len(train))\n\ntst_up = []\ntst_md = []\ntst_lw = []\n\n# Training Cycle\nfor fold in range(FOLDS):\n    if VERBOSE: print(\"Fold: \", fold)\n    # Currently using only testing last 3 weeks in OOF\n    # Base FVC has the highest feature importance\n    # therefore not going to include earlier weeks in same fold\n    # as this seems to introduce a lot of leak. Makes sense\n    # since there's only so much the lungs can decrease by.\n    \n    #trn = train[(train['fold']!=fold)|(train['val']==0)]\n    val = train[(train['fold']==fold)&(train['val']==1)]\n\n    trn = train[(train['fold']!=fold)]\n    #val = train[(train['fold']==fold)]\n    \n    trn_y = trn[y_cols]\n    trn_x = trn[x_cols]\n\n    val_y = val[y_cols]\n    val_x = val[x_cols]\n    \n    # For Test predictions\n    tst_x = test[x_cols]\n\n    fvc_model = GradientBoostingRegressor(loss='quantile', \n                                          alpha=ALPHA,\n                                          n_estimators=250, \n                                          max_depth=3,\n                                          learning_rate=.05, \n                                          min_samples_leaf=9,\n                                          min_samples_split=9,\n                                          subsample=0.5,\n                                          random_state=SEED)\n    \n    # Fit on the upper bound\n    fvc_model.fit(trn_x, trn_y)\n    trn_upper = fvc_model.predict(trn_x)\n    fvc_upper = fvc_model.predict(val_x)\n    tst_upper = fvc_model.predict(tst_x)\n\n    # Fit on the lower bound\n    fvc_model.set_params(alpha=1.0 - ALPHA)\n    fvc_model.fit(trn_x, trn_y)\n    trn_lower = fvc_model.predict(trn_x)\n    fvc_lower = fvc_model.predict(val_x)\n    tst_lower = fvc_model.predict(tst_x)\n\n    # Get the median\n    fvc_model.set_params(loss='ls')\n    fvc_model.fit(trn_x, trn_y)\n    trn_pred = fvc_model.predict(trn_x)\n    fvc_pred = fvc_model.predict(val_x)\n    tst_pred = fvc_model.predict(tst_x)\n\n    # Set the OOF predictions\n    fvc_up_prediction[val.index] = fvc_upper\n    fvc_md_prediction[val.index] = fvc_pred\n    fvc_lw_prediction[val.index] = fvc_lower\n    \n    # Get the Fold prediction score\n    print(\"Train Score: \", np.mean(metric(trn_pred, trn_upper - trn_lower, trn_y.FVC.tolist())))\n    print(\"Fold Score: \", np.mean(metric(fvc_pred, fvc_upper - fvc_lower, val_y.FVC.tolist())))\n    print()\n    \n    # Save the Test predictions\n    tst_up.append(tst_upper)\n    tst_md.append(tst_pred)\n    tst_lw.append(tst_lower)\n    \n    if VERBOSE:\n        # Plot Feature Importance\n        feature_importance = fvc_model.feature_importances_\n        sorted_idx = np.argsort(feature_importance)\n        pos = np.arange(sorted_idx.shape[0]) + .5\n        fig = plt.figure(figsize=(12, 6))\n        plt.subplot(1, 2, 1)\n        plt.barh(pos, feature_importance[sorted_idx], align='center')\n        plt.yticks(pos, np.array(x_cols)[sorted_idx])\n        plt.title(f'Fold {fold}: Feature Importance')\n\n    \n# Get the mean over the Folds\ntst_up_predictions = np.mean(tst_up, axis=0)\ntst_md_predictions = np.mean(tst_md, axis=0)\ntst_lw_predictions = np.mean(tst_lw, axis=0)\n\n# OOF Score\nprint(\"=\" * 40)\n\nval_idx = train[train['val']==1].index\nprint(\"OOF Score: \", np.mean(metric(fvc_md_prediction[val_idx], fvc_up_prediction[val_idx] - fvc_lw_prediction[val_idx], train.iloc[val_idx]['FVC'].tolist())))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Submission","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"sub = test[['Patient_Week']].copy()\nsub['FVC'] = tst_md_predictions\nsub['Confidence'] = tst_up_predictions - tst_lw_predictions","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}