{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"markdown","source":"## <u>Introduction</u>\n* Data preprocessing based on this kernel => https://www.kaggle.com/yasufuminakama/osic-ridge-baseline/output.","execution_count":null},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport math \nimport random\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport category_encoders as ce\nimport torch\nimport scipy as sp\nimport torch.nn as nn\nimport torch.nn.functional as F\n\nfrom logging import getLogger, INFO, StreamHandler, FileHandler, Formatter\nfrom functools import partial\nfrom sklearn.model_selection import StratifiedKFold, GroupKFold, KFold\nfrom sklearn.metrics import mean_squared_error\nfrom tqdm import tqdm\nfrom torch.autograd import Variable\n\nimport warnings\nwarnings.filterwarnings('ignore')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# learning parameters\nepochs = 50\nlr = 0.001","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## <u>Utils</u>","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_logger(filename='log'):\n    logger = getLogger(__name__)\n    logger.setLevel(INFO)\n    handler1 = StreamHandler()\n    handler1.setFormatter(Formatter(\"%(message)s\"))\n    handler2 = FileHandler(filename=f\"{filename}.log\")\n    handler2.setFormatter(Formatter(\"%(message)s\"))\n    logger.addHandler(handler1)\n    logger.addHandler(handler2)\n    return logger\n\nlogger = get_logger()\n\ndef seed_everything(seed=777):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## <u>Config</u>","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"OUTPUT_DICT = './'\n\nID = 'Patient_Week'\nTARGET = 'FVC'\nSEED = 42\nseed_everything(seed=SEED)\n\nN_FOLD = 4","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## <u>Data Loading</u>","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/train.csv')\ntrain.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# add a new `Patient_Week` column\ntrain[ID] = train['Patient'].astype(str) + '_' + train['Weeks'].astype(str)\nprint(train.shape)\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# construct train input\noutput = pd.DataFrame()\ngb = train.groupby('Patient')\ntk0 = tqdm(gb, total=len(gb))\nfor _, usr_df in tk0:\n    usr_output = pd.DataFrame()\n    for week, tmp in usr_df.groupby('Weeks'):\n        rename_cols = {\n            'Weeks': 'base_Week', 'FVC': 'base_FVC', \n            'Percent': 'base_Percent', 'Age': 'base_Age'\n        }\n        tmp = tmp.drop(columns='Patient_Week').rename(columns=rename_cols)\n        drop_cols = [\n            'Age', 'Sex', 'SmokingStatus', 'Percent'\n        ]\n        _usr_output = usr_df.drop(columns=drop_cols).rename(columns={\n            'Weeks': 'predict_Week'\n        }).merge(tmp, on='Patient')\n        _usr_output['Week_passed'] = _usr_output['predict_Week'] - _usr_output['base_Week']\n        usr_output = pd.concat([usr_output, _usr_output])\n    output = pd.concat([output, usr_output])\n    \ntrain = output[output['Week_passed']!=0].reset_index(drop=True)\nprint(train.shape)\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# construct test output\ntest = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/test.csv')\\\n        .rename(columns={'Weeks': 'base_Week', 'FVC': 'base_FVC', 'Percent': 'base_Percent', 'Age': 'base_Age'})\ntest.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/sample_submission.csv')\nsubmission['Patient'] = submission['Patient_Week'].apply(\n    lambda x: x.split('_')[0])\nsubmission['predict_Week'] = submission['Patient_Week'].apply(\n    lambda x: x.split('_')[1]).astype(int)\ntest = submission.drop(columns=['FVC', 'Confidence']).merge(test, on='Patient')\ntest['Week_passed'] = test['predict_Week'] - test['base_Week']\nprint(test.shape)\ntest.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/sample_submission.csv')\nprint(submission.shape)\nsubmission.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## <u>Prepare Folds</u>","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"folds = train[[ID, 'Patient', TARGET]].copy()\n\nFold = GroupKFold(n_splits=N_FOLD)\ngroups = folds['Patient'].values\nfor n, (train_index, val_index) in enumerate(Fold.split(folds, folds[TARGET], groups)):\n    folds.loc[val_index, 'fold'] = int(n)\nfolds['fold'] = folds['fold'].astype(int)\nprint(folds.shape)\nfolds.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## <u>Model</u>\n","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"class LinearRegressionModel(nn.Module):\n    def __init__(self, input_dim, output_dim):\n        super(LinearRegressionModel, self).__init__()\n        \n        self.fc1 = nn.Linear(input_dim, 64)\n        self.fc2 = nn.Linear(64, 128)\n        self.fc3 = nn.Linear(128, 256)\n        self.fc4 = nn.Linear(256, 512)\n        self.fc5 = nn.Linear(512, output_dim)\n        \n    def forward(self, x):\n        x = F.relu(self.fc1(x))\n        x = F.relu(self.fc2(x))\n        x = F.relu(self.fc3(x))\n        x = F.relu(self.fc4(x))\n        out = self.fc5(x)\n        return out","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def return_data(train_df, test_df, features, target, folds, fold_num):\n    trn_idx = folds[folds.fold!=fold_num].index\n    val_idx = folds[folds.fold==fold_num].index\n    \n    y_train = target.iloc[trn_idx].values\n    x_train = train_df.iloc[trn_idx][features].values\n    y_val = target.iloc[val_idx].values\n    x_val = train_df.iloc[val_idx][features].values\n    \n    oof = np.zeros(len(train_df))\n    predictions = np.zeros(len(test_df))\n    \n    x_train = torch.tensor(x_train, dtype=torch.float)\n    y_train = torch.tensor(y_train, dtype=torch.float)\n    x_val = torch.tensor(x_val, dtype=torch.float)\n    y_val = torch.tensor(y_val, dtype=torch.float)\n    \n#     x_train = x_train.t()\n#     y_train = y_train.t()\n#     x_val = x_val.t()\n#     y_val = y_val.t()\n    \n    return x_train.cuda(), y_train.cuda(), x_val.cuda(), y_val.cuda(), val_idx\n\ndef run_single_linear_nn(train_df, test_df, folds, features, \n                     target, fold_num):\n    x_train, y_train, x_val, y_val, val_idx = return_data(train, test, features, \n                                                 target, folds, fold_num)\n\n    ###### PyTorch Model ##########\n    input_dim = x_train.shape[1] # num features\n    output_dim = 1\n    model = LinearRegressionModel(input_dim, output_dim).cuda()\n#     print(model)\n    loss = torch.nn.MSELoss()\n    optimizer = torch.optim.Adam(model.parameters(), lr=lr)\n    ###### PyTorch Model ##########\n    \n\n    for epoch in tqdm(range(epochs)):\n        optimizer.zero_grad()\n        y_predicted = model(x_train)\n        current_loss = torch.sqrt(loss(y_predicted, y_train))\n        current_loss.backward()\n        optimizer.step()\n    \n    oof = torch.zeros(len(train_df), 1).cuda()\n#     print('SHAPE', oof.shape)\n    predictions = np.zeros(len(test_df))\n    test_df = test_df[features].values\n    test_df = torch.tensor(test_df, dtype=torch.float).cuda()\n#     test_df = test_df.t()\n    \n    \n    with torch.no_grad():\n        oof_preds = model(x_val)\n        oof[val_idx] = model(x_val)\n        oof = oof.reshape(-1)\n        preds = model(test_df)\n        preds = preds.t().reshape(-1, 1)\n        preds = torch.flatten(preds)\n        preds = preds.cpu().numpy()\n#         print(preds.size())\n        predictions += preds\n    \n    logger.info(f\"fold {fold_num} score: {np.sqrt(mean_squared_error(target[val_idx], oof[val_idx].cpu())):<8.5f}\")\n    \n    return oof, predictions\n\ndef run_kfold_linear_nn(train, test, folds, features, target, n_fold=5):\n    oof = torch.zeros(len(train)).cuda()\n    predictions = torch.zeros(len(test))\n    feature_importance_df = pd.DataFrame()\n#     print('TARGET', target.shape)\n#     print('target type', type(target))\n#     print('OOF', oof.shape)\n#     print('oof type', type(oof))\n#     print('PREDICTIONS', predictions.shape)\n#     print('predictions type', type(predictions))\n    for fold_ in range(n_fold):\n        logger.info(f\"fold {fold_}\")\n        _oof, _predictions = run_single_linear_nn(train, test, folds, features, target,\n                                              fold_num=fold_\n        )\n#         print('_OOF', _oof.shape)\n#         print('_oof type', type(_oof))\n#         print('_PREDICTIONS', _predictions.shape)\n#         print('_predictions type', type(_predictions))\n        oof += _oof\n        predictions += _predictions / n_fold\n        \n    logger.info(f\"CV score: {np.sqrt(mean_squared_error(target, oof.cpu())):<8.5f}\")\n        \n    return oof, predictions","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## <u>Predict FVC</u>","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"target = train[TARGET]\ntest[TARGET] = np.nan\n\n# features\ncat_features = ['Sex', 'SmokingStatus']\nnum_features = [c for c in test.columns if (test.dtypes[c] != 'object')\n                & (c not in cat_features)]\nfeatures = num_features + cat_features\ndrop_features = [ID, TARGET, 'predict_Week', 'base_Week']\nfeatures = [c for c in features if c not in drop_features]\n\nif cat_features:\n    ce_oe = ce.OrdinalEncoder(cols=cat_features, handle_unknown='impute')\n    ce_oe.fit(train)\n    train = ce_oe.transform(train)\n    test = ce_oe.transform(test)\n    \n\noof, predictions = run_kfold_linear_nn(train, test, folds, features, \n                                   target, n_fold=N_FOLD)\n\noof = oof.cpu().numpy()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['FVC_pred'] = oof\ntest['FVC_pred'] = predictions","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## <u>Make Confidence Labels</u>","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# baseline score\ntrain['Confidence'] = 100\ntrain['sigma_clipped'] = train['Confidence'].apply(\n    lambda x: max(x, 70))\ntrain['diff'] = abs(train['FVC'] - train['FVC_pred'])\ntrain['delta'] = train['diff'].apply(lambda x: min(x, 1000))\ntrain['score'] = -math.sqrt(2) * train['delta']/train['sigma_clipped'] - \\\nnp.log(math.sqrt(2)*train['sigma_clipped'])\nscore = train['score'].mean()\nprint(score)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def loss_func(weight, row):\n    confidence = weight\n    sigma_clipped = max(confidence, 70)\n    diff = abs(row['FVC'] - row['FVC_pred'])\n    delta = min(diff, 1000)\n    score = -math.sqrt(2)*delta/sigma_clipped - np.log(math.sqrt(2)*sigma_clipped)\n    return -score\n\nresults = []\ntk0 = tqdm(train.iterrows(), total=len(train))\nfor _, row in tk0:\n    loss_partial = partial(loss_func, row=row)\n    weight = [100]\n    result = sp.optimize.minimize(loss_partial, weight, method='SLSQP')\n    x = result['x']\n    results.append(x[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# optimized score\ntrain['Confidence'] = results\ntrain['sigma_clipped'] = train['Confidence'].apply(lambda x: max(x, 70))\ntrain['diff'] = abs(train['FVC'] - train['FVC_pred'])\ntrain['delta'] = train['diff'].apply(lambda x: min(x, 1000))\ntrain['score'] = -math.sqrt(2)*train['delta']/train['sigma_clipped'] - np.log(math.sqrt(2)*train['sigma_clipped'])\nscore = train['score'].mean()\nprint(score)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head(10)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## <u>Predict Confidence</u>","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"TARGET = 'Confidence'\n\ntarget = train[TARGET]\ntest[TARGET] = np.nan\n\n# features\ncat_features = ['Sex', 'SmokingStatus']\nnum_features = [c for c in test.columns if (test.dtypes[c] != 'object') \n                & (c not in cat_features)]\nfeatures = num_features + cat_features\ndrop_features = [ID, TARGET, 'predict_Week', 'base_Week', 'FVC', 'FVC_pred']\nfeatures  = [c for c in features if c not in drop_features]\n\noof, predictions = run_kfold_linear_nn(train, test, folds, features, \n                                   target, n_fold=N_FOLD)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['Confidence'] = oof.cpu().numpy()\ntrain['sigma_clipped'] = train['Confidence'].apply(lambda x: max(x, 70))\ntrain['diff'] = abs(train['FVC'] - train['FVC_pred'])\ntrain['delta'] = train['diff'].apply(lambda x: min(x, 1000))\ntrain['score'] = -math.sqrt(2)*train['delta']/train['sigma_clipped'] - \\\nnp.log(math.sqrt(2)*train['sigma_clipped'])\nscore = train['score'].mean()\nprint(score)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def lb_metric(train):\n    train['sigma_clipped'] = train['Confidence'].apply(lambda x: max(x, 70))\n    train['diff'] = abs(train['FVC'] - train['FVC_pred'])\n    train['delta'] = train['diff'].apply(lambda x: min(x, 1000))\n    train['score'] = -math.sqrt(2)*train['delta']/train['sigma_clipped'] - \\\n    np.log(math.sqrt(2)*train['sigma_clipped'])\n    score = train['score'].mean()\n    return score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"score = lb_metric(train)\nlogger.info(f\"Local Score: {score}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test['Confidence'] = predictions","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## <u>Submission</u>","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub = submission.drop(columns=['FVC', 'Confidence']).merge(test[[\n    'Patient_Week', 'FVC_pred', 'Confidence']], on='Patient_Week')\nsub.columns = submission.columns\nsub.to_csv('submission.csv', index=False)\nsub.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}