{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install fastai2 -q","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\n\nfrom sklearn.model_selection import KFold\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.metrics import mean_absolute_error\n\nimport plotly.express as px\nimport plotly.graph_objs as go\nimport plotly.figure_factory as ff\n\nimport missingno as msno\nimport tensorflow.keras.backend as K\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, Dense, Lambda\n\nimport matplotlib.pyplot as plt\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from fastai2.basics import *\nfrom fastai2.callback.all import *\nfrom fastai2.vision.all import *\nfrom fastai2.medical.imaging import *\n\nimport pydicom","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/train.csv')\ntest_df = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Missing values**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.subplot(121)\nmsno.bar(train_df)\nplt.title(\"Missing values in Training Data\", fontsize = 20, color = 'Red')\n\nplt.subplot(122)\nmsno.bar(test_df)\nplt.title(\"Missing values in Test Data\", fontsize = 20, color = 'Blue')\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Thus there are no missing values in either Training or Test data**","execution_count":null},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"print(f'Total unique patients are {train_df.Patient.nunique()} out of total {len(train_df.Patient)} patients')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**EDA**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"go.Figure(go.Pie(labels = train_df.Sex.value_counts().keys().tolist(), \n                       values = train_df.Sex.value_counts().values.tolist(), \n                       marker = dict(colors=['red']), hoverinfo = \"value\", pull=[0, 0.1]), \n          layout = go.Layout(title = {'text':\"Gender Distribution\", 'x':0.5}, font=dict(family=\"Courier New, monospace\",\n                                                                                                size=18,\n                                                                                                color=\"RebeccaPurple\")))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"go.Figure(go.Pie(labels = train_df.SmokingStatus.value_counts().keys().tolist(), \n                       values = train_df.SmokingStatus.value_counts().values.tolist(), \n                       marker = dict(colors=['pink', 'blue', 'purple']), hoverinfo = \"value\", hole = 0.3), \n          layout = go.Layout(title = {'text':\"Smoking Status\", 'x':0.425}, font=dict(family=\"Courier New, monospace\",\n                                                                                                size=18,\n                                                                                                color=\"RebeccaPurple\")))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.groupby('Sex')['SmokingStatus'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = go.Figure(data=[\n    go.Bar(name='Smoker', x= train_df.Sex.unique(), y = [train_df.groupby('Sex')['SmokingStatus'].value_counts().values[5],\n                                                         train_df.groupby('Sex')['SmokingStatus'].value_counts().values[2]]),\n    go.Bar(name='Non-Smoker', x= train_df.Sex.unique(), y = [train_df.groupby('Sex')['SmokingStatus'].value_counts().values[4],\n                                                             train_df.groupby('Sex')['SmokingStatus'].value_counts().values[0]]),\n    go.Bar(name='Ex-Smoker', x= train_df.Sex.unique(), y = [train_df.groupby('Sex')['SmokingStatus'].value_counts().values[3],\n                                                             train_df.groupby('Sex')['SmokingStatus'].value_counts().values[1]]),\n])\nfig.update_layout(title = {'text':\"Smoking Distribution by Sex\", 'x':0.5}, \n                  font = dict(family=\"Courier New, monospace\", size=18, color=\"RebeccaPurple\"),\n                  barmode ='group')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"count_df = pd.DataFrame(train_df['Patient'].value_counts())\ncount_df = count_df.reset_index()\ncount_df.rename(columns = {'index':'Patient ID', 'Patient':'No of Images'}, inplace = True)\n\nfig = px.bar(count_df, x='Patient ID',y ='No of Images',color='No of Images')\nfig.update_xaxes(showticklabels=False)\nfig.update_layout(title = {'text':\"Distribution of Images per Patient\", 'x':0.5}, \n                  font = dict(family=\"Courier New, monospace\", size=18, color=\"RebeccaPurple\"))\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Let's quickly check DICOM files also","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"dicom_ids = os.listdir('../input/osic-pulmonary-fibrosis-progression/train/')\npatient_sizes = [len(os.listdir('../input/osic-pulmonary-fibrosis-progression/train/' + d)) for d in dicom_ids]\ndicom_df = pd.DataFrame({'Dicom_ID':dicom_ids, 'Dicom Files':patient_sizes})\n\nfig = px.bar(dicom_df, x='Dicom_ID',y ='Dicom Files',color='Dicom Files')\nfig.update_xaxes(showticklabels=False)\nfig.update_layout(title = {'text':\"Distribution of Dicom Files per Dicom ID\", 'x':0.5}, \n                  font = dict(family=\"Courier New, monospace\", size=18, color=\"RebeccaPurple\"))\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Most of the DICOMs have less than 200 files","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = ff.create_distplot([train_df.Age.values], ['Age'], colors = ['red'])\nfig.update_layout(title = {'text':\"Age Distribution\", 'x':0.5}, \n                  font = dict(family=\"Courier New, monospace\", size=18, color=\"RebeccaPurple\"))\nfig.show()\n\nfig = ff.create_distplot([train_df.Weeks.values], ['Weeks'], colors = ['blue'])\nfig.update_layout(title = {'text':\"Weeks Distribution\", 'x':0.5}, \n                  font = dict(family=\"Courier New, monospace\", size=18, color=\"RebeccaPurple\"))\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = ff.create_distplot([train_df.Percent.values], ['Percent'], colors = ['purple'])\nfig.update_layout(title = {'text':\"Percent Distribution\", 'x':0.5}, \n                  font = dict(family=\"Courier New, monospace\", size=18, color=\"RebeccaPurple\"))\nfig.show()\n\nfig = ff.create_distplot([train_df.FVC.values], ['FVC'], colors = ['green'])\nfig.update_layout(title = {'text':\"FVC Distribution\", 'x':0.5}, \n                  font = dict(family=\"Courier New, monospace\", size=18, color=\"RebeccaPurple\"))\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.histogram(train_df, x='Age', color='SmokingStatus', marginal=\"box\", \n                   color_discrete_map={'Ex-smoker':'green','Never smoked':'light green','Currently smokes':'orange'})\nfig.update_traces(marker_line_color='cyan',marker_line_width=1, opacity=0.8)\nfig.update_layout(title = {'text':\"Smoking Status by Age\", 'x':0.4}, \n                  font = dict(family=\"Courier New, monospace\", size=18, color=\"RebeccaPurple\"))\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.histogram(train_df, x='Age', color='Sex',marginal=\"box\", color_discrete_map={'Male':'blue','Female':'light green'})\nfig.update_traces(marker_line_color='cyan',marker_line_width=1, opacity=0.8)\nfig.update_layout(title = {'text':\"Sex Distribution by Age\", 'x':0.45}, \n                  font = dict(family=\"Courier New, monospace\", size=18, color=\"RebeccaPurple\"))\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.histogram(train_df, x='FVC', color='Sex', marginal=\"rug\", \n                   color_discrete_map={'Male':'DarkKhaki','Female':'MediumSpringGreen'})\nfig.update_traces(marker_line_color='LightSlateGrey',marker_line_width=1, opacity=0.8)\nfig.update_layout(title = {'text':\"Gender Distribution in FVC\", 'x':0.45}, \n                  font = dict(family=\"Courier New, monospace\", size=18, color=\"RebeccaPurple\"))\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.histogram(train_df, x='FVC', color='SmokingStatus', marginal=\"box\",\n                   color_discrete_map={'Ex-smoker':'#393E46','Never smoked':'MediumTurquoise','Currently smokes':'Linen'})\nfig.update_traces(marker_line_color = 'black',marker_line_width = 1, opacity = 0.8)\nfig.update_layout(title = {'text':\"SmokingStatus Distribution in FVC\", 'x':0.45}, \n                  font = dict(family=\"Courier New, monospace\", size=18, color=\"RebeccaPurple\"))\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Inspect DICOM files with FastAI**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"source_path = Path('../input/osic-pulmonary-fibrosis-progression')\nsource_files = os.listdir(source_path)\nprint(source_files)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_path = source_path/'train'\ntrain_files = get_dicom_files(train_path)\ntrain_files","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dicom_img = dcmread(train_files[0])\ndicom_img","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dicom_img.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tensor_dicom = pixels(dicom_img) #convert into tensor\n\nprint(f'RescaleIntercept: {dicom_img.RescaleIntercept:1f}\\nRescaleSlope: {dicom_img.RescaleSlope:1f}\\nMax pixel: '\n      f'{tensor_dicom.max()}\\nMin pixel: {tensor_dicom.min()}\\nShape: {tensor_dicom.shape}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tensor_dicom_scaled = scaled_px(dicom_img)\nplt.hist(tensor_dicom_scaled.flatten(), color='c')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Largest bin value is -1000 which represent Air on Hounsfield Scale(https://en.wikipedia.org/wiki/Hounsfield_scale)","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Viewing Cancellous Bone Area\ndicom_img.show(min_px = 300, max_px = 400, figsize=(10, 10))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Fat Area\ndicom_img.show(min_px = -120, max_px = -90, figsize=(10, 10))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Water Based Area\ndicom_img.show(max_px=None, min_px=0, figsize=(10, 10))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Air Based Area\ndicom_img.show(max_px=None, min_px=-1000, figsize=(10, 10))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Data Inspection**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.shape","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Check for duplicates patient records**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df[train_df.duplicated(subset = ['Patient','Weeks'])]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.drop_duplicates(keep=False, inplace = True, subset=['Patient','Weeks'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_df = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"temp_sub_df = submission_df['Patient_Week'].str.split('_', expand = True)\ntemp_sub_df.rename(columns = {0: 'Patient', 1: 'Weeks'}, inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_df = pd.concat([submission_df, temp_sub_df], axis = 1)\nsubmission_df = submission_df[['Patient','Weeks','Confidence','Patient_Week']]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_df = submission_df.merge(test_df.drop('Weeks', axis = 1), on = 'Patient')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['data_type'] = 'Train'\ntest_df['data_type'] = 'Val'\nsubmission_df['data_type'] = 'Test'\ncombined_df = train_df.append([test_df, submission_df])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_type = ['Train', 'Val', 'Test']\nfor type in data_type:\n    data = combined_df.query(\"data_type == @type\")\n    print(type, \"shape in combined data is \", data.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Minimum Week for each patient\ncombined_df['Min_Weeks'] = combined_df['Weeks']\ncombined_df.loc[combined_df.data_type == 'Test','Min_Weeks'] = np.nan\ncombined_df['Min_Weeks'] = combined_df.groupby('Patient')['Min_Weeks'].transform('min')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"base = combined_df.loc[combined_df.Weeks == combined_df.Min_Weeks]\nbase = base[['Patient','FVC']].rename(columns = {'FVC':'min_FVC'})\nbase.drop_duplicates(keep = 'first', inplace = True, subset = ['Patient'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"combined_df.Weeks = combined_df.Weeks.astype(int)\ncombined_df.Min_Weeks = combined_df.Min_Weeks.astype(float)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"combined_df = combined_df.merge(base, on='Patient', how='left')\ncombined_df['Deviation_Weeks'] = combined_df['Weeks'] - combined_df['Min_Weeks']\ndel base","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"combined_df = pd.concat([combined_df, pd.get_dummies(combined_df[['Sex','SmokingStatus']])], axis = 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"scaler = MinMaxScaler()\nscaled = pd.DataFrame(scaler.fit_transform(combined_df[['Age','Percent','min_FVC','Deviation_Weeks']]), \n                      columns = ['scaled_Age', 'scaled_Percent', 'scaled_FVC', 'scaled_Deviation_Weeks'])\ncombined_df = pd.concat([combined_df, scaled], axis = 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"combined_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"feature_columns = ['Sex_Male','Sex_Female','SmokingStatus_Ex-smoker','SmokingStatus_Never smoked','SmokingStatus_Currently smokes',\n                   'scaled_Age','scaled_Percent','scaled_Deviation_Weeks','scaled_FVC']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = combined_df.loc[combined_df.data_type == 'Train']\ntest_df = combined_df.loc[combined_df.data_type == 'Val']\nsubmission_df = combined_df.loc[combined_df.data_type == 'Test']\ndel combined_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.shape, test_df.shape, submission_df.shape","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Model**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"C1, C2 = tf.constant(70, dtype='float32'), tf.constant(1000, dtype=\"float32\")\n\ndef score(y_true, y_pred):\n    tf.dtypes.cast(y_true, tf.float32)\n    tf.dtypes.cast(y_pred, tf.float32)\n    sigma = y_pred[:, 2] - y_pred[:, 0]\n    fvc_pred = y_pred[:, 1]\n    \n    sigma_clip = tf.maximum(sigma, C1)\n    delta = tf.abs(y_true[:, 0] - fvc_pred)\n    delta = tf.minimum(delta, C2)\n    sq2 = tf.sqrt( tf.dtypes.cast(2, dtype=tf.float32) )\n    metric = (delta / sigma_clip)*sq2 + tf.math.log(sigma_clip* sq2)\n    return K.mean(metric)\n\ndef qloss(y_true, y_pred):\n    qs = [0.2, 0.50, 0.8]\n    q = tf.constant(np.array([qs]), dtype=tf.float32)\n    e = y_true - y_pred\n    v = tf.maximum(q*e, (q-1)*e)\n    return K.mean(v)\n\ndef mloss(_lambda):\n    def loss(y_true, y_pred):\n        return _lambda * qloss(y_true, y_pred) + (1 - _lambda)*score(y_true, y_pred)\n    return loss\n\ndef make_model():\n    x1 = Input((9,), name=\"Patient\")\n    x2 = Dense(100, activation=\"relu\", name=\"d1\")(x1)\n    x3 = Dense(100, activation=\"relu\", name=\"d2\")(x2)\n    \n    p1 = Dense(3, activation=\"relu\", name=\"p1\")(x3)\n    p2 = Dense(3, activation=\"relu\", name=\"p2\")(x3)\n    \n    preds = Lambda(lambda x3: x3[0] + tf.cumsum(x3[1], axis=1), \n                     name=\"preds\")([p1, p2])\n    \n    model = Model(x1, preds, name=\"CNN\")\n   \n    model.compile(loss = mloss(0.8), optimizer = tf.keras.optimizers.Adam(lr=0.1, beta_1=0.9, beta_2=0.999, epsilon=None, decay=0.005,\n                                                                          amsgrad=False), metrics=[score])\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = make_model()\nprint(model.summary())\nprint(model.count_params())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y = train_df['FVC'].values\nz = train_df[feature_columns].values\nsub = submission_df[feature_columns].values\npe = np.zeros((sub.shape[0], 3))\npred = np.zeros((z.shape[0], 3))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"NFOLD = 5\nBATCH_SIZE=128\nkf = KFold(n_splits=NFOLD)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\ncnt = 0\nfor tr_idx, val_idx in kf.split(z):\n    cnt += 1\n    print(f\"FOLD {cnt}\")\n    model = make_model()\n    model.fit(z[tr_idx], y[tr_idx], batch_size=BATCH_SIZE, epochs=800, \n            validation_data=(z[val_idx], y[val_idx]), verbose=0) #\n    print(\"train\", model.evaluate(z[tr_idx], y[tr_idx], verbose=0, batch_size=BATCH_SIZE))\n    print(\"val\", model.evaluate(z[val_idx], y[val_idx], verbose=0, batch_size=BATCH_SIZE))\n    print(\"predict val...\")\n    pred[val_idx] = model.predict(z[val_idx], batch_size=BATCH_SIZE, verbose=0)\n    print(\"predict test...\")\n    pe += model.predict(sub, batch_size=BATCH_SIZE, verbose=0) / NFOLD","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sigma_opt = mean_absolute_error(y, pred[:, 1])\nunc = pred[:,2] - pred[:, 0]\nsigma_mean = np.mean(unc)\nprint(sigma_opt, sigma_mean)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"idxs = np.random.randint(0, y.shape[0], 100)\nplt.plot(y[idxs], label=\"ground truth\")\nplt.plot(pred[idxs, 0], label=\"q25\")\nplt.plot(pred[idxs, 1], label=\"q50\")\nplt.plot(pred[idxs, 2], label=\"q75\")\nplt.legend(loc=\"best\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(unc.min(), unc.mean(), unc.max(), (unc>=0).mean())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.hist(unc)\nplt.title(\"uncertainty in prediction\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pe[:, 1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_df['FVC1'] = pe[:, 1]\nsubmission_df['Confidence1'] = pe[:, 2] - pe[:, 0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"subm = submission_df[['Patient_Week','FVC','Confidence','FVC1','Confidence1']].copy()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"subm.loc[~subm.FVC1.isnull()].head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"subm.loc[~subm.FVC1.isnull(),'FVC'] = subm.loc[~subm.FVC1.isnull(),'FVC1']\nif sigma_mean<70:\n    subm['Confidence'] = sigma_opt\nelse:\n    subm.loc[~subm.FVC1.isnull(),'Confidence'] = subm.loc[~subm.FVC1.isnull(),'Confidence1']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"subm.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"subm.describe().T","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"otest = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/test.csv')\nfor i in range(len(otest)):\n    subm.loc[subm['Patient_Week']==otest.Patient[i]+'_'+str(otest.Weeks[i]), 'FVC'] = otest.FVC[i]\n    subm.loc[subm['Patient_Week']==otest.Patient[i]+'_'+str(otest.Weeks[i]), 'Confidence'] = 0.1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"subm[[\"Patient_Week\",\"FVC\",\"Confidence\"]].to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}