{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-15T16:17:08.959705Z","iopub.execute_input":"2023-01-15T16:17:08.962216Z","iopub.status.idle":"2023-01-15T16:17:09.977998Z","shell.execute_reply.started":"2023-01-15T16:17:08.962146Z","shell.execute_reply":"2023-01-15T16:17:09.976337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"directory = '/kaggle/input/osic-pulmonary-fibrosis-progression/'\ntrain_df = pd.read_csv(directory + '/train.csv')\ntest_df = pd.read_csv(directory + '/test.csv')","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:17:09.980338Z","iopub.execute_input":"2023-01-15T16:17:09.980810Z","iopub.status.idle":"2023-01-15T16:17:10.018938Z","shell.execute_reply.started":"2023-01-15T16:17:09.980769Z","shell.execute_reply":"2023-01-15T16:17:10.017562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['Patient'].count()","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:17:10.021114Z","iopub.execute_input":"2023-01-15T16:17:10.021593Z","iopub.status.idle":"2023-01-15T16:17:10.031656Z","shell.execute_reply.started":"2023-01-15T16:17:10.021554Z","shell.execute_reply":"2023-01-15T16:17:10.030538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['Patient'].count()","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:17:10.034087Z","iopub.execute_input":"2023-01-15T16:17:10.034592Z","iopub.status.idle":"2023-01-15T16:17:10.049086Z","shell.execute_reply.started":"2023-01-15T16:17:10.034547Z","shell.execute_reply":"2023-01-15T16:17:10.047396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pathlib import Path\n\nlen(list(Path(directory+'/train/').rglob(\"*\")))","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:17:10.052857Z","iopub.execute_input":"2023-01-15T16:17:10.053321Z","iopub.status.idle":"2023-01-15T16:17:11.733308Z","shell.execute_reply.started":"2023-01-15T16:17:10.053285Z","shell.execute_reply":"2023-01-15T16:17:11.731822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(list(Path(directory+'/test/').rglob(\"*\")))","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:17:11.735351Z","iopub.execute_input":"2023-01-15T16:17:11.735842Z","iopub.status.idle":"2023-01-15T16:17:11.761571Z","shell.execute_reply.started":"2023-01-15T16:17:11.735800Z","shell.execute_reply":"2023-01-15T16:17:11.760210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:17:11.763352Z","iopub.execute_input":"2023-01-15T16:17:11.764092Z","iopub.status.idle":"2023-01-15T16:17:11.780685Z","shell.execute_reply.started":"2023-01-15T16:17:11.764041Z","shell.execute_reply":"2023-01-15T16:17:11.779176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isna().mean()","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:17:11.782512Z","iopub.execute_input":"2023-01-15T16:17:11.783188Z","iopub.status.idle":"2023-01-15T16:17:11.798622Z","shell.execute_reply.started":"2023-01-15T16:17:11.783147Z","shell.execute_reply":"2023-01-15T16:17:11.797105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.isna().mean()","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:17:11.800295Z","iopub.execute_input":"2023-01-15T16:17:11.800857Z","iopub.status.idle":"2023-01-15T16:17:11.816195Z","shell.execute_reply.started":"2023-01-15T16:17:11.800811Z","shell.execute_reply":"2023-01-15T16:17:11.814699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['Patient'].nunique()","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:17:11.822318Z","iopub.execute_input":"2023-01-15T16:17:11.822908Z","iopub.status.idle":"2023-01-15T16:17:11.833233Z","shell.execute_reply.started":"2023-01-15T16:17:11.822847Z","shell.execute_reply":"2023-01-15T16:17:11.831768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['Patient'].nunique()","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:17:11.834924Z","iopub.execute_input":"2023-01-15T16:17:11.835429Z","iopub.status.idle":"2023-01-15T16:17:11.846577Z","shell.execute_reply.started":"2023-01-15T16:17:11.835364Z","shell.execute_reply":"2023-01-15T16:17:11.844988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:17:11.848638Z","iopub.execute_input":"2023-01-15T16:17:11.849691Z","iopub.status.idle":"2023-01-15T16:17:11.857433Z","shell.execute_reply.started":"2023-01-15T16:17:11.849623Z","shell.execute_reply":"2023-01-15T16:17:11.855794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['Weeks'].describe()","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:17:11.859233Z","iopub.execute_input":"2023-01-15T16:17:11.859603Z","iopub.status.idle":"2023-01-15T16:17:11.878523Z","shell.execute_reply.started":"2023-01-15T16:17:11.859570Z","shell.execute_reply":"2023-01-15T16:17:11.877469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['Sex'].hist(bins=3)","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:17:11.880240Z","iopub.execute_input":"2023-01-15T16:17:11.881410Z","iopub.status.idle":"2023-01-15T16:17:12.105295Z","shell.execute_reply.started":"2023-01-15T16:17:11.881346Z","shell.execute_reply":"2023-01-15T16:17:12.103921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['SmokingStatus'].hist(bins=3)","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:17:12.106958Z","iopub.execute_input":"2023-01-15T16:17:12.107823Z","iopub.status.idle":"2023-01-15T16:17:12.267082Z","shell.execute_reply.started":"2023-01-15T16:17:12.107771Z","shell.execute_reply":"2023-01-15T16:17:12.266024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.groupby('Sex')['Age'].hist(bins=20,histtype='step')","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:17:12.269262Z","iopub.execute_input":"2023-01-15T16:17:12.270204Z","iopub.status.idle":"2023-01-15T16:17:12.486686Z","shell.execute_reply.started":"2023-01-15T16:17:12.270141Z","shell.execute_reply":"2023-01-15T16:17:12.485330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['FVC'].describe()","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:17:12.489062Z","iopub.execute_input":"2023-01-15T16:17:12.489651Z","iopub.status.idle":"2023-01-15T16:17:12.507008Z","shell.execute_reply.started":"2023-01-15T16:17:12.489598Z","shell.execute_reply":"2023-01-15T16:17:12.504665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['Percent'].describe()","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:17:12.508602Z","iopub.execute_input":"2023-01-15T16:17:12.509080Z","iopub.status.idle":"2023-01-15T16:17:12.525205Z","shell.execute_reply.started":"2023-01-15T16:17:12.509036Z","shell.execute_reply":"2023-01-15T16:17:12.523732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.groupby('SmokingStatus')['Age'].hist(bins=10,histtype='step')","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:17:12.527078Z","iopub.execute_input":"2023-01-15T16:17:12.528003Z","iopub.status.idle":"2023-01-15T16:17:12.774136Z","shell.execute_reply.started":"2023-01-15T16:17:12.527952Z","shell.execute_reply":"2023-01-15T16:17:12.772924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from typing import Dict\n\ndef extract_dicom_meta_data(filename: str) -> Dict:\n    # Load image\n    \n    image_data = pydicom.read_file(filename)\n    img=np.array(image_data.pixel_array).flatten()\n    row = {\n        'Patient': image_data.PatientID,\n        'image_position_patient': image_data.ImagePositionPatient,\n        'pixel_spacing': image_data.PixelSpacing,\n        'BitsStored': image_data.BitsStored,\n        'HighBit': image_data.HighBit,\n        'img_min': np.min(img),\n        'img_max': np.max(img),\n        'img_mean': np.mean(img),\n        'img_std': np.std(img)}\n\n    return row","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:17:12.775750Z","iopub.execute_input":"2023-01-15T16:17:12.776208Z","iopub.status.idle":"2023-01-15T16:17:12.786515Z","shell.execute_reply.started":"2023-01-15T16:17:12.776173Z","shell.execute_reply":"2023-01-15T16:17:12.784412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nimport tqdm \nimport pydicom\n\ntrain_image_files = glob.glob(os.path.join(directory, 'train/', '*', '*.dcm'))\n\nmeta_data_df = []\nfor filename in tqdm.tqdm(train_image_files):\n    try:\n        meta_data_df.append(extract_dicom_meta_data(filename))\n    except Exception as e:\n        print(e)\n        continue","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:17:12.788227Z","iopub.execute_input":"2023-01-15T16:17:12.789666Z","iopub.status.idle":"2023-01-15T16:21:46.406843Z","shell.execute_reply.started":"2023-01-15T16:17:12.789606Z","shell.execute_reply":"2023-01-15T16:21:46.404136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_data_df = pd.DataFrame.from_dict(meta_data_df)\nmeta_data_df","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:21:46.408919Z","iopub.status.idle":"2023-01-15T16:21:46.409849Z","shell.execute_reply.started":"2023-01-15T16:21:46.409586Z","shell.execute_reply":"2023-01-15T16:21:46.409617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df=train_df.merge(train_df.groupby('Patient')['Weeks'].min(),on='Patient',suffixes=('','_min'))","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:21:46.411419Z","iopub.status.idle":"2023-01-15T16:21:46.412444Z","shell.execute_reply.started":"2023-01-15T16:21:46.412160Z","shell.execute_reply":"2023-01-15T16:21:46.412190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_df=train_df.drop(columns='weeks_min',axis=1)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:21:46.413820Z","iopub.status.idle":"2023-01-15T16:21:46.414460Z","shell.execute_reply.started":"2023-01-15T16:21:46.414244Z","shell.execute_reply":"2023-01-15T16:21:46.414266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def negate(minal):\n    if minal<1: \n        return -minal\n    else:\n        return 0","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:21:46.415628Z","iopub.status.idle":"2023-01-15T16:21:46.416244Z","shell.execute_reply.started":"2023-01-15T16:21:46.416036Z","shell.execute_reply":"2023-01-15T16:21:46.416058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['Weeks_passed'] = train_df['Weeks'] + train_df['Weeks_min'].apply(negate)","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:21:46.417437Z","iopub.status.idle":"2023-01-15T16:21:46.418087Z","shell.execute_reply.started":"2023-01-15T16:21:46.417884Z","shell.execute_reply":"2023-01-15T16:21:46.417906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:21:46.419326Z","iopub.status.idle":"2023-01-15T16:21:46.420038Z","shell.execute_reply.started":"2023-01-15T16:21:46.419819Z","shell.execute_reply":"2023-01-15T16:21:46.419842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_image(dirc, n=5, rows=2, cols=5, i=0):\n    plt.figure(figsize=(16, 9))\n    \n    for path in os.listdir(dirc):\n        for flnm in os.listdir(dirc + path):\n            image = pydicom.read_file(dirc + path + '/' + flnm)\n            image = image.pixel_array\n            i+=1\n            \n            if i>0:\n                plt.subplot(rows, cols, i)\n                plt.imshow(image)\n            \n            if i==n:\n                break\n        if i==n:\n            break","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:21:46.421248Z","iopub.status.idle":"2023-01-15T16:21:46.421914Z","shell.execute_reply.started":"2023-01-15T16:21:46.421698Z","shell.execute_reply":"2023-01-15T16:21:46.421722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_image(directory + 'train/',i=-40)\nshow_image(directory + 'test/', i=-30)","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:21:46.423243Z","iopub.status.idle":"2023-01-15T16:21:46.424294Z","shell.execute_reply.started":"2023-01-15T16:21:46.424004Z","shell.execute_reply":"2023-01-15T16:21:46.424046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:21:46.425721Z","iopub.status.idle":"2023-01-15T16:21:46.426195Z","shell.execute_reply.started":"2023-01-15T16:21:46.425977Z","shell.execute_reply":"2023-01-15T16:21:46.425999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#LABEL ENCODER\nlabel_encoder = LabelEncoder()\n\nto_encode = ['Sex','SmokingStatus']\nencoded_all = []\n\nfor column in to_encode:\n    encoded = label_encoder.fit_transform(train_df[column])\n    encoded_all.append(encoded)\n    \ntrain_df['Sex'] = encoded_all[0]\ntrain_df['SmokingStatus'] = encoded_all[1]","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:21:46.427726Z","iopub.status.idle":"2023-01-15T16:21:46.428173Z","shell.execute_reply.started":"2023-01-15T16:21:46.427964Z","shell.execute_reply":"2023-01-15T16:21:46.427986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:21:46.429971Z","iopub.status.idle":"2023-01-15T16:21:46.430425Z","shell.execute_reply.started":"2023-01-15T16:21:46.430199Z","shell.execute_reply":"2023-01-15T16:21:46.430219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#LABEL ENCODER\nlabel_encoder = LabelEncoder()\n\nto_encode = ['Sex','SmokingStatus']\nencoded_all = []\n\nfor column in to_encode:\n    encoded = label_encoder.fit_transform(test_df[column])\n    encoded_all.append(encoded)\n    \ntest_df['Sex'] = encoded_all[0]\ntest_df['SmokingStatus'] = encoded_all[1]","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:21:46.433329Z","iopub.status.idle":"2023-01-15T16:21:46.434895Z","shell.execute_reply.started":"2023-01-15T16:21:46.434544Z","shell.execute_reply":"2023-01-15T16:21:46.434582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:21:46.437267Z","iopub.status.idle":"2023-01-15T16:21:46.438912Z","shell.execute_reply.started":"2023-01-15T16:21:46.438534Z","shell.execute_reply":"2023-01-15T16:21:46.438571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from catboost import CatBoostRegressor\n\ncat_features=['Weeks','Percent','Age','Sex','SmokingStatus']\n\ncat = CatBoostRegressor()\ncat.fit(train_df[cat_features], train_df['FVC'])","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:21:46.441008Z","iopub.status.idle":"2023-01-15T16:21:46.442185Z","shell.execute_reply.started":"2023-01-15T16:21:46.441934Z","shell.execute_reply":"2023-01-15T16:21:46.441962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_preds = cat.predict(test_df[cat_features])\n\nsubmission = pd.DataFrame()\nsubmission['cat_FVC'] = cat_preds\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:21:46.444054Z","iopub.status.idle":"2023-01-15T16:21:46.444595Z","shell.execute_reply.started":"2023-01-15T16:21:46.444326Z","shell.execute_reply":"2023-01-15T16:21:46.444349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def competition_metric(trueFVC, predFVC, predSTD):\n    clipSTD = np.clip(predSTD, 70 , 9e9)  \n    deltaFVC = np.clip(np.abs(trueFVC - predFVC), 0 , 1000)  \n    return np.mean(-1 * (np.sqrt(2) * deltaFVC / clipSTD) - np.log(np.sqrt(2) * clipSTD))\n    \n\nprint(\n    'Competition metric: ', \n    competition_metric(submission['cat_FVC'].values, test_df['FVC'], 275) \n)","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:21:46.447510Z","iopub.status.idle":"2023-01-15T16:21:46.448562Z","shell.execute_reply.started":"2023-01-15T16:21:46.448225Z","shell.execute_reply":"2023-01-15T16:21:46.448257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Try to merge metadata BD and train BD\n\n#for idntf in train_df['Patient'].unique():\n#    meta_train_df['Patient']=idntf\n#    meta_train_df. train_df[train_df['Patient'==idntf]\n#meta_train_df=pd.DataFrame({'Weeks': [5]})\n#mt_train_df=train_df.loc[train_df['Weeks']==5,'Patient'].tolist()\n#meta_train_df['Patient']=mt_train_df","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:21:46.450203Z","iopub.status.idle":"2023-01-15T16:21:46.450832Z","shell.execute_reply.started":"2023-01-15T16:21:46.450526Z","shell.execute_reply":"2023-01-15T16:21:46.450557Z"},"trusted":true},"execution_count":null,"outputs":[]}]}