{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfrom os import listdir\nimport pandas as pd\nimport numpy as np\nimport glob\nimport tqdm\nfrom typing import Dict\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\n#plotly\n!pip install chart_studio\nimport plotly.express as px\nimport chart_studio.plotly as py\nimport plotly.graph_objs as go\nfrom plotly.offline import iplot\nimport cufflinks\ncufflinks.go_offline()\ncufflinks.set_config_file(world_readable=True, theme='pearl')\n\n#color\nfrom colorama import Fore, Back, Style\n\nimport seaborn as sns\nsns.set(style=\"whitegrid\")\n\n#pydicom\nimport pydicom\n\n# Suppress warnings \nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Settings for pretty nice plots\nplt.style.use('fivethirtyeight')\nplt.show()","metadata":{"_kg_hide-input":false,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-04-07T07:45:15.770221Z","iopub.execute_input":"2023-04-07T07:45:15.77057Z","iopub.status.idle":"2023-04-07T07:45:27.693703Z","shell.execute_reply.started":"2023-04-07T07:45:15.77054Z","shell.execute_reply":"2023-04-07T07:45:27.692431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# List files available\nlist(os.listdir(\"../input/osic-pulmonary-fibrosis-progression\"))","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:27.69793Z","iopub.execute_input":"2023-04-07T07:45:27.698284Z","iopub.status.idle":"2023-04-07T07:45:27.704168Z","shell.execute_reply.started":"2023-04-07T07:45:27.698253Z","shell.execute_reply":"2023-04-07T07:45:27.703065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_PATH = \"../input/osic-pulmonary-fibrosis-progressiont/\"\n\ntrain_df = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/train.csv')\ntest_df = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/test.csv')\n\nprint(Fore.YELLOW + 'Training data shape: ',Style.RESET_ALL,train_df.shape)\ntrain_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:27.706111Z","iopub.execute_input":"2023-04-07T07:45:27.706805Z","iopub.status.idle":"2023-04-07T07:45:27.760786Z","shell.execute_reply.started":"2023-04-07T07:45:27.706758Z","shell.execute_reply":"2023-04-07T07:45:27.759738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.groupby(['SmokingStatus']).count()['Sex'].to_frame()","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2023-04-07T07:45:27.76242Z","iopub.execute_input":"2023-04-07T07:45:27.762868Z","iopub.status.idle":"2023-04-07T07:45:27.783218Z","shell.execute_reply.started":"2023-04-07T07:45:27.762832Z","shell.execute_reply":"2023-04-07T07:45:27.781968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## General Info","metadata":{}},{"cell_type":"code","source":"# Null values and Data types\nprint(Fore.YELLOW + 'Train Set !!',Style.RESET_ALL)\nprint(train_df.info())\nprint('-------------')\nprint(Fore.BLUE + 'Test Set !!',Style.RESET_ALL)\nprint(test_df.info())","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:27.786847Z","iopub.execute_input":"2023-04-07T07:45:27.787199Z","iopub.status.idle":"2023-04-07T07:45:27.806954Z","shell.execute_reply.started":"2023-04-07T07:45:27.787166Z","shell.execute_reply":"2023-04-07T07:45:27.805989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The type of Percent column is float64.","metadata":{}},{"cell_type":"markdown","source":"### Missing values","metadata":{}},{"cell_type":"code","source":"train_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:27.812381Z","iopub.execute_input":"2023-04-07T07:45:27.812719Z","iopub.status.idle":"2023-04-07T07:45:27.821758Z","shell.execute_reply.started":"2023-04-07T07:45:27.812688Z","shell.execute_reply":"2023-04-07T07:45:27.820594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:27.823249Z","iopub.execute_input":"2023-04-07T07:45:27.823551Z","iopub.status.idle":"2023-04-07T07:45:27.83668Z","shell.execute_reply.started":"2023-04-07T07:45:27.823524Z","shell.execute_reply":"2023-04-07T07:45:27.835378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There is no missing values in train_df and test_df.","metadata":{}},{"cell_type":"code","source":"# Total number of Patient in the dataset(train+test)\n\nprint(Fore.YELLOW +\"Total Patients in Train set: \",Style.RESET_ALL,train_df['Patient'].count())\nprint(Fore.BLUE +\"Total Patients in Test set: \",Style.RESET_ALL,test_df['Patient'].count())","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:27.837994Z","iopub.execute_input":"2023-04-07T07:45:27.83841Z","iopub.status.idle":"2023-04-07T07:45:27.84897Z","shell.execute_reply.started":"2023-04-07T07:45:27.838378Z","shell.execute_reply":"2023-04-07T07:45:27.848088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" `5` : Patients in Test Set","metadata":{}},{"cell_type":"markdown","source":"## Unique Patients(Ids)","metadata":{}},{"cell_type":"code","source":"print(Fore.YELLOW + \"The total patient ids are\",Style.RESET_ALL,f\"{train_df['Patient'].count()},\", Fore.BLUE + \"from those the unique ids are\", Style.RESET_ALL, f\"{train_df['Patient'].value_counts().shape[0]}.\")","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:27.850059Z","iopub.execute_input":"2023-04-07T07:45:27.850401Z","iopub.status.idle":"2023-04-07T07:45:27.860725Z","shell.execute_reply.started":"2023-04-07T07:45:27.85037Z","shell.execute_reply":"2023-04-07T07:45:27.859898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_patient_ids = set(train_df['Patient'].unique())\ntest_patient_ids = set(test_df['Patient'].unique())\n\ntrain_patient_ids.intersection(test_patient_ids)","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:27.861887Z","iopub.execute_input":"2023-04-07T07:45:27.862385Z","iopub.status.idle":"2023-04-07T07:45:27.877091Z","shell.execute_reply.started":"2023-04-07T07:45:27.862353Z","shell.execute_reply":"2023-04-07T07:45:27.876268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We already see `5` patients in test set that can be found in train set as well.","metadata":{}},{"cell_type":"code","source":"columns = train_df.keys()\ncolumns = list(columns)\nprint(columns)","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:27.878426Z","iopub.execute_input":"2023-04-07T07:45:27.878911Z","iopub.status.idle":"2023-04-07T07:45:27.885704Z","shell.execute_reply.started":"2023-04-07T07:45:27.878878Z","shell.execute_reply":"2023-04-07T07:45:27.884462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Patient Counts","metadata":{}},{"cell_type":"code","source":"train_df['Patient'].value_counts().max()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:27.887279Z","iopub.execute_input":"2023-04-07T07:45:27.887612Z","iopub.status.idle":"2023-04-07T07:45:27.899821Z","shell.execute_reply.started":"2023-04-07T07:45:27.887581Z","shell.execute_reply":"2023-04-07T07:45:27.898907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In train set, there are multiple rows for one 'Patient'. Because Patient has different weeks, FVC, Percent.","metadata":{}},{"cell_type":"code","source":"test_df['Patient'].value_counts().max()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:27.901466Z","iopub.execute_input":"2023-04-07T07:45:27.901917Z","iopub.status.idle":"2023-04-07T07:45:27.914145Z","shell.execute_reply.started":"2023-04-07T07:45:27.901873Z","shell.execute_reply":"2023-04-07T07:45:27.913143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In test set, we can see one Patient and it mean Patient id is unique.","metadata":{}},{"cell_type":"code","source":"np.quantile(train_df['Patient'].value_counts(), 0.75) - np.quantile(test_df['Patient'].value_counts(), 0.25)","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:27.915999Z","iopub.execute_input":"2023-04-07T07:45:27.916551Z","iopub.status.idle":"2023-04-07T07:45:27.930978Z","shell.execute_reply.started":"2023-04-07T07:45:27.916502Z","shell.execute_reply":"2023-04-07T07:45:27.930167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(np.quantile(train_df['Patient'].value_counts(), 0.95))\nprint(np.quantile(test_df['Patient'].value_counts(), 0.95))","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:27.932164Z","iopub.execute_input":"2023-04-07T07:45:27.932603Z","iopub.status.idle":"2023-04-07T07:45:27.942793Z","shell.execute_reply.started":"2023-04-07T07:45:27.93256Z","shell.execute_reply":"2023-04-07T07:45:27.942006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Number of Patients and Images in Training Images Folder\n* https://www.kaggle.com/yeayates21/osic-simple-image-eda","metadata":{}},{"cell_type":"code","source":"files = folders = 0\n\npath = \"/kaggle/input/osic-pulmonary-fibrosis-progression/train\"\n\nfor _, dirnames, filenames in os.walk(path):\n  # ^ this idiom means \"we won't be using this value\"\n    files += len(filenames)\n    folders += len(dirnames)\n#print(Fore.YELLOW +\"Total Patients in Train set: \",Style.RESET_ALL,train_df['Patient'].count())\nprint(Fore.YELLOW +f'{files:,}',Style.RESET_ALL,\"files/images, \" + Fore.BLUE + f'{folders:,}',Style.RESET_ALL ,'folders/patients')","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:27.943863Z","iopub.execute_input":"2023-04-07T07:45:27.944272Z","iopub.status.idle":"2023-04-07T07:45:43.328962Z","shell.execute_reply.started":"2023-04-07T07:45:27.944239Z","shell.execute_reply":"2023-04-07T07:45:43.328223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files = []\nfor _, dirnames, filenames in os.walk(path):\n  # ^ this idiom means \"we won't be using this value\"\n    files.append(len(filenames))\n\nprint(Fore.YELLOW +f'{round(np.mean(files)):,}',Style.RESET_ALL,'average files/images per patient')\nprint(Fore.BLUE +f'{round(np.max(files)):,}',Style.RESET_ALL, 'max files/images per patient')","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:43.330244Z","iopub.execute_input":"2023-04-07T07:45:43.330753Z","iopub.status.idle":"2023-04-07T07:45:43.448878Z","shell.execute_reply.started":"2023-04-07T07:45:43.33071Z","shell.execute_reply":"2023-04-07T07:45:43.447912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating Individual Patient Dataframe\n\nfor `175` unique patients, we make new dataframe\n\nThanks [@wjdanalharthi](https://www.kaggle.com/wjdanalharthi)","metadata":{}},{"cell_type":"code","source":"patient_df = train_df[['Patient', 'Age', 'Sex', 'SmokingStatus']].drop_duplicates()\npatient_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:43.450355Z","iopub.execute_input":"2023-04-07T07:45:43.450657Z","iopub.status.idle":"2023-04-07T07:45:43.468874Z","shell.execute_reply.started":"2023-04-07T07:45:43.450626Z","shell.execute_reply":"2023-04-07T07:45:43.468004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"You can use this:\n- https://www.kaggle.com/redwankarimsony/pulmonary-fibrosis-progression-interactive-eda","metadata":{}},{"cell_type":"code","source":"# Creating unique patient lists and their properties. \ntrain_dir = '../input/osic-pulmonary-fibrosis-progression/train/'\ntest_dir = '../input/osic-pulmonary-fibrosis-progression/test/'\n\npatient_ids = os.listdir(train_dir)\npatient_ids = sorted(patient_ids)\n\n#Creating new rows\nno_of_instances = []\nage = []\nsex = []\nsmoking_status = []\n\nfor patient_id in patient_ids:\n    patient_info = train_df[train_df['Patient'] == patient_id].reset_index()\n    no_of_instances.append(len(os.listdir(train_dir + patient_id)))\n    age.append(patient_info['Age'][0])\n    sex.append(patient_info['Sex'][0])\n    smoking_status.append(patient_info['SmokingStatus'][0])\n\n#Creating the dataframe for the patient info    \npatient_df = pd.DataFrame(list(zip(patient_ids, no_of_instances, age, sex, smoking_status)), \n                                 columns =['Patient', 'no_of_instances', 'Age', 'Sex', 'SmokingStatus'])\nprint(patient_df.info())\npatient_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:43.47031Z","iopub.execute_input":"2023-04-07T07:45:43.470602Z","iopub.status.idle":"2023-04-07T07:45:43.910867Z","shell.execute_reply.started":"2023-04-07T07:45:43.470573Z","shell.execute_reply":"2023-04-07T07:45:43.910009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Exploring the 'SmokingStatus' column","metadata":{}},{"cell_type":"code","source":"patient_df['SmokingStatus'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:43.912583Z","iopub.execute_input":"2023-04-07T07:45:43.912895Z","iopub.status.idle":"2023-04-07T07:45:43.922667Z","shell.execute_reply.started":"2023-04-07T07:45:43.912865Z","shell.execute_reply":"2023-04-07T07:45:43.921553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patient_df['SmokingStatus'].value_counts().iplot(kind='bar',\n                                              yTitle='Counts', \n                                              linecolor='black', \n                                              opacity=0.7,\n                                              color='blue',\n                                              theme='pearl',\n                                              bargap=0.5,\n                                              gridcolor='white',\n                                              title='Distribution of the SmokingStatus column in the Unique Patient Set')","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:43.924423Z","iopub.execute_input":"2023-04-07T07:45:43.924931Z","iopub.status.idle":"2023-04-07T07:45:44.681367Z","shell.execute_reply.started":"2023-04-07T07:45:43.924888Z","shell.execute_reply":"2023-04-07T07:45:44.680297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"`118` : Ex-smoker\n\n`49` : Never smoked\n\n`9` : Currently smokes","metadata":{}},{"cell_type":"markdown","source":"## Weeks distribution","metadata":{}},{"cell_type":"code","source":"train_df['Weeks'].value_counts().head()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:44.682965Z","iopub.execute_input":"2023-04-07T07:45:44.683754Z","iopub.status.idle":"2023-04-07T07:45:44.69377Z","shell.execute_reply.started":"2023-04-07T07:45:44.683706Z","shell.execute_reply":"2023-04-07T07:45:44.692908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['Weeks'].value_counts().iplot(kind='barh',\n                                      xTitle='Counts(Weeks)', \n                                      linecolor='black', \n                                      opacity=0.7,\n                                      color='#FB8072',\n                                      theme='pearl',\n                                      bargap=0.2,\n                                      gridcolor='white',\n                                      title='Distribution of the Weeks in the training set')","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:44.695235Z","iopub.execute_input":"2023-04-07T07:45:44.695935Z","iopub.status.idle":"2023-04-07T07:45:44.757034Z","shell.execute_reply.started":"2023-04-07T07:45:44.695891Z","shell.execute_reply":"2023-04-07T07:45:44.755896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['Weeks'].iplot(kind='hist',\n                              xTitle='Weeks', \n                              yTitle='Counts',\n                              linecolor='black', \n                              opacity=0.7,\n                              color='#FB8072',\n                              theme='pearl',\n                              bargap=0.2,\n                              gridcolor='white',\n                              title='Distribution of the Weeks in the training set')","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:44.758586Z","iopub.execute_input":"2023-04-07T07:45:44.759237Z","iopub.status.idle":"2023-04-07T07:45:44.852904Z","shell.execute_reply.started":"2023-04-07T07:45:44.759192Z","shell.execute_reply":"2023-04-07T07:45:44.851813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are some negative values for Weeks. \n\nBecause Weeks is the relative number of weeks pre/post the baseline CT.","metadata":{}},{"cell_type":"markdown","source":"## Distribution Age over Week","metadata":{}},{"cell_type":"code","source":"fig = px.scatter(train_df, x=\"Weeks\", y=\"Age\", color='Sex')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:44.854457Z","iopub.execute_input":"2023-04-07T07:45:44.855055Z","iopub.status.idle":"2023-04-07T07:45:44.964508Z","shell.execute_reply.started":"2023-04-07T07:45:44.854994Z","shell.execute_reply":"2023-04-07T07:45:44.963407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## FVC - The forced vital capacity","metadata":{}},{"cell_type":"markdown","source":" The forced vital capacity (FVC), i.e. the volume of air exhaled\n - the recorded lung capacity in ml","metadata":{}},{"cell_type":"code","source":"train_df['FVC'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:44.966308Z","iopub.execute_input":"2023-04-07T07:45:44.966718Z","iopub.status.idle":"2023-04-07T07:45:44.982537Z","shell.execute_reply.started":"2023-04-07T07:45:44.966675Z","shell.execute_reply":"2023-04-07T07:45:44.981383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['FVC'].iplot(kind='hist',\n                      xTitle='Lung Capacity(ml)', \n                      linecolor='black', \n                      opacity=0.8,\n                      color='#FB8072',\n                      bargap=0.5,\n                      gridcolor='white',\n                      title='Distribution of the FVC in the training set')","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:44.984354Z","iopub.execute_input":"2023-04-07T07:45:44.984824Z","iopub.status.idle":"2023-04-07T07:45:45.06311Z","shell.execute_reply.started":"2023-04-07T07:45:44.984782Z","shell.execute_reply":"2023-04-07T07:45:45.062084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### FVC vs Percent","metadata":{}},{"cell_type":"code","source":"fig = px.scatter(train_df, x=\"FVC\", y=\"Percent\", color='Age')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:45.065051Z","iopub.execute_input":"2023-04-07T07:45:45.06547Z","iopub.status.idle":"2023-04-07T07:45:45.360151Z","shell.execute_reply.started":"2023-04-07T07:45:45.065423Z","shell.execute_reply":"2023-04-07T07:45:45.359056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"FVC seems to related Percent linearly. Makes sense as both terms are proportional.","metadata":{}},{"cell_type":"markdown","source":"### FVC vs Age","metadata":{}},{"cell_type":"code","source":"fig = px.scatter(train_df, x=\"FVC\", y=\"Age\", color='Sex')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:45.361451Z","iopub.execute_input":"2023-04-07T07:45:45.361849Z","iopub.status.idle":"2023-04-07T07:45:45.438681Z","shell.execute_reply.started":"2023-04-07T07:45:45.361808Z","shell.execute_reply":"2023-04-07T07:45:45.437783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Males have higher FVC than females irrespective of age","metadata":{}},{"cell_type":"markdown","source":"### FVC vs Weeks","metadata":{}},{"cell_type":"code","source":"fig = px.scatter(train_df, x=\"FVC\", y=\"Weeks\", color='SmokingStatus')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:45.44024Z","iopub.execute_input":"2023-04-07T07:45:45.440545Z","iopub.status.idle":"2023-04-07T07:45:45.522127Z","shell.execute_reply.started":"2023-04-07T07:45:45.440515Z","shell.execute_reply":"2023-04-07T07:45:45.521166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Pick one patient for FVC vs Weeks","metadata":{}},{"cell_type":"code","source":"patient = train_df[train_df.Patient == 'ID00228637202259965313869']\nfig = px.line(patient, x=\"Weeks\", y=\"FVC\", color='SmokingStatus')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:45.52349Z","iopub.execute_input":"2023-04-07T07:45:45.523788Z","iopub.status.idle":"2023-04-07T07:45:45.602453Z","shell.execute_reply.started":"2023-04-07T07:45:45.523761Z","shell.execute_reply":"2023-04-07T07:45:45.601341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Person never smoked has FVC lower than smoker. Some Ex-smoker have very high FVC.","metadata":{}},{"cell_type":"markdown","source":"## Percent","metadata":{}},{"cell_type":"markdown","source":"A computed field which approximates the patient's FVC as a percent of the typical FVC for a person of similar characteristics","metadata":{}},{"cell_type":"code","source":"train_df['Percent'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:45.610313Z","iopub.execute_input":"2023-04-07T07:45:45.610625Z","iopub.status.idle":"2023-04-07T07:45:45.619562Z","shell.execute_reply.started":"2023-04-07T07:45:45.610596Z","shell.execute_reply":"2023-04-07T07:45:45.618663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['Percent'].iplot(kind='hist',bins=30,color='blue',xTitle='Percent distribution',yTitle='Count')","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:45.620918Z","iopub.execute_input":"2023-04-07T07:45:45.621244Z","iopub.status.idle":"2023-04-07T07:45:45.702003Z","shell.execute_reply.started":"2023-04-07T07:45:45.621214Z","shell.execute_reply":"2023-04-07T07:45:45.701036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Percent vs SmokingStatus In Patient Dataframe","metadata":{}},{"cell_type":"code","source":"df = train_df\nfig = px.violin(df, y='Percent', x='SmokingStatus', box=True, color='Sex', points=\"all\",\n          hover_data=train_df.columns)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:45.703741Z","iopub.execute_input":"2023-04-07T07:45:45.704276Z","iopub.status.idle":"2023-04-07T07:45:45.926671Z","shell.execute_reply.started":"2023-04-07T07:45:45.704232Z","shell.execute_reply":"2023-04-07T07:45:45.922941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16, 6))\nax = sns.violinplot(x = train_df['SmokingStatus'], y = train_df['Percent'], palette = 'Reds')\nax.set_xlabel(xlabel = 'Smoking Habit', fontsize = 15)\nax.set_ylabel(ylabel = 'Percent', fontsize = 15)\nax.set_title(label = 'Distribution of Smoking Status Over Percentage', fontsize = 20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:45.928211Z","iopub.execute_input":"2023-04-07T07:45:45.928535Z","iopub.status.idle":"2023-04-07T07:45:46.216686Z","shell.execute_reply.started":"2023-04-07T07:45:45.928504Z","shell.execute_reply":"2023-04-07T07:45:46.215712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.scatter(train_df, x=\"Age\", y=\"Percent\", color='SmokingStatus')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:46.217943Z","iopub.execute_input":"2023-04-07T07:45:46.218352Z","iopub.status.idle":"2023-04-07T07:45:46.302439Z","shell.execute_reply.started":"2023-04-07T07:45:46.218319Z","shell.execute_reply":"2023-04-07T07:45:46.301636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Pick one patient for FVC vs Weeks","metadata":{}},{"cell_type":"code","source":"patient = train_df[train_df.Patient == 'ID00228637202259965313869']\nfig = px.line(patient, x=\"Weeks\", y=\"Percent\", color='SmokingStatus')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:46.303584Z","iopub.execute_input":"2023-04-07T07:45:46.304071Z","iopub.status.idle":"2023-04-07T07:45:46.364259Z","shell.execute_reply.started":"2023-04-07T07:45:46.304Z","shell.execute_reply":"2023-04-07T07:45:46.363537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Age Distribution of Unique Patients","metadata":{}},{"cell_type":"code","source":"patient_df['Age'].iplot(kind='hist',bins=30,color='red',xTitle='Ages of distribution',yTitle='Count')","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:46.365448Z","iopub.execute_input":"2023-04-07T07:45:46.365916Z","iopub.status.idle":"2023-04-07T07:45:46.426887Z","shell.execute_reply.started":"2023-04-07T07:45:46.365865Z","shell.execute_reply":"2023-04-07T07:45:46.425849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Distribution of Age vs SmokingStatus In Patient Dataframe","metadata":{}},{"cell_type":"code","source":"patient_df['SmokingStatus'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:46.428366Z","iopub.execute_input":"2023-04-07T07:45:46.428666Z","iopub.status.idle":"2023-04-07T07:45:46.436926Z","shell.execute_reply.started":"2023-04-07T07:45:46.428635Z","shell.execute_reply":"2023-04-07T07:45:46.435998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16, 6))\nsns.kdeplot(patient_df.loc[patient_df['SmokingStatus'] == 'Ex-smoker', 'Age'], label = 'Ex-smoker',shade=True)\nsns.kdeplot(patient_df.loc[patient_df['SmokingStatus'] == 'Never smoked', 'Age'], label = 'Never smoked',shade=True)\nsns.kdeplot(patient_df.loc[patient_df['SmokingStatus'] == 'Currently smokes', 'Age'], label = 'Currently smokes', shade=True)\n\n# Labeling of plot\nplt.xlabel('Age (years)'); plt.ylabel('Density'); plt.title('Distribution of Ages');","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:46.438658Z","iopub.execute_input":"2023-04-07T07:45:46.439081Z","iopub.status.idle":"2023-04-07T07:45:46.774738Z","shell.execute_reply.started":"2023-04-07T07:45:46.439037Z","shell.execute_reply":"2023-04-07T07:45:46.773672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16, 6))\nax = sns.violinplot(x = patient_df['SmokingStatus'], y = patient_df['Age'], palette = 'Reds')\nax.set_xlabel(xlabel = 'Smoking habit', fontsize = 15)\nax.set_ylabel(ylabel = 'Age', fontsize = 15)\nax.set_title(label = 'Distribution of Smokers over Age', fontsize = 20)\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-07T07:45:46.77647Z","iopub.execute_input":"2023-04-07T07:45:46.776804Z","iopub.status.idle":"2023-04-07T07:45:46.987276Z","shell.execute_reply.started":"2023-04-07T07:45:46.776763Z","shell.execute_reply":"2023-04-07T07:45:46.98616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Distribution of Age vs Gender In Patient Dataframe","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16, 6))\nsns.kdeplot(patient_df.loc[patient_df['Sex'] == 'Male', 'Age'], label = 'Male',shade=True)\nsns.kdeplot(patient_df.loc[patient_df['Sex'] == 'Female', 'Age'], label = 'Female',shade=True)\nplt.xlabel('Age (years)'); plt.ylabel('Density'); plt.title('Distribution of Ages');","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:46.988947Z","iopub.execute_input":"2023-04-07T07:45:46.989405Z","iopub.status.idle":"2023-04-07T07:45:47.321572Z","shell.execute_reply.started":"2023-04-07T07:45:46.989361Z","shell.execute_reply":"2023-04-07T07:45:47.320146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Gender Distribution","metadata":{}},{"cell_type":"code","source":"patient_df['Sex'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:47.322859Z","iopub.execute_input":"2023-04-07T07:45:47.323184Z","iopub.status.idle":"2023-04-07T07:45:47.33155Z","shell.execute_reply.started":"2023-04-07T07:45:47.323155Z","shell.execute_reply":"2023-04-07T07:45:47.330623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patient_df['Sex'].value_counts().iplot(kind='bar',\n                                          yTitle='Count', \n                                          linecolor='black', \n                                          opacity=0.7,\n                                          color='blue',\n                                          theme='pearl',\n                                          bargap=0.8,\n                                          gridcolor='white',\n                                          title='Distribution of the Sex column in Patient Dataframe')","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:47.333013Z","iopub.execute_input":"2023-04-07T07:45:47.333361Z","iopub.status.idle":"2023-04-07T07:45:47.389935Z","shell.execute_reply.started":"2023-04-07T07:45:47.333329Z","shell.execute_reply":"2023-04-07T07:45:47.388866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"`139` : Male\n\n`37` : Female","metadata":{}},{"cell_type":"markdown","source":"### Gender vs SmokingStatus In Patient Dataframe","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16, 6))\na = sns.countplot(data=patient_df, x='SmokingStatus', hue='Sex')\n\nfor p in a.patches:\n    a.annotate(format(p.get_height(), ','), \n           (p.get_x() + p.get_width() / 2., \n            p.get_height()), ha = 'center', va = 'center', \n           xytext = (0, 4), textcoords = 'offset points')\n\nplt.title('Gender split by SmokingStatus', fontsize=16)\nsns.despine(left=True, bottom=True);","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:47.391396Z","iopub.execute_input":"2023-04-07T07:45:47.391695Z","iopub.status.idle":"2023-04-07T07:45:47.606851Z","shell.execute_reply.started":"2023-04-07T07:45:47.391665Z","shell.execute_reply":"2023-04-07T07:45:47.605816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.box(patient_df, x=\"Sex\", y=\"Age\", points=\"all\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:47.608712Z","iopub.execute_input":"2023-04-07T07:45:47.609164Z","iopub.status.idle":"2023-04-07T07:45:47.691179Z","shell.execute_reply.started":"2023-04-07T07:45:47.609119Z","shell.execute_reply":"2023-04-07T07:45:47.690144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Patient Overlap","metadata":{"trusted":true}},{"cell_type":"code","source":"# Extract patient id's for the training set\nids_train = train_df.Patient.values\n# Extract patient id's for the validation set\nids_test = test_df.Patient.values\n#print(Fore.YELLOW +\"Total Patients in Train set: \",Style.RESET_ALL,train_df['Patient'].count())\n# Create a \"set\" datastructure of the training set id's to identify unique id's\nids_train_set = set(ids_train)\nprint(Fore.YELLOW + \"There are\",Style.RESET_ALL,f'{len(ids_train_set)}', Fore.BLUE + 'unique Patient IDs',Style.RESET_ALL,'in the training set')\n# Create a \"set\" datastructure of the validation set id's to identify unique id's\nids_test_set = set(ids_test)\nprint(Fore.YELLOW + \"There are\", Style.RESET_ALL, f'{len(ids_test_set)}', Fore.BLUE + 'unique Patient IDs',Style.RESET_ALL,'in the test set')\n\n# Identify patient overlap by looking at the intersection between the sets\npatient_overlap = list(ids_train_set.intersection(ids_test_set))\nn_overlap = len(patient_overlap)\nprint(Fore.YELLOW + \"There are\", Style.RESET_ALL, f'{n_overlap}', Fore.BLUE + 'Patient IDs',Style.RESET_ALL, 'in both the training and test sets')\nprint('')\nprint(Fore.CYAN + 'These patients are in both the training and test datasets:', Style.RESET_ALL)\nprint(f'{patient_overlap}')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-07T07:45:47.693242Z","iopub.execute_input":"2023-04-07T07:45:47.693663Z","iopub.status.idle":"2023-04-07T07:45:47.705489Z","shell.execute_reply.started":"2023-04-07T07:45:47.693619Z","shell.execute_reply":"2023-04-07T07:45:47.704311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"`5` patients are in both the training and test datasets.","metadata":{}},{"cell_type":"markdown","source":"## Heatmap for train.csv","metadata":{}},{"cell_type":"code","source":"corrmat = train_df.corr() \nf, ax = plt.subplots(figsize =(9, 8)) \nsns.heatmap(corrmat, ax = ax, cmap = 'RdYlBu_r', linewidths = 0.5) ","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:47.708115Z","iopub.execute_input":"2023-04-07T07:45:47.70847Z","iopub.status.idle":"2023-04-07T07:45:47.956691Z","shell.execute_reply.started":"2023-04-07T07:45:47.708416Z","shell.execute_reply":"2023-04-07T07:45:47.954734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Please compare with the previous visualization information. And we may compare to Pandas Profiling below.","metadata":{}},{"cell_type":"markdown","source":"# 6. <a id='visual'>Visualising Images : DECOM 🗺️</a>  ","metadata":{}},{"cell_type":"markdown","source":"`1` type of images containing the information:\n\n- `.dcm` files: [DICOM files](https://en.wikipedia.org/wiki/DICOM). It's saved in the \"Digital Imaging and Communications in Medicine\" format. It contains an image from a medical scan, such as an ultrasound or MRI + information about the patient.","metadata":{}},{"cell_type":"code","source":"print(Fore.YELLOW + 'Train .dcm number of images:',Style.RESET_ALL, len(list(os.listdir('../input/osic-pulmonary-fibrosis-progression/train'))), '\\n' +\n      Fore.BLUE + 'Test .dcm number of images:',Style.RESET_ALL, len(list(os.listdir('../input/osic-pulmonary-fibrosis-progression/test'))), '\\n' +\n      '--------------------------------', '\\n' +\n      'There is the same number of images as in train/ test .csv datasets')","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:47.957993Z","iopub.execute_input":"2023-04-07T07:45:47.958353Z","iopub.status.idle":"2023-04-07T07:45:47.976673Z","shell.execute_reply.started":"2023-04-07T07:45:47.958302Z","shell.execute_reply":"2023-04-07T07:45:47.974041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's look at the DICOM images.","metadata":{}},{"cell_type":"code","source":"def plot_pixel_array(dataset, figsize=(5,5)):\n    plt.figure(figsize=figsize)\n    plt.grid(False)\n    plt.imshow(dataset.pixel_array, cmap='gray') # cmap=plt.cm.bone)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:47.978163Z","iopub.execute_input":"2023-04-07T07:45:47.978585Z","iopub.status.idle":"2023-04-07T07:45:47.984054Z","shell.execute_reply.started":"2023-04-07T07:45:47.97855Z","shell.execute_reply":"2023-04-07T07:45:47.982945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/schlerp/getting-to-know-dicom-and-the-data\ndef show_dcm_info(dataset):\n    print(Fore.YELLOW + \"Filename.........:\",Style.RESET_ALL,file_path)\n    print()\n\n    pat_name = dataset.PatientName\n    display_name = pat_name.family_name + \", \" + pat_name.given_name\n    print(Fore.BLUE + \"Patient's name......:\",Style.RESET_ALL, display_name)\n    print(Fore.BLUE + \"Patient id..........:\",Style.RESET_ALL, dataset.PatientID)\n    print(Fore.BLUE + \"Patient's Sex.......:\",Style.RESET_ALL, dataset.PatientSex)\n    print(Fore.YELLOW + \"Modality............:\",Style.RESET_ALL, dataset.Modality)\n    print(Fore.GREEN + \"Body Part Examined..:\",Style.RESET_ALL, dataset.BodyPartExamined)\n    \n    if 'PixelData' in dataset:\n        rows = int(dataset.Rows)\n        cols = int(dataset.Columns)\n        print(Fore.BLUE + \"Image size.......:\",Style.RESET_ALL,\" {rows:d} x {cols:d}, {size:d} bytes\".format(\n            rows=rows, cols=cols, size=len(dataset.PixelData)))\n        if 'PixelSpacing' in dataset:\n            print(Fore.YELLOW + \"Pixel spacing....:\",Style.RESET_ALL,dataset.PixelSpacing)\n            dataset.PixelSpacing = [1, 1]\n        plt.figure(figsize=(10, 10))\n        plt.imshow(dataset.pixel_array, cmap='gray')\n        plt.show()\nfor file_path in glob.glob('../input/osic-pulmonary-fibrosis-progression/train/*/*.dcm'):\n    dataset = pydicom.dcmread(file_path)\n    show_dcm_info(dataset)\n    break # Comment this out to see all","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-07T07:45:47.985428Z","iopub.execute_input":"2023-04-07T07:45:47.98581Z","iopub.status.idle":"2023-04-07T07:45:48.722587Z","shell.execute_reply.started":"2023-04-07T07:45:47.985751Z","shell.execute_reply":"2023-04-07T07:45:48.721614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/yeayates21/osic-simple-image-eda\n\nimdir = \"/kaggle/input/osic-pulmonary-fibrosis-progression/train/ID00123637202217151272140\"\nprint(\"total images for patient ID00123637202217151272140: \", len(os.listdir(imdir)))\n\n# view first (columns*rows) images in order\nfig=plt.figure(figsize=(12, 12))\ncolumns = 4\nrows = 5\nimglist = os.listdir(imdir)\nfor i in range(1, columns*rows +1):\n    filename = imdir + \"/\" + str(i) + \".dcm\"\n    ds = pydicom.dcmread(filename)\n    fig.add_subplot(rows, columns, i)\n    plt.imshow(ds.pixel_array, cmap='gray')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:48.723863Z","iopub.execute_input":"2023-04-07T07:45:48.7242Z","iopub.status.idle":"2023-04-07T07:45:53.574527Z","shell.execute_reply.started":"2023-04-07T07:45:48.72416Z","shell.execute_reply":"2023-04-07T07:45:53.573485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/yeayates21/osic-simple-image-eda\n\nimdir = \"/kaggle/input/osic-pulmonary-fibrosis-progression/train/ID00123637202217151272140\"\nprint(\"total images for patient ID00123637202217151272140: \", len(os.listdir(imdir)))\n\n# view first (columns*rows) images in order\nfig=plt.figure(figsize=(12, 12))\ncolumns = 4\nrows = 5\nimglist = os.listdir(imdir)\nfor i in range(1, columns*rows +1):\n    filename = imdir + \"/\" + str(i) + \".dcm\"\n    ds = pydicom.dcmread(filename)\n    fig.add_subplot(rows, columns, i)\n    plt.imshow(ds.pixel_array, cmap='jet')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:53.575922Z","iopub.execute_input":"2023-04-07T07:45:53.576419Z","iopub.status.idle":"2023-04-07T07:45:58.238858Z","shell.execute_reply.started":"2023-04-07T07:45:53.576376Z","shell.execute_reply":"2023-04-07T07:45:58.238049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualization using gif\n* https://www.kaggle.com/danpresil1/dicom-basic-preprocessing-and-visualization","metadata":{}},{"cell_type":"code","source":"apply_resample = False\n\ndef load_scan(path):\n    slices = [pydicom.read_file(path + '/' + s) for s in os.listdir(path)]\n    slices.sort(key = lambda x: float(x.ImagePositionPatient[2]))\n    try:\n        slice_thickness = np.abs(slices[0].ImagePositionPatient[2] - slices[1].ImagePositionPatient[2])\n    except:\n        slice_thickness = np.abs(slices[0].SliceLocation - slices[1].SliceLocation)\n        \n    for s in slices:\n        s.SliceThickness = slice_thickness\n        \n    return slices","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-07T07:45:58.239904Z","iopub.execute_input":"2023-04-07T07:45:58.240194Z","iopub.status.idle":"2023-04-07T07:45:58.249135Z","shell.execute_reply.started":"2023-04-07T07:45:58.24016Z","shell.execute_reply":"2023-04-07T07:45:58.24784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_scan(path):\n    slices = [pydicom.read_file(path + '/' + s) for s in os.listdir(path)]\n    slices.sort(key = lambda x: float(x.ImagePositionPatient[2]))\n    try:\n        slice_thickness = np.abs(slices[0].ImagePositionPatient[2] - slices[1].ImagePositionPatient[2])\n    except:\n        slice_thickness = np.abs(slices[0].SliceLocation - slices[1].SliceLocation)\n        \n    for s in slices:\n        s.SliceThickness = slice_thickness\n        \n    return slices","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-07T07:45:58.250965Z","iopub.execute_input":"2023-04-07T07:45:58.251392Z","iopub.status.idle":"2023-04-07T07:45:58.261226Z","shell.execute_reply.started":"2023-04-07T07:45:58.251349Z","shell.execute_reply":"2023-04-07T07:45:58.260161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_pixels_hu(slices):\n    image = np.stack([s.pixel_array for s in slices])\n    # Convert to int16 (from sometimes int16), \n    # should be possible as values should always be low enough (<32k)\n    image = image.astype(np.int16)\n\n    # Set outside-of-scan pixels to 0\n    # The intercept is usually -1024, so air is approximately 0\n    image[image == -2000] = 0\n    \n    # Convert to Hounsfield units (HU)\n    for slice_number in range(len(slices)):\n        \n        intercept = slices[slice_number].RescaleIntercept\n        slope = slices[slice_number].RescaleSlope\n        \n        if slope != 1:\n            image[slice_number] = slope * image[slice_number].astype(np.float64)\n            image[slice_number] = image[slice_number].astype(np.int16)\n            \n        image[slice_number] += np.int16(intercept)\n    \n    return np.array(image, dtype=np.int16)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-07T07:45:58.26314Z","iopub.execute_input":"2023-04-07T07:45:58.263568Z","iopub.status.idle":"2023-04-07T07:45:58.274344Z","shell.execute_reply.started":"2023-04-07T07:45:58.263524Z","shell.execute_reply":"2023-04-07T07:45:58.273484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_lungwin(img, hu=[-1200., 600.]):\n    lungwin = np.array(hu)\n    newimg = (img-lungwin[0]) / (lungwin[1]-lungwin[0])\n    newimg[newimg < 0] = 0\n    newimg[newimg > 1] = 1\n    newimg = (newimg * 255).astype('uint8')\n    return newimg","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-07T07:45:58.276131Z","iopub.execute_input":"2023-04-07T07:45:58.276457Z","iopub.status.idle":"2023-04-07T07:45:58.289466Z","shell.execute_reply.started":"2023-04-07T07:45:58.276426Z","shell.execute_reply":"2023-04-07T07:45:58.288462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scans = load_scan('../input/osic-pulmonary-fibrosis-progression/train/ID00007637202177411956430/')\nscan_array = set_lungwin(get_pixels_hu(scans))","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:58.29092Z","iopub.execute_input":"2023-04-07T07:45:58.291412Z","iopub.status.idle":"2023-04-07T07:45:58.897573Z","shell.execute_reply.started":"2023-04-07T07:45:58.291376Z","shell.execute_reply":"2023-04-07T07:45:58.896767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Resample to 1mm (An optional step, it may not be relevant to this competition because of the large slice thickness on the z axis)\n\nfrom scipy.ndimage.interpolation import zoom\n\ndef resample(imgs, spacing, new_spacing):\n    new_shape = np.round(imgs.shape * spacing / new_spacing)\n    true_spacing = spacing * imgs.shape / new_shape\n    resize_factor = new_shape / imgs.shape\n    imgs = zoom(imgs, resize_factor, mode='nearest')\n    return imgs, true_spacing, new_shape\n\nspacing_z = (scans[-1].ImagePositionPatient[2] - scans[0].ImagePositionPatient[2]) / len(scans)\n\nif apply_resample:\n    scan_array_resample = resample(scan_array, np.array(np.array([spacing_z, *scans[0].PixelSpacing])), np.array([1.,1.,1.]))[0]","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-07T07:45:58.898885Z","iopub.execute_input":"2023-04-07T07:45:58.899313Z","iopub.status.idle":"2023-04-07T07:45:58.907747Z","shell.execute_reply.started":"2023-04-07T07:45:58.899256Z","shell.execute_reply":"2023-04-07T07:45:58.906625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import imageio\nfrom IPython.display import Image\n\nimageio.mimsave(\"/tmp/gif.gif\", scan_array, duration=0.0001)\nImage(filename=\"/tmp/gif.gif\", format='png')","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:45:58.909617Z","iopub.execute_input":"2023-04-07T07:45:58.910109Z","iopub.status.idle":"2023-04-07T07:46:00.355411Z","shell.execute_reply.started":"2023-04-07T07:45:58.910002Z","shell.execute_reply":"2023-04-07T07:46:00.351554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualization using Animation\n* https://www.kaggle.com/pranavkasela/animating-the-lung-ct-scan","metadata":{}},{"cell_type":"markdown","source":"For animation, I used scan_array in 'visualization using gif' section.","metadata":{}},{"cell_type":"code","source":"import matplotlib.animation as animation\n\nfig = plt.figure()\n\nims = []\nfor image in scan_array:\n    im = plt.imshow(image, animated=True, cmap=\"Greys\")\n    plt.axis(\"off\")\n    ims.append([im])\n\nani = animation.ArtistAnimation(fig, ims, interval=100, blit=False,\n                                repeat_delay=1000)\n","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-04-07T07:46:00.357256Z","iopub.execute_input":"2023-04-07T07:46:00.357704Z","iopub.status.idle":"2023-04-07T07:46:00.965272Z","shell.execute_reply.started":"2023-04-07T07:46:00.35767Z","shell.execute_reply":"2023-04-07T07:46:00.963905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"HTML(ani.to_jshtml())","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:46:00.966863Z","iopub.execute_input":"2023-04-07T07:46:00.967322Z","iopub.status.idle":"2023-04-07T07:46:03.828811Z","shell.execute_reply.started":"2023-04-07T07:46:00.967287Z","shell.execute_reply":"2023-04-07T07:46:03.827179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"HTML(ani.to_html5_video())","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:46:03.830587Z","iopub.execute_input":"2023-04-07T07:46:03.830973Z","iopub.status.idle":"2023-04-07T07:46:06.202133Z","shell.execute_reply.started":"2023-04-07T07:46:03.830937Z","shell.execute_reply":"2023-04-07T07:46:06.201279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 7. <a id='extract'>Extracting DIOCOM files information in a dataframe 🌊</a>","metadata":{}},{"cell_type":"code","source":"def extract_dicom_meta_data(filename: str) -> Dict:\n    # Load image\n    \n    image_data = pydicom.read_file(filename)\n    img=np.array(image_data.pixel_array).flatten()\n    row = {\n        'Patient': image_data.PatientID,\n        'body_part_examined': image_data.BodyPartExamined,\n        'image_position_patient': image_data.ImagePositionPatient,\n        'image_orientation_patient': image_data.ImageOrientationPatient,\n        'photometric_interpretation': image_data.PhotometricInterpretation,\n        'rows': image_data.Rows,\n        'columns': image_data.Columns,\n        'pixel_spacing': image_data.PixelSpacing,\n        'window_center': image_data.WindowCenter,\n        'window_width': image_data.WindowWidth,\n        'modality': image_data.Modality,\n        'StudyInstanceUID': image_data.StudyInstanceUID,\n        'SeriesInstanceUID': image_data.StudyInstanceUID,\n        'StudyID': image_data.StudyInstanceUID, \n        'SamplesPerPixel': image_data.SamplesPerPixel,\n        'BitsAllocated': image_data.BitsAllocated,\n        'BitsStored': image_data.BitsStored,\n        'HighBit': image_data.HighBit,\n        'PixelRepresentation': image_data.PixelRepresentation,\n        'RescaleIntercept': image_data.RescaleIntercept,\n        'RescaleSlope': image_data.RescaleSlope,\n        'img_min': np.min(img),\n        'img_max': np.max(img),\n        'img_mean': np.mean(img),\n        'img_std': np.std(img)}\n\n    return row","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:46:06.203495Z","iopub.execute_input":"2023-04-07T07:46:06.203911Z","iopub.status.idle":"2023-04-07T07:46:06.214376Z","shell.execute_reply.started":"2023-04-07T07:46:06.203876Z","shell.execute_reply":"2023-04-07T07:46:06.213232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_image_path = '/kaggle/input/osic-pulmonary-fibrosis-progression/train'\ntrain_image_files = glob.glob(os.path.join(train_image_path, '*', '*.dcm'))\n\nmeta_data_df = []\nfor filename in tqdm.tqdm(train_image_files):\n    try:\n        meta_data_df.append(extract_dicom_meta_data(filename))\n    except Exception as e:\n        continue","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:46:06.215846Z","iopub.execute_input":"2023-04-07T07:46:06.216424Z","iopub.status.idle":"2023-04-07T07:58:09.864788Z","shell.execute_reply.started":"2023-04-07T07:46:06.216378Z","shell.execute_reply":"2023-04-07T07:58:09.863832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert to a pd.DataFrame from dict\nmeta_data_df = pd.DataFrame.from_dict(meta_data_df)\nmeta_data_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:58:09.866108Z","iopub.execute_input":"2023-04-07T07:58:09.866501Z","iopub.status.idle":"2023-04-07T07:58:10.301338Z","shell.execute_reply.started":"2023-04-07T07:58:09.866468Z","shell.execute_reply":"2023-04-07T07:58:10.300363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The code below still makes sense, so I leave it.","metadata":{}},{"cell_type":"code","source":"# source: https://www.kaggle.com/c/siim-isic-melanoma-classification/discussion/154658\nfolder='train'\nPATH='../input/osic-pulmonary-fibrosis-progression/'\n\nlast_index = 2\n\ncolumn_names = ['image_name', 'dcm_ImageOrientationPatient', \n                'dcm_ImagePositionPatient', 'dcm_PatientID',\n                'dcm_PatientName', 'dcm_PatientSex'\n                'dcm_rows', 'dcm_columns']\n\ndef extract_DICOM_attributes(folder):\n    patients_folder = list(os.listdir(os.path.join(PATH, folder)))\n    df = pd.DataFrame()\n    \n    i = 0\n    \n    for patient_id in patients_folder:\n   \n        img_path = os.path.join(PATH, folder, patient_id)\n        \n        print(img_path)\n        \n        images = list(os.listdir(img_path))\n        \n        #df = pd.DataFrame()\n\n        for image in images:\n            image_name = image.split(\".\")[0]\n\n            dicom_file_path = os.path.join(img_path,image)\n            dicom_file_dataset = pydicom.read_file(dicom_file_path)\n                \n            '''\n            print(dicom_file_dataset.dir(\"pat\"))\n            print(dicom_file_dataset.data_element(\"ImageOrientationPatient\"))\n            print(dicom_file_dataset.data_element(\"ImagePositionPatient\"))\n            print(dicom_file_dataset.data_element(\"PatientID\"))\n            print(dicom_file_dataset.data_element(\"PatientName\"))\n            print(dicom_file_dataset.data_element(\"PatientSex\"))\n            '''\n            \n            imageOrientationPatient = dicom_file_dataset.ImageOrientationPatient\n            #imagePositionPatient = dicom_file_dataset.ImagePositionPatient\n            patientID = dicom_file_dataset.PatientID\n            patientName = dicom_file_dataset.PatientName\n            patientSex = dicom_file_dataset.PatientSex\n        \n            rows = dicom_file_dataset.Rows\n            cols = dicom_file_dataset.Columns\n            \n            #print(rows)\n            #print(columns)\n            \n            temp_dict = {'image_name': image_name, \n                                    'dcm_ImageOrientationPatient': imageOrientationPatient,\n                                    #'dcm_ImagePositionPatient':imagePositionPatient,\n                                    'dcm_PatientID': patientID, \n                                    'dcm_PatientName': patientName,\n                                    'dcm_PatientSex': patientSex,\n                                    'dcm_rows': rows,\n                                    'dcm_columns': cols}\n\n\n            df = df.append([temp_dict])\n            \n        i += 1\n        \n        if i == last_index:\n            break\n            \n    return df\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-07T07:58:10.302926Z","iopub.execute_input":"2023-04-07T07:58:10.303265Z","iopub.status.idle":"2023-04-07T07:58:10.31715Z","shell.execute_reply.started":"2023-04-07T07:58:10.303233Z","shell.execute_reply":"2023-04-07T07:58:10.316135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"extract_DICOM_attributes('train')","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:58:10.319061Z","iopub.execute_input":"2023-04-07T07:58:10.319439Z","iopub.status.idle":"2023-04-07T07:58:23.385657Z","shell.execute_reply.started":"2023-04-07T07:58:10.319406Z","shell.execute_reply":"2023-04-07T07:58:23.384738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas_profiling as pdp","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:58:23.387114Z","iopub.execute_input":"2023-04-07T07:58:23.387424Z","iopub.status.idle":"2023-04-07T07:58:24.342395Z","shell.execute_reply.started":"2023-04-07T07:58:23.387395Z","shell.execute_reply":"2023-04-07T07:58:24.341399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/train.csv')\ntest_df = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/test.csv')","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:58:24.343907Z","iopub.execute_input":"2023-04-07T07:58:24.344233Z","iopub.status.idle":"2023-04-07T07:58:24.363936Z","shell.execute_reply.started":"2023-04-07T07:58:24.344203Z","shell.execute_reply":"2023-04-07T07:58:24.362854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"profile_train_df = pdp.ProfileReport(train_df)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-04-07T07:58:24.365718Z","iopub.execute_input":"2023-04-07T07:58:24.367225Z","iopub.status.idle":"2023-04-07T07:58:37.711508Z","shell.execute_reply.started":"2023-04-07T07:58:24.367175Z","shell.execute_reply":"2023-04-07T07:58:37.7102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"profile_train_df","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:58:37.713683Z","iopub.execute_input":"2023-04-07T07:58:37.714004Z","iopub.status.idle":"2023-04-07T07:58:38.796364Z","shell.execute_reply.started":"2023-04-07T07:58:37.713969Z","shell.execute_reply":"2023-04-07T07:58:38.79545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"profile_test_df = pdp.ProfileReport(test_df)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-04-07T07:58:38.79818Z","iopub.execute_input":"2023-04-07T07:58:38.798538Z","iopub.status.idle":"2023-04-07T07:58:43.414002Z","shell.execute_reply.started":"2023-04-07T07:58:38.798505Z","shell.execute_reply":"2023-04-07T07:58:43.412912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"profile_test_df","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:58:43.415331Z","iopub.execute_input":"2023-04-07T07:58:43.415604Z","iopub.status.idle":"2023-04-07T07:58:43.728866Z","shell.execute_reply.started":"2023-04-07T07:58:43.415578Z","shell.execute_reply":"2023-04-07T07:58:43.72794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}