{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-22T15:36:48.252258Z","iopub.execute_input":"2022-12-22T15:36:48.253173Z","iopub.status.idle":"2022-12-22T15:37:05.828433Z","shell.execute_reply.started":"2022-12-22T15:36:48.253133Z","shell.execute_reply":"2022-12-22T15:37:05.827086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"directory = '/kaggle/input/osic-pulmonary-fibrosis-progression/'\ntrain_df = pd.read_csv(directory + '/train.csv')\ntest_df = pd.read_csv(directory + '/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-12-22T15:37:05.830841Z","iopub.execute_input":"2022-12-22T15:37:05.831485Z","iopub.status.idle":"2022-12-22T15:37:05.866262Z","shell.execute_reply.started":"2022-12-22T15:37:05.831437Z","shell.execute_reply":"2022-12-22T15:37:05.865111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['Patient'].count()","metadata":{"execution":{"iopub.status.busy":"2022-12-22T15:37:05.867894Z","iopub.execute_input":"2022-12-22T15:37:05.868638Z","iopub.status.idle":"2022-12-22T15:37:05.888099Z","shell.execute_reply.started":"2022-12-22T15:37:05.868590Z","shell.execute_reply":"2022-12-22T15:37:05.886790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['Patient'].count()","metadata":{"execution":{"iopub.status.busy":"2022-12-22T15:37:05.892791Z","iopub.execute_input":"2022-12-22T15:37:05.893209Z","iopub.status.idle":"2022-12-22T15:37:05.900786Z","shell.execute_reply.started":"2022-12-22T15:37:05.893174Z","shell.execute_reply":"2022-12-22T15:37:05.899625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2022-12-22T15:37:05.902457Z","iopub.execute_input":"2022-12-22T15:37:05.903373Z","iopub.status.idle":"2022-12-22T15:37:05.932817Z","shell.execute_reply.started":"2022-12-22T15:37:05.903337Z","shell.execute_reply":"2022-12-22T15:37:05.931603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pathlib import Path\nlen(list(Path(directory+'/train/').rglob(\"*\")))","metadata":{"execution":{"iopub.status.busy":"2022-12-22T15:37:05.934824Z","iopub.execute_input":"2022-12-22T15:37:05.935301Z","iopub.status.idle":"2022-12-22T15:37:06.512538Z","shell.execute_reply.started":"2022-12-22T15:37:05.935256Z","shell.execute_reply":"2022-12-22T15:37:06.511078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(list(Path(directory+'/test/').rglob(\"*\")))","metadata":{"execution":{"iopub.status.busy":"2022-12-22T15:37:06.514178Z","iopub.execute_input":"2022-12-22T15:37:06.515017Z","iopub.status.idle":"2022-12-22T15:37:06.539193Z","shell.execute_reply.started":"2022-12-22T15:37:06.514969Z","shell.execute_reply":"2022-12-22T15:37:06.538015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2022-12-22T15:37:06.540832Z","iopub.execute_input":"2022-12-22T15:37:06.541195Z","iopub.status.idle":"2022-12-22T15:37:06.555555Z","shell.execute_reply.started":"2022-12-22T15:37:06.541163Z","shell.execute_reply":"2022-12-22T15:37:06.554293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isna().mean()","metadata":{"execution":{"iopub.status.busy":"2022-12-22T15:37:06.557358Z","iopub.execute_input":"2022-12-22T15:37:06.557914Z","iopub.status.idle":"2022-12-22T15:37:06.569042Z","shell.execute_reply.started":"2022-12-22T15:37:06.557874Z","shell.execute_reply":"2022-12-22T15:37:06.567696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.isna().mean()","metadata":{"execution":{"iopub.status.busy":"2022-12-22T15:37:06.572969Z","iopub.execute_input":"2022-12-22T15:37:06.573320Z","iopub.status.idle":"2022-12-22T15:37:06.584180Z","shell.execute_reply.started":"2022-12-22T15:37:06.573288Z","shell.execute_reply":"2022-12-22T15:37:06.582866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['Patient'].nunique()","metadata":{"execution":{"iopub.status.busy":"2022-12-22T15:37:06.585570Z","iopub.execute_input":"2022-12-22T15:37:06.586520Z","iopub.status.idle":"2022-12-22T15:37:06.598844Z","shell.execute_reply.started":"2022-12-22T15:37:06.586469Z","shell.execute_reply":"2022-12-22T15:37:06.597898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['Patient'].nunique()","metadata":{"execution":{"iopub.status.busy":"2022-12-22T15:37:06.601746Z","iopub.execute_input":"2022-12-22T15:37:06.602091Z","iopub.status.idle":"2022-12-22T15:37:06.608633Z","shell.execute_reply.started":"2022-12-22T15:37:06.602062Z","shell.execute_reply":"2022-12-22T15:37:06.607842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-12-22T15:37:06.610140Z","iopub.execute_input":"2022-12-22T15:37:06.610696Z","iopub.status.idle":"2022-12-22T15:37:06.621278Z","shell.execute_reply.started":"2022-12-22T15:37:06.610663Z","shell.execute_reply":"2022-12-22T15:37:06.620200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['Weeks'].describe()","metadata":{"execution":{"iopub.status.busy":"2022-12-22T15:37:06.622776Z","iopub.execute_input":"2022-12-22T15:37:06.623557Z","iopub.status.idle":"2022-12-22T15:37:06.639383Z","shell.execute_reply.started":"2022-12-22T15:37:06.623522Z","shell.execute_reply":"2022-12-22T15:37:06.638166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['Sex'].hist(bins=4)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T15:37:06.640892Z","iopub.execute_input":"2022-12-22T15:37:06.641484Z","iopub.status.idle":"2022-12-22T15:37:06.886981Z","shell.execute_reply.started":"2022-12-22T15:37:06.641440Z","shell.execute_reply":"2022-12-22T15:37:06.885863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['SmokingStatus'].hist(bins=10)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T15:37:06.889881Z","iopub.execute_input":"2022-12-22T15:37:06.890636Z","iopub.status.idle":"2022-12-22T15:37:07.102650Z","shell.execute_reply.started":"2022-12-22T15:37:06.890597Z","shell.execute_reply":"2022-12-22T15:37:07.101865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.groupby('Sex')['Age'].hist(bins=20,histtype='step')","metadata":{"execution":{"iopub.status.busy":"2022-12-22T15:37:07.104204Z","iopub.execute_input":"2022-12-22T15:37:07.104508Z","iopub.status.idle":"2022-12-22T15:37:07.355133Z","shell.execute_reply.started":"2022-12-22T15:37:07.104479Z","shell.execute_reply":"2022-12-22T15:37:07.352809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.groupby('Sex')['Age'].hist(bins=10,histtype='step')","metadata":{"execution":{"iopub.status.busy":"2022-12-22T15:37:07.356853Z","iopub.execute_input":"2022-12-22T15:37:07.357419Z","iopub.status.idle":"2022-12-22T15:37:07.590680Z","shell.execute_reply.started":"2022-12-22T15:37:07.357385Z","shell.execute_reply":"2022-12-22T15:37:07.589722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['FVC'].describe()","metadata":{"execution":{"iopub.status.busy":"2022-12-22T15:37:07.591865Z","iopub.execute_input":"2022-12-22T15:37:07.592470Z","iopub.status.idle":"2022-12-22T15:37:07.605282Z","shell.execute_reply.started":"2022-12-22T15:37:07.592434Z","shell.execute_reply":"2022-12-22T15:37:07.603839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['Percent'].describe()","metadata":{"execution":{"iopub.status.busy":"2022-12-22T15:37:07.606943Z","iopub.execute_input":"2022-12-22T15:37:07.607289Z","iopub.status.idle":"2022-12-22T15:37:07.624534Z","shell.execute_reply.started":"2022-12-22T15:37:07.607259Z","shell.execute_reply":"2022-12-22T15:37:07.623266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.groupby('SmokingStatus')['Age'].hist(bins=10,histtype='step')","metadata":{"execution":{"iopub.status.busy":"2022-12-22T15:37:07.625799Z","iopub.execute_input":"2022-12-22T15:37:07.626196Z","iopub.status.idle":"2022-12-22T15:37:07.865382Z","shell.execute_reply.started":"2022-12-22T15:37:07.626162Z","shell.execute_reply":"2022-12-22T15:37:07.864179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from typing import Dict\n\ndef extract_dicom_meta_data(filename: str) -> Dict:\n    # Load image\n    \n    image_data = pydicom.read_file(filename)\n    img=np.array(image_data.pixel_array).flatten()\n    row = {\n        'Patient': image_data.PatientID,\n        'body_part_examined': image_data.BodyPartExamined,\n        'image_position_patient': image_data.ImagePositionPatient,\n        'image_orientation_patient': image_data.ImageOrientationPatient,\n        'photometric_interpretation': image_data.PhotometricInterpretation,\n        'rows': image_data.Rows,\n        'columns': image_data.Columns,\n        'pixel_spacing': image_data.PixelSpacing,\n        'window_center': image_data.WindowCenter,\n        'window_width': image_data.WindowWidth,\n        'modality': image_data.Modality,\n        'StudyInstanceUID': image_data.StudyInstanceUID,\n        'SeriesInstanceUID': image_data.StudyInstanceUID,\n        'StudyID': image_data.StudyInstanceUID, \n        'SamplesPerPixel': image_data.SamplesPerPixel,\n        'BitsAllocated': image_data.BitsAllocated,\n        'BitsStored': image_data.BitsStored,\n        'HighBit': image_data.HighBit,\n        'PixelRepresentation': image_data.PixelRepresentation,\n        'RescaleIntercept': image_data.RescaleIntercept,\n        'RescaleSlope': image_data.RescaleSlope,\n        'img_min': np.min(img),\n        'img_max': np.max(img),\n        'img_mean': np.mean(img),\n        'img_std': np.std(img)}\n\n    return row","metadata":{"execution":{"iopub.status.busy":"2022-12-22T15:37:07.866796Z","iopub.execute_input":"2022-12-22T15:37:07.867209Z","iopub.status.idle":"2022-12-22T15:37:07.878280Z","shell.execute_reply.started":"2022-12-22T15:37:07.867174Z","shell.execute_reply":"2022-12-22T15:37:07.876687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nimport tqdm \nimport pydicom\n\ntrain_image_files = glob.glob(os.path.join(directory, 'train/', '*', '*.dcm'))\n\nmeta_data_df = []\nfor filename in tqdm.tqdm(train_image_files):\n    try:\n        meta_data_df.append(extract_dicom_meta_data(filename))\n    except Exception as e:\n        print(e)\n        continue","metadata":{"execution":{"iopub.status.busy":"2022-12-22T15:37:07.879682Z","iopub.execute_input":"2022-12-22T15:37:07.880360Z","iopub.status.idle":"2022-12-22T15:48:58.254397Z","shell.execute_reply.started":"2022-12-22T15:37:07.880324Z","shell.execute_reply":"2022-12-22T15:48:58.253012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_data_df = pd.DataFrame.from_dict(meta_data_df)\nmeta_data_df","metadata":{"execution":{"iopub.status.busy":"2022-12-22T15:48:58.256624Z","iopub.execute_input":"2022-12-22T15:48:58.257985Z","iopub.status.idle":"2022-12-22T15:48:58.717923Z","shell.execute_reply.started":"2022-12-22T15:48:58.257945Z","shell.execute_reply":"2022-12-22T15:48:58.716825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = \"/kaggle/input/osic-pulmonary-fibrosis-progression/train/ID00020637202178344345685/284.dcm\"\ndataset = pydicom.dcmread(path)\n\n\nplt.figure(figsize = (16, 9))\nplt.imshow(dataset.pixel_array, cmap=\"gray\")","metadata":{"execution":{"iopub.status.busy":"2022-12-22T15:48:59.749655Z","iopub.execute_input":"2022-12-22T15:48:59.750212Z","iopub.status.idle":"2022-12-22T15:49:00.116294Z","shell.execute_reply.started":"2022-12-22T15:48:59.750171Z","shell.execute_reply":"2022-12-22T15:49:00.115269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_image(dataframe, n=5, rows=1, cols=5):\n    plt.figure(figsize=(16, 9))\n    \n    for i, path in enumerate(dataframe['/kaggle/input/osic-pulmonary-fibrosis-progression/train/ID00020637202178344345685/284.dcm'][:n]):\n        image = pydicom.read_file(path)\n        image = image.pixel_array\n        \n        plt.subplot(rows, cols, i+1)\n        plt.imshow(image)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T15:50:34.556273Z","iopub.execute_input":"2022-12-22T15:50:34.556668Z","iopub.status.idle":"2022-12-22T15:50:34.563993Z","shell.execute_reply.started":"2022-12-22T15:50:34.556635Z","shell.execute_reply":"2022-12-22T15:50:34.562740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imdir = \"/kaggle/input/osic-pulmonary-fibrosis-progression/train/ID00123637202217151272140\"\n\nfig=plt.figure(figsize=(16, 9))\ncolumns = 3\nrows = 2\nimglist = os.listdir(imdir)\nfor i in range(1, columns*rows +1):\n    filename = imdir + \"/\" + str(i) + \".dcm\"\n    ds = pydicom.dcmread(filename)\n    fig.add_subplot(rows, columns, i)\n    plt.imshow(ds.pixel_array, cmap='gray')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-22T15:49:28.179958Z","iopub.execute_input":"2022-12-22T15:49:28.180389Z","iopub.status.idle":"2022-12-22T15:49:29.145738Z","shell.execute_reply.started":"2022-12-22T15:49:28.180344Z","shell.execute_reply":"2022-12-22T15:49:29.144540Z"},"trusted":true},"execution_count":null,"outputs":[]}]}