{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# import ipympl\nimport matplotlib.pyplot as plt\nimport matplotlib\nimport pydicom\nimport pandas_profiling as pdp\n\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"TRAIN_CSV_PATH = '/kaggle/input/osic-pulmonary-fibrosis-progression/train.csv'\nTEST_CSV_PATH = '/kaggle/input/osic-pulmonary-fibrosis-progression/test.csv'\n\nTRAIN_DICOM_DIR = '/kaggle/input/osic-pulmonary-fibrosis-progression/train/'\nTEST_DICOM_DIR = '/kaggle/input/osic-pulmonary-fibrosis-progression/test/'\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.read_csv(TRAIN_CSV_PATH)\ndf.head(4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Unique patients\ndf['Patient'].nunique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df['Sex'].value_counts().plot(kind='bar')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Histogram of Age\ndf.groupby('Sex')['Age'].plot(kind='hist', legend=True, alpha=0.5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Histogram of measurements/patient\ndf['Patient'].value_counts().plot(kind='hist', bins=np.arange(df['Patient'].value_counts().max()+2))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Histogram of FVC values\ndf['FVC'].plot(kind='hist')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.groupby( ['Sex','SmokingStatus'] )['FVC'].agg( ['mean','std','count'] )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"i = df.Patient == 'ID00007637202177411956430'\ndf[i][['Weeks', 'FVC']].plot(kind='line', x='Weeks', y='FVC', style=['d-'])\nplt.title(f'Patient {df[i][\"Patient\"][0]}:\\n{df[i][\"Sex\"][0]} / {df[i][\"Age\"][0]} / {df[i][\"SmokingStatus\"][0]}')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Generate report (https://www.kaggle.com/piantic/osic-pulmonary-fibrosis-progression-basic-eda)\nreport = pdp.ProfileReport(df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"report","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## DICOM data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"series = np.array(\n    [\n        [\n            (os.path.join(dp, f), pydicom.dcmread(os.path.join(dp, f), stop_before_pixels = True))\n            for f in files\n        ]\n        for dp,_,files in os.walk(TRAIN_DICOM_DIR) if len(files) != 0\n    ]\n)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"series[0][0][1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Total files\ninstances = [f for l in series for f in l]\nlen(instances)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Patients\npatient_ids = np.unique([inst[1].PatientID for inst in instances])\nlen(patient_ids)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Series (3D volumes)\nlen(series)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# How many studies?\nstudies = {}\n\nfor s in series:\n    studies.setdefault(s[0][1].StudyInstanceUID, []).append(s)\n\nlen(studies)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Studies per patient\n[len([st for st in studies.values() if st[0][0][1].PatientID == p]) for p in patient_ids]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Series per study\nseries_per_study = [(len(sr), sr[0][0][1].PatientID) for sr in studies.values()]\nseries_per_study","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"So, one CT per patient","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Images per series\nimg_per_series = [len(s) for s in series]\nprint(img_per_series)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"res = {}\nspc = {}\nthck = {}\n\nfor sr in series:\n    try:\n        sr.sort(key = lambda inst: int(inst[1].ImagePositionPatient[2]))\n    except:\n        sr.sort(key = lambda inst: int(inst[1].InstanceNumber))\n    \n    dcm = sr[0][1]\n    key = str(dcm.PixelSpacing)\n    spc.setdefault(key, [])\n    spc[key].append((dcm.PatientID, ))#dcm.StudyDescription, dcm.StudyDate, dcm.SeriesDescription))\n    \n    key = str((dcm.Rows, dcm.Columns))\n    res.setdefault(key, [])\n    res[key].append((dcm.PatientID, ))#dcm.StudyDescription, dcm.StudyDate, dcm.SeriesDescription))\n    \n    try:\n        key = str(np.abs(sr[0][1].ImagePositionPatient[2] - sr[1][1].ImagePositionPatient[2]))\n        thck.setdefault(key, [])\n        thck[key].append((dcm.PatientID, ))#dcm.StudyDescription, dcm.StudyDate, dcm.SeriesDescription))\n    except:\n        print(dcm.PatientID)\n        continue\n    \n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"thck.keys()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"res.keys()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"spc.keys()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"res","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for key in res.keys():\n    print(f'{key}:{len(res[key])}')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Possible outliers","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"for key in ['(1302, 1302)', '(843, 888)', '(632, 632)', '(1100, 888)', '(788, 888)', '(734, 888)', '(752, 888)', '(733, 888)']:\n    print(f'{key}: {res[key]}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"spc","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for key in spc.keys():\n    print(f'{key}:{len(spc[key])}')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Visualize some images:","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"seq1 = r\"ID00210637202257228694086\"\nseq1_slices = [pydicom.dcmread(os.path.join(TRAIN_DICOM_DIR, seq1, f)) for f in os.listdir(os.path.join(TRAIN_DICOM_DIR, seq1))]\nseq1_slices.sort(key = lambda inst: int(inst.ImagePositionPatient[2]))\n\nseq2 = r\"ID00132637202222178761324\"\nseq2_slices = [pydicom.dcmread(os.path.join(TRAIN_DICOM_DIR, seq2, f)) for f in os.listdir(os.path.join(TRAIN_DICOM_DIR, seq2))]\ntry:\n    seq2_slices.sort(key = lambda inst: int(inst.ImagePositionPatient[2]))\nexcept:\n    seq2_slices.sort(key = lambda inst: int(inst.InstanceNumber))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"seq1 = np.stack([s.pixel_array for s in seq1_slices])\nseq2 = np.stack([s.pixel_array for s in seq2_slices])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"[ipp.ImagePositionPatient for ipp in seq1_slices]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"seq1.shape, seq2.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.imshow(seq1[150,:,:], cmap='gray')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.imshow(seq2[150,:,:], cmap='gray')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}