{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Source for basic EDA and dicom processing: https://www.kaggle.com/piantic/osic-pulmonary-fibrosis-progression-basic-eda\n\n### Author of source: Heroseo","execution_count":null},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import os\nfrom os import listdir\nimport pandas as pd\nimport numpy as np\nimport glob\nimport tqdm\nfrom typing import Dict\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\n#plotly\n!pip install chart_studio\nimport plotly.express as px\nimport chart_studio.plotly as py\nimport plotly.graph_objs as go\nfrom plotly.offline import iplot\nimport cufflinks\ncufflinks.go_offline()\ncufflinks.set_config_file(world_readable=True,theme='pearl')\n\n#color\nfrom colorama import Fore, Back, Style\nimport seaborn as sns\nsns.set(style='whitegrid')\n\n#supress warnings\nimport warnings\nwarnings.filterwarnings('ignore')\n\n#settings for pretty nice plots\nplt.style.use('fivethirtyeight')\nplt.show()\n\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Reading the train.csv","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"#List files available\nlist(os.listdir(\"../input/osic-pulmonary-fibrosis-progression\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"IMAGE_PATH = \"../input/osic-pulmonary-fibrosis-progression/\"\ntrain_df = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/train.csv')\ntest_df = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/test.csv')\nprint(Fore.YELLOW + 'Training data shape: ',Style.RESET_ALL,train_df.shape)\ntrain_df.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.groupby(['SmokingStatus']).count()['Sex'].to_frame()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Basic Data Exploration","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"## General Info","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"#Null values and Data types\nprint(Fore.YELLOW + 'Train set !!',Style.RESET_ALL)\nprint(train_df.info())\nprint('-----------------')\nprint(Fore.BLUE + 'Test set !!',Style.RESET_ALL)\nprint(test_df.info())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Missing values","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Total number of patients in the dataset(train+test)\nprint(Fore.YELLOW + \"Total Patients in Train Set: \",Style.RESET_ALL,train_df['Patient'].count())\nprint(Fore.BLUE + \"Total Patients in Test set: \",Style.RESET_ALL,test_df['Patient'].count())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Unique Patients(IDs)","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"print(Fore.YELLOW + \"The total patient IDs are \",Style.RESET_ALL,f\"{train_df['Patient'].count()},\",\n      Fore.BLUE + \"from those the unique IDs are\",Style.RESET_ALL,f\"{train_df['Patient'].value_counts().shape[0]}.\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_patient_ids = set(train_df['Patient'].unique())\ntest_patient_ids = set(test_df['Patient'].unique())\n\ntrain_patient_ids.intersection(test_patient_ids)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"columns = train_df.keys()\ncolumns = list(columns)\nprint(columns)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['Patient'].value_counts().max()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df['Patient'].value_counts().max()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Number of Patients and Images in Training Images Folder","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"files = folders = 0\npath = \"/kaggle/input/osic-pulmonary-fibrosis-progression/train\"\n\nfor _, dirnames, filenames in os.walk(path):\n    # ^ this idion means \"we won't be using this value\"\n    files += len(filenames)\n    folders += len(dirnames)\n#print (Fore.YELLOW + \"Total Patients in Train set: \",Style.RESET_ALL,train_df['Patient'].count())\nprint(Fore.YELLOW + f'{files:,}',Style.RESET_ALL,\"files/images \" + Fore.BLUE + f'{folders:,}',Style.RESET_ALL,'folder/patients')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"files = []\nfor _,dirnames,filenames in os.walk(path):\n    files.append(len(filenames))\nprint(Fore.YELLOW +f'{round(np.mean(files)):,}',Style.RESET_ALL,'average files/images per patient')\nprint(Fore.BLUE + f'{round(np.max(files)):,}',Style.RESET_ALL,'max files/images per patient')\nprint(Fore.GREEN + f'{round(np.min(files)):,}',Style.RESET_ALL,'min files/images per patient')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Data Exploration in Details","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"## Creating Individual Patient Dataframe","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_df = train_df[['Patient','Age','Sex','SmokingStatus']].drop_duplicates()\npatient_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Creating unique patient lists and their properties\ntrain_dir = '../input/osic-pulmonary-fibrosis-progression/train/'\ntest_dir = '../input/osic-pulmonary-fibrosis-progression/test/'\n\npatient_ids = os.listdir(train_dir)\npatient_ids = sorted(patient_ids)\n\n#Creating new rows\nno_of_instances = []\nage = []\nsex = []\nsmoking_status = []\n\nfor patient_id in patient_ids:\n    patient_info = train_df[train_df['Patient'] == patient_id].reset_index()\n    no_of_instances.append(len(os.listdir(train_dir + patient_id)))\n    age.append(patient_info['Age'][0])\n    sex.append(patient_info['Sex'][0])\n    smoking_status.append(patient_info['SmokingStatus'][0])\n\n#Creating the dataframe for the patient info\npatient_df = pd.DataFrame(list(zip(patient_ids,no_of_instances,age,sex,smoking_status)),\n                         columns = ['Patient','no_of_instances','Age','Sex','SmokingStatus'])\nprint(patient_df.info())\npatient_df.head()\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Exploring 'SmokingStatus' Column","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_df['SmokingStatus'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_df['SmokingStatus'].value_counts().iplot(kind='bar',yTitle='Counts',linecolor='black',opacity=0.7,color='blue',theme='pearl',bargap=0.5,gridcolor='white',title='Distribution of the SmokingStatus column in the Unique Patient Set')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Weeks distribution","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['Weeks'].value_counts().iplot(kind = 'barh',xTitle = 'Counts(Weeks)',linecolor='black',opacity=0.7,color='#FB8072',theme='pearl',bargap=0.2,gridcolor='white',title='Distribution of the Weeks in the training set')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['Weeks'].iplot(kind='hist',\n                       xTitle='Weeks',\n                       yTitle='Counts',\n                       linecolor='black',\n                       opacity=0.7,\n                       color='#FB8072',\n                       theme='PEARL',\n                       bargap=0.2,\n                       gridcolor='white',\n                       title='Distribution of the Weeks in the training set')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Distrubtion Age over Week","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.scatter(train_df,x=\"Weeks\",y=\"Age\",color='Sex')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## FVC - The forced vital capacity","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['FVC'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['FVC'].iplot(kind='hist',\n                     xTitle='Lung Capacity(mL)',\n                     linecolor='black',\n                     opacity=0.8,\n                     color='#FB8072',\n                     bargap=0.5,\n                     gridcolor='white',\n                     title='Distribution of the FVC in the training set')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## FVC vs Percent","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"fig=px.scatter(train_df,x=\"FVC\",y=\"Percent\",color='Age')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## FVC vs Age","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.scatter(train_df,x=\"FVC\",y=\"Age\",color='Sex')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## FVC vs Weeks","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.scatter(train_df,x=\"FVC\",y=\"Weeks\",color='SmokingStatus')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Pick one patient for FVC vs weeks","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"patient = train_df[train_df.Patient == 'ID00228637202259965313869']\nfig = px.line(patient,x=\"Weeks\",y=\"FVC\",color='SmokingStatus')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Percent\n#### A computed field which approximates the patient's FVC as a percent of the typical FVC for a person of similiar characteristics\n","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['Percent'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['Percent'].iplot(kind='hist',bins=30,color='blue',xTitle='Percent distribution',yTitle='Count')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Percent vs SmokingStatus in Patient Dataframe","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"df = train_df\nfig = px.violin(df,y='Percent',x='SmokingStatus',box=True,color='Sex',points=\"all\",\n               hover_data=train_df.columns)\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(16,6))\nax = sns.violinplot(x = train_df['SmokingStatus'],y=train_df['Percent'],palette = 'Reds')\nax.set_xlabel(xlabel='Smoking Habit',fontsize=15)\nax.set_ylabel(ylabel='Percent',fontsize=15)\nax.set_title(label='Distribution of Smoking Status Over Percentage',fontsize=20)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.scatter(train_df,x=\"Age\",y=\"Percent\",color='SmokingStatus')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Pick one patient for FVC vs Weeks","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"patient = train_df[train_df.Patient=='ID00228637202259965313869']\nfig = px.line(patient,x=\"Weeks\",y=\"Percent\",color='SmokingStatus')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Age Distribution of Unique Patients","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_df['Age'].iplot(kind='hist',bins=30,color='red',xTitle='Ages of distribution',yTitle='Count')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Distribution of Age vs SmokingStatus in Patient Dataframe","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_df['SmokingStatus'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(16,6))\nsns.kdeplot(patient_df.loc[patient_df['SmokingStatus']=='Ex-smoker','Age'],label='Ex-smoker',shade=True)\nsns.kdeplot(patient_df.loc[patient_df['SmokingStatus']=='Never smoked','Age'],label='Never smoked',shade=True)\nsns.kdeplot(patient_df.loc[patient_df['SmokingStatus']=='Currently smokes','Age'],label='Currently smokes',shade=True)\n#labeling plot\nplt.xlabel('Age(years)');plt.ylabel('Density');plt.title('Distribution of Ages');\n\nplt.figure(figsize=(16,6))\nax = sns.violinplot(x = patient_df['SmokingStatus'],y=train_df['Age'],palette = 'Reds')\nax.set_xlabel(xlabel='Smoking Habit',fontsize=15)\nax.set_ylabel(ylabel='Age',fontsize=15)\nax.set_title(label='Distribution of Smoking Status Over Age',fontsize=20)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Distribution of Age vs Gender in Patient Dataframe","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(16,6))\nsns.kdeplot(patient_df.loc[patient_df['Sex']=='Male','Age'],label='Male',shade=True)\nsns.kdeplot(patient_df.loc[patient_df['Sex']=='Female','Age'],label='Female',shade=True)\nplt.xlabel('Age (years)');plt.ylabel('Density');plt.title('Distribution of Ages');","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Gender distribution","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_df['Sex'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_df['Sex'].value_counts().iplot(kind='bar',yTitle='Count',linecolor='black',opacity=0.7,color='blue',theme='pearl',bargap=0.8,gridcolor='white',title='Distribution of the Sex column in Patient Dataframe')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Gender vs SmokingStatus in Patient Dataframe","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(16,6))\na = sns.countplot(data=patient_df,x='SmokingStatus',hue='Sex')\nfor p in a.patches:\n    a.annotate(format(p.get_height(),','),\n               (p.get_x() + p.get_width()/2.,\n               p.get_height()),ha='center',va='center',xytext=(0,4),textcoords='offset points')\nplt.title('Gender split by SmokingStatus',fontsize=16)\nsns.despine(left=True,bottom=True)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig=px.box(patient_df,x=\"Sex\",y=\"Age\",points=\"all\")\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_patient_ids = set(train_df['Patient'].unique())\ntest_patient_ids = set(test_df['Patient'].unique())\ntrain_patient_ids.intersection(test_patient_ids)\n\nprint(Fore.YELLOW + \"There are \",Style.RESET_ALL,f\"{train_df['Patient'].value_counts().shape[0]}\",\n      Fore.BLUE + \"unique Patient IDs\",Fore.WHITE + \"in the training set.\")\nprint(Fore.YELLOW + \"There are \",Style.RESET_ALL,f\"{test_df['Patient'].value_counts().shape[0]}\",\n      Fore.BLUE + \"unique Patient IDs\",Fore.WHITE + \"in the test set.\")\nprint(Fore.YELLOW + \"There are \",Style.RESET_ALL,f\"{len(train_patient_ids.intersection(test_patient_ids))}\",\n     Fore.BLUE,\"Patient IDs\",Fore.WHITE + \"in both the training and test sets\")\nprint(Fore.CYAN + \"These patients are in both the training and test data sets:\")\nprint(Fore.WHITE,Style.RESET_ALL,f\"{train_patient_ids.intersection(test_patient_ids)}\")\n\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Heatmap for train.csv","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"corrmat = train_df.corr()\nf,ax = plt.subplots(figsize = (9,8))\nsns.heatmap(corrmat,ax=ax,cmap='RdYlBu_r',linewidths=0.5)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Visualizing Images: DECOM\n\n##### 1 type of images containing the information:\n* .dcm files: DICOM files. It's saved in the \"Digital Imagining and Communications in Medicine\" format. It contains an image from a medical scan, such as an ultrasound or MRI + information about the patient.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"print(Fore.YELLOW + 'Train .dcm number of images:',Style.RESET_ALL,len(list(os.listdir('../input/osic-pulmonary-fibrosis-progression/train'))), '\\n' +\nFore.BLUE + 'Test .dcm number of images:',Style.RESET_ALL,len(list(os.listdir('../input/osic-pulmonary-fibrosis-progression/test'))), '\\n' +\n'-----------------------------------','\\n' +\n'There is the same number of images as in train/test .csv datasets')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_pixel_array(dataset,figsize=(5,5)):\n    plt.figure(figsize=figsize)\n    plt.grid(False)\n    plt.imshow(dataset.pixel_array,cmap='gray')\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import pydicom\nprint(__doc__)\nfilename = '../input/osic-pulmonary-fibrosis-progression/train/ID00123637202217151272140/137.dcm'\ndataset = pydicom.dcmread(filename)\n#Normal mode:\nprint()\nprint(\"Filename.........:\", filename)\npat_name = dataset.PatientName\ndisplay_name = pat_name.family_name\nprint(\"Patient's name...:\", display_name)\nprint(\"Patient id.......:\", dataset.PatientID)\nprint(\"Modality.........:\", dataset.Modality)\nprint(\"Body Part Examined.........:\", dataset.BodyPartExamined)\nif 'PixelData' in dataset:\n    rows = int(dataset.Rows)\n    cols = int(dataset.Columns)\n    print(\"Image size.........: {rows:d} x {cols:d}, {size:d} bytes\".format(rows=rows,cols=cols,size=len(dataset.PixelData)))\n    if 'PixelSpacing' in dataset:\n        print(\"Pixel spacing.........\",dataset.PixelSpacing)\n#use .get() if not sure the item exists, and want a default value if missing\nprint(\"Slice location.........\",dataset.get('SliceLocation',\"(missing)\"))\n#plot the image using matplotlib\nplt.imshow(dataset.pixel_array,cmap=plt.cm.bone)\nplt.show()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"imdir = \"../input/osic-pulmonary-fibrosis-progression/train/ID00123637202217151272140\"\nprint(\"total images for patient ID00123637202217151272140:\",len(os.listdir(imdir)))\n#view first (columns*rows) images in order\nfig=plt.figure(figsize=(12,12))\ncolumns=4\nrows=5\nimlist=os.listdir(imdir)\nfor i in range(1,columns*rows+1):\n    filename = imdir + \"/\" + str(i) + \".dcm\"\n    ds = pydicom.dcmread(filename)\n    fig.add_subplot(rows,columns,i)\n    plt.imshow(ds.pixel_array,cmap='gray')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"imdir = \"../input/osic-pulmonary-fibrosis-progression/train/ID00123637202217151272140\"\nprint(\"total images for patient ID00123637202217151272140:\",len(os.listdir(imdir)))\n#view first (columns*rows) images in order\nfig=plt.figure(figsize=(12,12))\ncolumns=4\nrows=5\nimlist=os.listdir(imdir)\nfor i in range(1,columns*rows+1):\n    filename = imdir + \"/\" + str(i) + \".dcm\"\n    ds = pydicom.dcmread(filename)\n    fig.add_subplot(rows,columns,i)\n    plt.imshow(ds.pixel_array,cmap='jet')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Visualization using gif","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_scan(path):\n    slices=[pydicom.read_file(path+'/'+s) for s in os.listdir(path)]\n    slices.sort(key=lambda x: int(x.InstanceNumber))\n    try:\n        slice_thickness = np.abs(slices[0].ImagePositionPatient[2]-slices[1].ImagePositionPatient[2])\n    except:\n        slice_thickness = np.abs(slices[0].SliceLocation-slices[1].SliceLocation)\n    for s in slices:\n        s.SliceThickness = slice_thickness\n    return slices\ndef get_pixels_hu(scans):\n    image=np.stack([s.pixel_array for s in scans])\n    #convert to int16 (from sometimes int16),\n    #should be possible as values should always be low enough (<32k)\n    image = image.astype(np.int16)\n    #set outside-of-scan pixels to 1\n    #the intercept is usually -1024, so air is approximately 0\n    image[image == -2000]=0\n    #convert to hounsfield units (HU)\n    intercept = scans[0].RescaleIntercept\n    slope = scans[0].RescaleSlope\n    if slope != 1:\n        image = slope*image.astype(np.float64)\n        image = image.astype(np.int16)\n    image += np.int16(intercept)\n    return np.array(image,dtype=np.int16)\ndef set_lungwin(img,hu=[-1200.,600.]):\n    lungwin = np.array(hu)\n    newimg = (img-lungwin[0])/(lungwin[1]-lungwin[0])\n    newimg[newimg < 0] = 0\n    newimg[newimg > 1] = 1\n    newimg = (newimg * 255).astype('uint8')\n    return newimg\nid=0\nscans=load_scan('../input/osic-pulmonary-fibrosis-progression/train/ID00123637202217151272140')\nscan_array = set_lungwin(get_pixels_hu(scans))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import imageio\nfrom IPython.display import Image\nimageio.mimsave(\"/tmp/gif.gif\",scan_array,duration=0.0001)\nImage(filename=\"/tmp/gif.gif\",format='png')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Visualization using Animation","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.animation as animation\nfrom IPython.display import HTML\nfig = plt.figure()\nims = []\nfor image in scan_array:\n    im = plt.imshow(image,animated=True,cmap=\"Greys\")\n    plt.axis(\"off\")\n    ims.append([im])\nani = animation.ArtistAnimation(fig,ims,interval=100,blit=False,repeat_delay=1000)\nHTML(ani.to_jshtml())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"HTML(ani.to_html5_video())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Extracting DICOM files information in a dataframe","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def extract_dicom_meta_data(filename: str) -> Dict:\n    # Load image\n    \n    image_data = pydicom.read_file(filename)\n    img=np.array(image_data.pixel_array).flatten()\n    row = {\n        'Patient': image_data.PatientID,\n        'body_part_examined': image_data.BodyPartExamined,\n        'image_position_patient': image_data.ImagePositionPatient,\n        'image_orientation_patient': image_data.ImageOrientationPatient,\n        'photometric_interpretation': image_data.PhotometricInterpretation,\n        'rows': image_data.Rows,\n        'columns': image_data.Columns,\n        'pixel_spacing': image_data.PixelSpacing,\n        'window_center': image_data.WindowCenter,\n        'window_width': image_data.WindowWidth,\n        'modality': image_data.Modality,\n        'StudyInstanceUID': image_data.StudyInstanceUID,\n        'SeriesInstanceUID': image_data.StudyInstanceUID,\n        'StudyID': image_data.StudyInstanceUID, \n        'SamplesPerPixel': image_data.SamplesPerPixel,\n        'BitsAllocated': image_data.BitsAllocated,\n        'BitsStored': image_data.BitsStored,\n        'HighBit': image_data.HighBit,\n        'PixelRepresentation': image_data.PixelRepresentation,\n        'RescaleIntercept': image_data.RescaleIntercept,\n        'RescaleSlope': image_data.RescaleSlope,\n        'img_min': np.min(img),\n        'img_max': np.max(img),\n        'img_mean': np.mean(img),\n        'img_std': np.std(img)}\n\n    return row","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_image_path = '/kaggle/input/osic-pulmonary-fibrosis-progression/train'\ntrain_image_files = glob.glob(os.path.join(train_image_path, '*', '*.dcm'))\n\nmeta_data_df = []\nfor filename in tqdm.tqdm(train_image_files):\n    try:\n        meta_data_df.append(extract_dicom_meta_data(filename))\n    except Exception as e:\n        continue\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#convert to a pd.DataFrame from dict\nmeta_data_df = pd.DataFrame.from_dict(meta_data_df)\n#meta_data_df.head()\ndisplay(meta_data_df)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Changing size of DICOM images for CNN\n### Source: Sentdex\n### Link: https://www.kaggle.com/sentdex/first-pass-through-data-w-3d-convnet","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"data_dir = '../input/osic-pulmonary-fibrosis-progression/train/'\npatients = os.listdir(data_dir)\nlabels_df = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/train.csv')\nfor patient in patients[:20]:\n    try:\n        path = data_dir + patient\n        slices = [pydicom.read_file(path + '/' + s) for s in os.listdir(path)]\n        slices.sort(key = lambda x: int(x.InstanceNumber))\n        print(len(slices),slices[0].pixel_array.shape)\n    \n    except Exception as e:\n        # again, some patients are not labeled, but JIC we still want the error if something\n        # else is wrong with our code\n        print(str(e))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Resize DICOM images to 30 x 30 x 15\n### Preparing for convoluted neural network model","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"import cv2\nimport math\ndef chunks(l,n):\n    # Credit: Ned Batchelder\n    # Link: http://stackoverflow.com/questions/312443/how-do-you-split-a-list-into-evenly-sized-chunks\n    \"\"\"Yield successive n-sized chunks from 1.\"\"\"\n    for i in range(0,len(l),n):\n        yield l[i:i+n]\ndef mean(l):\n    return sum(l)/len(l)\nIMG_PX_SIZE = 30\nHM_SLICES = 15\ndata_dir = '../input/osic-pulmonary-fibrosis-progression/train/'\npatients = os.listdir(data_dir)\nlabels_df = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/train.csv')\nnew_slices = []\n\n\nfor patient in patients[:100]:\n    try:\n        path = data_dir + patient\n        slices = [pydicom.read_file(path + '/' + s) for s in os.listdir(path)]\n        slices.sort(key = lambda x: int(x.InstanceNumber))\n        new_slices = []\n\n        slices = [cv2.resize(np.array(each_slice.pixel_array),(IMG_PX_SIZE,IMG_PX_SIZE)) for each_slice in slices]\n\n        chunk_sizes = math.ceil(len(slices) / HM_SLICES)\n\n\n        for slice_chunk in chunks(slices, chunk_sizes):\n            slice_chunk = list(map(mean, zip(*slice_chunk)))\n            new_slices.append(slice_chunk)\n\n        if len(new_slices) < HM_SLICES:\n            while len(new_slices) < HM_SLICES:\n                new_slices.append(new_slices[-1])\n\n        if len(new_slices) > HM_SLICES:\n            while len(new_slices) > HM_SLICES:\n                new_val = list(map(mean, zip(*[new_slices[HM_SLICES-1],new_slices[HM_SLICES],])))\n                del new_slices[HM_SLICES]\n                new_slices[HM_SLICES-1] = new_val\n\n        print(len(slices), len(new_slices))\n    \n    except Exception as e:\n        # again, some patients are not labeled, but JIC we still want the error if something\n        # else is wrong with our code\n        print(str(e))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Etc. - Pandas Profiling","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas_profiling as pdp\ntrain_df = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/train.csv')\ntest_df = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/test.csv')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"profile_train_df = pdp.ProfileReport(train_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"profile_train_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"profile_test_df = pdp.ProfileReport(test_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"profile_test_df","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}