{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# import the libararies"},{"metadata":{"trusted":true},"cell_type":"code","source":"from tqdm import tqdm\nimport os\nfrom random import randint\n\nimport numpy as np\nimport pandas as pd\n\nimport nibabel as nib\nimport pydicom as pdm\nimport nilearn as nl\nimport nilearn.plotting as nlplt\n#import nrrd\nimport h5py\nimport pydicom\n\nfrom glob import glob\n\nimport imageio\nfrom IPython.display import Image\n\nimport matplotlib.pyplot as plt\nfrom matplotlib import cm\nimport matplotlib.animation as anim\n\nimport imageio\nfrom skimage.transform import resize\nfrom skimage.util import montage\n\nfrom IPython.display import Image as show_gif\nimport numpy.ma as ma\nimport warnings\nwarnings.simplefilter(\"ignore\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# to display dicom file \npath = r\"../input/osic-pulmonary-fibrosis-progression/train\"\nseries = np.array([[(os.path.join(dp, f), pydicom.dcmread(os.path.join(dp, f), stop_before_pixels = True)) for f in files]\n                   for dp,_,files in os.walk(path) if len(files) != 0])\n\n# to make list of dicom files \nPATH='../input/osic-pulmonary-fibrosis-progression/train'\n## First, read all of my DICOM files into a list\nmydicoms = glob(\"../input/osic-pulmonary-fibrosis-progression/train/*/*.dcm\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# to show the meta_Data of any patient "},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"mydicoms [0][3][1]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def specific_dicom_display(path=path):\n    dcm=pydicom.dcmread(path)\n    pat_name = dcm.PatientName\n    display_name = pat_name.family_name + \", \" + pat_name.given_name\n    print(\"Patient's name................:\", display_name)\n    print(\"Patient id....................:\", RefDs.PatientID)\n    print(\"Modality......................:\", RefDs.Modality)\n    print(\"BodyPartExamined..............:\", RefDs.BodyPartExamined)  \n    print(\"Image Position    (Patient)...:\", RefDs.ImagePositionPatient)\n    print(\"Image Orientation (Patient)...:\", RefDs.ImageOrientationPatient)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"specific_dicom_display('../input/osic-pulmonary-fibrosis-progression/train/ID00007637202177411956430/1.dcm')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"instances = [f for l in series for f in l]\nprint(len(instances))\npatient_ids = np.unique([inst[1].PatientID for inst in instances])\nprint(\"the number of patients :\",len(patient_ids))\n#-------------------------------------------------------------------------------\nprint(\"Number of DICOM files:\", len(os.listdir(\"../input/osic-pulmonary-fibrosis-progression/train\")))\n\n#---------------------------------------------------------------\n\nprint(\"the number of series :\",len(series))\n\n\n#-----------------------------------------------------------\n# How many studies?\n\nstudies = {}\n\nfor s in series:\n    studies.setdefault(s[0][1].StudyInstanceUID, []).append(s)\nprint(\"the number of studies :\",len(studies))\n\n#---------------------------------------------------\n[len([st for st in studies.values() if st[0][0][1].PatientID == p]) for p in patient_ids]\n\n\n\n#-------------------------------------------------\n# Let's see how many series per study\n\nseries_per_study = [(len(sr), sr[0][0][1].PatientID) for sr in studies.values()]\n#print(\"the number of series per study :\",series_per_study)\n\n\n#-----------------------------------------------------\nimg_per_series = [len(s) for s in series]\n\nprint(\"the number of images per series :\",img_per_series)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# the outliers for number of images \n**3 for studies more than 600 image\n\n**4 for studies more_500_less_600 image"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.hist(img_per_series)\nplt.show()\nmore_600=[]\nless_100=[]\nmore_100_less_200=[]\nmore_200_less_300=[]\nmore_300_less_400=[]\nmore_400_less_500=[]\nmore_500_less_600=[]\nfor i in range(len(img_per_series)):\n    \n    if img_per_series[i]<100:\n        less_100.append(img_per_series[i])\n     \n    elif((img_per_series[i]>=100) and (img_per_series[i]<200)):\n        more_100_less_200.append(img_per_series[i])\n        \n    elif((img_per_series[i]>=200) and (img_per_series[i]<300)):\n        more_200_less_300.append(img_per_series[i])\n        \n    elif((img_per_series[i]>=300) and (img_per_series[i]<400)):\n        more_300_less_400.append(img_per_series[i])\n        \n    elif((img_per_series[i]>=400) and (img_per_series[i]<500)):\n        more_400_less_500.append(img_per_series[i])\n                                 \n    elif((img_per_series[i]>=500) and (img_per_series[i]<600)):\n        more_500_less_600.append(img_per_series[i])\n                                 \n    else:\n        more_600.append(img_per_series[i])\n                                 \n                                 \nprint(\"the length of studies more than 600 image is:{} \",len(more_600))\nprint(\"the length of studies less than 100 image is:{} \",len(less_100))\nprint(\"the length of studies more_100_less_200 image is:{} \",len(more_100_less_200))                                 \nprint(\"the length of studies more_200_less_300 image is:{} \",len(more_200_less_300))                                 \nprint(\"the length of studies more_300_less_400 image is:{} \",len(more_300_less_400))\nprint(\"the length of studies more_400_less_500 image is:{} \",len(more_400_less_500))\nprint(\"the length of studies more_500_less_600 image is:{} \",len(more_500_less_600))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def process_data(path):\n    data = pd.DataFrame([{'path': filepath} for filepath in glob(PATH+path)])\n    data['file'] = data['path'].map(os.path.basename)\n    for i in mydicoms:\n        dicom_read = pydicom.dcmread(i)\n    data['ID'] = dicom_read.PatientID\n    #data['Age'] = dicom_read.PatientAge\n    data['Modality'] = dicom_read.Modality\n    #data['description']=dicom_read.StudyDescription\n    return data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dicom_data = process_data('/*/*.dcm')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dicom_data.shape[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dim = dicom_data.shape[0]\nsize = 6\ndata2Dlist = [[0 for x in range(size)] for y in range(dim)] \nlstFilesDCM =mydicoms\nfor i in range(0,dim):\n    #print(i, lstFilesDCM[i])\n    data2Dlist[i][0] = pydicom.dcmread(lstFilesDCM[i]).PatientID\n    data2Dlist[i][1] = pydicom.dcmread(lstFilesDCM[i]).Modality\n    data2Dlist[i][2] = pydicom.dcmread(lstFilesDCM[i]).BodyPartExamined\n    data2Dlist[i][3] = pydicom.dcmread(lstFilesDCM[i]).InstanceNumber\n    data2Dlist[i][4] = '73'\n    data2Dlist[i][5] = 'male'\n                      \n\ndata2Dlist\ndf = pd.DataFrame(data2Dlist, columns=['ID', 'Modality', 'BPE', 'Slice', 'Age', 'Sex'])\nprint(df.head(10))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data=np.concatenate([df,dicom_data],axis=1)\n\ndata=pd.DataFrame(data,columns=['ID', 'Modality', 'BPE', 'Slice', 'Age', 'Sex','DIRECTORY','FILE','I','M'])\n\nprint(df.shape[0],dicom_data.shape[0])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data.head(10)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data=data.drop(['I'], axis = 1) \ndata=data.drop(['M'], axis = 1) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_train = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/train.csv')\ndata_test = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/test.csv')\ndata_train.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def subplotting(column):\n    fig, axs = plt.subplots(1, 2, sharey=True, tight_layout=True)\n    axs[0].hist(data_train[column] )\n    plt.title(\"TRAIN\")\n    axs[1].hist(data_test[column])\n    plt.title(\"TEST\")\n    plt.suptitle(\"the train vs test for %s \"%column)\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"subplotting(\"Weeks\")\nsubplotting(\"FVC\")\nsubplotting(\"Percent\")\nsubplotting(\"Age\")\nsubplotting(\"Sex\")\nsubplotting(\"SmokingStatus\")\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import pydicom\nfrom pydicom.data import get_testdata_files\nimport numpy as np\n\nprint(__doc__)\n\nPathDicom = '/kaggle/input/osic-pulmonary-fibrosis-progression/'\nlstFilesDCM = []  # create an empty list\nfor dirName, subdirList, fileList in os.walk(PathDicom):\n    for filename in fileList:\n        if \".dcm\" in filename.lower():  # check whether the file's DICOM\n            lstFilesDCM.append(os.path.join(dirName,filename))\n\n\nfig, axs = plt.subplots(4,4, figsize=(25,25))\nfor i in range(0,16):\n    RefDs = pydicom.dcmread(lstFilesDCM[i])\n    axs[i//4, i%4].imshow(RefDs.pixel_array, cmap='gray') \n    axs[i//4, i%4].margins(x=0.5, y=-0.25)   # Values in (-0.5, 0.0) zooms in to center\n    axs[i//4, i%4].set_axis_off()\n    axs[i//4, i%4].set_title('Modality: {} BPE: {}\\n Slice: {} Age: {} Sex: {}'.format(df.Modality[i],df.BPE[i],df.Slice[i],df.Age[i],df.Sex[i]))\n\nplt.axis(\"off\") \nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"the visulaization of dicom file \n"},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_scan(path):\n    \"\"\"\n    Loads scans from a folder and into a list.\n    \n    Parameters: path (Folder path)\n    \n    Returns: slices (List of slices)\n    \"\"\"\n    \n    slices = [pydicom.read_file(path + '/' + s) for s in os.listdir(path)]\n    slices.sort(key = lambda x: int(x.InstanceNumber))\n    \n    try:\n        slice_thickness = np.abs(slices[0].ImagePositionPatient[2] - slices[1].ImagePositionPatient[2])\n    except:\n        slice_thickness = np.abs(slices[0].SliceLocation - slices[1].SliceLocation)\n        \n    for s in slices:\n        s.SliceThickness = slice_thickness\n        \n    return slices\ndef get_pixels_hu(scans):\n    \"\"\"\n    Converts raw images to Hounsfield Units (HU).\n    \n    Parameters: scans (Raw images)\n    \n    Returns: image (NumPy array)\n    \"\"\"\n    \n    image = np.stack([s.pixel_array for s in scans])\n    image = image.astype(np.int16)\n\n    # Since the scanning equipment is cylindrical in nature and image output is square,\n    # we set the out-of-scan pixels to 0\n    image[image == -2000] = 0\n    \n    \n    # HU = m*P + b\n    intercept = scans[0].RescaleIntercept\n    slope = scans[0].RescaleSlope\n    \n    if slope != 1:\n        image = slope * image.astype(np.float64)\n        image = image.astype(np.int16)\n        \n    image += np.int16(intercept)\n    \n    return np.array(image, dtype=np.int16)\ndef set_lungwin(img, hu=[-1200., 600.]):\n    lungwin = np.array(hu)\n    newimg = (img-lungwin[0]) / (lungwin[1]-lungwin[0])\n    newimg[newimg < 0] = 0\n    newimg[newimg > 1] = 1\n    newimg = (newimg * 255).astype('uint8')\n    return newimg\n\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"scans = load_scan('../input/osic-pulmonary-fibrosis-progression/train/ID00007637202177411956430/')\nscan_array = set_lungwin(get_pixels_hu(scans))\n\nimageio.mimsave(\"/tmp/gif.gif\", scan_array, duration=0.0001)\nImage(filename=\"/tmp/gif.gif\", format='png')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}