{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\n%matplotlib inline\nfrom os import listdir\n\n\nimport plotly.express as px\nimport plotly.graph_objs as go\n\n\nimport pydicom\nimport glob\nimport imageio\nfrom IPython.display import Image\n\n\nimport warnings\nwarnings.filterwarnings('ignore')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    i=0\n    for filename in filenames:\n        i+=1        \n        print(os.path.join(dirname, filename))\nprint(str(i))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(os.listdir('/kaggle/input/osic-pulmonary-fibrosis-progression/train/'))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Number of image directories in the folder set. Each directory consists of images of one unique patient. So the unique patients are expected to be 176","execution_count":null},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"list(os.listdir(\"../input/osic-pulmonary-fibrosis-progression\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"traindf = pd.read_csv(\"../input/osic-pulmonary-fibrosis-progression/train.csv\")\ntestdf = pd.read_csv(\"../input/osic-pulmonary-fibrosis-progression/test.csv\")\nsample_submissiondf = pd.read_csv(\"../input/osic-pulmonary-fibrosis-progression/sample_submission.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"traindf.describe()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"The Mininum age of the patients is 49 years and the maximum is 88 years old.\n","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"traindf.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"testdf.info()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"no missing values seen in the train or test data.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"testdf.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"traindf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"traindf['Patient'].count()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"patients_with_multiple_data = traindf[\"Patient\"].value_counts()\n\npatients_with_multiple_data.count()\npatients_with_multiple_data.describe()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"This means that there are 176 unique patient ids and the data of 1549 images is from these 176 patients. \nThe maximum number of images corresponding to a single patient is 10 and the minimum is 6. \n\nHence, each patient has between 6 and 10 image files and data files available in the data.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"new_df = traindf.groupby([traindf.Patient,traindf.Age,traindf.Sex, traindf.SmokingStatus])['Patient'].count()\nnew_df.index = new_df.index.set_names(['id','Age','Sex','SmokingStatus'])\nnew_df = new_df.reset_index()\nnew_df.rename(columns = {'Patient': 'freq'},inplace = True)\nnew_df.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"new_df.shape","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Data Distribution","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.bar(new_df, x=\"id\", y=\"freq\", color='freq')\nfig.update_layout(xaxis={'categoryorder':'total ascending'},title='number of data entries for each patient')\nfig.update_xaxes(showticklabels=False)\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Age Distribution","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"hst = px.histogram(new_df, x='Age', nbins=40, opacity=0.7,title='Age of patients against the number of records', labels={'Age':'Age of patients', 'freq':'Records available for each patient'})\nhst.update_traces(marker_color='rgb(123,125,222)', marker_line_color='rgb(9,4,21)', marker_line_width=1.5)\nhst.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Most people in the training data appear to be in the age between 63 to 74","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"# Sex Distribution","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"agehst = px.histogram(new_df, x='Sex')\nagehst.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Out of 176 patients, 139 are males and 37 are females","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"sexdf = new_df['Sex'].value_counts()\ntotal = new_df.shape[0]\nmale = sexdf['Male']\n\nfemale = sexdf['Female']\nprint('Out of %d patients' % total)\nprint('%.2f percent are male' % (male/total))\nprint('%.2f percent are females' % (female/total))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# The Dicom files\nThe DICOM file or \"Digital Imaging and Communications in Medicine\" format contains an image from a medical scan, like a CT scan + information about the patient.\n\nDICOM is the standard for the medical imaging information and related data. Hence, it is expected that all medical related analyses will include scans in this format.\n","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"# Looking at a sample image","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"image = '/kaggle/input/osic-pulmonary-fibrosis-progression/train/ID00060637202187965290703/107.dcm'\nds = pydicom.dcmread(image)\nplt.figure(figsize=(12,12))\nplt.imshow(ds.pixel_array, cmap=plt.cm.bone)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Showing multiple images of one patient","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"imgdir = '/kaggle/input/osic-pulmonary-fibrosis-progression/train/ID00060637202187965290703/'\nimg_list = os.listdir(imgdir)\nimg_list.sort()\nlen(img_list)\n\n\nfig=plt.figure(figsize=(15,20))\ncolumns = 5\nrows = 10\n\nfor i in range(1, columns*rows +1):\n    filename = imgdir + \"/\" + str(i) + \".dcm\"\n    ds = pydicom.dcmread(filename)\n    fig.add_subplot(rows, columns, i)\n    plt.imshow(ds.pixel_array, cmap=plt.cm.bone)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Trying out an animation\nTrying to learn to use animations after having a look at https://www.kaggle.com/danpresil1/dicom-basic-preprocessing-and-visualization and https://www.kaggle.com/twinkle0705/your-starter-notebook-for-osic\nThanks to these works :)","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_scan(path):\n    slices = [pydicom.read_file(path + '/' + s) for s in os.listdir(path)]\n    slices.sort(key = lambda x: float(x.ImagePositionPatient[2]))\n    try:\n        slice_thickness = np.abs(slices[0].ImagePositionPatient[2] - slices[1].ImagePositionPatient[2])\n    except:\n        slice_thickness = np.abs(slices[0].SliceLocation - slices[1].SliceLocation)\n        \n    for s in slices:\n        s.SliceThickness = slice_thickness\n        \n    return slices\n\n\ndef get_pixels_hu(slices):\n    image = np.stack([s.pixel_array for s in slices])\n    # Convert to int16 (from sometimes int16), \n    # should be possible as values should always be low enough (<32k)\n    image = image.astype(np.int16)\n\n    # Set outside-of-scan pixels to 0\n    # The intercept is usually -1024, so air is approximately 0\n    image[image == -2000] = 0\n    \n    # Convert to Hounsfield units (HU)\n    for slice_number in range(len(slices)):\n        \n        intercept = slices[slice_number].RescaleIntercept\n        slope = slices[slice_number].RescaleSlope\n        \n        if slope != 1:\n            image[slice_number] = slope * image[slice_number].astype(np.float64)\n            image[slice_number] = image[slice_number].astype(np.int16)\n            \n        image[slice_number] += np.int16(intercept)\n    \n    return np.array(image, dtype=np.int16)\n\ndef set_lungwin(img, hu=[-1200., 600.]):\n    lungwin = np.array(hu)\n    newimg = (img-lungwin[0]) / (lungwin[1]-lungwin[0])\n    newimg[newimg < 0] = 0\n    newimg[newimg > 1] = 1\n    newimg = (newimg * 255).astype('uint8')\n    return newimg\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"scans = load_scan('../input/osic-pulmonary-fibrosis-progression/train/ID00060637202187965290703/')\nscan_array = set_lungwin(get_pixels_hu(scans))\n\n\n\nimageio.mimsave(\"/tmp/gif.gif\", scan_array, duration=0.0001)\nImage(filename=\"/tmp/gif.gif\", format='png')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Thanks to other notebooks, I have been learning about dicom processing. \n","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"","execution_count":null}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}