{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\ntrain_df = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/train.csv')\ntest_df = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import os\nfrom os import listdir\n\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\n#plotly\n!pip install chart_studio\nimport plotly.express as px\nimport chart_studio.plotly as py\nimport plotly.graph_objs as go\nfrom plotly.offline import iplot\nimport cufflinks\ncufflinks.go_offline()\ncufflinks.set_config_file(world_readable=True, theme='pearl')\n\nimport seaborn as sns\nsns.set(style=\"whitegrid\")\n\n#pydicom\nimport pydicom\n\n# Suppress warnings \nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Settings for pretty nice plots\nplt.style.use('fivethirtyeight')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.groupby(['SmokingStatus']).count()['Sex'].to_frame()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Null values and Data types\nprint('Train Set !!')\nprint(train_df.info())\nprint('-------------')\nprint('Test Set !!')\nprint(test_df.info())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Total number of Patient in the dataset(train+test)\nprint(\"Total Patient in Train set: \",train_df['Patient'].count())\nprint(\"Total Patient in Test set: \",test_df['Patient'].count())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"columns = train_df.keys()\ncolumns = list(columns)\nprint(columns)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['SmokingStatus'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['SmokingStatus'].value_counts(normalize=True).iplot(kind='bar',\n                                                      yTitle='Percentage', \n                                                      linecolor='black', \n                                                      opacity=0.7,\n                                                      color='red',\n                                                      theme='pearl',\n                                                      bargap=0.8,\n                                                      gridcolor='white',\n                                                     \n                                                      title='Distribution of the SmokingStatus column in the training set')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['Weeks'].value_counts().sort_values().iplot(kind='barh',\n                                                      xTitle='Counts(Weeks)', \n                                                      linecolor='black', \n                                                      opacity=0.7,\n                                                      color='#FB8072',\n                                                      theme='pearl',\n                                                      bargap=0.2,\n                                                      gridcolor='white',\n                                                      title='Distribution of the Weeks in the training set')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"z=train_df.groupby(['SmokingStatus','Weeks'])['FVC'].count().to_frame().reset_index().head()\nz.style.background_gradient(cmap='Reds') ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['FVC'].value_counts().iplot(kind='barh',\n                                      xTitle='Lung Capacity(ml)', \n                                      linecolor='black', \n                                      opacity=0.8,\n                                      color='#FB8072',\n                                      bargap=0.5,\n                                      gridcolor='white',\n                                      title='Distribution of the FVC in the training set')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['Percent'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['Percent'].iplot(kind='hist',bins=30,color='blue',xTitle='Percent distribution',yTitle='Count')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['Age'].iplot(kind='hist',bins=30,color='red',xTitle='Age distribution',yTitle='Count')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(16, 6))\nsns.kdeplot(train_df.loc[train_df['SmokingStatus'] == 'Ex-smoker', 'Age'], label = 'Ex-smoker',shade=True)\nsns.kdeplot(train_df.loc[train_df['SmokingStatus'] == 'Never smoked', 'Age'], label = 'Never smoked',shade=True)\nsns.kdeplot(train_df.loc[train_df['SmokingStatus'] == 'Currently smokes', 'Age'], label = 'Currently smokes',shade=True)\n\n# Labeling of plot\nplt.xlabel('Age (years)'); plt.ylabel('Density'); plt.title('Distribution of Ages');\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(16, 6))\nsns.kdeplot(train_df.loc[train_df['Sex'] == 'Male', 'Age'], label = 'Male',shade=True)\nsns.kdeplot(train_df.loc[train_df['Sex'] == 'Female', 'Age'], label = 'Female',shade=True)\nplt.xlabel('Age (years)'); plt.ylabel('Density'); plt.title('Distribution of Ages');","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['Sex'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(16, 6))\na = sns.countplot(data=train_df, x='SmokingStatus', hue='Sex')\n\nfor p in a.patches:\n    a.annotate(format(p.get_height(), ','), \n           (p.get_x() + p.get_width() / 2., \n            p.get_height()), ha = 'center', va = 'center', \n           xytext = (0, 4), textcoords = 'offset points')\n\nplt.title('Gender split by SmokingStatus', fontsize=16)\nsns.despine(left=True, bottom=True);","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_pixel_array(dataset, figsize=(5,5)):\n    plt.figure(figsize=figsize)\n    plt.grid(False)\n    plt.imshow(dataset.pixel_array, cmap='gray') # cmap=plt.cm.bone)\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Train .dcm number of images:', len(list(os.listdir('../input/osic-pulmonary-fibrosis-progression/train'))), '\\n' +\n      'Test .dcm number of images:', len(list(os.listdir('../input/osic-pulmonary-fibrosis-progression/test'))), '\\n' +\n      '--------------------------------', '\\n' +\n      'There is the same number of images as in train/ test .csv datasets')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# https://www.kaggle.com/schlerp/getting-to-know-dicom-and-the-data\ndef show_dcm_info(dataset):\n    print(\"Filename.........:\", file_path)\n\n    pat_name = dataset.PatientName\n    display_name = pat_name.family_name + \", \" + pat_name.given_name\n    print(\"Patient's name......:\", display_name)\n    \n    print(dataset.data_element(\"ImageOrientationPatient\"))\n    print(dataset.data_element(\"ImagePositionPatient\"))\n    print(dataset.data_element(\"PatientID\"))\n    print(dataset.data_element(\"PatientName\"))\n    print(dataset.data_element(\"PatientSex\"))\n   \n    \n    if 'PixelData' in dataset:\n        rows = int(dataset.Rows)\n        cols = int(dataset.Columns)\n        print(\"Image size.......: {rows:d} x {cols:d}, {size:d} bytes\".format(\n            rows=rows, cols=cols, size=len(dataset.PixelData)))\n        if 'PixelSpacing' in dataset:\n            print(\"Pixel spacing....:\", dataset.PixelSpacing)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"i = 1\nnum_to_plot = 2\nfor folder_name in os.listdir('../input/osic-pulmonary-fibrosis-progression/train/'):\n        patient_path = os.path.join('../input/osic-pulmonary-fibrosis-progression/train/',folder_name)\n        \n        for i in range(1, num_to_plot+1):     \n            file_path = os.path.join(patient_path, str(i) + '.dcm')\n\n            dataset = pydicom.dcmread(file_path)\n            show_dcm_info(dataset)\n            plot_pixel_array(dataset)\n\n        break","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# https://www.kaggle.com/yeayates21/osic-simple-image-eda\n\nimdir = \"/kaggle/input/osic-pulmonary-fibrosis-progression/train/ID00123637202217151272140\"\nprint(\"total images for patient ID00123637202217151272140: \", len(os.listdir(imdir)))\n\n# view first (columns*rows) images in order\nw=10\nh=10\nfig=plt.figure(figsize=(12, 12))\ncolumns = 4\nrows = 5\nimglist = os.listdir(imdir)\nfor i in range(1, columns*rows +1):\n    filename = imdir + \"/\" + str(i) + \".dcm\"\n    ds = pydicom.dcmread(filename)\n    fig.add_subplot(rows, columns, i)\n    plt.imshow(ds.pixel_array, cmap='gray')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# source: https://www.kaggle.com/c/siim-isic-melanoma-classification/discussion/154658\nfolder='train'\nPATH='../input/osic-pulmonary-fibrosis-progression/'\n\nlast_index = 2\n\ncolumn_names = ['image_name', 'dcm_ImageOrientationPatient', \n                'dcm_ImagePositionPatient', 'dcm_PatientID',\n                'dcm_PatientName', 'dcm_PatientSex'\n                'dcm_rows', 'dcm_columns']\n\ndef extract_DICOM_attributes(folder):\n    patients_folder = list(os.listdir(os.path.join(PATH, folder)))\n    df = pd.DataFrame()\n    \n    i = 0\n    \n    for patient_id in patients_folder:\n   \n        img_path = os.path.join(PATH, folder, patient_id)\n        \n        print(img_path)\n        \n        images = list(os.listdir(img_path))\n        \n        #df = pd.DataFrame()\n\n        for image in images:\n            image_name = image.split(\".\")[0]\n\n            dicom_file_path = os.path.join(img_path,image)\n            dicom_file_dataset = pydicom.read_file(dicom_file_path)\n                \n            '''\n            print(dicom_file_dataset.dir(\"pat\"))\n            print(dicom_file_dataset.data_element(\"ImageOrientationPatient\"))\n            print(dicom_file_dataset.data_element(\"ImagePositionPatient\"))\n            print(dicom_file_dataset.data_element(\"PatientID\"))\n            print(dicom_file_dataset.data_element(\"PatientName\"))\n            print(dicom_file_dataset.data_element(\"PatientSex\"))\n            '''\n            \n            imageOrientationPatient = dicom_file_dataset.ImageOrientationPatient\n            #imagePositionPatient = dicom_file_dataset.ImagePositionPatient\n            patientID = dicom_file_dataset.PatientID\n            patientName = dicom_file_dataset.PatientName\n            patientSex = dicom_file_dataset.PatientSex\n        \n            rows = dicom_file_dataset.Rows\n            cols = dicom_file_dataset.Columns\n            \n            #print(rows)\n            #print(columns)\n            \n            temp_dict = {'image_name': image_name, \n                                    'dcm_ImageOrientationPatient': imageOrientationPatient,\n                                    #'dcm_ImagePositionPatient':imagePositionPatient,\n                                    'dcm_PatientID': patientID, \n                                    'dcm_PatientName': patientName,\n                                    'dcm_PatientSex': patientSex,\n                                    'dcm_rows': rows,\n                                    'dcm_columns': cols}\n\n\n            df = df.append([temp_dict])\n            \n        i += 1\n        \n        if i == last_index:\n            break\n            \n    return df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"extract_DICOM_attributes('train')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas_profiling as pdp","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/train.csv')\ntest_df = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"profile_train_df = pdp.ProfileReport(train_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"profile_train_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"profile_test_df = pdp.ProfileReport(test_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"profile_test_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!ls /kaggle/input/osic-pulmonary-fibrosis-progression/","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub   = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i in range(len(test_df)):\n    sub.loc[sub['Patient_Week'].str.contains(test_df.Patient[i]), 'FVC'] = test_df.FVC[i]\n    \nsub.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}