{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport plotly.express as px\nimport plotly.graph_objs as go\n\nimport pydicom\nfrom pydicom.data import get_testdata_files","metadata":{"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=pd.read_csv('../input/osic-pulmonary-fibrosis-progression/train.csv')\ntest=pd.read_csv('../input/osic-pulmonary-fibrosis-progression/test.csv')","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"smokers=train.loc[train.SmokingStatus=='Currently smokes']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.pairplot(train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.pairplot(train,hue='Sex')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The distribution of FVC of males is shifted to higher FVC values indicating males have a higher FVC on average","metadata":{}},{"cell_type":"code","source":"sns.pairplot(train,hue='SmokingStatus')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Age distribution of currently smoking patients has multiple small peaks across all ages and, a single high peak near age 70(like other two categories). This indicates pulmonary fibrosis affects middle aged people who are currently smokers.","metadata":{}},{"cell_type":"code","source":"sns.distplot(smokers['Age'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.distplot(train['Age'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"\ndf= train.groupby([train.Patient,train.Age,train.Sex, train.SmokingStatus])['Patient'].count()\ndf.index = df.index.set_names(['id','Age','Sex','SmokingStatus'])\ndf = df.reset_index()\ndf.rename(columns = {'Patient': 'freq'},inplace = True)\nprint(df.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.bar(df, x='freq',y ='id',color='freq')\nfig.update_layout(yaxis={'categoryorder':'total ascending'},title='No. of observations for each patient')\nfig.update_yaxes(showticklabels=False)\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=[10,8])\nplt.style.use('ggplot')\nsns.countplot(x='Sex',data=train,hue='SmokingStatus')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.corr()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(train.corr(),annot=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.line(train.loc[train['Patient']=='ID00422637202311677017371'],x='Weeks',y='FVC')\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.line(train.loc[train['Patient']=='ID00426637202313170790466'],x='Weeks',y='FVC')\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"p1=train.loc[train['Patient']=='ID00426637202313170790466']\np2=train.loc[train['Patient']=='ID00007637202177411956430']\np3=train.loc[train['Patient']=='ID00355637202295106567614']\npati=pd.concat([p1,p2,p3])\npx.line(pati,x='Weeks',y='FVC',line_group='Patient',color='Patient')\n\n\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PathDicom = '/kaggle/input/osic-pulmonary-fibrosis-progression/'\nlstFilesDCM = []  # create an empty list\nfor dirName, subdirList, fileList in os.walk(PathDicom):\n    for filename in fileList:\n        if \".dcm\" in filename.lower():  # check whether the file's DICOM\n            lstFilesDCM.append(os.path.join(dirName,filename))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(lstFilesDCM[0])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RefDs = pydicom.dcmread(lstFilesDCM[4])\nprint(RefDs)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Patient id is {}'.format(RefDs.PatientID))\nprint('Sex..........{}'.format(RefDs.PatientSex))\nprint('Image Position {}'.format(RefDs.ImagePositionPatient))\nprint('Image Orientation {}'.format(RefDs.ImageOrientationPatient))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def MakeDF(lst):\n    dictDf={}\n    refd=pydicom.dcmread(lst[0])\n    dictDf['Patient']=[refd.PatientID]\n    dictDf['rows']=[refd.Rows]\n    for i in range(1,len(lst)):\n        refd=pydicom.dcmread(lst[i])\n        dictDf['Patient'].append(refd.PatientID)\n        dictDf['rows'].append(refd.Rows)\n        #print(dictDf)\n    return(dictDf)    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd_dict=MakeDF(lstFilesDCM)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=pd.DataFrame.from_dict(pd_dict)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=[10,10])\nplt.imshow(RefDs.pixel_array, cmap=plt.cm.bone)\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"approch is to use autoencoder to encode image data and concat it with tabular data and then use rnn for predicting time series data","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}