{"cells":[{"metadata":{},"cell_type":"markdown","source":"Used https://www.kaggle.com/andradaolteanu/pulmonary-fibrosis-competition-eda-dicom-prep as a reference","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"DATA_DIR = \"/kaggle/input/osic-pulmonary-fibrosis-progression/\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(os.path.join(DATA_DIR, \"train.csv\"))\ntest_df = pd.read_csv(os.path.join(DATA_DIR, \"test.csv\"))\nsub_df = pd.read_csv(os.path.join(DATA_DIR, \"sample_submission.csv\"))\n# remove the duplicates from the train_df\ntrain_df.drop_duplicates(keep=False, inplace=True, subset=['Patient', 'Weeks'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# get the unique portion of the data for the patients\ndata = train_df.groupby(by=\"Patient\")[['Patient', 'Age', 'Sex', 'SmokingStatus']].first().reset_index(drop=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, (ax1, ax2, ax3) = plt.subplots(1, 3, figsize=(16,6))\n\nax1.set_title(\"Gender Frequency\")\ndata['Sex'].hist(ax=ax1)\nax2.set_title(\"Patient Age Distribution\")\ndata['Age'].hist(ax=ax2)\nax3.set_title(\"Smoking Status\")\ndata['SmokingStatus'].hist(ax=ax3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(20, 7))\nax1.set_title(\"FVC Distribution\")\ntrain_df['FVC'].hist(ax=ax1)\nax2.set_title(\"Percent Distribution\")\ntrain_df['Percent'].hist(ax=ax2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(16,6))\ntrain_df['Weeks'].hist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# FVC vs Percent plot\nfig, (ax1, ax2, ax3) = plt.subplots(1, 3, figsize=(16,6))\n\n# FVC vs Percent\nmale_df = train_df[train_df['Sex'] == 'Male']\nfemale_df = train_df[train_df['Sex'] == 'Female']\n\nmale_df.plot.scatter('FVC', 'Percent', c='red', label='Male', ax=ax1)\nfemale_df.plot.scatter('FVC', 'Percent', c='green', ax=ax1, label=\"Female\")\n\n# FVC vs Age\nmale_df.plot.scatter('FVC', 'Age', c='red', label='Male', ax=ax2)\nfemale_df.plot.scatter('FVC', 'Age', c='green', label='Female', ax=ax2)\n\n# Percent vs Age\nmale_df.plot.scatter('Percent', 'Age', c='red', label='Male', ax=ax3)\nfemale_df.plot.scatter('Percent', 'Age', c='green', label='Female', ax=ax3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(16,6))\n\nax1.bar(train_df.groupby('SmokingStatus').mean()['FVC'].index, train_df.groupby('SmokingStatus').mean()['FVC'])\nax2.bar(train_df.groupby('SmokingStatus').mean()['Percent'].index, train_df.groupby('SmokingStatus').mean()['Percent'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# change in FVC in patients\nplt.figure(figsize=(20,20))\nplt.title(\"FVC decrease with Weeks\")\ntrain_df.groupby('Patient').apply(lambda grp: plt.plot(grp['Weeks'], grp['FVC']))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"number_of_ct = {x: len(os.listdir(os.path.join(DATA_DIR, \"train\", x))) for x in os.listdir(os.path.join(DATA_DIR, \"train\"))}\n# sort by number_of_ct scans\nnumber_of_ct = {k: v for k, v in sorted(number_of_ct.items(), key=lambda item: item[1])}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(20, 10))\nplt.bar(number_of_ct.keys(), number_of_ct.values())\nplt.title('Number of CT scans per patients')\nplt.tick_params(axis='x', which='both', bottom=False, top=False, labelbottom=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}