{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport pydicom\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport px\n\nfrom colorama import Fore, Back, Style\n\n# Set Color Palettes for the notebook\ncustom_colors = ['#74a09e','#86c1b2','#98e2c6','#f3c969','#f2a553', '#d96548', '#c14953']\nsns.palplot(sns.color_palette(custom_colors))\n\n# Set Style\nsns.set_style(\"whitegrid\")\nsns.despine(left=True, bottom=True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**#Goals:**\n1. Data exploration:\n- Answer what columns of the relational database there are\n- How many unique rows\n- How many unique values each of the columns\n- Are there any missing values\n- Explore the relationships between the columns (plot the relationships and give analysis)\n- Plot the dcm out\n- Build dataloaders for the dcms\n2. Build baseline models\n3. ..."},{"metadata":{"trusted":true},"cell_type":"code","source":"#Load data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/osic-pulmonary-fibrosis-progression/train.csv')\ntest = pd.read_csv('/kaggle/input/osic-pulmonary-fibrosis-progression/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# How many unique values of rows (identifier is patient)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Missing values??\ntrain.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def unique_values_of_each_col(df):\n    for col in df.columns:\n        print(f\"{col} has {len(df[col].unique())} unique values\")\nunique_values_of_each_col(train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.info()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Number of images"},{"metadata":{"trusted":true},"cell_type":"code","source":"path = \"/kaggle/input/osic-pulmonary-fibrosis-progression/train\"\nnumFiles = 0\nnumFolders = 0\nfiles = []\nfor x, dirnames, filenames in os.walk(path):\n    numFiles += len(filenames)\n    numFolders += len(dirnames)\n    files.append(len(filenames))\n\nprint(\"number of files: \", numFiles)\nprint(\"number of folders: \", numFolders)\nprint(f\"there are {np.mean(files)} in average per patient\")\nprint(f\"there are {np.max(files)} max images for a patient\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f, axes = plt.subplots(2, 2, figsize = (7, 7), sharex = False, sharey=False)\nsns.distplot(train[\"Weeks\"], ax = axes[0, 0])\nsns.distplot(train[\"FVC\"], ax = axes[0, 1])\nsns.distplot(train[\"Percent\"], ax = axes[1, 0])\nsns.distplot(train[\"Age\"], ax = axes[1, 1])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#Relationships between the columns"},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.catplot(x=\"SmokingStatus\", y=\"FVC\", kind=\"bar\", data=train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.catplot(x=\"Sex\", y=\"FVC\", kind=\"bar\", data=train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# This is how to get values for a specific value in another column\ntrain.loc[train['SmokingStatus'] == 'Ex-smoker', 'Age']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(16, 6))\nsns.kdeplot(train.loc[train['SmokingStatus'] == 'Ex-smoker', 'Age'], label = 'Ex-smoker',shade=True)\nsns.kdeplot(train.loc[train['SmokingStatus'] == 'Never smoked', 'Age'], label = 'Never smoked',shade=True)\nsns.kdeplot(train.loc[train['SmokingStatus'] == 'Currently smokes', 'Age'], label = 'Currently smokes', shade=True)\n\n# Labeling of plot\nplt.xlabel('Age (years)');\nplt.ylabel('Density');\nplt.title('Distribution of Ages over SmokingStatus');","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(16, 6))\nsns.kdeplot(train.loc[train['SmokingStatus'] == 'Ex-smoker', 'FVC'], label = 'Ex-smoker',shade=True)\nsns.kdeplot(train.loc[train['SmokingStatus'] == 'Never smoked', 'FVC'], label = 'Never smoked',shade=True)\nsns.kdeplot(train.loc[train['SmokingStatus'] == 'Currently smokes', 'FVC'], label = 'Currently smokes', shade=True)\n\n# Labeling of plot\nplt.xlabel('FVC');\nplt.ylabel('Density');\nplt.title('Distribution of FVC over SmokingStatus');","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Use px.scatter to analyze relationships between features**"},{"metadata":{"trusted":true},"cell_type":"code","source":"import plotly.express as px\nfig = px.scatter(train, x=\"Weeks\", y=\"FVC\", color=\"Age\")\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.scatter(train, x=\"Weeks\", y=\"FVC\", color=\"SmokingStatus\")\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.scatter(train, x=\"FVC\", y=\"Percent\", color='SmokingStatus')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#One patient to lineplot","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"patient = train[train.Patient == 'ID00228637202259965313869']\nfig = px.line(patient, x=\"Weeks\", y=\"FVC\", color='SmokingStatus')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Value counts for the features"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.hist(train[\"Sex\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.countplot(data=train, x=\"SmokingStatus\", hue=\"Sex\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Visualizing dcm!!"},{"metadata":{"trusted":true},"cell_type":"code","source":"first_patient_path = \"/kaggle/input/osic-pulmonary-fibrosis-progression/train/ID00123637202217151272140/\"\nimages = os.listdir(first_patient_path)\nnum = []\nfor image in images:\n    num.append(int(image.split(\".\")[0]))\nprint(sorted(num))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig=plt.figure(figsize=(12, 12))\ncolumns = 4\nrows = 5\nfor i in range(1, columns*rows +1):\n    filename = first_patient_path + \"/\" + str(num[i]) + \".dcm\"\n    ds = pydicom.dcmread(filename)\n    fig.add_subplot(rows, columns, i)\n    plt.imshow(ds.pixel_array, cmap='jet')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def dump(obj):\n    dic = dict()\n    for attr in dir(obj):\n        dic[attr] = getattr(obj, attr)\n    return dic\n\nsample_dcm = pydicom.dcmread(\"/kaggle/input/osic-pulmonary-fibrosis-progression/train/ID00123637202217151272140/2.dcm\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sample_dic = dump(sample_dcm)\nsample_dic","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.imshow(sample_dcm.pixel_array, cmap =\"jet\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}