{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":20604,"databundleVersionId":1357052,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-29T02:28:20.452019Z","iopub.execute_input":"2025-08-29T02:28:20.452376Z","iopub.status.idle":"2025-08-29T02:30:09.046173Z","shell.execute_reply.started":"2025-08-29T02:28:20.452346Z","shell.execute_reply":"2025-08-29T02:30:09.045239Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/osic-pulmonary-fibrosis-progression/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/osic-pulmonary-fibrosis-progression/test.csv\")\ntrain.head()\ntest.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-29T02:30:09.047764Z","iopub.execute_input":"2025-08-29T02:30:09.048144Z","iopub.status.idle":"2025-08-29T02:30:09.103691Z","shell.execute_reply.started":"2025-08-29T02:30:09.048122Z","shell.execute_reply":"2025-08-29T02:30:09.102804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isnull().sum() #check any NaN value\ntrain.duplicated().sum() #check any duplicated data\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-29T02:30:09.104623Z","iopub.execute_input":"2025-08-29T02:30:09.104952Z","iopub.status.idle":"2025-08-29T02:30:09.121293Z","shell.execute_reply.started":"2025-08-29T02:30:09.104923Z","shell.execute_reply":"2025-08-29T02:30:09.120411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-29T02:30:09.122264Z","iopub.execute_input":"2025-08-29T02:30:09.122664Z","iopub.status.idle":"2025-08-29T02:30:10.058514Z","shell.execute_reply.started":"2025-08-29T02:30:09.122640Z","shell.execute_reply":"2025-08-29T02:30:10.057613Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# GENERAL ","metadata":{}},{"cell_type":"code","source":"\nfig, (ax1, ax2, ax3) = plt.subplots(1, 3, figsize=(12, 6))\n\n\nax1.hist(train[\"Age\"], bins=20, edgecolor=\"black\", color=\"skyblue\")\nax1.set_title(\"Age Distribution\", fontsize=14)\nax1.set_xlabel(\"Age\")\nax1.set_ylabel(\"Frequency\")\n\n\nsex_counts = train[\"Sex\"].value_counts()\nax2.bar(sex_counts.index, sex_counts.values, color=[\"lightcoral\", \"lightblue\"])\nax2.set_title(\"Sex Distribution\", fontsize=14)\nax2.set_xlabel(\"Sex\")\nax2.set_ylabel(\"Count\")\n\n\nsmoke_counts = train[\"SmokingStatus\"].value_counts()\nax3.bar(smoke_counts.index, smoke_counts.values, color=[\"orange\", \"green\", \"blue\"])\nax3.set_title(\"Smoking Status\", fontsize=14)\nax3.set_xlabel(\"Smoking Status\")\nax3.set_ylabel(\"Count\")\n\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-29T02:30:10.060048Z","iopub.execute_input":"2025-08-29T02:30:10.060450Z","iopub.status.idle":"2025-08-29T02:30:10.727073Z","shell.execute_reply.started":"2025-08-29T02:30:10.060429Z","shell.execute_reply":"2025-08-29T02:30:10.726142Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# FVC\n📌Remember:\n\n* FVC: most values lie between 1,000 and 5,000 ml. There are also some very high outliers above 5,000 and little values that lie below 1,000.\n","metadata":{}},{"cell_type":"code","source":"outliers = train[(train['FVC'] < 1000) | (train['FVC'] > 5000)]\noutliers\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-29T02:30:10.728318Z","iopub.execute_input":"2025-08-29T02:30:10.728673Z","iopub.status.idle":"2025-08-29T02:30:10.745219Z","shell.execute_reply.started":"2025-08-29T02:30:10.728642Z","shell.execute_reply":"2025-08-29T02:30:10.744294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(12, 6))\n\n# FVC distribution\nax[0].hist(train[\"FVC\"], bins=50, edgecolor=\"black\")\nax[0].set_title(\"FVC Distribution\")\nax[0].set_xlabel(\"FVC\")\nax[0].set_ylabel(\"Frequency\")\n\n\nax[1].scatter(train[\"FVC\"], train[\"Percent\"], alpha=0.6)\nax[1].set_title(\"FVC vs Percent\")\nax[1].set_xlabel(\"FVC\")\nax[1].set_ylabel(\"Percent\")\n\nfig.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-29T02:30:10.746176Z","iopub.execute_input":"2025-08-29T02:30:10.746489Z","iopub.status.idle":"2025-08-29T02:30:11.210039Z","shell.execute_reply.started":"2025-08-29T02:30:10.746462Z","shell.execute_reply":"2025-08-29T02:30:11.209194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#allocate patients into each age group\nbins = [50,60,65,70,80,90]\nlabels = [\"50-59\", \"60-65\", \"65-70\", \"71-80\", \"81-90\"]\ntrain['AgeGroup'] = pd.cut(train[\"Age\"], bins=bins, labels=labels, right=False, include_lowest=True)\n\n#find the mean of fvc in each age group\navg = train.groupby(\"AgeGroup\",observed=True)[\"FVC\"].mean()\n\n#plot\navg.plot(figsize=(16,10));\nplt.xlabel(\"Age Group\")\nplt.ylabel(\"The mean of FVC\")\nplt.title(\"Average FVC by Age Group\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-29T03:12:31.647532Z","iopub.execute_input":"2025-08-29T03:12:31.648393Z","iopub.status.idle":"2025-08-29T03:12:31.908319Z","shell.execute_reply.started":"2025-08-29T03:12:31.648366Z","shell.execute_reply":"2025-08-29T03:12:31.907526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"g50_59  = train[train['AgeGroup'] == \"50-59\"]\ng60_65  = train[train['AgeGroup'] == \"60-65\"]\ng65_70  = train[train['AgeGroup'] == \"65-70\"]\ng71_80  = train[train['AgeGroup'] == \"71-80\"]\ng81_90  = train[train['AgeGroup'] == \"81-90\"]\nfig, ax = plt.subplots(nrows=2, ncols=3, figsize=(12,6))\n\nax = ax.flatten()\n\n\nax[0].boxplot(g50_59['FVC'])\nax[0].set_title(\"50-59\", fontsize=8)\n\nax[1].boxplot(g60_65['FVC'])\nax[1].set_title(\"60-65\", fontsize=8)\n\nax[2].boxplot(g65_70['FVC'])\nax[2].set_title(\"65-70\", fontsize=8)\n\nax[3].boxplot(g71_80['FVC'])\nax[3].set_title(\"71-80\", fontsize=8)\n\nax[4].boxplot(g81_90['FVC'])\nax[4].set_title(\"81-90\", fontsize=8)\n\n\nfig.delaxes(ax[5])\n\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-29T03:22:57.469873Z","iopub.execute_input":"2025-08-29T03:22:57.470172Z","iopub.status.idle":"2025-08-29T03:22:58.203895Z","shell.execute_reply.started":"2025-08-29T03:22:57.470149Z","shell.execute_reply":"2025-08-29T03:22:58.203115Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}