{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3"},"language_info":{"name":"python"},"colab":{"provenance":[],"authorship_tag":"ABX9TyN0klX1iPZl9/UY0mxIRyAH"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\n\nimport pandas as pd\n\nimport matplotlib.pyplot as plt\n\nimport seaborn as sns","metadata":{"id":"jDspBeRNBUAK","executionInfo":{"status":"ok","timestamp":1733947115037,"user_tz":-60,"elapsed":266,"user":{"displayName":"abderazzak ghalgaoui","userId":"04332753540926458919"}}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train=pd.read_csv('/content/train.csv')\n\ndf_test=pd.read_csv('/content/test.csv')","metadata":{"id":"HKTiwjuCKkck","executionInfo":{"status":"ok","timestamp":1733947115724,"user_tz":-60,"elapsed":422,"user":{"displayName":"abderazzak ghalgaoui","userId":"04332753540926458919"}}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train\n","metadata":{"id":"B15nLytqBpjh","executionInfo":{"status":"ok","timestamp":1733947115725,"user_tz":-60,"elapsed":27,"user":{"displayName":"abderazzak ghalgaoui","userId":"04332753540926458919"}},"outputId":"d7ba8df4-46c4-4fa8-d949-c9418d7b0585"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test","metadata":{"id":"RywXu27iKaCR","executionInfo":{"status":"ok","timestamp":1733947115726,"user_tz":-60,"elapsed":23,"user":{"displayName":"abderazzak ghalgaoui","userId":"04332753540926458919"}},"outputId":"de53b6c3-8814-4649-ed85-80fdae60d282"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.shape","metadata":{"id":"caP1jJ4SF55j","executionInfo":{"status":"ok","timestamp":1733947115727,"user_tz":-60,"elapsed":22,"user":{"displayName":"abderazzak ghalgaoui","userId":"04332753540926458919"}},"outputId":"e7dd9400-98bd-4652-c159-1e6983909e20"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.columns","metadata":{"id":"JYaNkuByIeLo","executionInfo":{"status":"ok","timestamp":1733947115727,"user_tz":-60,"elapsed":17,"user":{"displayName":"abderazzak ghalgaoui","userId":"04332753540926458919"}},"outputId":"725a72b1-aae4-49c1-e6a8-2d7718a048b8"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.info()","metadata":{"id":"3taJMTWhIoRI","executionInfo":{"status":"ok","timestamp":1733947115727,"user_tz":-60,"elapsed":15,"user":{"displayName":"abderazzak ghalgaoui","userId":"04332753540926458919"}},"outputId":"0f66ebaf-527e-4fd4-df39-a7b594b4f8f7"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.isnull().sum()","metadata":{"id":"jX2uEpOCIx_U","executionInfo":{"status":"ok","timestamp":1733947115727,"user_tz":-60,"elapsed":12,"user":{"displayName":"abderazzak ghalgaoui","userId":"04332753540926458919"}},"outputId":"2ee4e6ee-45c5-48bb-f325-3eef62c603c6"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.heatmap(df_train.isnull())","metadata":{"id":"LHd1KPmoI-Ne","executionInfo":{"status":"ok","timestamp":1733947144550,"user_tz":-60,"elapsed":1798,"user":{"displayName":"abderazzak ghalgaoui","userId":"04332753540926458919"}},"outputId":"72f35c4f-0b50-4de7-9460-51961fd4e8e5"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.dtypes","metadata":{"id":"gsgHpl68JKGS","executionInfo":{"status":"ok","timestamp":1733949245993,"user_tz":-60,"elapsed":312,"user":{"displayName":"abderazzak ghalgaoui","userId":"04332753540926458919"}},"outputId":"6851ddac-4ea8-452e-bb47-fc3099571d68"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install polars\n\nimport polars as pl","metadata":{"id":"s0Ax_6PncSV9","executionInfo":{"status":"ok","timestamp":1733949477306,"user_tz":-60,"elapsed":3670,"user":{"displayName":"abderazzak ghalgaoui","userId":"04332753540926458919"}},"outputId":"75d0b623-d205-4ea8-d536-6c815764fb7d"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert the pandas DataFrame to a Polars DataFrame\n\ndf_train_pl = pl.DataFrame(df_train)\n\n\n\n# Calculate the sum of null values for each column using Polars\n\ntrain_null_counts = df_train_pl.select(pl.col(\"*\").is_null().sum())\n","metadata":{"id":"egBgf3DxcRmD","executionInfo":{"status":"ok","timestamp":1733949751382,"user_tz":-60,"elapsed":313,"user":{"displayName":"abderazzak ghalgaoui","userId":"04332753540926458919"}}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_null_counts","metadata":{"id":"KaRyGMSpbQtb","executionInfo":{"status":"ok","timestamp":1733949752943,"user_tz":-60,"elapsed":332,"user":{"displayName":"abderazzak ghalgaoui","userId":"04332753540926458919"}},"outputId":"890034dc-9778-42b3-86a0-a9c96e6931a8"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert the pandas DataFrame to a Polars DataFrame\n\ndf_test_pl = pl.DataFrame(df_test)\n\n\n\n# Calculate the sum of null values for each column using Polars\n\ntest_null_counts = df_test_pl.select(pl.col(\"*\").is_null().sum())\n","metadata":{"id":"0mLK0TJQbvk1","executionInfo":{"status":"ok","timestamp":1733949779128,"user_tz":-60,"elapsed":286,"user":{"displayName":"abderazzak ghalgaoui","userId":"04332753540926458919"}}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_null_counts","metadata":{"id":"PZeyHy9ldeoS","executionInfo":{"status":"ok","timestamp":1733949788523,"user_tz":-60,"elapsed":318,"user":{"displayName":"abderazzak ghalgaoui","userId":"04332753540926458919"}},"outputId":"509ede3a-ded2-4f1b-ecb0-93b6243e3d9d"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from matplotlib.ticker import PercentFormatter","metadata":{"id":"YaPChp2Ke5a7","executionInfo":{"status":"ok","timestamp":1733950155849,"user_tz":-60,"elapsed":362,"user":{"displayName":"abderazzak ghalgaoui","userId":"04332753540926458919"}}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#check missingness for all features\n\nmissing_count = (\n\n    df_train_pl # Use the Polars DataFrame instead of the Pandas DataFrame\n\n    .null_count() # Use the Polars null_count method\n\n    .transpose(include_header=True,\n\n               header_name='feature',\n\n               column_names=['train_null_count'])\n\n    .sort('train_null_count', descending=True)\n\n    .with_columns((pl.col('train_null_count') / len(df_train)).alias('null_ratio'))\n\n)\n\nplt.figure(figsize=(5, 15))\n\nplt.title('Missing values for training dataset')\n\nplt.barh(np.arange(len(missing_count)), missing_count.get_column('null_ratio'), color='navy', label='missing')\n\nplt.barh(np.arange(len(missing_count)),\n\n         1 - missing_count.get_column('null_ratio'),\n\n         left=missing_count.get_column('null_ratio'),\n\n         color='orange', label='available')\n\nplt.yticks(np.arange(len(missing_count)), missing_count.get_column('feature'))\n\n# Assuming PercentFormatter is imported, if not, import it\n\n# from matplotlib.ticker import PercentFormatter\n\nplt.gca().xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\n\nplt.xlim(0, 1)\n\nplt.legend()\n\nplt.show()\n","metadata":{"id":"MEAAPbO3dg6g","executionInfo":{"status":"ok","timestamp":1733950159664,"user_tz":-60,"elapsed":2619,"user":{"displayName":"abderazzak ghalgaoui","userId":"04332753540926458919"}},"outputId":"b48e914b-a07d-4674-f93f-9e30876eec37"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"supervised_usable = (\n\n    df_train_pl  # Use the Polars DataFrame instead of the Pandas DataFrame\n\n    .filter(pl.col('sii').is_not_null())\n\n)\n\n\n\nmissing_count = (\n\n    supervised_usable\n\n    .null_count()\n\n    .transpose(include_header=True,\n\n               header_name='feature',\n\n               column_names=['null_count'])\n\n    .sort('null_count', descending=True)\n\n    .with_columns((pl.col('null_count') / len(supervised_usable)).alias('null_ratio'))\n\n)\n\nplt.figure(figsize=(6, 15))\n\nplt.title(f'Missing values for the {len(supervised_usable)} observations with the target (sii)')\n\nplt.barh(np.arange(len(missing_count)), missing_count.get_column('null_ratio'), color='navy', label='missing')\n\nplt.barh(np.arange(len(missing_count)),\n\n         1 - missing_count.get_column('null_ratio'),\n\n         left=missing_count.get_column('null_ratio'),\n\n         color='orange', label='available')\n\nplt.yticks(np.arange(len(missing_count)), missing_count.get_column('feature'))\n\nplt.gca().xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\n\nplt.xlim(0, 1)\n\nplt.legend()\n\nplt.show()\n","metadata":{"id":"rrZBH6BBeJ-H","executionInfo":{"status":"ok","timestamp":1733950363766,"user_tz":-60,"elapsed":2382,"user":{"displayName":"abderazzak ghalgaoui","userId":"04332753540926458919"}},"outputId":"ffb80c9e-a4e4-4c0a-a1e4-15db2dee7a20"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"vc = df_train['Basic_Demos-Enroll_Season'].value_counts()\n\n\n\n# Extract the seasons and counts from the index and values of the Series\n\nseasons = vc.index\n\ncounts = vc.values\n\n\n\n# Sort the seasons and counts based on the counts in descending order\n\nsort = sorted(range(len(counts)), key=lambda i: counts[i], reverse=True)\n\nseasons = [seasons[i] for i in sort]\n\ncounts = [counts[i] for i in sort]\n\n\n\nplt.bar(seasons, counts, color=\"orange\")\n\nplt.xlabel(\"Seasons\")\n\nplt.ylabel(\"Counts\")\n\nplt.title(\"Enrollment season\")\n\nplt.show()","metadata":{"id":"GooP1TBQfaSv","executionInfo":{"status":"ok","timestamp":1733950498832,"user_tz":-60,"elapsed":305,"user":{"displayName":"abderazzak ghalgaoui","userId":"04332753540926458919"}},"outputId":"88c8fb81-5880-4530-98c5-8e544068d235"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"_, axs = plt.subplots(2, 1, sharex=True, sharey=True)\n\n# Define target_labels here\n\ntarget_labels = ['No impairment', 'Mild impairment', 'Moderate impairment', 'Severe impairment']  # Example labels, replace with actual labels if different\n\n\n\nfor sex in range(2):\n\n    ax = axs.ravel()[sex]\n\n    # Filter using pandas syntax instead of polars\n\n    vc = df_train[df_train['Basic_Demos-Sex'] == sex]['sii'].value_counts()\n\n    ax.bar(vc.index,\n\n           vc.values / vc.values.sum(),\n\n           color=['orange', 'navy'][sex],\n\n           label=['Male', 'Female'][sex])\n\n    ax.set_xticks(np.arange(4), target_labels)\n\n    ax.yaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\n\n    ax.set_ylabel('Count')\n\n    ax.legend()\n\nplt.suptitle('Severity impairment index by sex')\n\naxs.ravel()[1].set_xlabel('Severity impairment index (sii)')\n\nplt.show()","metadata":{"id":"k_oUSgMjgePj","executionInfo":{"status":"ok","timestamp":1733950820494,"user_tz":-60,"elapsed":717,"user":{"displayName":"abderazzak ghalgaoui","userId":"04332753540926458919"}},"outputId":"799668cb-b72f-4275-8c42-b0d5e6b0f512"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"_, axs = plt.subplots(2, 1, sharex=True)\n\nfor sex in range(2):\n\n    ax = axs.ravel()[sex]\n\n    # Filter using pandas syntax instead of polars\n\n    vc = df_train[df_train['Basic_Demos-Sex'] == sex]['Basic_Demos-Age'].value_counts()  # Changed this line\n\n    ax.bar(vc.index,\n\n           vc.values,  # Removed the unnecessary division\n\n           color=['orange', 'navy'][sex],\n\n           label=['Male', 'Female'][sex])\n\n    ax.xaxis.set_major_locator(plt.MaxNLocator(integer=True))  # Use plt.MaxNLocator\n\n    ax.set_ylabel('count')\n\n    ax.legend()\n\nplt.suptitle('Age distribution')\n\naxs.ravel()[1].set_xlabel('years')\n\nplt.show()","metadata":{"id":"nlM4tcGjhq-8","executionInfo":{"status":"ok","timestamp":1733950947594,"user_tz":-60,"elapsed":1413,"user":{"displayName":"abderazzak ghalgaoui","userId":"04332753540926458919"}},"outputId":"c5baaf4d-92c7-4b85-ad04-4a9f86b9d4bc"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"(df_train_pl  # Use df_train_pl (Polars DataFrame) instead of df_train\n\n .select(pl.col('PCIAT-PCIAT_Total'))\n\n .group_by(df_train_pl.get_column('sii')) # Use df_train_pl here as well\n\n .agg(pl.col('PCIAT-PCIAT_Total').min().alias('PCIAT-PCIAT_Total min'),\n\n      pl.col('PCIAT-PCIAT_Total').max().alias('PCIAT-PCIAT_Total max'),\n\n      pl.col('PCIAT-PCIAT_Total').len().alias('count'))\n\n .sort('sii')\n\n)\n\n#the dataset is imbalanced. almost all samples are in class 0, very few in class 3.","metadata":{"id":"v0aOeBRTiJQr","executionInfo":{"status":"ok","timestamp":1733951037623,"user_tz":-60,"elapsed":348,"user":{"displayName":"abderazzak ghalgaoui","userId":"04332753540926458919"}},"outputId":"7d47eaa3-956a-423d-d309-745195f1e26b"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#checking BMI difference for those who had it measured 2x\n\nbmi_ratio = df_train_pl.select(pl.col('Physical-BMI') / pl.col('BIA-BIA_BMI')).to_series()  # Use df_train_pl instead of df_train\n\ncolor = (bmi_ratio < 0.7) | (bmi_ratio > 1.3) # red if difference > 30 %\n\n\n\nplt.scatter(df_train_pl.get_column('Physical-BMI'),  # Use df_train_pl instead of df_train\n\n            df_train_pl.get_column('BIA-BIA_BMI'),  # Use df_train_pl instead of df_train\n\n            s=6,\n\n            cmap='coolwarm',\n\n            c=color)\n\nplt.gca().set_aspect('equal')\n\nplt.xlabel('Physical-BMI')\n\nplt.ylabel('BIA-BIA_BMI')\n\nplt.title('How much can BMI change in a year?')\n\nplt.show()","metadata":{"id":"2s4IGL0uij4v","executionInfo":{"status":"ok","timestamp":1733951149087,"user_tz":-60,"elapsed":705,"user":{"displayName":"abderazzak ghalgaoui","userId":"04332753540926458919"}},"outputId":"fd456994-2e0e-487e-855e-820dffe93093"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Assuming df_train_pl is your Polars DataFrame\n\ncolor = df_train_pl.get_column('Physical-Systolic_BP') <= df_train_pl.get_column('Physical-Diastolic_BP')  # Use df_train_pl instead of df_train and train\n\n\n\nplt.scatter(df_train_pl.get_column('Physical-Diastolic_BP'),  # Use df_train_pl instead of df_train\n\n            df_train_pl.get_column('Physical-Systolic_BP'),  # Use df_train_pl instead of df_train\n\n            s=6,\n\n            cmap='coolwarm',\n\n            c=color)\n\nplt.gca().set_aspect('equal')\n\nplt.plot([0, 200], [0, 200], color='gray')\n\nplt.xlabel('Diastolic blood pressure')\n\nplt.ylabel('Systolic blood pressure')\n\nplt.title('Systolic vs. diastolic blood pressure')\n\nplt.xticks(np.linspace(0, 200, 5))\n\nplt.yticks(np.linspace(0, 200, 5))\n\nplt.show()","metadata":{"id":"9iKlisHKi9gG","executionInfo":{"status":"ok","timestamp":1733951253610,"user_tz":-60,"elapsed":796,"user":{"displayName":"abderazzak ghalgaoui","userId":"04332753540926458919"}},"outputId":"11b4021c-4c6d-4293-9284-d070fa02ee2f"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#this is meant to check for outliers, but it's only cutting off at 50 on the left tail\n\n#would it be better to use IQR to check for outliers instead?\n\n# Use df_train_pl (Polars DataFrame) instead of df_train\n\nvc = df_train_pl.get_column('Physical-HeartRate').value_counts()\n\n# Access values using the 'to_numpy()' method of the Polars Series\n\n# color is now defined using the 'Physical-HeartRate' column, which contains the original values\n\ncolor = np.where(vc.get_column('Physical-HeartRate').to_numpy() < 50, 'r', 'navy')\n\nplt.figure(figsize=(10, 3))\n\nplt.title('Heart rate with outliers')\n\n# Access values using the 'to_numpy()' method of the Polars Series,\n\n# and using the correct column name for counts, which is the original column name in this case\n\nplt.bar(vc.get_column('Physical-HeartRate').to_numpy(), vc.get_column(vc.columns[1]).to_numpy(), color=color)\n\nplt.xlabel('Physical-HeartRate')\n\nplt.ylabel('count')\n\nplt.show()","metadata":{"id":"d8GZ9eBQjtWm","executionInfo":{"status":"ok","timestamp":1733951478738,"user_tz":-60,"elapsed":989,"user":{"displayName":"abderazzak ghalgaoui","userId":"04332753540926458919"}},"outputId":"d8d0c2d3-03ab-4985-95c5-a9f93d6071be"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"vc = df_train['SDS-SDS_Total_Raw'].value_counts()\n\nplt.figure(figsize=(6, 2))\n\nplt.title('Sleep disturbance scale')\n\nplt.bar(vc.index, vc.values, color='navy')  # Use index and values for pandas Series\n\nplt.xlabel('SDS-SDS_Total_Raw')\n\nplt.ylabel('count')\n\nplt.show()\n\n\n\nplt.title('Sleep disturbance: conversion from raw to T')\n\nplt.scatter(df_train['SDS-SDS_Total_Raw'],  # Use bracket notation to access columns\n\n            df_train['SDS-SDS_Total_T'],  # Use bracket notation to access columns\n\n            color='navy')\n\nplt.xlabel('SDS-SDS_Total_Raw')\n\nplt.ylabel('SDS-SDS_Total_T')\n\nplt.show()\n\n\n\nvc = df_train['SDS-SDS_Total_T'].value_counts()\n\nplt.figure(figsize=(6, 2))\n\nplt.title('Sleep disturbance scale')\n\nplt.bar(vc.index, vc.values, color='navy')  # Use index and values for pandas Series\n\nplt.xlabel('SDS-SDS_Total_T')\n\nplt.ylabel('count')\n\nplt.show()","metadata":{"id":"LUjMYy95kRkJ","executionInfo":{"status":"ok","timestamp":1733951608527,"user_tz":-60,"elapsed":1275,"user":{"displayName":"abderazzak ghalgaoui","userId":"04332753540926458919"}},"outputId":"1b065dd0-773e-48be-ea9d-1257a6fb0b70"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"id":"SftmVOIikdAY"},"outputs":[],"execution_count":null}]}