{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# This notebook will be a comprehensive EDA done for the competition\nThe notebook has First level data understanding that includes : \n1. Basic data information\n2. Descriptive statistics\n3. Missing value analysis\n4. Data distribution\n5. Distribution of label with other columns\n6. First submission","metadata":{}},{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import polars as pl\nimport pandas as pd\npd.set_option('display.max_columns',None)\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport glob\nimport numpy as np\nimport os\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.preprocessing import StandardScaler\nfrom matplotlib.ticker import PercentFormatter\nimport warnings\nwarnings.filterwarnings('ignore')\nfrom statsmodels.tsa.seasonal import seasonal_decompose\nfrom statsmodels.graphics.tsaplots import plot_acf, plot_pacf\n\nfrom sklearn.decomposition import PCA\nfrom sklearn.manifold import TSNE\nfrom sklearn.cluster import KMeans, DBSCAN\nfrom sklearn.ensemble import RandomForestClassifier\nimport shap\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-03T07:02:32.856226Z","iopub.execute_input":"2024-11-03T07:02:32.856671Z","iopub.status.idle":"2024-11-03T07:02:37.862225Z","shell.execute_reply.started":"2024-11-03T07:02:32.856626Z","shell.execute_reply":"2024-11-03T07:02:37.860740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading the data","metadata":{}},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"Stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:02:37.863777Z","iopub.execute_input":"2024-11-03T07:02:37.864646Z","iopub.status.idle":"2024-11-03T07:02:37.874689Z","shell.execute_reply.started":"2024-11-03T07:02:37.864600Z","shell.execute_reply":"2024-11-03T07:02:37.873384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 1: Read the tabular data\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\nprint(\"Step 1 : Read the tabular data completed\")\n\n# Step 2: Read the data dictionary\ndata_dict = pl.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')\nprint(\"Step 2 : Read the data dictionary completed\")\n\n# Step 3: Read the actigraphy data\ntrain_actigraphy_df = load_time_series('/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet')\ntest_actigraphy_df = load_time_series('/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet')\nprint(\"Step 3: Read the actigraphy data completed\")\n# Step 4: Merge the tabular data with the actigraphy data based on `id`\ntrain_merged_df = pd.merge(train, train_actigraphy_df, how=\"left\", on='id')\ntest_merged_df = pd.merge(test, test_actigraphy_df, how=\"left\", on='id')\nprint(\"Step 4: Merge the tabular data with the actigraphy data based on `id` completed\")\nprint(\"Train data shape : \",train_merged_df.shape)\nprint(\"Test data shape : \",test_merged_df.shape)","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:02:37.876415Z","iopub.execute_input":"2024-11-03T07:02:37.877333Z","iopub.status.idle":"2024-11-03T07:04:21.984305Z","shell.execute_reply.started":"2024-11-03T07:02:37.877276Z","shell.execute_reply":"2024-11-03T07:04:21.983047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data_columns = set(test_merged_df.columns)\ntrain_data_columns = set(train_merged_df.columns)\nprint(len(test_data_columns))\ncolumns_not_in_test = train_data_columns - test_data_columns\nprint(\"Columns not in test : \")\nprint(columns_not_in_test)\n# Select only the columns present in the test dataset plus the `sii` column\ncolumns_to_keep = list(test_data_columns) + ['sii']\ntrain_merged_df = train_merged_df[columns_to_keep]\nprint(len(columns_to_keep))","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:21.986940Z","iopub.execute_input":"2024-11-03T07:04:21.987335Z","iopub.status.idle":"2024-11-03T07:04:21.998618Z","shell.execute_reply.started":"2024-11-03T07:04:21.987295Z","shell.execute_reply":"2024-11-03T07:04:21.997507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# First level EDA","metadata":{}},{"cell_type":"markdown","source":"## 1. Basic Data Information","metadata":{}},{"cell_type":"code","source":"print(\"Basic Data Information\")\nprint(train_merged_df.info())","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:22.000459Z","iopub.execute_input":"2024-11-03T07:04:22.000869Z","iopub.status.idle":"2024-11-03T07:04:22.026682Z","shell.execute_reply.started":"2024-11-03T07:04:22.000826Z","shell.execute_reply":"2024-11-03T07:04:22.025351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_merged_df.head())","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:22.028325Z","iopub.execute_input":"2024-11-03T07:04:22.028781Z","iopub.status.idle":"2024-11-03T07:04:22.101200Z","shell.execute_reply.started":"2024-11-03T07:04:22.028731Z","shell.execute_reply":"2024-11-03T07:04:22.099830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Descriptive Statistics","metadata":{}},{"cell_type":"code","source":"print(train_merged_df.describe(include='all'))","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:22.102864Z","iopub.execute_input":"2024-11-03T07:04:22.103368Z","iopub.status.idle":"2024-11-03T07:04:22.475752Z","shell.execute_reply.started":"2024-11-03T07:04:22.103315Z","shell.execute_reply":"2024-11-03T07:04:22.474394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Insights<br>\n1. There are 11 categorical columns out of which one is id which we will drop and other 10 are season columns only.\n2. There are 2 integer columns i.e. Age and Sex.\n3. Rest other are float columns only including sii.\n","metadata":{}},{"cell_type":"markdown","source":"## 3. Missing Values Analysis","metadata":{}},{"cell_type":"code","source":"print(\"Missing Values Analysis\")\nmissing_values = train_merged_df.isnull().sum()\nmissing_percentage = (missing_values / len(train_merged_df)) * 100\nmissing_data = pd.DataFrame({'Missing Values': missing_values,'Total_values' : len(train_merged_df), 'Percentage': missing_percentage})\nprint(missing_data[missing_data['Missing Values'] > 0])","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:22.477086Z","iopub.execute_input":"2024-11-03T07:04:22.477477Z","iopub.status.idle":"2024-11-03T07:04:22.493403Z","shell.execute_reply.started":"2024-11-03T07:04:22.477439Z","shell.execute_reply":"2024-11-03T07:04:22.492236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Assuming `train_merged_df` is your pandas DataFrame\nmissing_count = train_merged_df.isnull().sum().reset_index()\nmissing_count.columns = ['feature', 'null_count']\nmissing_count['null_ratio'] = missing_count['null_count'] / len(train_merged_df)\nmissing_count = missing_count.sort_values(by='null_ratio', ascending=False)\n\n# Plotting with seaborn\nplt.figure(figsize=(15, 35))\nplt.title('Missing values over the whole training dataset', fontsize=16)\nsns.barplot(\n    y='feature', \n    x='null_ratio', \n    data=missing_count, \n    palette='viridis'\n)\nplt.xlabel('Missing Ratio', fontsize=14)\nplt.ylabel('Feature', fontsize=14)\nplt.gca().xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\nplt.xlim(0, 1)\n\n# Increase the distance between y-ticks\nplt.gca().yaxis.set_tick_params(pad=10)\n\nfor index, value in enumerate(missing_count['null_ratio']):\n    plt.text(value, index, f'{value:.1%}', va='center', ha='left', fontsize=10, color='black')\n\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:22.496934Z","iopub.execute_input":"2024-11-03T07:04:22.497336Z","iopub.status.idle":"2024-11-03T07:04:25.141124Z","shell.execute_reply.started":"2024-11-03T07:04:22.497297Z","shell.execute_reply":"2024-11-03T07:04:25.139841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Insights<br>\n1. Majority of the columns have more than 70% of the data as null\n2. Label sii has around 30% of null Values.\n3. We can build a model with labeled data and then can use unlabelled data for unsupervised model.\n4. We can also explore the semi supervised approach","metadata":{}},{"cell_type":"code","source":"train_df_sii_non_null = train_merged_df.dropna(subset = 'sii')\n\n# Assuming `train_df_sii_non_null` is your pandas DataFrame\nmissing_count = train_df_sii_non_null.isnull().sum().reset_index()\nmissing_count.columns = ['feature', 'null_count']\nmissing_count['null_ratio'] = missing_count['null_count'] / len(train_merged_df)\nmissing_count = missing_count.sort_values(by='null_ratio', ascending=False)\n\n# Plotting with seaborn\nplt.figure(figsize=(15, 35))\nplt.title('Missing values over non missing sii data', fontsize=16)\nsns.barplot(\n    y='feature', \n    x='null_ratio', \n    data=missing_count, \n    palette='viridis'\n)\nplt.xlabel('Missing Ratio', fontsize=14)\nplt.ylabel('Feature', fontsize=14)\nplt.gca().xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\nplt.xlim(0, 1)\n\n# Increase the distance between y-ticks\nplt.gca().yaxis.set_tick_params(pad=10)\n\nfor index, value in enumerate(missing_count['null_ratio']):\n    plt.text(value, index, f'{value:.1%}', va='center', ha='left', fontsize=10, color='black')\n\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:25.142870Z","iopub.execute_input":"2024-11-03T07:04:25.143521Z","iopub.status.idle":"2024-11-03T07:04:27.599337Z","shell.execute_reply.started":"2024-11-03T07:04:25.143462Z","shell.execute_reply":"2024-11-03T07:04:27.598189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Data Distribution","metadata":{}},{"cell_type":"code","source":"# 1. Overall Distribution of Sex\nplt.figure(figsize=(6, 4))\nsns.countplot(x='Basic_Demos-Sex', data=train_merged_df, palette='viridis')\nplt.title('Overall Distribution of Sex', fontsize=16)\nplt.xlabel('Sex', fontsize=14)\nplt.ylabel('Count', fontsize=14)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:27.601518Z","iopub.execute_input":"2024-11-03T07:04:27.601995Z","iopub.status.idle":"2024-11-03T07:04:27.816516Z","shell.execute_reply.started":"2024-11-03T07:04:27.601947Z","shell.execute_reply":"2024-11-03T07:04:27.815420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 2. Distribution of Sex for Each Individual Value\n# Assuming we want to see the distribution of sex against another feature, e.g., 'age'\nplt.figure(figsize=(6, 4))\nsns.histplot(data=train_merged_df, x='Basic_Demos-Age', hue='Basic_Demos-Sex', multiple='stack', palette='viridis')\nplt.title('Distribution of Age by Sex', fontsize=16)\nplt.xlabel('Age', fontsize=14)\nplt.ylabel('Count', fontsize=14)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:27.818154Z","iopub.execute_input":"2024-11-03T07:04:27.818554Z","iopub.status.idle":"2024-11-03T07:04:28.236217Z","shell.execute_reply.started":"2024-11-03T07:04:27.818517Z","shell.execute_reply":"2024-11-03T07:04:28.235127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 3. Calculate the distribution of seasons\nseason_counts = train_merged_df['Basic_Demos-Enroll_Season'].value_counts()\n\n# Plotting the pie chart\nplt.figure(figsize=(4, 6))\nplt.pie(season_counts, labels=season_counts.index, autopct='%1.1f%%', startangle=140, colors=sns.color_palette('viridis', len(season_counts)))\nplt.title('Distribution of Seasons', fontsize=16)\nplt.axis('equal')  # Equal aspect ratio ensures that pie is drawn as a circle.\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:28.241480Z","iopub.execute_input":"2024-11-03T07:04:28.241975Z","iopub.status.idle":"2024-11-03T07:04:28.391997Z","shell.execute_reply.started":"2024-11-03T07:04:28.241932Z","shell.execute_reply":"2024-11-03T07:04:28.390584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 4. Plotting the distribution of the target label 'sii'\nplt.figure(figsize=(10, 6))\nsns.countplot(x='sii', data=train_merged_df, palette='viridis')\nplt.title('Distribution of Severity Impairment Index (sii)', fontsize=16)\nplt.xlabel('Severity Impairment Index (sii)', fontsize=14)\nplt.ylabel('Count', fontsize=14)\n\n# Adding percentage labels on top of the bars\ntotal = len(train_merged_df)\nfor p in plt.gca().patches:\n    percentage = f'{100 * p.get_height() / total:.1f}%'\n    plt.gca().annotate(percentage, (p.get_x() + p.get_width() / 2., p.get_height()), \n                       ha='center', va='center', fontsize=12, color='black', xytext=(0, 10), \n                       textcoords='offset points')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:28.394148Z","iopub.execute_input":"2024-11-03T07:04:28.394667Z","iopub.status.idle":"2024-11-03T07:04:28.701375Z","shell.execute_reply.started":"2024-11-03T07:04:28.394613Z","shell.execute_reply":"2024-11-03T07:04:28.700217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Insights\n1. Distribution of sex is not balanced\n2. Most of the childer are in the age group 7 to 12\n3. Distribution of seasons is very much balanced\n4. Distribution of label sii is imbalanced so will have to take care if we build a classification model","metadata":{}},{"cell_type":"markdown","source":"## 5. Distribution of label with respect to other columns","metadata":{}},{"cell_type":"code","source":"# 1. Distribution of `sii` with Age\nplt.figure(figsize=(10, 6))\nsns.histplot(data=train_merged_df, x='Basic_Demos-Age', hue='sii', multiple='stack', palette='viridis')\nplt.title('Distribution of Age by Severity Impairment Index (sii)', fontsize=16)\nplt.xlabel('Age', fontsize=14)\nplt.ylabel('Count', fontsize=14)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:28.702756Z","iopub.execute_input":"2024-11-03T07:04:28.703115Z","iopub.status.idle":"2024-11-03T07:04:29.570734Z","shell.execute_reply.started":"2024-11-03T07:04:28.703078Z","shell.execute_reply":"2024-11-03T07:04:29.569565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 2. Distribution of `sii` with Sex\nplt.figure(figsize=(10, 6))\nsns.countplot(x='Basic_Demos-Sex', hue='sii', data=train_merged_df, palette='viridis')\nplt.title('Distribution of Sex by Severity Impairment Index (sii)', fontsize=16)\nplt.xlabel('Sex', fontsize=14)\nplt.ylabel('Count', fontsize=14)\nplt.legend(title='sii')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:29.572135Z","iopub.execute_input":"2024-11-03T07:04:29.572520Z","iopub.status.idle":"2024-11-03T07:04:29.817957Z","shell.execute_reply.started":"2024-11-03T07:04:29.572481Z","shell.execute_reply":"2024-11-03T07:04:29.816799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 3. Distribution of `sii` with Season\nplt.figure(figsize=(10, 6))\nsns.countplot(x='Basic_Demos-Enroll_Season', hue='sii', data=train_merged_df, palette='viridis')\nplt.title('Distribution of Seasons by Severity Impairment Index (sii)', fontsize=16)\nplt.xlabel('Season', fontsize=14)\nplt.ylabel('Count', fontsize=14)\nplt.legend(title='sii')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:29.819566Z","iopub.execute_input":"2024-11-03T07:04:29.820049Z","iopub.status.idle":"2024-11-03T07:04:30.140832Z","shell.execute_reply.started":"2024-11-03T07:04:29.819998Z","shell.execute_reply":"2024-11-03T07:04:30.139692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 6. Time Series Analysis (for actigraphy data)","metadata":{}},{"cell_type":"code","source":"# lets load one ids data for actigraphy data analysis\nsample_data =pd.read_parquet('/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id=00115b9f/part-0.parquet')\nsample_id = \"00115b9f\"\n# Ensure ENMO is calculated if not already present\nif 'enmo' not in sample_data.columns:\n    sample_data['enmo'] = np.sqrt(sample_data['X']**2 + sample_data['Y']**2 + sample_data['Z']**2) - 1\n    sample_data['enmo'] = sample_data['enmo'].apply(lambda x: max(x, 0))\n\n# Plot the time series for ENMO\nplt.figure(figsize=(10, 6))\nplt.plot(sample_data['step'], sample_data['enmo'])\nplt.title(f'ENMO over Time for Participant {sample_id}')\nplt.xlabel('Time Step')\nplt.ylabel('ENMO')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:30.142252Z","iopub.execute_input":"2024-11-03T07:04:30.142591Z","iopub.status.idle":"2024-11-03T07:04:30.423034Z","shell.execute_reply.started":"2024-11-03T07:04:30.142558Z","shell.execute_reply":"2024-11-03T07:04:30.421891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Time Series Decomposition\ndecomposition = seasonal_decompose(sample_data['enmo'], period=24, model='additive')\nfig = decomposition.plot()\nfig.set_size_inches(10, 6)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:30.424334Z","iopub.execute_input":"2024-11-03T07:04:30.424662Z","iopub.status.idle":"2024-11-03T07:04:31.504594Z","shell.execute_reply.started":"2024-11-03T07:04:30.424629Z","shell.execute_reply":"2024-11-03T07:04:31.503183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Rolling Statistics\nrolling_mean = sample_data['enmo'].rolling(window=24).mean()\nrolling_std = sample_data['enmo'].rolling(window=24).std()\n\nplt.figure(figsize=(10, 6))\nplt.plot(sample_data['step'], sample_data['enmo'], label='Original')\nplt.plot(sample_data['step'], rolling_mean, label='Rolling Mean')\nplt.plot(sample_data['step'], rolling_std, label='Rolling Std')\nplt.title('Rolling Mean & Standard Deviation')\nplt.xlabel('Time Step')\nplt.ylabel('ENMO')\nplt.legend()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:31.506249Z","iopub.execute_input":"2024-11-03T07:04:31.506622Z","iopub.status.idle":"2024-11-03T07:04:31.961509Z","shell.execute_reply.started":"2024-11-03T07:04:31.506586Z","shell.execute_reply":"2024-11-03T07:04:31.960364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Autocorrelation and Partial Autocorrelation\nfig, axes = plt.subplots(1, 2, figsize=(14, 6))\nplot_acf(sample_data['enmo'], ax=axes[0])\nplot_pacf(sample_data['enmo'], ax=axes[1])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:31.962945Z","iopub.execute_input":"2024-11-03T07:04:31.963327Z","iopub.status.idle":"2024-11-03T07:04:32.948781Z","shell.execute_reply.started":"2024-11-03T07:04:31.963288Z","shell.execute_reply":"2024-11-03T07:04:32.947624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fourier Transform\nfft_values = np.fft.fft(sample_data['enmo'])\nfft_freq = np.fft.fftfreq(len(fft_values))\n\nplt.figure(figsize=(10, 6))\nplt.plot(fft_freq, np.abs(fft_values))\nplt.title('Fourier Transform')\nplt.xlabel('Frequency')\nplt.ylabel('Amplitude')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:32.950381Z","iopub.execute_input":"2024-11-03T07:04:32.950848Z","iopub.status.idle":"2024-11-03T07:04:33.240357Z","shell.execute_reply.started":"2024-11-03T07:04:32.950795Z","shell.execute_reply":"2024-11-03T07:04:33.239207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize Other Features\nfeatures_to_plot = ['anglez', 'non-wear_flag', 'light', 'battery_voltage']\nfor feature in features_to_plot:\n    plt.figure(figsize=(10, 6))\n    plt.plot(sample_data['step'], sample_data[feature])\n    plt.title(f'{feature} over Time for Participant {sample_id}')\n    plt.xlabel('Time Step')\n    plt.ylabel(feature)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:33.241683Z","iopub.execute_input":"2024-11-03T07:04:33.242016Z","iopub.status.idle":"2024-11-03T07:04:34.512062Z","shell.execute_reply.started":"2024-11-03T07:04:33.241983Z","shell.execute_reply":"2024-11-03T07:04:34.510700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from statsmodels.tsa.stattools import grangercausalitytests\n\n# Granger Causality Test\nmax_lag = 12\ntest_result = grangercausalitytests(sample_data[['enmo', 'light']], max_lag, verbose=True)","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:34.513846Z","iopub.execute_input":"2024-11-03T07:04:34.514357Z","iopub.status.idle":"2024-11-03T07:04:36.215697Z","shell.execute_reply.started":"2024-11-03T07:04:34.514302Z","shell.execute_reply":"2024-11-03T07:04:36.214453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from statsmodels.tsa.api import VAR\n\n# Vector Autoregression (VAR)\nmodel = VAR(sample_data[['enmo', 'light', 'anglez']])\nmodel_fitted = model.fit(5)\n\n# Forecasting\nforecast_input = sample_data[['enmo', 'light', 'anglez']].values[-5:]\nforecast = model_fitted.forecast(y=forecast_input, steps=10)\nforecast_df = pd.DataFrame(forecast, index=range(len(sample_data), len(sample_data) + 10), columns=['enmo', 'light', 'anglez'])\n\n# Plotting the forecast\nplt.figure(figsize=(12, 6))\nplt.plot(sample_data['step'], sample_data['enmo'], label='Original')\nplt.plot(forecast_df.index, forecast_df['enmo'], label='Forecast', color='red')\nplt.title('VAR Forecasting')\nplt.xlabel('Time Step')\nplt.ylabel('ENMO')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:36.217668Z","iopub.execute_input":"2024-11-03T07:04:36.218495Z","iopub.status.idle":"2024-11-03T07:04:36.873323Z","shell.execute_reply.started":"2024-11-03T07:04:36.218442Z","shell.execute_reply":"2024-11-03T07:04:36.872143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import IsolationForest\n\n# Anomaly Detection using Isolation Forest\niso_forest = IsolationForest(contamination=0.1)\nsample_data['anomaly'] = iso_forest.fit_predict(sample_data[['enmo']])\n\n# Plotting anomalies\nplt.figure(figsize=(12, 6))\nplt.plot(sample_data['step'], sample_data['enmo'], label='ENMO')\nplt.scatter(sample_data[sample_data['anomaly'] == -1]['step'], sample_data[sample_data['anomaly'] == -1]['enmo'], color='red', label='Anomaly')\nplt.title('Anomaly Detection using Isolation Forest')\nplt.xlabel('Time Step')\nplt.ylabel('ENMO')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:36.874942Z","iopub.execute_input":"2024-11-03T07:04:36.875696Z","iopub.status.idle":"2024-11-03T07:04:39.537959Z","shell.execute_reply.started":"2024-11-03T07:04:36.875642Z","shell.execute_reply":"2024-11-03T07:04:39.536551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 7. PCA","metadata":{}},{"cell_type":"code","source":"print(\"Principal Component Analysis (PCA)\")\n# List of columns to drop\ncolumns_to_drop = ['sii']  # Add more column names as needed\n\n# Select numerical columns and drop the specified columns\nfeatures = train_merged_df.select_dtypes(include=[np.number]).drop(columns=columns_to_drop).columns\n\nx = train_merged_df[features].fillna(0)\nx = StandardScaler().fit_transform(x)\npca = PCA(n_components=2)\nprincipal_components = pca.fit_transform(x)\npca_df = pd.DataFrame(data=principal_components, columns=['PC1', 'PC2'])\npca_df['sii'] = train_merged_df['sii']\n\nplt.figure(figsize=(10, 6))\nsns.scatterplot(x='PC1', y='PC2', hue='sii', data=pca_df, palette='viridis')\nplt.title('PCA of Features')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:39.539794Z","iopub.execute_input":"2024-11-03T07:04:39.540337Z","iopub.status.idle":"2024-11-03T07:04:40.153726Z","shell.execute_reply.started":"2024-11-03T07:04:39.540282Z","shell.execute_reply":"2024-11-03T07:04:40.152407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 8. t-SNE","metadata":{}},{"cell_type":"code","source":"# 9. t-Distributed Stochastic Neighbor Embedding (t-SNE)\nprint(\"t-Distributed Stochastic Neighbor Embedding (t-SNE)\")\ntsne = TSNE(n_components=2, perplexity=30, n_iter=300)\ntsne_results = tsne.fit_transform(x)\ntsne_df = pd.DataFrame(data=tsne_results, columns=['TSNE1', 'TSNE2'])\ntsne_df['sii'] = train_merged_df['sii']\n\nplt.figure(figsize=(10, 6))\nsns.scatterplot(x='TSNE1', y='TSNE2', hue='sii', data=tsne_df, palette='viridis')\nplt.title('t-SNE of Features')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:40.155517Z","iopub.execute_input":"2024-11-03T07:04:40.156032Z","iopub.status.idle":"2024-11-03T07:04:46.712377Z","shell.execute_reply.started":"2024-11-03T07:04:40.155977Z","shell.execute_reply":"2024-11-03T07:04:46.711238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Insights\n1. We can clearly see there are some clusters forming in PCA results so we can try some clustering on top of PCA data to confirm.\n2. t-SNE is not able to create distinct clusters as all the types of SII is distributed in each cluster","metadata":{}},{"cell_type":"markdown","source":"## 9. K-means","metadata":{}},{"cell_type":"code","source":"print(\"Clustering K-Means\")\nkmeans = KMeans(n_clusters=4)\nkmeans_labels = kmeans.fit_predict(x)\npca_df['kmeans_cluster'] = kmeans_labels\n\nplt.figure(figsize=(10, 6))\nsns.scatterplot(x='PC1', y='PC2', hue='kmeans_cluster', data=pca_df, palette='viridis')\nplt.title('K-Means Clustering on PCA Components')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:46.714026Z","iopub.execute_input":"2024-11-03T07:04:46.714868Z","iopub.status.idle":"2024-11-03T07:04:49.425014Z","shell.execute_reply.started":"2024-11-03T07:04:46.714812Z","shell.execute_reply":"2024-11-03T07:04:49.423805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 10. DBScan","metadata":{}},{"cell_type":"code","source":"dbscan = DBSCAN(eps=0.5, min_samples=5)\ndbscan_labels = dbscan.fit_predict(x)\npca_df['dbscan_cluster'] = dbscan_labels\n\nplt.figure(figsize=(10, 6))\nsns.scatterplot(x='PC1', y='PC2', hue='dbscan_cluster', data=pca_df, palette='viridis')\nplt.title('DBSCAN Clustering on PCA Components')\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:04:49.426677Z","iopub.execute_input":"2024-11-03T07:04:49.427032Z","iopub.status.idle":"2024-11-03T07:04:50.127756Z","shell.execute_reply.started":"2024-11-03T07:04:49.426996Z","shell.execute_reply":"2024-11-03T07:04:50.126614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 11. Feature Importance using Random Forest","metadata":{}},{"cell_type":"code","source":"print(\"Feature Importance using Random Forest\")\n\nrf_data = train_merged_df.dropna(subset = 'sii')\nx_data = rf_data[features].fillna(0)\nrf = RandomForestClassifier(n_estimators=100)\nrf.fit(x_data, rf_data['sii'])\nimportances = rf.feature_importances_\nfeature_importance_df = pd.DataFrame({'Feature': features, 'Importance': importances})\nfeature_importance_df = feature_importance_df.sort_values(by='Importance', ascending=False)\n\nplt.figure(figsize=(15, 35))\nsns.barplot(x='Importance', y='Feature', data=feature_importance_df)\nplt.title('Feature Importance from Random Forest')\n\n# Increase the distance between y-ticks\nplt.gca().yaxis.set_tick_params(pad=10)\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:19:26.388293Z","iopub.execute_input":"2024-11-03T07:19:26.389547Z","iopub.status.idle":"2024-11-03T07:19:30.027651Z","shell.execute_reply.started":"2024-11-03T07:19:26.389495Z","shell.execute_reply":"2024-11-03T07:19:30.026354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Insights\nNo Stat features are there in top 20 features","metadata":{}},{"cell_type":"markdown","source":"## 12. Pair Plot Analysis","metadata":{}},{"cell_type":"code","source":"print(\"Pair Plot Analysis\")\nsns.pairplot(rf_data, hue='sii', vars=features[:5], palette='viridis')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:23:38.731438Z","iopub.execute_input":"2024-11-03T07:23:38.732022Z","iopub.status.idle":"2024-11-03T07:23:50.336219Z","shell.execute_reply.started":"2024-11-03T07:23:38.731962Z","shell.execute_reply":"2024-11-03T07:23:50.335002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 14. SHAP Values for Model Interpretability","metadata":{}},{"cell_type":"code","source":"print(\"SHAP Values for Model Interpretability\")\nexplainer = shap.TreeExplainer(rf)\nshap_values = explainer.shap_values(x_data)\nshap.summary_plot(shap_values, x_data, plot_type=\"bar\")\nshap.summary_plot(shap_values, x_data)","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:28:19.363945Z","iopub.execute_input":"2024-11-03T07:28:19.364426Z","iopub.status.idle":"2024-11-03T07:32:00.648409Z","shell.execute_reply.started":"2024-11-03T07:28:19.364383Z","shell.execute_reply":"2024-11-03T07:32:00.647150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":" # Preprocess the test data\nx_test = test_merged_df[features].fillna(0)\n\n# Make predictions on the test data\ntest_predictions = rf.predict(x_test)\n\n# Prepare the submission file\nsubmission = pd.DataFrame({\n    'id': sample['id'],  # Assuming 'id' is the identifier column in the test data\n    'sii': test_predictions\n})\n\nprint(submission.head())\n\n# Save the submission file\nsubmission.to_csv('submission.csv', index=False)\nprint(\"Submission file saved as 'submission.csv'\")","metadata":{"execution":{"iopub.status.busy":"2024-11-03T07:41:02.406726Z","iopub.execute_input":"2024-11-03T07:41:02.407195Z","iopub.status.idle":"2024-11-03T07:41:02.435183Z","shell.execute_reply.started":"2024-11-03T07:41:02.407136Z","shell.execute_reply":"2024-11-03T07:41:02.434005Z"},"trusted":true},"execution_count":null,"outputs":[]}]}