{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":10002004,"sourceType":"datasetVersion","datasetId":6156586},{"sourceId":177355,"sourceType":"modelInstanceVersion","modelInstanceId":151085,"modelId":173563}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div style=\"color:#016FD0;margin:0;font-size:32px;font-family:Georgia;text-align:center;display:fill;border-radius:5px;overflow:hidden;font-weight:600;\">CMI - Problematic Internet Use: EDA & Modeling</div>\n\n<div style=\"text-align:center\">\n    <img width=\"800\" alt=\"image\" src=\"https://images.unsplash.com/photo-1610571648632-7500e6e7c0e7?q=80&w=3570&auto=format&fit=crop&ixlib=rb-4.0.3&ixid=M3wxMjA3fDB8MHxwaG90by1wYWdlfHx8fGVufDB8fHx8fA%3D%3D\">\n</div>\n<div style=\"text-align:center\">\n    <a href=\"https://unsplash.com/photos/brown-wooden-i-love-you-print-board-on-green-grass-during-daytime-Ag0fAuFtH6I\">Photo from Unsplash</a>\n</div>","metadata":{}},{"cell_type":"markdown","source":"## <span style=\"color: #016FD0;\">Work In Progress</span>","metadata":{}},{"cell_type":"markdown","source":"# <div style=\"padding:20px;color:white;margin:0;font-size:30px;font-family:Georgia;text-align:left;display:fill;border-radius:5px;background-color:#016FD0;overflow:hidden\">Import Python Libraries</div>","metadata":{}},{"cell_type":"code","source":"# Data Manipulation and Analysis\nimport pandas as pd\nimport numpy as np\nimport math\n\n# Plotting and Visualization\nimport matplotlib.pyplot as plt\nfrom matplotlib.lines import Line2D\nimport seaborn as sns\nimport plotly.express as px\nimport plotly.subplots as sp\nfrom plotly.subplots import make_subplots\nimport plotly.graph_objects as go\n\n# Machine Learning and Preprocessing\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import StratifiedKFold, KFold\nfrom sklearn.metrics import (\n    cohen_kappa_score, \n    classification_report, \n    confusion_matrix, \n    accuracy_score, \n    mean_squared_error\n)\nfrom sklearn.ensemble import VotingClassifier\n\n# Machine Learning Libraries\nfrom xgboost import XGBClassifier\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostClassifier\nfrom catboost import CatBoostRegressor\nfrom lightgbm import LGBMClassifier\nimport lightgbm as lgb\n\n# Optimization\nimport optuna\n\n# File and OS Utilities\nimport os\nfrom glob import glob\n\n# Suppress Warnings\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:16:35.589633Z","iopub.execute_input":"2024-12-08T11:16:35.590517Z","iopub.status.idle":"2024-12-08T11:16:35.596658Z","shell.execute_reply.started":"2024-12-08T11:16:35.59048Z","shell.execute_reply":"2024-12-08T11:16:35.595777Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <div style=\"padding:20px;color:white;margin:0;font-size:30px;font-family:Georgia;text-align:left;display:fill;border-radius:5px;background-color:#016FD0;overflow:hidden\">Dataset Loading</div>","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ndata_dictionary = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:14:30.308071Z","iopub.execute_input":"2024-12-08T11:14:30.308556Z","iopub.status.idle":"2024-12-08T11:14:30.356885Z","shell.execute_reply.started":"2024-12-08T11:14:30.308526Z","shell.execute_reply":"2024-12-08T11:14:30.35614Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:14:30.358344Z","iopub.execute_input":"2024-12-08T11:14:30.358627Z","iopub.status.idle":"2024-12-08T11:14:30.364785Z","shell.execute_reply.started":"2024-12-08T11:14:30.358601Z","shell.execute_reply":"2024-12-08T11:14:30.363879Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <div style=\"padding:20px;color:white;margin:0;font-size:30px;font-family:Georgia;text-align:left;display:fill;border-radius:5px;background-color:#016FD0;overflow:hidden\">Exploratory Data Analysis</div>\n\n## <span style=\"color: #016FD0;\">Data Profiling</span>","metadata":{}},{"cell_type":"code","source":"train_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:14:31.659284Z","iopub.execute_input":"2024-12-08T11:14:31.659673Z","iopub.status.idle":"2024-12-08T11:14:31.665347Z","shell.execute_reply.started":"2024-12-08T11:14:31.659628Z","shell.execute_reply":"2024-12-08T11:14:31.664446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:14:32.225085Z","iopub.execute_input":"2024-12-08T11:14:32.225399Z","iopub.status.idle":"2024-12-08T11:14:32.231768Z","shell.execute_reply.started":"2024-12-08T11:14:32.225374Z","shell.execute_reply":"2024-12-08T11:14:32.230936Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dictionary.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:14:32.670662Z","iopub.execute_input":"2024-12-08T11:14:32.671032Z","iopub.status.idle":"2024-12-08T11:14:32.68452Z","shell.execute_reply.started":"2024-12-08T11:14:32.671002Z","shell.execute_reply":"2024-12-08T11:14:32.683411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"list(data_dictionary['Type'].unique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:14:33.166529Z","iopub.execute_input":"2024-12-08T11:14:33.166965Z","iopub.status.idle":"2024-12-08T11:14:33.172874Z","shell.execute_reply.started":"2024-12-08T11:14:33.166933Z","shell.execute_reply":"2024-12-08T11:14:33.171916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dictionary[\ndata_dictionary['Type'].str.contains('categorical')\n].shape[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:14:33.551533Z","iopub.execute_input":"2024-12-08T11:14:33.551898Z","iopub.status.idle":"2024-12-08T11:14:33.558905Z","shell.execute_reply.started":"2024-12-08T11:14:33.551867Z","shell.execute_reply":"2024-12-08T11:14:33.557865Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dictionary[\ndata_dictionary['Type'].str.contains('str')\n].shape[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:14:34.041543Z","iopub.execute_input":"2024-12-08T11:14:34.041913Z","iopub.status.idle":"2024-12-08T11:14:34.048635Z","shell.execute_reply.started":"2024-12-08T11:14:34.041881Z","shell.execute_reply":"2024-12-08T11:14:34.047784Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dictionary[\ndata_dictionary['Type'].str.contains('float')\n].shape[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:14:34.491079Z","iopub.execute_input":"2024-12-08T11:14:34.491417Z","iopub.status.idle":"2024-12-08T11:14:34.498636Z","shell.execute_reply.started":"2024-12-08T11:14:34.491389Z","shell.execute_reply":"2024-12-08T11:14:34.497575Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dictionary[\ndata_dictionary['Type'].str.contains('int') & ~data_dictionary['Type'].str.contains('categorical') \n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:14:34.86689Z","iopub.execute_input":"2024-12-08T11:14:34.867208Z","iopub.status.idle":"2024-12-08T11:14:34.879684Z","shell.execute_reply.started":"2024-12-08T11:14:34.86718Z","shell.execute_reply":"2024-12-08T11:14:34.878698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(data_dictionary[\ndata_dictionary['Field'].str.contains('PreInt_EduHx-computerinternet_hoursday')\n][\"Value Labels\"].iloc[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:14:35.311782Z","iopub.execute_input":"2024-12-08T11:14:35.312518Z","iopub.status.idle":"2024-12-08T11:14:35.318848Z","shell.execute_reply.started":"2024-12-08T11:14:35.312486Z","shell.execute_reply":"2024-12-08T11:14:35.317783Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <span style=\"color: #016FD0;\">Understanding the Distribution of the Target Variable (sii)</span>","metadata":{}},{"cell_type":"code","source":"# Define the mapping for `sii` levels including \"Missing\" for NaNs\nsii_map = {\n    0: \"None\",\n    1: \"Mild\",\n    2: \"Moderate\",\n    3: \"Severe\",\n    \"missing\": \"Missing Data\"\n}\n\n# Replace NaN values with \"missing\" to count them as a separate category\ntrain_df['sii_filled'] = train_df['sii'].fillna(\"missing\")\ntrain_df['sii_label'] = train_df['sii_filled'].map(sii_map)\n\n# Count the occurrences for each `sii` level, including missing data\nsii_counts = train_df['sii_label'].value_counts()\nsii_percentages = (sii_counts / sii_counts.sum()) * 100\n\n# Create DataFrame for plotting with counts and percentages, placing \"Missing Data\" at the end\nsii_data = pd.DataFrame({\n    'SII Level': sii_counts.index,\n    'Count': sii_counts.values,\n    'Percentage': sii_percentages.values\n})\n\n# Reorder to ensure \"Missing Data\" is at the end\nsii_data['SII Level'] = pd.Categorical(sii_data['SII Level'], categories=[\"None\", \"Mild\", \"Moderate\", \"Severe\", \"Missing Data\"], ordered=True)\nsii_data = sii_data.sort_values('SII Level')\n\n# Define color mapping based on sii levels\ncolors = {\n    \"None\": \"#4B9CD3\",        # Blue\n    \"Mild\": \"#0074D9\",        # Darker Blue\n    \"Moderate\": \"#17BECF\",    # Teal\n    \"Severe\": \"#FF4136\",      # Red\n    \"Missing Data\": \"#A9A9A9\" # Gray\n}\n\n# Plot the data in Plotly and apply the colors based on SII Level\nfig = px.bar(\n    sii_data,\n    x='SII Level',\n    y='Count',\n    text='Count',\n    title=\"Distribution of SII Target Variable\",\n    labels={'Count': 'Number of Participants', 'SII Level': 'SII Severity Level'},\n    hover_data={'Percentage': ':.2f'}\n)\n\n# Apply custom colors to each bar based on SII Level\nfig.update_traces(marker=dict(color=[colors[level] for level in sii_data['SII Level']]))\n\n# Customize the layout to match the previous settings\nfig.update_layout(\n    title_font=dict(size=20, color=\"#004080\"),\n    plot_bgcolor='white',\n    xaxis=dict(showgrid=False, showline=False, zeroline=False, tickfont=dict(size=12)),\n    yaxis=dict(showgrid=False, showline=False, zeroline=False, tickfont=dict(size=12)),\n    margin=dict(t=100, l=50, r=50, b=80)\n)\n\nfig.show()\n","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-08T11:14:36.055388Z","iopub.execute_input":"2024-12-08T11:14:36.05608Z","iopub.status.idle":"2024-12-08T11:14:36.430623Z","shell.execute_reply.started":"2024-12-08T11:14:36.056047Z","shell.execute_reply":"2024-12-08T11:14:36.429801Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a KDE plot of PCIAT-PCIAT_Total\nfig_kde = px.histogram(\n    train_df, \n    x=\"PCIAT-PCIAT_Total\", \n    nbins=50, \n    opacity=0.7,\n    histnorm=\"density\",\n    marginal=\"violin\", \n    title=\"Density Plot of PCIAT-PCIAT_Total\",\n    labels={\"PCIAT-PCIAT_Total\": \"PCIAT-PCIAT_Total Score\", \"density\": \"Density\"}\n)\n\n# Customize layout for consistency with previous styling\nfig_kde.update_layout(\n    title_font=dict(size=20, color=\"#004080\"),  # Title color and size\n    plot_bgcolor='white',                      # White background\n    xaxis=dict(showgrid=False, showline=False, zeroline=False, tickfont=dict(size=12)),\n    yaxis=dict(showgrid=False, showline=False, zeroline=False, tickfont=dict(size=12)),\n    margin=dict(t=80, l=50, r=50, b=50)\n)\n\nfig_kde.show()\n","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-08T11:14:36.780187Z","iopub.execute_input":"2024-12-08T11:14:36.780786Z","iopub.status.idle":"2024-12-08T11:14:36.85551Z","shell.execute_reply.started":"2024-12-08T11:14:36.780751Z","shell.execute_reply":"2024-12-08T11:14:36.854797Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set up the figure and plot\nplt.figure(figsize=(10, 6))\nsns.kdeplot(\n    data=train_df,\n    x=\"PCIAT-PCIAT_Total\",\n    fill=True,\n    color=\"#4B9CD3\",  # Consistent blue color\n    linewidth=2\n)\n\n# Customize the plot\nplt.title(\"Density Plot of PCIAT-PCIAT_Total\", fontsize=20, color=\"#004080\")\nplt.xlabel(\"PCIAT-PCIAT_Total Score\", fontsize=14)\nplt.ylabel(\"Density\", fontsize=14)\nplt.tight_layout()\n\nplt.show()\n","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-08T11:14:37.219618Z","iopub.execute_input":"2024-12-08T11:14:37.220311Z","iopub.status.idle":"2024-12-08T11:14:37.669107Z","shell.execute_reply.started":"2024-12-08T11:14:37.220262Z","shell.execute_reply":"2024-12-08T11:14:37.668207Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <span style=\"color: #016FD0;\">Gender Distribution</span>\n- A bar plot showing the count of each gender (Male and Female)","metadata":{"_kg_hide-input":true}},{"cell_type":"code","source":"train_df['Basic_Demos-Sex'].value_counts(dropna=False, normalize=True)","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-08T11:14:38.171352Z","iopub.execute_input":"2024-12-08T11:14:38.172206Z","iopub.status.idle":"2024-12-08T11:14:38.180236Z","shell.execute_reply.started":"2024-12-08T11:14:38.172156Z","shell.execute_reply":"2024-12-08T11:14:38.17955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Map for the sex categories\nsex_map = {0: \"Male\", 1: \"Female\"}\n\n# Apply the mapping to create a new column for visualization\ntrain_df['Sex'] = train_df['Basic_Demos-Sex'].map(sex_map)\n\n# Count occurrences for each category and calculate percentages\nsex_counts = train_df['Sex'].value_counts().sort_index()\nsex_percentages = (sex_counts / sex_counts.sum()) * 100\n\n# Create a DataFrame for plotting\nsex_data = pd.DataFrame({\n    'Sex': sex_counts.index,\n    'Count': sex_counts.values,\n    'Percentage': sex_percentages.values\n})\n\n# Plot the data\nfig = px.bar(\n    sex_data,\n    x='Sex',\n    y='Count',\n    title=\"Distribution of Participants' Sex\",\n    labels={'Count': 'Number of Participants', 'Sex': 'Sex'},\n    hover_data={'Percentage': ':.2f'}\n)\n\n# Use the same blue color for both categories\nfig.update_traces(marker_color=\"#4B9CD3\", marker_line_width=0)  # Blue color for both Male and Female\nfig.update_layout(\n    title_font=dict(size=20, color=\"#004080\"),\n    plot_bgcolor='white',\n    xaxis=dict(showgrid=False, showline=False, zeroline=False, tickfont=dict(size=12)),\n    yaxis=dict(showgrid=False, showline=False, zeroline=False, tickfont=dict(size=12)),\n    margin=dict(t=80, l=50, r=50, b=80)\n)\n\nfig.show()\n","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-08T11:14:38.578836Z","iopub.execute_input":"2024-12-08T11:14:38.57915Z","iopub.status.idle":"2024-12-08T11:14:38.643774Z","shell.execute_reply.started":"2024-12-08T11:14:38.579123Z","shell.execute_reply":"2024-12-08T11:14:38.642678Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <span style=\"color: #016FD0;\">Insights into Youth Internet Habits Based on Parental Feedback</span>","metadata":{}},{"cell_type":"code","source":"# Extract columns that match \"PCIAT-PCIAT_\" followed by a number (not including \"PCIAT-PCIAT_Total\")\npciat_columns = [col for col in train_df.columns if col.startswith('PCIAT-PCIAT_') and col != 'PCIAT-PCIAT_Total']\n\n# Check for NaN values in each of these columns\nnan_counts = train_df[pciat_columns].isna().sum()\n\n# Filter to show only columns with missing values\nmissing_values = nan_counts[nan_counts > 0]\n\n# Display the results\nprint(\"Columns with NaN values in PCIAT-PCIAT_ fields:\")\nprint(missing_values)","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-08T11:14:39.57843Z","iopub.execute_input":"2024-12-08T11:14:39.579214Z","iopub.status.idle":"2024-12-08T11:14:39.586656Z","shell.execute_reply.started":"2024-12-08T11:14:39.57918Z","shell.execute_reply":"2024-12-08T11:14:39.585676Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Filter for PCIAT-related fields in the data dictionary\npciat_dict_df = data_dictionary[data_dictionary['Field'].str.startswith('PCIAT-PCIAT_')]\n\n# Create the map from field names to short descriptions\npciat_question_map = dict(zip(pciat_dict_df['Field'], pciat_dict_df['Description']))\n\n\n# Define the explanation for each response category, including \"Missing\" at the beginning\nscale_explanation = {\n    \"missing\": \"Missing\",\n    0: \"Does Not Apply\",\n    1: \"Rarely\",\n    2: \"Occasionally\",\n    3: \"Frequently\",\n    4: \"Often\",\n    5: \"Always\"\n}\n\n# Define a function to insert line breaks based on character limit\ndef wrap_text(text, char_limit=25, padding_lines=0):\n    words = text.split()\n    wrapped_text = \"\"\n    line = \"\"\n    for word in words:\n        if len(line) + len(word) + 1 > char_limit:\n            wrapped_text += line + \"<br>\"\n            line = word\n        else:\n            line += (\" \" if line else \"\") + word\n    wrapped_text += line\n    # Add extra line breaks for padding\n    wrapped_text += \"<br>\" * padding_lines\n    return wrapped_text\n\n# Wrap long titles to fit within specified width and add padding\nsubplot_titles = [wrap_text(pciat_question_map[col], char_limit=25, padding_lines=0) for col in pciat_columns]\n\n# Set the number of columns to 2 for better readability\nnum_cols = 2  \nnum_rows = (len(pciat_columns) + num_cols - 1) // num_cols  # Calculate rows needed\n\n# Colors based on the Child Mind Institute logo\nbar_color = \"#0082c8\"  # Blue color for bars\nmissing_color = \"#A9A9A9\"  # Gray color for \"Missing\" category\ntitle_color = \"#004080\"  # Darker blue shade for titles\n\n# Create subplot grid with wrapped titles and minimal spacing\nfig = sp.make_subplots(rows=num_rows, cols=num_cols, subplot_titles=subplot_titles)\n\n# Add a bar plot for each PCIAT question\nfor i, col in enumerate(pciat_columns):\n    row = i // num_cols + 1\n    col_pos = i % num_cols + 1\n    \n    # Count the responses, including NaN as \"Missing\"\n    response_counts = train_df[col].value_counts().sort_index()\n    missing_count = train_df[col].isna().sum()\n    \n    # Create a DataFrame for plotting with the missing count added at the start\n    response_data = pd.DataFrame({\n        'Response': ['Missing'] + response_counts.index.map(scale_explanation).tolist(),\n        'Count': [missing_count] + response_counts.values.tolist()\n    })\n    response_data['Percentage'] = (response_data['Count'] / response_data['Count'].sum()) * 100\n    \n    # Create a bar plot for the current question with percentages in the hover\n    bar_fig = px.bar(response_data, x='Response', y='Count', \n                     labels={'Count': 'Frequency', 'Response': 'Response'},\n                     hover_data={'Percentage': ':.2f'},\n                     title=pciat_question_map[col])\n    \n    # Customize each bar plot's appearance, setting a different color for \"Missing\"\n    colors = [missing_color if x == \"Missing\" else bar_color for x in response_data['Response']]\n    bar_fig.update_traces(marker_color=colors)  # Apply custom colors\n    bar_fig.update_layout(\n        showlegend=False,  # Hide legend\n        xaxis=dict(showgrid=False, showline=False, zeroline=False, tickfont=dict(size=10, color=title_color)),  # Smaller x-axis labels with blue color\n        yaxis=dict(showgrid=False, showline=False, zeroline=False, tickfont=dict(size=10))  # Remove y-axis grid and lines\n    )\n\n    # Add the bar plot to the grid\n    for trace in bar_fig.data:\n        fig.add_trace(trace, row=row, col=col_pos)\n\n# Update layout for readability, spacing, and background color\nfig.update_layout(\n    height=300 * num_rows,  # Set a fixed height per row\n    width=1000,  # Increase width slightly to give more space\n    title_text=\"Distribution of Responses for Parent-Child Internet Addiction Test Questions\",\n    title_font=dict(size=20, color=title_color),  # Set main title font size\n    margin=dict(t=140, l=50, r=50, b=100),  # Increase top margin\n    plot_bgcolor='white',\n    paper_bgcolor='white'\n)\n\n# Rotate x-axis labels slightly\nfig.update_xaxes(tickangle=15)\nfig.update_yaxes(tickfont=dict(size=10))\n\n# Set font size for subplot titles\nfig.update_annotations(font=dict(size=12, color=title_color))  # Adjust subplot title font size and color\n\nfig.show()\n","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-08T11:14:40.010431Z","iopub.execute_input":"2024-12-08T11:14:40.011071Z","iopub.status.idle":"2024-12-08T11:14:41.190197Z","shell.execute_reply.started":"2024-12-08T11:14:40.011038Z","shell.execute_reply":"2024-12-08T11:14:41.189262Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <span style=\"color: #016FD0;\">Daily Internet Usage</span>","metadata":{}},{"cell_type":"code","source":"train_df[\"PreInt_EduHx-computerinternet_hoursday\"].value_counts(dropna=False, normalize=True)","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-08T11:14:41.192124Z","iopub.execute_input":"2024-12-08T11:14:41.192788Z","iopub.status.idle":"2024-12-08T11:14:41.201604Z","shell.execute_reply.started":"2024-12-08T11:14:41.192714Z","shell.execute_reply":"2024-12-08T11:14:41.200545Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the mapping for hours per day categories with a label for missing values\ninternet_hours_map = {\n    0: \"Less than 1h/day\",\n    1: \"Around 1h/day\",\n    2: \"Around 2hs/day\",\n    3: \"More than 3hs/day\",\n    \"missing\": \"Missing Data\"\n}\n\n# Replace NaN values with \"missing\" to count them as a separate category\ntrain_df['Internet_Use_Hours'] = train_df['PreInt_EduHx-computerinternet_hoursday'].map(\n    lambda x: internet_hours_map.get(x, \"Missing Data\")\n)\n\n# Convert 'Internet_Use_Hours' to a categorical type with the specified order\ncategory_order = list(internet_hours_map.values())\ntrain_df['Internet_Use_Hours'] = pd.Categorical(train_df['Internet_Use_Hours'], categories=category_order, ordered=True)\n\n# Count occurrences of each category (including \"Missing Data\") and calculate percentages\ninternet_use_counts = train_df['Internet_Use_Hours'].value_counts(sort=False)\ninternet_use_percentages = (internet_use_counts / internet_use_counts.sum()) * 100\n\n# Create DataFrame for plotting\ninternet_use_data = pd.DataFrame({\n    'Hours per Day': internet_use_counts.index,\n    'Count': internet_use_counts.values,\n    'Percentage': internet_use_percentages.values\n})\n\n# Define custom colors, using gray for \"Missing Data\" and blue for other categories\ncolors = [\"#A9A9A9\" if category == \"Missing Data\" else \"#4B9CD3\" for category in internet_use_data['Hours per Day']]\n\n# Plot the data\nfig = px.bar(\n    internet_use_data,\n    x='Hours per Day',\n    y='Count',\n    title=\"Daily Internet Usage Distribution Among Youth\",\n    labels={'Count': 'Number of Respondents', 'Hours per Day': 'Daily Internet Usage'},\n    hover_data={'Percentage': ':.2f'}\n)\n\n# Apply custom colors and layout\nfig.update_traces(marker_color=colors, marker_line_width=0)\nfig.update_layout(\n    title_font=dict(size=20, color=\"#004080\"),\n    plot_bgcolor='white',\n    xaxis=dict(showgrid=False, showline=False, zeroline=False, tickfont=dict(size=12)),\n    yaxis=dict(showgrid=False, showline=False, zeroline=False, tickfont=dict(size=12)),\n    margin=dict(t=80, l=50, r=50, b=80)\n)\n\nfig.show()\n","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-08T11:14:41.203758Z","iopub.execute_input":"2024-12-08T11:14:41.204294Z","iopub.status.idle":"2024-12-08T11:14:41.27794Z","shell.execute_reply.started":"2024-12-08T11:14:41.204247Z","shell.execute_reply":"2024-12-08T11:14:41.277145Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <span style=\"color: #016FD0;\">Analysis of Fitness and Body Composition Metrics</span>","metadata":{}},{"cell_type":"code","source":"\n# Define mappings for each field's response categories, with no special colors for specific categories\nfitness_bio_categories = {\n    'FGC-FGC_CU_Zone': {0: \"Needs Improvement\", 1: \"Healthy Fitness Zone\"},\n    'FGC-FGC_GSND_Zone': {1: \"Weak\", 2: \"Normal\", 3: \"Strong\"},\n    'FGC-FGC_GSD_Zone': {1: \"Weak\", 2: \"Normal\", 3: \"Strong\"},\n    'FGC-FGC_PU_Zone': {0: \"Needs Improvement\", 1: \"Healthy Fitness Zone\"},\n    'FGC-FGC_SRL_Zone': {0: \"Needs Improvement\", 1: \"Healthy Fitness Zone\"},\n    'FGC-FGC_SRR_Zone': {0: \"Needs Improvement\", 1: \"Healthy Fitness Zone\"},\n    'FGC-FGC_TL_Zone': {0: \"Needs Improvement\", 1: \"Healthy Fitness Zone\"},\n    'BIA-BIA_Activity_Level_num': {1: \"Very Light\", 2: \"Light\", 3: \"Moderate\", 4: \"Heavy\", 5: \"Very Heavy\"},\n    'BIA-BIA_Frame_num': {1: \"Small\", 2: \"Medium\", 3: \"Large\"}\n}\n\n# Titles for each field for visualization\nfitness_bio_titles = {\n    'FGC-FGC_CU_Zone': \"Curl up fitness zone\",\n    'FGC-FGC_GSND_Zone': \"Grip Strength fitness zone (non-dominant)\",\n    'FGC-FGC_GSD_Zone': \"Grip Strength fitness zone (dominant)\",\n    'FGC-FGC_PU_Zone': \"Push-up fitness zone\",\n    'FGC-FGC_SRL_Zone': \"Sit & Reach fitness zone (left side)\",\n    'FGC-FGC_SRR_Zone': \"Sit & Reach fitness zone (right side)\",\n    'FGC-FGC_TL_Zone': \"Trunk lift fitness zone\",\n    'BIA-BIA_Activity_Level_num': \"Activity Level\",\n    'BIA-BIA_Frame_num': \"Body Frame\"\n}\n\n# Set the number of columns for the subplot grid\nnum_cols = 2\nnum_rows = (len(fitness_bio_categories) + num_cols - 1) // num_cols  # Calculate number of rows\n\n# Colors\nbar_color = \"#4B9CD3\"  # Blue for all categories\nmissing_color = \"#A9A9A9\"  # Gray for missing values\n\n# Create subplot grid with titles\nfig = sp.make_subplots(rows=num_rows, cols=num_cols, subplot_titles=[fitness_bio_titles[col] for col in fitness_bio_categories.keys()])\n\n# Add a bar plot for each field\nfor i, (col, categories) in enumerate(fitness_bio_categories.items()):\n    row = i // num_cols + 1\n    col_pos = i % num_cols + 1\n    \n    # Count the responses, including NaN as \"Missing\"\n    response_counts = train_df[col].value_counts(dropna=False).sort_index()\n    \n    # Create DataFrame for plotting with mapped categories and calculate percentages\n    response_data = pd.DataFrame({\n        'Response': response_counts.index.map(lambda x: \"Missing\" if pd.isna(x) else categories.get(x, str(x))),\n        'Count': response_counts.values\n    })\n    response_data['Percentage'] = (response_data['Count'] / response_data['Count'].sum()) * 100\n    \n    # Define categorical order based on the fitness_bio_categories order, including \"Missing\" at the beginning\n    category_order = [\"Missing\"] + list(categories.values())\n    response_data['Response'] = pd.Categorical(response_data['Response'], categories=category_order, ordered=True)\n    response_data = response_data.sort_values('Response')  # Sort by the defined order\n    \n    # Create bar plot for the current field with percentages in the hover\n    bar_fig = px.bar(\n        response_data,\n        x='Response',\n        y='Count',\n        labels={'Count': 'Frequency', 'Response': 'Response'},\n        hover_data={'Percentage': ':.2f'}\n    )\n    \n    # Customize colors, setting \"Missing\" category to gray and others to blue\n    colors = [missing_color if resp == \"Missing\" else bar_color for resp in response_data['Response']]\n    bar_fig.update_traces(marker_color=colors)  # Apply custom colors\n    \n    # Customize each bar plot appearance\n    bar_fig.update_layout(\n        showlegend=False,\n        xaxis=dict(showgrid=False, showline=False, zeroline=False, tickfont=dict(size=10)),\n        yaxis=dict(showgrid=False, showline=False, zeroline=False, tickfont=dict(size=10))\n    )\n\n    # Add the bar plot to the subplot\n    for trace in bar_fig.data:\n        fig.add_trace(trace, row=row, col=col_pos)\n\n# Update layout for readability, spacing, and background color\nfig.update_layout(\n    height=250 * num_rows,  # Set a smaller height per row for a compact layout\n    width=1000,\n    title_text=\"Distribution of FitnessGram and Bio-electric Impedance Analysis Zones\",\n    title_font=dict(size=20, color=\"#004080\"),\n    margin=dict(t=120, l=50, r=50, b=100),\n    plot_bgcolor='white',\n    paper_bgcolor='white'\n)\n\n# Rotate x-axis labels slightly and reduce font size for subplot titles\nfig.update_xaxes(tickangle=15)\nfig.update_annotations(font=dict(size=12, color=\"#004080\"))\n\nfig.show()\n","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-08T11:14:41.986528Z","iopub.execute_input":"2024-12-08T11:14:41.986863Z","iopub.status.idle":"2024-12-08T11:14:42.44907Z","shell.execute_reply.started":"2024-12-08T11:14:41.986834Z","shell.execute_reply":"2024-12-08T11:14:42.44816Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n## <span style=\"color: #016FD0;\">Analysis of Continuous Variables by SII Target Distribution</span>","metadata":{}},{"cell_type":"code","source":"\n# List of continuous columns based on your data\ncontinuous_columns = [\n    'Basic_Demos-Age', 'Physical-BMI', 'Physical-Height', 'Physical-Weight',\n    'FGC-FGC_GSND', 'FGC-FGC_GSD', 'FGC-FGC_SRL', 'FGC-FGC_SRR',\n    'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE',\n    'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI',\n    'BIA-BIA_Fat', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST',\n    'BIA-BIA_SMM', 'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total', 'PAQ_C-PAQ_C_Total'\n]\n\n# Select only the continuous columns and the sii target\ndata_continuous = train_df[continuous_columns + ['sii']]\n\n# Remove rows where sii is NaN, but keep NaNs in continuous variables\ndata_continuous = data_continuous.dropna(subset=['sii'])\n\n# Melt the data to long format for Seaborn\ndata_long = pd.melt(data_continuous, id_vars='sii', var_name=\"variable\", value_name=\"value\")\n\n# Set up the FacetGrid for KDE plots with increased figure size\ng = sns.FacetGrid(data_long, col=\"variable\", col_wrap=4, height=3.5, aspect=1.2, sharex=False, sharey=False)\n\n# Map KDE plots onto the grid, using hue for sii target\ng.map_dataframe(sns.kdeplot, x=\"value\", hue=\"sii\", fill=True, common_norm=False, palette=\"Set2\", alpha=0.4, linewidth=1.5)\n\n# Add title, adjust layout\ng.fig.suptitle(\"Distribution of Continuous Fitness and Health Metrics by SII Target\", y=1.05, fontsize=18, color=\"#004080\")\ng.set_titles(\"{col_name}\")\ng.set_axis_labels(\"Value\", \"Density\")\n\n# Create custom legend\n# Define the labels and colors (matching `palette=\"Set2\"` used above)\nsii_labels = sorted(data_continuous['sii'].dropna().unique())\ncolors = sns.color_palette(\"Set2\", len(sii_labels))\nlegend_elements = [Line2D([0], [0], color=color, lw=4, label=f'SII {label}') for color, label in zip(colors, sii_labels)]\n\n# Position the custom legend outside of the grid\ng.fig.legend(handles=legend_elements, loc=\"upper center\", ncol=4, bbox_to_anchor=(0.5, 1.15), frameon=False)\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-08T11:14:42.941021Z","iopub.execute_input":"2024-12-08T11:14:42.941349Z","iopub.status.idle":"2024-12-08T11:14:50.246973Z","shell.execute_reply.started":"2024-12-08T11:14:42.941321Z","shell.execute_reply":"2024-12-08T11:14:50.246032Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <span style=\"color: #016FD0;\">Correlation Heatmap Between Continuous Variables and sii</span>","metadata":{}},{"cell_type":"code","source":"\n# Calculate the correlation matrix\ncontinuous_columns = [\n    'Basic_Demos-Age', 'Physical-BMI', 'Physical-Height', 'Physical-Weight',\n    'FGC-FGC_GSND', 'FGC-FGC_GSD', 'FGC-FGC_SRL', 'FGC-FGC_SRR',\n    'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE',\n    'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI',\n    'BIA-BIA_Fat', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST',\n    'BIA-BIA_SMM', 'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total', 'PAQ_C-PAQ_C_Total', 'sii'\n]\n\ncorrelation_matrix = train_df[continuous_columns].corr()\n\n# Plot the heatmap with improved readability\nplt.figure(figsize=(18, 15))  # Increase figure size\nsns.heatmap(\n    correlation_matrix, \n    annot=True, \n    fmt=\".2f\",  # Reduce the number of decimal places\n    cmap=\"coolwarm\", \n    center=0, \n    annot_kws={\"size\": 8},  # Set annotation font size smaller\n    cbar_kws={\"shrink\": 0.8}  # Shrink color bar slightly\n)\n\n# Adjust the labels for readability\nplt.xticks(rotation=90, fontsize=10)  # Rotate x labels for better fit\nplt.yticks(rotation=0, fontsize=10)   # Keep y labels horizontal\nplt.title(\"Correlation Heatmap of Continuous Variables and SII\", fontsize=18, color=\"#004080\")\nplt.tight_layout()  # Ensure everything fits within the figure area\n\nplt.show()\n","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-08T11:14:50.248845Z","iopub.execute_input":"2024-12-08T11:14:50.249195Z","iopub.status.idle":"2024-12-08T11:14:52.060598Z","shell.execute_reply.started":"2024-12-08T11:14:50.249159Z","shell.execute_reply":"2024-12-08T11:14:52.059626Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <span style=\"color: #016FD0;\">Descriptive Statistics of Discrete Variables</span>","metadata":{"_kg_hide-input":false}},{"cell_type":"code","source":"# List of integer variables to analyze\ninteger_columns = list(data_dictionary[\n    data_dictionary['Type'].str.contains('int') & ~data_dictionary['Type'].str.contains('categorical')\n][\"Field\"].unique())\n\n# Calculate the statistics for each column\ncolumn_stats = {\n    col: {\n        'Null Percentage': train_df[col].isna().mean() * 100,\n        'Unique Values': train_df[col].nunique(),\n        'Mean': train_df[col].mean(),\n        'Std': train_df[col].std(),\n        'Min': train_df[col].min(),\n        '25%': train_df[col].quantile(0.25),\n        '50%': train_df[col].median(),\n        '75%': train_df[col].quantile(0.75),\n        'Max': train_df[col].max()\n    } \n    for col in integer_columns\n}\n\n# Create a DataFrame to display the results\nstats_df = pd.DataFrame.from_dict(column_stats, orient='index').reset_index()\nstats_df.columns = [\n    'Column', 'Null Percentage', 'Unique Values', \n    'Mean', 'Std', 'Min', '25%', '50%', '75%', 'Max'\n]\nstats_df = stats_df.sort_values(by='Null Percentage', ascending=False).reset_index(drop=True)\n\n# Display the DataFrame\nstats_df","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-08T11:14:52.061763Z","iopub.execute_input":"2024-12-08T11:14:52.062058Z","iopub.status.idle":"2024-12-08T11:14:52.117068Z","shell.execute_reply.started":"2024-12-08T11:14:52.062028Z","shell.execute_reply":"2024-12-08T11:14:52.116294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate the number of rows and columns for the grid\nnum_cols = 3  # Adjust the number of columns as needed\nnum_rows = math.ceil(len(integer_columns) / num_cols)\n\n# Create subplots\nfig = make_subplots(rows=num_rows, cols=num_cols, subplot_titles=integer_columns)\n\n# Add a box plot for each integer column\nfor i, col in enumerate(integer_columns):\n    row = (i // num_cols) + 1\n    col_pos = (i % num_cols) + 1\n    fig.add_trace(\n        go.Box(y=train_df[col], name=col, marker_color=\"#0082c8\"),  # Drop NaNs for box plot\n        row=row,\n        col=col_pos\n    )\n\n# Update layout\nfig.update_layout(\n    height=300 * num_rows,  # Adjust height per row\n    width=1000,\n    title_text=\"Box Plot Distribution of Discrete Variables\",\n    title_font=dict(size=20, color=\"#004080\"),\n    showlegend=False,\n    plot_bgcolor='white',\n    paper_bgcolor='white'\n)\n\n# Adjust title font size for each subplot\nfor annotation in fig['layout']['annotations']:\n    annotation['font'] = dict(size=12)\n\nfig.show()","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-08T11:14:52.118782Z","iopub.execute_input":"2024-12-08T11:14:52.119067Z","iopub.status.idle":"2024-12-08T11:14:52.206869Z","shell.execute_reply.started":"2024-12-08T11:14:52.119039Z","shell.execute_reply":"2024-12-08T11:14:52.206032Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate the number of rows and columns for the grid\nnum_cols = 3  # Adjust the number of columns as needed\nnum_rows = math.ceil(len(integer_columns) / num_cols)\n\n# Create subplots\nfig = make_subplots(rows=num_rows, cols=num_cols, subplot_titles=integer_columns)\n\n# Define color mapping for each sii category (e.g., 0 = None, 1 = Mild, etc.)\nsii_colors = {\n    0: \"#4B9CD3\",   # Color for 'None'\n    1: \"#F7A072\",   # Color for 'Mild'\n    2: \"#A3C4BC\",   # Color for 'Moderate'\n    3: \"#FF6F61\",   # Color for 'Severe'\n}\n\n# Add a box plot for each integer column, split by sii category\nfor i, col in enumerate(integer_columns):\n    row = (i // num_cols) + 1\n    col_pos = (i % num_cols) + 1\n    \n    # Add box plots for each sii category\n    for sii_value, color in sii_colors.items():\n        fig.add_trace(\n            go.Box(\n                y=train_df[train_df['sii'] == sii_value][col].dropna(),\n                name=f\"sii {sii_value}\",\n                marker_color=color,\n                boxmean=True,\n                showlegend=(i == 0)  # Show legend only for the first variable to avoid duplicates\n            ),\n            row=row,\n            col=col_pos\n        )\n\n# Update layout\nfig.update_layout(\n    height=300 * num_rows,  # Adjust height per row\n    width=1000,\n    title_text=\"Box Plot Distribution of Discrete Variables by SII\",\n    title_font=dict(size=20, color=\"#004080\"),\n    showlegend=True,\n    plot_bgcolor='white',\n    paper_bgcolor='white'\n)\n\n# Hide x-axis labels for all subplots\nfig.update_xaxes(showticklabels=False)\n\n# Update legend to make it centered and add legend title\nfig.update_layout(\n    legend=dict(\n        title=\"SII Levels\",\n        x=0.5,\n        xanchor=\"center\",\n        orientation=\"h\"\n    )\n)\n\n# Adjust title font size for each subplot\nfor annotation in fig['layout']['annotations']:\n    annotation['font'] = dict(size=12)\n\nfig.show()\n","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-08T11:14:52.208189Z","iopub.execute_input":"2024-12-08T11:14:52.208513Z","iopub.status.idle":"2024-12-08T11:14:52.390451Z","shell.execute_reply.started":"2024-12-08T11:14:52.208469Z","shell.execute_reply":"2024-12-08T11:14:52.389599Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <div style=\"padding:20px;color:white;margin:0;font-size:30px;font-family:Georgia;text-align:left;display:fill;border-radius:5px;background-color:#016FD0;overflow:hidden\">Step 1: Load the Time Series (Actigraphy Files) and Tabular Data</div>\n\nDuring their participation in the HBN study, some participants were given an accelerometer to wear for up to 30 days continually while at home and going about their regular daily lives.\n\nTo process the actigraphy time series data for modeling, here’s the general approach:\n\n## <span style=\"color: #016FD0;\">Non-wear detection</span>\n\n- `non-wear_flag`:A flag (`0`: **watch is being worn**, `1`: the watch is **not** worn) to help determine periods when the watch has been removed, based on the GGIR definition, which uses the standard deviation and range of the accelerometer data.\n\nIt can happen that a study participant does not wear the accelerometer. This can happen for a variety of reasons: getting tired of wearing the accelerometer, forgetting to put the accelerometer back on after a short moment of not wearing it, or getting instructed by the researcher to take it off. However, when accelerometer are not worn, they still collect data. When the accelerometer is lying still, the data collected looks like as if the participant that is supposed to wear it does is not moving. If left undetected, not wearing the accelerometer would bias the estimates of time spent in inactive behaviours.\n\nStrategies for Handling **Non-Wear Periods:**\n\n1. **Filter Out Non-Wear Data:** Remove data where non_wear_flag == 1. This ensures only valid, reliable data is used in your analysis and feature engineering. This ensures the dataset only includes meaningful data.\n\n1. **Use Non-Wear Periods as a Feature:** Instead of removing or imputing non-wear data, summarize the non-wear periods as a separate feature. This could indicate behavior patterns, such as extended periods of inactivity.\n    - **Derived Features:**\n        - Total duration of non-wear periods (non_wear_flag.sum()).\n        - Percentage of total time spent in non-wear periods.\n\n\n## <span style=\"color: #016FD0;\">Summary of Approach</span>\n\n**Objective**: To preprocess actigraphy data from wearable accelerometers and generate reliable features for machine learning, ensuring non-wear periods do not bias the analysis.\n\n**Key Steps:**\n\n1. **Non-Wear Data Handling:** Filtered out data where the device was not worn (non_wear_flag == 1) to ensure only reliable wear data is used.\n\n1. **Feature Aggregation:**\n    - Computed statistical features (mean, std, min, max, median) for activity metrics (`X`, `Y`, `Z`, `enmo`, `anglez`, `light`).\n    - Derived activity intensity: $$\n\\text{Activity Intensity} = \\sqrt{X^2 + Y^2 + Z^2}\n$$\n\n    - Separated daytime (6 AM–6 PM) and nighttime (6 PM–6 AM) activity, calculating intensity means and their ratio.\n\n1. **Processing Efficiency:** Processed participants one at a time to handle large datasets efficiently, aggregating each participant’s data into a single row.\n\n**Output:** A final dataset where each row represents one participant, containing meaningful features derived from filtered wear data, ready for modeling.","metadata":{}},{"cell_type":"code","source":"class DataProcessingPipeline:\n    def __init__(self, random_state: int = 42):\n        \"\"\"\n        Initialize the pipeline with random state for reproducibility.\n\n        Args:\n            random_state (int): Random state for reproducibility.\n        \"\"\"\n        self.scaler = None\n        self.label_encoders = {}\n        self.feature_types = {\"categorical\": [], \"numerical\": []}\n\n    def load_data(self, train_path: str, test_path: str, data_dict_path: str) -> tuple:\n        \"\"\"\n        Load train, test, and data dictionary datasets.\n\n        Args:\n            train_path (str): Path to the training dataset.\n            test_path (str): Path to the testing dataset.\n            data_dict_path (str): Path to the data dictionary.\n\n        Returns:\n            tuple: Train, test, and data dictionary as pandas DataFrames.\n        \"\"\"\n        train_df = pd.read_csv(train_path)\n        test_df = pd.read_csv(test_path)\n        data_dict = pd.read_csv(data_dict_path)\n        return train_df, test_df, data_dict\n\n    def preprocess_data(\n        self, train_df: pd.DataFrame, test_df: pd.DataFrame, data_dict: pd.DataFrame, target_column: str, ts_train_path: str, ts_test_path: str\n    ) -> tuple:\n        \"\"\"\n        Preprocess the data, including merging time-series data, handling missing values, and scaling.\n\n        Args:\n            train_df (pd.DataFrame): Training dataset.\n            test_df (pd.DataFrame): Testing dataset.\n            data_dict (pd.DataFrame): Data dictionary.\n            target_column (str): Name of the target column.\n            ts_train_path (str): Path to time-series training data.\n            ts_test_path (str): Path to time-series testing data.\n\n        Returns:\n            tuple: Processed train features, train target, test features, and feature names.\n        \"\"\"\n        # Get common features between train and test datasets\n        common_features = list(set(train_df.columns) & set(test_df.columns))\n        sii_df =  train_df[train_df['sii'].notnull()]['sii']\n        train_df = train_df[common_features + [target_column]]\n        test_df = test_df[common_features]\n\n        # Separate feature types using the data dictionary\n        for _, row in data_dict.iterrows():\n            if row[\"Field\"] in common_features and row[\"Field\"] != \"id\":\n                if row[\"Type\"] in [\"str\"]:\n                    self.feature_types[\"categorical\"].append(row[\"Field\"])\n                else:  # Exclude 'id' from numerical\n                    self.feature_types[\"numerical\"].append(row[\"Field\"])\n\n        # Drop rows with missing target values\n        train_df = train_df[train_df[target_column].notnull()]\n\n        # Handle missing values\n        num_imputer = SimpleImputer(strategy=\"median\")\n        cat_imputer = SimpleImputer(strategy=\"most_frequent\")\n\n        # Impute numerical features\n        train_df[self.feature_types[\"numerical\"]] = num_imputer.fit_transform(train_df[self.feature_types[\"numerical\"]])\n        test_df[self.feature_types[\"numerical\"]] = num_imputer.transform(test_df[self.feature_types[\"numerical\"]])\n\n        # Impute categorical features\n        train_df[self.feature_types[\"categorical\"]] = cat_imputer.fit_transform(train_df[self.feature_types[\"categorical\"]])\n        test_df[self.feature_types[\"categorical\"]] = cat_imputer.transform(test_df[self.feature_types[\"categorical\"]])\n\n        # Encode categorical features\n        for col in self.feature_types[\"categorical\"]:\n            le = LabelEncoder()\n            train_df[col] = le.fit_transform(train_df[col].astype(str))\n            test_df[col] = le.transform(test_df[col].astype(str))\n            self.label_encoders[col] = le\n\n        # Scale numerical features\n        self.scaler = StandardScaler()\n        train_df[self.feature_types[\"numerical\"]] = self.scaler.fit_transform(train_df[self.feature_types[\"numerical\"]])\n        test_df[self.feature_types[\"numerical\"]] = self.scaler.transform(test_df[self.feature_types[\"numerical\"]])\n\n        # Aggregate time-series data\n        train_ts_data = self.aggregate_time_series(ts_train_path)\n        test_ts_data = self.aggregate_time_series(ts_test_path)\n\n        # Merge time-series data with tabular data\n        train_df = pd.merge(train_df, train_ts_data, how=\"left\", on=\"id\")\n        test_df = pd.merge(test_df, test_ts_data, how=\"left\", on=\"id\")\n\n        # Replace missing values in time-series data with -1\n        train_df = train_df.fillna(-1)\n        test_df = test_df.fillna(-1)\n\n        # Extract features and target\n        features = [\"id\"] + self.feature_types[\"categorical\"] + self.feature_types[\"numerical\"] + list(train_ts_data.columns.drop(\"id\"))\n        X_train = train_df[features]\n        y_train = train_df[target_column]\n        X_test = test_df[features]\n\n        time_series_features = list(train_ts_data.columns)\n        \n        return X_train, y_train, X_test, features, time_series_features, sii_df\n\n    def aggregate_time_series(self, data_path: str) -> pd.DataFrame:\n        \"\"\"\n        Aggregate time series data for all participants into a single row per participant.\n\n        Args:\n            data_path (str): Path to the folder containing time-series parquet files.\n\n        Returns:\n            pd.DataFrame: Aggregated data with one row per participant.\n        \"\"\"\n        participant_folders = glob(os.path.join(data_path, \"id=*\"))\n        aggregated_data = []\n\n        for participant_folder in participant_folders:\n            # Extract participant ID from folder name\n            participant_id = os.path.basename(participant_folder).split(\"=\")[1]\n\n            # Path to the parquet file\n            file_path = os.path.join(participant_folder, \"part-0.parquet\")\n\n            participant_data = self._process_participant_time_series(file_path, participant_id)\n\n            # Append to the main list if data is not empty\n            if not participant_data.empty:\n                aggregated_data.append(participant_data)\n\n        return pd.concat(aggregated_data, ignore_index=True)\n\n    def _process_participant_time_series(self, file_path: str, participant_id: str) -> pd.DataFrame:\n        \"\"\"\n        Process and aggregate features from a single participant's time-series data.\n\n        Args:\n            file_path (str): Path to the participant's parquet file.\n            participant_id (str): Participant ID.\n\n        Returns:\n            pd.DataFrame: A single-row DataFrame with aggregated features for the participant.\n        \"\"\"\n        try:\n            # Load the participant's parquet file\n            df = pd.read_parquet(file_path)\n    \n            # Convert time_of_day to datetime\n            try:\n                df['timestamp'] = pd.to_datetime(df['time_of_day'], unit='ns')  # Assuming nanoseconds\n            except ValueError:\n                df['timestamp'] = pd.to_datetime(df['time_of_day'], unit='s')\n    \n            df = df[df['non-wear_flag'] == 0]\n            # Set the timestamp as the index for time-based calculations\n            df = df.set_index('timestamp')\n    \n            # Derived feature: Total activity intensity\n            df['activity_intensity'] = np.sqrt(df['X']**2 + df['Y']**2 + df['Z']**2)\n    \n            # Identify daytime (6 AM to 6 PM) vs nighttime\n            df['hour'] = df.index.hour\n            df['is_daytime'] = ((df['hour'] >= 6) & (df['hour'] < 18)).astype(int)\n    \n            # Aggregate features\n            aggregated = {\n                'id': participant_id,\n                # Basic statistics for X, Y, Z\n                'X_mean': df['X'].mean(),\n                'X_std': df['X'].std(),\n                'X_max': df['X'].max(),\n                'X_min': df['X'].min(),\n                'X_median': df['X'].median(),\n                'Y_mean': df['Y'].mean(),\n                'Y_std': df['Y'].std(),\n                'Y_max': df['Y'].max(),\n                'Y_min': df['Y'].min(),\n                'Y_median': df['Y'].median(),\n                'Z_mean': df['Z'].mean(),\n                'Z_std': df['Z'].std(),\n                'Z_max': df['Z'].max(),\n                'Z_min': df['Z'].min(),\n                'Z_median': df['Z'].median(),\n                # ENMO and Anglez\n                'enmo_mean': df['enmo'].mean(),\n                'enmo_std': df['enmo'].std(),\n                'enmo_max': df['enmo'].max(),\n                'enmo_min': df['enmo'].min(),\n                'anglez_mean': df['anglez'].mean(),\n                'anglez_std': df['anglez'].std(),\n                'anglez_max': df['anglez'].max(),\n                'anglez_min': df['anglez'].min(),\n                # Light exposure\n                'light_mean': df['light'].mean(),\n                'light_std': df['light'].std(),\n                'light_max': df['light'].max(),\n                'light_min': df['light'].min(),\n                'light_median': df['light'].median(),\n                # Battery voltage\n                'battery_voltage_mean': df['battery_voltage'].mean(),\n                'battery_voltage_std': df['battery_voltage'].std(),\n                # Derived features\n                'activity_intensity_mean': df['activity_intensity'].mean(),\n                'activity_intensity_std': df['activity_intensity'].std(),\n                'activity_intensity_max': df['activity_intensity'].max(),\n                'activity_intensity_min': df['activity_intensity'].min(),\n                # Daytime vs nighttime activity\n                'daytime_activity_mean': df.loc[df['is_daytime'] == 1, 'activity_intensity'].mean(),\n                'nighttime_activity_mean': df.loc[df['is_daytime'] == 0, 'activity_intensity'].mean(),\n                'daytime_nighttime_activity_ratio': (\n                    df.loc[df['is_daytime'] == 1, 'activity_intensity'].mean() /\n                    (df.loc[df['is_daytime'] == 0, 'activity_intensity'].mean() + 1e-6)  # Avoid division by zero\n                ),\n                # Variance of activity throughout the day\n                'activity_variance': df['activity_intensity'].var()\n            }\n    \n            return pd.DataFrame([aggregated])  # Return as a single-row DataFrame\n    \n        except Exception as e:\n            print(f\"Error processing file {file_path} for participant {participant_id}: {e}\")\n            return pd.DataFrame()  # Return an empty DataFrame in case of error","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T12:11:30.441103Z","iopub.execute_input":"2024-12-08T12:11:30.441416Z","iopub.status.idle":"2024-12-08T12:11:30.465787Z","shell.execute_reply.started":"2024-12-08T12:11:30.441391Z","shell.execute_reply":"2024-12-08T12:11:30.464971Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_path = '/kaggle/input/child-mind-institute-problematic-internet-use/train.csv'\ntest_path = '/kaggle/input/child-mind-institute-problematic-internet-use/test.csv'\ndata_dict_path = '/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv'\nts_train_path = \"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\"\nts_test_path = \"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T12:11:31.175267Z","iopub.execute_input":"2024-12-08T12:11:31.176278Z","iopub.status.idle":"2024-12-08T12:11:31.180276Z","shell.execute_reply.started":"2024-12-08T12:11:31.176239Z","shell.execute_reply":"2024-12-08T12:11:31.179403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Usage\npipeline = DataProcessingPipeline()\n\n# Load data\ntrain_df, test_df, data_dict = pipeline.load_data(train_path, test_path, data_dict_path)\n\n# Preprocess data\nX_train, y_train, X_test, features, time_series_features, sii_df = pipeline.preprocess_data(\n    train_df, test_df, data_dict, target_column=\"PCIAT-PCIAT_Total\", ts_train_path=ts_train_path, ts_test_path=ts_test_path\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T12:11:31.912486Z","iopub.execute_input":"2024-12-08T12:11:31.912952Z","iopub.status.idle":"2024-12-08T12:12:49.024018Z","shell.execute_reply.started":"2024-12-08T12:11:31.912921Z","shell.execute_reply":"2024-12-08T12:12:49.022971Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <span style=\"color: #016FD0;\">Data Validation Checks</span>\n","metadata":{}},{"cell_type":"code","source":"train_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:16:09.185787Z","iopub.execute_input":"2024-12-08T11:16:09.186068Z","iopub.status.idle":"2024-12-08T11:16:09.192026Z","shell.execute_reply.started":"2024-12-08T11:16:09.186042Z","shell.execute_reply":"2024-12-08T11:16:09.191034Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['id'].nunique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:16:09.195081Z","iopub.execute_input":"2024-12-08T11:16:09.195349Z","iopub.status.idle":"2024-12-08T11:16:09.205741Z","shell.execute_reply.started":"2024-12-08T11:16:09.195323Z","shell.execute_reply":"2024-12-08T11:16:09.204772Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:16:09.206862Z","iopub.execute_input":"2024-12-08T11:16:09.207202Z","iopub.status.idle":"2024-12-08T11:16:09.229907Z","shell.execute_reply.started":"2024-12-08T11:16:09.20716Z","shell.execute_reply":"2024-12-08T11:16:09.229037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:16:09.231014Z","iopub.execute_input":"2024-12-08T11:16:09.231338Z","iopub.status.idle":"2024-12-08T11:16:09.244874Z","shell.execute_reply.started":"2024-12-08T11:16:09.231303Z","shell.execute_reply":"2024-12-08T11:16:09.243805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train['id'].nunique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:16:09.245703Z","iopub.execute_input":"2024-12-08T11:16:09.245983Z","iopub.status.idle":"2024-12-08T11:16:09.257425Z","shell.execute_reply.started":"2024-12-08T11:16:09.24596Z","shell.execute_reply":"2024-12-08T11:16:09.256711Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:16:09.258396Z","iopub.execute_input":"2024-12-08T11:16:09.258693Z","iopub.status.idle":"2024-12-08T11:16:09.267004Z","shell.execute_reply.started":"2024-12-08T11:16:09.258653Z","shell.execute_reply":"2024-12-08T11:16:09.26612Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:16:09.268328Z","iopub.execute_input":"2024-12-08T11:16:09.268744Z","iopub.status.idle":"2024-12-08T11:16:09.283868Z","shell.execute_reply.started":"2024-12-08T11:16:09.268687Z","shell.execute_reply":"2024-12-08T11:16:09.283079Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:16:09.284931Z","iopub.execute_input":"2024-12-08T11:16:09.285258Z","iopub.status.idle":"2024-12-08T11:16:09.295369Z","shell.execute_reply.started":"2024-12-08T11:16:09.285222Z","shell.execute_reply":"2024-12-08T11:16:09.2939Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:16:09.296391Z","iopub.execute_input":"2024-12-08T11:16:09.297126Z","iopub.status.idle":"2024-12-08T11:16:09.317544Z","shell.execute_reply.started":"2024-12-08T11:16:09.2971Z","shell.execute_reply":"2024-12-08T11:16:09.316547Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <div style=\"padding:20px;color:white;margin:0;font-size:30px;font-family:Georgia;text-align:left;display:fill;border-radius:5px;background-color:#016FD0;overflow:hidden\">Step 2: Modeling with LightGBM Regressor</div>","metadata":{}},{"cell_type":"code","source":"# Prepare data\nX = X_train.drop(columns=['id'])  # Features\ny = y_train  # Continuous target variable","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:16:09.318574Z","iopub.execute_input":"2024-12-08T11:16:09.318916Z","iopub.status.idle":"2024-12-08T11:16:09.327716Z","shell.execute_reply.started":"2024-12-08T11:16:09.31887Z","shell.execute_reply":"2024-12-08T11:16:09.326672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define thresholds to map predictions to `sii` categories\ndef map_to_sii(predictions):\n    \"\"\"\n    Map continuous predictions to sii categories based on thresholds.\n    \"\"\"\n    return np.digitize(predictions, bins=[30, 49, 79])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:16:09.328838Z","iopub.execute_input":"2024-12-08T11:16:09.329088Z","iopub.status.idle":"2024-12-08T11:16:09.33767Z","shell.execute_reply.started":"2024-12-08T11:16:09.329063Z","shell.execute_reply":"2024-12-08T11:16:09.336899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Define Optuna objective with CV for regression\ndef objective(trial):\n    \"\"\"\n    Objective function for Optuna hyperparameter optimization using cross-validation.\n    \"\"\"\n    # Define hyperparameters to tune\n    params = {\n        'objective': 'regression',\n        'boosting_type': 'gbdt',\n        'metric': 'rmse',\n        'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.3),\n        'num_leaves': trial.suggest_int('num_leaves', 20, 150),\n        'max_depth': trial.suggest_int('max_depth', 3, 12),\n        'min_data_in_leaf': trial.suggest_int('min_data_in_leaf', 10, 100),\n        'feature_fraction': trial.suggest_float('feature_fraction', 0.6, 1.0),\n        'bagging_fraction': trial.suggest_float('bagging_fraction', 0.6, 1.0),\n        'bagging_freq': trial.suggest_int('bagging_freq', 1, 7),\n        'lambda_l1': trial.suggest_float('lambda_l1', 0.0, 10.0),\n        'lambda_l2': trial.suggest_float('lambda_l2', 0.0, 10.0),\n        'verbosity': -1,\n        'seed': 42\n    }\n    \n    # Cross-validation setup\n    kf = KFold(n_splits=5, shuffle=True, random_state=42)\n    rmse_scores = []\n\n    for train_index, val_index in kf.split(X):\n        X_train, X_val = X.iloc[train_index], X.iloc[val_index]\n        y_train, y_val = y.iloc[train_index], y.iloc[val_index]\n\n        # Prepare LightGBM datasets\n        train_data = lgb.Dataset(X_train, label=y_train)\n        val_data = lgb.Dataset(X_val, label=y_val)\n\n        # Train model\n        model = lgb.train(\n            params,\n            train_data,\n            valid_sets=[val_data],\n            valid_names=['validation'],\n            num_boost_round=1000,\n            # early_stopping_rounds=50,\n            # verbose_eval=False\n        )\n        \n        # Predict on validation set\n        y_pred = model.predict(X_val)\n\n        # Calculate RMSE\n        rmse = np.sqrt(mean_squared_error(y_val, y_pred))\n        rmse_scores.append(rmse)\n\n    # Return the mean RMSE score across folds\n    return np.mean(rmse_scores)\n\n# # Create the Optuna study\n# study = optuna.create_study(direction='minimize')  # Aim to minimize RMSE\n# study.optimize(objective, n_trials=50)  # Adjust n_trials based on available time/resources\n\n# #Output the best parameters and RMSE score\n# print(\"Best RMSE Score:\", study.best_value)\n# print(\"Best Parameters:\", study.best_params)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:16:09.339255Z","iopub.execute_input":"2024-12-08T11:16:09.339568Z","iopub.status.idle":"2024-12-08T11:16:09.348387Z","shell.execute_reply.started":"2024-12-08T11:16:09.339532Z","shell.execute_reply":"2024-12-08T11:16:09.34777Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the final model with the best parameters\n#best_params = study.best_params\n\nbest_params = {'learning_rate': 0.01219638891487601,\n 'num_leaves': 20,\n 'max_depth': 3,\n 'min_data_in_leaf': 67,\n 'feature_fraction': 0.7343189362001207,\n 'bagging_fraction': 0.9792862620910322,\n 'bagging_freq': 7,\n 'lambda_l1': 3.89260781654852,\n 'lambda_l2': 9.76055925002488}\n\nbest_params.update({\n    'objective': 'regression',\n    'metric': 'rmse',\n    'verbosity': -1,\n    'seed': 42\n})\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:16:09.349385Z","iopub.execute_input":"2024-12-08T11:16:09.349643Z","iopub.status.idle":"2024-12-08T11:16:09.362272Z","shell.execute_reply.started":"2024-12-08T11:16:09.349618Z","shell.execute_reply":"2024-12-08T11:16:09.361592Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train final model on the entire dataset\nfinal_kf = KFold(n_splits=5, shuffle=True, random_state=42)\nfinal_rmse_scores = []\n\nfor train_index, val_index in final_kf.split(X):\n    X_train, X_val = X.iloc[train_index], X.iloc[val_index]\n    y_train, y_val = y.iloc[train_index], y.iloc[val_index]\n\n    # Prepare LightGBM datasets\n    train_data = lgb.Dataset(X_train, label=y_train)\n    val_data = lgb.Dataset(X_val, label=y_val)\n\n    # Train model\n    final_model = lgb.train(\n        best_params,\n        train_data,\n        valid_sets=[val_data],\n        valid_names=['validation'],\n        num_boost_round=1000,\n        # early_stopping_rounds=50,\n        # verbose_eval=50\n    )\n\n    # Predict on validation set\n    y_pred = final_model.predict(X_val)\n\n    # Map predictions to `sii` categories\n    y_pred_sii = map_to_sii(y_pred)\n\n    # Calculate RMSE for regression\n    rmse = np.sqrt(mean_squared_error(y_val, y_pred))\n    final_rmse_scores.append(rmse)\n\n# Output the final cross-validated RMSE score\nprint(\"\\nFinal Cross-Validated RMSE Score:\", np.mean(final_rmse_scores))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:16:09.363284Z","iopub.execute_input":"2024-12-08T11:16:09.36362Z","iopub.status.idle":"2024-12-08T11:16:13.185565Z","shell.execute_reply.started":"2024-12-08T11:16:09.363581Z","shell.execute_reply":"2024-12-08T11:16:13.18479Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def xgb_objective(trial):\n    \"\"\"\n    Objective function for Optuna hyperparameter optimization for XGBoost.\n    \"\"\"\n    params = {\n        'objective': 'reg:squarederror',\n        'booster': 'gbtree',\n        'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.3),\n        'max_depth': trial.suggest_int('max_depth', 3, 12),\n        'min_child_weight': trial.suggest_float('min_child_weight', 1, 10),\n        'subsample': trial.suggest_float('subsample', 0.5, 1.0),\n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0),\n        'lambda': trial.suggest_float('lambda', 0.0, 10.0),\n        'alpha': trial.suggest_float('alpha', 0.0, 10.0),\n        'seed': 42\n    }\n    \n    # Cross-validation setup\n    kf = KFold(n_splits=5, shuffle=True, random_state=42)\n    rmse_scores = []\n\n    for train_index, val_index in kf.split(X):\n        X_train, X_val = X.iloc[train_index], X.iloc[val_index]\n        y_train, y_val = y.iloc[train_index], y.iloc[val_index]\n\n        # Train XGBoost model\n        model = XGBRegressor(**params)\n        model.fit(X_train, y_train, eval_set=[(X_val, y_val)], early_stopping_rounds=50, verbose=False)\n\n        # Predict and calculate RMSE\n        y_pred = model.predict(X_val)\n        rmse = np.sqrt(mean_squared_error(y_val, y_pred))\n        rmse_scores.append(rmse)\n\n    return np.mean(rmse_scores)\n\n# # Optimize XGBoost parameters\n# xgb_study = optuna.create_study(direction='minimize')\n# xgb_study.optimize(xgb_objective, n_trials=50)\n\n# print(\"Best RMSE for XGBoost:\", xgb_study.best_value)\n# print(\"Best Parameters for XGBoost:\", xgb_study.best_params)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:16:13.186484Z","iopub.execute_input":"2024-12-08T11:16:13.186831Z","iopub.status.idle":"2024-12-08T11:16:13.196332Z","shell.execute_reply.started":"2024-12-08T11:16:13.186798Z","shell.execute_reply":"2024-12-08T11:16:13.195649Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# xgb_best_params = xgb_study.best_params\n# xgb_best_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:16:13.19784Z","iopub.execute_input":"2024-12-08T11:16:13.198254Z","iopub.status.idle":"2024-12-08T11:16:13.211866Z","shell.execute_reply.started":"2024-12-08T11:16:13.198221Z","shell.execute_reply":"2024-12-08T11:16:13.211156Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train final XGBoost model on entire dataset\n#xgb_best_params = xgb_study.best_params\nxgb_best_params = {'learning_rate': 0.10188630468562546,\n 'max_depth': 3,\n 'min_child_weight': 9.947640099695992,\n 'subsample': 0.6679228912766382,\n 'colsample_bytree': 0.5045549968675882,\n 'lambda': 4.535357944947238,\n 'alpha': 9.989245566519957\n}\nxgb_best_params.update({'objective': 'reg:squarederror', 'seed': 42})\n\nfinal_xgb_model = XGBRegressor(**xgb_best_params)\nfinal_xgb_model.fit(X, y)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:16:13.214253Z","iopub.execute_input":"2024-12-08T11:16:13.214536Z","iopub.status.idle":"2024-12-08T11:16:13.546976Z","shell.execute_reply.started":"2024-12-08T11:16:13.214505Z","shell.execute_reply":"2024-12-08T11:16:13.546217Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Extract tabular features only (excluding time-series features)\ntabular_features = [col for col in X.columns if col not in time_series_features]\n\nfrom sklearn.metrics import cohen_kappa_score\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    \"\"\"\n    Calculate the Quadratic Weighted Kappa (QWK) metric.\n    \"\"\"\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n\ndef catboost_qwk_objective(trial):\n    \"\"\"\n    Objective function for Optuna hyperparameter optimization for CatBoostClassifier, using QWK as the metric.\n    \"\"\"\n    params = {\n        'iterations': trial.suggest_int('iterations', 500, 2000),\n        'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.3),\n        'depth': trial.suggest_int('depth', 4, 10),\n        'l2_leaf_reg': trial.suggest_float('l2_leaf_reg', 1e-3, 10.0),\n        'random_strength': trial.suggest_float('random_strength', 1e-3, 10.0),\n        'bagging_temperature': trial.suggest_float('bagging_temperature', 0.0, 1.0),\n        'border_count': trial.suggest_int('border_count', 32, 255),\n        'auto_class_weights': trial.suggest_categorical('auto_class_weights', ['Balanced', None]),\n        'verbose': 0,\n        'random_seed': 42\n    }\n    \n    # Cross-validation setup\n    kf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n    qwk_scores = []\n    y = sii_df\n    for train_index, val_index in kf.split(X[tabular_features], y):\n        X_train, X_val = X.iloc[train_index][tabular_features], X.iloc[val_index][tabular_features]\n        y_train, y_val = y.iloc[train_index], y.iloc[val_index]\n\n        # Train CatBoost model\n        model = CatBoostClassifier(**params)\n        model.fit(X_train, y_train, eval_set=(X_val, y_val), early_stopping_rounds=50, verbose=False)\n\n        # Predict and calculate QWK\n        y_pred = model.predict(X_val)\n        qwk = quadratic_weighted_kappa(y_val, y_pred)\n        qwk_scores.append(qwk)\n\n    return np.mean(qwk_scores)\n\n# # Optimize CatBoost parameters\n# catboost_study = optuna.create_study(direction='maximize')  # Maximize QWK\n# catboost_study.optimize(catboost_qwk_objective, n_trials=50)\n\n# print(\"Best QWK for CatBoost:\", catboost_study.best_value)\n# print(\"Best Parameters for CatBoost:\", catboost_study.best_params)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T12:25:35.140342Z","iopub.execute_input":"2024-12-08T12:25:35.1407Z","iopub.status.idle":"2024-12-08T12:25:35.149577Z","shell.execute_reply.started":"2024-12-08T12:25:35.140668Z","shell.execute_reply":"2024-12-08T12:25:35.148643Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Train the final CatBoostClassifier with the best parameters\ncatboost_best_params = {'iterations': 1201, 'learning_rate': 0.013039841744365226, 'depth': 10, 'l2_leaf_reg': 5.867074906373046, 'random_strength': 3.631971312147143, 'bagging_temperature': 0.8544193697083458, 'border_count': 103, 'auto_class_weights': 'Balanced'}\nfinal_catboost_clf = CatBoostClassifier(**catboost_best_params, random_seed=42, verbose=100)\nfinal_catboost_clf.fit(X[tabular_features], sii_df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T12:22:27.02184Z","iopub.execute_input":"2024-12-08T12:22:27.022208Z","iopub.status.idle":"2024-12-08T12:23:42.889691Z","shell.execute_reply.started":"2024-12-08T12:22:27.022176Z","shell.execute_reply":"2024-12-08T12:23:42.888696Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.stats import mode\n\n# Prepare data (replace df with your dataset)\ntest_ids = X_test['id'] \n\nX_pred = X_test.drop(columns=['id'])  # Features\n\n# Predict on test dataset using LightGBM\ntest_pred_lgb = final_model.predict(X_pred)\ntest_pred_lgb_sii = map_to_sii(test_pred_lgb)  # Map to `sii` categories\n\n\n# Predict on test dataset using XGBoost\ntest_pred_xgb = final_xgb_model.predict(X_pred)\ntest_pred_xgb_sii = map_to_sii(test_pred_xgb)  # Map to `sii` categories\n\n\n# Predict on test dataset using CatBoost (only tabular features)\ntest_pred_catboost = final_catboost_clf.predict(X_pred[tabular_features])\ntest_pred_catboost_sii = np.squeeze(test_pred_catboost)  # Remove extra dimension\n\n\n# Average predictions from all three models\n#test_pred_ensemble = (test_pred_lgb + test_pred_xgb + test_pred_catboost) / 3\n\n# Map predictions to `sii` categories\n#test_pred_sii_ensemble = map_to_sii(test_pred_ensemble)\n\n\n# Combine all discrete predictions\nall_predictions = np.vstack([\n    test_pred_lgb_sii,\n    test_pred_xgb_sii,\n    test_pred_catboost_sii\n])\n\n# Perform majority voting\ntest_pred_majority = mode(all_predictions, axis=0).mode.flatten()\n\n\n\n\n# Prepare submission file\nsubmission = pd.DataFrame({\n    'id': test_ids,\n    'sii': test_pred_majority\n})\nsubmission.to_csv('submission.csv', index=False)\nprint(\"Submission file created: submission_ensemble.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T12:33:05.549992Z","iopub.execute_input":"2024-12-08T12:33:05.550615Z","iopub.status.idle":"2024-12-08T12:33:05.607007Z","shell.execute_reply.started":"2024-12-08T12:33:05.550577Z","shell.execute_reply":"2024-12-08T12:33:05.606267Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Prepare data (replace df with your dataset)\n# test_ids = X_test['id'] \n\n# X_pred = X_test.drop(columns=['id'])  # Features\n\n# # Predict on test dataset\n# test_pred = final_model.predict(X_pred)  # Continuous predictions\n# test_pred_sii = map_to_sii(test_pred)    # Map to `sii` categories\n\n# # Prepare submission file\n# submission = pd.DataFrame({\n#     'id': test_ids,  # Use the ID column from the test dataset\n#     'sii': test_pred_sii\n# })\n# submission.to_csv('submission.csv', index=False)\n# print(\"Submission file created: submission.csv\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T12:33:07.949572Z","iopub.execute_input":"2024-12-08T12:33:07.950471Z","iopub.status.idle":"2024-12-08T12:33:07.95901Z","shell.execute_reply.started":"2024-12-08T12:33:07.950414Z","shell.execute_reply":"2024-12-08T12:33:07.958009Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}