{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.impute import KNNImputer\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nfrom concurrent.futures import ThreadPoolExecutor\nfrom scipy.optimize import minimize\nimport os\nimport pyarrow.parquet as pq\nfrom datetime import timedelta\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.cluster import KMeans\nfrom sklearn.metrics import silhouette_score\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.experimental import enable_iterative_imputer \nfrom sklearn.impute import IterativeImputer\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:18:50.684172Z","iopub.execute_input":"2024-12-22T12:18:50.684589Z","iopub.status.idle":"2024-12-22T12:18:52.678440Z","shell.execute_reply.started":"2024-12-22T12:18:50.684548Z","shell.execute_reply":"2024-12-22T12:18:52.677259Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_file(filename, dirname):\n    \n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    \n \n    \n    # Ensure the datetime format for time-of-day trends\n    df['time_of_day'] = pd.to_datetime(df['time_of_day'], format='%H:%M:%S.%f',errors='coerce')\n    df['hour'] = df['time_of_day'].dt.hour \n    df['weekday'] = df['weekday']         \n    df['quarter'] = df['quarter']        \n\n    # Aggregate daily metrics\n    daily_agg = df.groupby('relative_date_PCIAT').agg({\n        'X': ['mean', 'std', 'max'],\n        'Y': ['mean', 'std', 'max'],\n        'Z': ['mean', 'std', 'max'],\n        'enmo': ['mean', 'std', 'max'],\n        'anglez': ['mean', 'std', 'max'],\n        # 'non_wear_flag': lambda x: (x == 1).sum()\n        'light': ['mean', 'max']                  \n    }).reset_index()\n    \n    \n   \n\n    daily_agg.columns = ['_'.join(col).strip('_') for col in daily_agg.columns]\n\n    enmo_variability = daily_agg[['enmo_std', 'enmo_mean']].apply(\n        lambda row: row['enmo_std'] / row['enmo_mean'] if row['enmo_mean'] > 0 else 0, axis=1\n    )\n    daily_agg['enmo_variability'] = enmo_variability\n\n\n    time_context_features = df.groupby(['hour', 'weekday', 'quarter'])['enmo'].mean().reset_index()\n\n    daily_agg_columns = [f\"{col}_{date}\" for date in daily_agg['relative_date_PCIAT']\n                         for col in daily_agg.columns if col != 'relative_date_PCIAT']\n    time_context_columns = [\n        f\"enmo_hour{hour}_weekday{weekday}_quarter{quarter}\"\n        for hour, weekday, quarter in zip(time_context_features['hour'],\n                                          time_context_features['weekday'],\n                                          time_context_features['quarter'])\n    ]\n    \n    flattened_stats = np.concatenate([\n        daily_agg.drop('relative_date_PCIAT', axis=1).values.flatten(),  # Daily metrics\n        time_context_features['enmo'].values.flatten()  # Time-based trends\n    ])\n\n    column_names = daily_agg_columns + time_context_columns\n\n    # return flattened_stats, filename.split('=')[1]\n    # return df.describe().values.reshape(-1), filename.split('=')[1]\n    return flattened_stats, column_names, filename.split('=')[1]\n\n\n\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    max_columns = max(len(columns) for _, columns, _ in results)\n\n    padded_stats = []\n    all_column_names = None\n    for stats, columns, idx in results:\n        if len(stats) < max_columns:\n            stats = np.pad(stats, (0, max_columns - len(stats)), constant_values=np.nan)\n        padded_stats.append(stats)\n        if all_column_names is None:\n            all_column_names = columns + [f\"missing_col_{i}\" for i in range(len(columns), max_columns)]\n    \n    df = pd.DataFrame(padded_stats, columns=all_column_names)\n    df['id'] = [idx for _, _, idx in results]\n    return df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:18:52.679580Z","iopub.execute_input":"2024-12-22T12:18:52.680143Z","iopub.status.idle":"2024-12-22T12:18:52.693163Z","shell.execute_reply.started":"2024-12-22T12:18:52.680103Z","shell.execute_reply":"2024-12-22T12:18:52.691980Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n\ntrain_series = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_series = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:18:52.694627Z","iopub.execute_input":"2024-12-22T12:18:52.695076Z","iopub.status.idle":"2024-12-22T12:26:52.989343Z","shell.execute_reply.started":"2024-12-22T12:18:52.695031Z","shell.execute_reply":"2024-12-22T12:26:52.988277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_series)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:26:52.990436Z","iopub.execute_input":"2024-12-22T12:26:52.990767Z","iopub.status.idle":"2024-12-22T12:26:53.009166Z","shell.execute_reply.started":"2024-12-22T12:26:52.990742Z","shell.execute_reply":"2024-12-22T12:26:53.007935Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_series = train_series.drop(columns=train_series.filter(regex='^missing_col').columns)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:26:53.012118Z","iopub.execute_input":"2024-12-22T12:26:53.012437Z","iopub.status.idle":"2024-12-22T12:26:53.021641Z","shell.execute_reply.started":"2024-12-22T12:26:53.012398Z","shell.execute_reply":"2024-12-22T12:26:53.020672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_series)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:26:53.026593Z","iopub.execute_input":"2024-12-22T12:26:53.026936Z","iopub.status.idle":"2024-12-22T12:26:53.048779Z","shell.execute_reply.started":"2024-12-22T12:26:53.026908Z","shell.execute_reply":"2024-12-22T12:26:53.047720Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"nan_percentage = train_series.isna().mean() * 100\n\nhigh_nan_columns = nan_percentage[nan_percentage > 80]\n\ntrain_series = train_series.drop(columns=high_nan_columns.index)\n\nprint(\"Columns with more than 80% NaN values removed:\", high_nan_columns.index.tolist())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:26:53.049948Z","iopub.execute_input":"2024-12-22T12:26:53.050255Z","iopub.status.idle":"2024-12-22T12:26:53.073659Z","shell.execute_reply.started":"2024-12-22T12:26:53.050217Z","shell.execute_reply":"2024-12-22T12:26:53.072475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.merge(train, train_series, how=\"left\", on='id')\ntest = pd.merge(test, test_series, how=\"left\", on='id')\ntrain_df=train\ntest_df=test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:26:53.074764Z","iopub.execute_input":"2024-12-22T12:26:53.075145Z","iopub.status.idle":"2024-12-22T12:26:53.114098Z","shell.execute_reply.started":"2024-12-22T12:26:53.075109Z","shell.execute_reply":"2024-12-22T12:26:53.113156Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:26:53.115311Z","iopub.execute_input":"2024-12-22T12:26:53.115725Z","iopub.status.idle":"2024-12-22T12:26:53.121598Z","shell.execute_reply.started":"2024-12-22T12:26:53.115686Z","shell.execute_reply":"2024-12-22T12:26:53.120430Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def knn_impute(df, numerical_cols):\n    imputer = KNNImputer(n_neighbors=3)\n    df[numerical_cols] = imputer.fit_transform(df[numerical_cols])\n    return df\n\ndef mice_impute_onehot_with_reverse(df, categorical_cols):\n    onehot_mappings = {} \n    original_cols = df.columns.tolist() \n    \n    df_onehot = pd.get_dummies(df, columns=categorical_cols, dummy_na=True)\n    \n    imputer = IterativeImputer(max_iter=10, random_state=0)\n    df_onehot[:] = imputer.fit_transform(df_onehot)\n    \n    df_reversed = df.copy()\n    for col in categorical_cols:\n        relevant_cols = [c for c in df_onehot.columns if c.startswith(f\"{col}_\")]\n        mappings = {idx: cat.replace(f\"{col}_\", \"\") for idx, cat in enumerate(relevant_cols)}\n        onehot_mappings[col] = mappings\n        \n        df_reversed[col] = df_onehot[relevant_cols].idxmax(axis=1).str.replace(f\"{col}_\", \"\")\n        \n        df_onehot.drop(columns=relevant_cols, inplace=True)\n    \n    return onehot_mappings, df_onehot, df_reversed","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:26:53.122511Z","iopub.execute_input":"2024-12-22T12:26:53.122754Z","iopub.status.idle":"2024-12-22T12:26:53.141956Z","shell.execute_reply.started":"2024-12-22T12:26:53.122731Z","shell.execute_reply":"2024-12-22T12:26:53.140693Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_cols = train.select_dtypes(include=['int64', 'float64']).columns.tolist()\n\ncategorical_cols= train.select_dtypes(include=['object', 'category']).columns.tolist()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:26:53.143075Z","iopub.execute_input":"2024-12-22T12:26:53.143345Z","iopub.status.idle":"2024-12-22T12:26:53.193061Z","shell.execute_reply.started":"2024-12-22T12:26:53.143322Z","shell.execute_reply":"2024-12-22T12:26:53.191860Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = knn_impute(train, numerical_cols)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:26:53.194191Z","iopub.execute_input":"2024-12-22T12:26:53.194588Z","iopub.status.idle":"2024-12-22T12:28:16.546915Z","shell.execute_reply.started":"2024-12-22T12:26:53.194541Z","shell.execute_reply":"2024-12-22T12:28:16.545160Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:28:16.548519Z","iopub.execute_input":"2024-12-22T12:28:16.548810Z","iopub.status.idle":"2024-12-22T12:28:16.572086Z","shell.execute_reply.started":"2024-12-22T12:28:16.548783Z","shell.execute_reply":"2024-12-22T12:28:16.570894Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def mice_impute_label_encode_onehot(df, categorical_cols):\n    label_encoders = {}  # Store LabelEncoders for each column\n    original_values = {}  # Store original column mappings\n    df_encoded = df.copy()\n\n    for col in categorical_cols:\n        le = LabelEncoder()\n        df_encoded[col] = le.fit_transform(df_encoded[col].astype(str).fillna(\"missing\"))\n        label_encoders[col] = le\n        original_values[col] = dict(zip(le.transform(le.classes_), le.classes_))\n\n    imputer = IterativeImputer(max_iter=10, random_state=0)\n    df_encoded[categorical_cols] = imputer.fit_transform(df_encoded[categorical_cols])\n\n    df_onehot = pd.get_dummies(df_encoded, columns=categorical_cols, drop_first=False)\n\n    # Step 4: Create a mapping for one-hot encoded columns to original categories\n    onehot_mappings = {}\n    for col in categorical_cols:\n        onehot_mappings[col] = [\n            f\"{col}_{cat}\" for cat in label_encoders[col].classes_\n        ]\n\n    return label_encoders, original_values, onehot_mappings, df_encoded, df_onehot","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:28:16.575874Z","iopub.execute_input":"2024-12-22T12:28:16.576224Z","iopub.status.idle":"2024-12-22T12:28:16.588434Z","shell.execute_reply.started":"2024-12-22T12:28:16.576195Z","shell.execute_reply":"2024-12-22T12:28:16.587385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train.columns)\ncategorical_cols = [col for col in categorical_cols if col != 'id']\nprint(categorical_cols)\nprint(numerical_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:48:39.494923Z","iopub.execute_input":"2024-12-22T12:48:39.495341Z","iopub.status.idle":"2024-12-22T12:48:39.502614Z","shell.execute_reply.started":"2024-12-22T12:48:39.495311Z","shell.execute_reply":"2024-12-22T12:48:39.501510Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_encoders, original_values, onehot_mappings, df_encoded, df_onehot = mice_impute_label_encode_onehot(train, categorical_cols)\ntrain = df_onehot","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:33:15.455720Z","iopub.execute_input":"2024-12-22T12:33:15.456155Z","iopub.status.idle":"2024-12-22T12:33:15.705156Z","shell.execute_reply.started":"2024-12-22T12:33:15.456121Z","shell.execute_reply":"2024-12-22T12:33:15.700946Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(onehot_mappings)\nprint(label_encoders)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T13:39:18.342284Z","iopub.execute_input":"2024-12-22T13:39:18.343019Z","iopub.status.idle":"2024-12-22T13:39:18.363684Z","shell.execute_reply.started":"2024-12-22T13:39:18.342972Z","shell.execute_reply":"2024-12-22T13:39:18.362487Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_for_cluster = train.drop(columns=['id'])\n\nscaler = StandardScaler()\nscaled_features = scaler.fit_transform(train_for_cluster)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:33:58.234016Z","iopub.execute_input":"2024-12-22T12:33:58.234427Z","iopub.status.idle":"2024-12-22T12:33:58.550229Z","shell.execute_reply.started":"2024-12-22T12:33:58.234383Z","shell.execute_reply":"2024-12-22T12:33:58.549068Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\ninertia = []\nsilhouette = []\nk_values = range(2, 15)\nfor k in k_values:\n    kmeans = KMeans(n_clusters=k, random_state=42)\n    kmeans.fit(scaled_features)\n    inertia.append(kmeans.inertia_)\n    silhouette.append(silhouette_score(scaled_features, kmeans.labels_))\n\nplt.figure(figsize=(12, 5))\nplt.subplot(1, 2, 1)\nplt.plot(k_values, inertia, marker='o')\nplt.title('Elbow Method')\nplt.xlabel('Number of Clusters')\nplt.ylabel('Inertia')\n\nplt.subplot(1, 2, 2)\nplt.plot(k_values, silhouette, marker='o')\nplt.title('Silhouette Scores')\nplt.xlabel('Number of Clusters')\nplt.ylabel('Silhouette Score')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:34:02.037507Z","iopub.execute_input":"2024-12-22T12:34:02.037867Z","iopub.status.idle":"2024-12-22T12:34:46.122439Z","shell.execute_reply.started":"2024-12-22T12:34:02.037837Z","shell.execute_reply":"2024-12-22T12:34:46.121395Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data_after_clustering = train\n\noptimal_k = 8  \nkmeans = KMeans(n_clusters=optimal_k, random_state=42)\nclusters = kmeans.fit_predict(scaled_features)\n\ntrain_data_after_clustering['Cluster'] = clusters\n\n# print(train_data_after_clustering.groupby('Cluster').mean())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:36:49.012933Z","iopub.execute_input":"2024-12-22T12:36:49.013280Z","iopub.status.idle":"2024-12-22T12:36:51.912936Z","shell.execute_reply.started":"2024-12-22T12:36:49.013254Z","shell.execute_reply":"2024-12-22T12:36:51.911999Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_data_after_clustering)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:45:18.838504Z","iopub.execute_input":"2024-12-22T12:45:18.838955Z","iopub.status.idle":"2024-12-22T12:45:18.856450Z","shell.execute_reply.started":"2024-12-22T12:45:18.838922Z","shell.execute_reply":"2024-12-22T12:45:18.855394Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"custom_colors = ['#FF0000', '#0000FF', '#00FF00', '#FFA500', '#800080', '#00FFFF', '#FFC0CB', '#808080']\n\nplt.figure(figsize=(10, 6))\nsns.histplot(\n    data=train_data_after_clustering,\n    x='sii',\n    hue='Cluster',\n    kde=True,\n    bins=30,\n    palette=custom_colors  ,\n    alpha=0.5 \n)\nplt.title(\"Distribution of 'sii' Across Clusters\")\nplt.xlabel(\"sii\")\nplt.ylabel(\"Density\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T13:07:56.922561Z","iopub.execute_input":"2024-12-22T13:07:56.923059Z","iopub.status.idle":"2024-12-22T13:07:57.760975Z","shell.execute_reply.started":"2024-12-22T13:07:56.923019Z","shell.execute_reply":"2024-12-22T13:07:57.758899Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Key Observations:\nClusters Distribution:\n\nThe variable sii has values concentrated around specific points (e.g., near 0, 1.0, and 2.0).\nDifferent clusters (indicated by the hue with unique colors) overlap in these regions, suggesting that some clusters share similar ranges for sii.\n\nCluster Peaks:\n\nClusters 6 and 7 (dark purple and light purple) dominate the sii values at 0 and around 1.0, as seen by their higher density in those regions.\nOther clusters (e.g., clusters 1 and 2) contribute less significantly in the histogram bins.\n\nDensity Trends:\n\nWhile the histogram shows stacked bar heights for each cluster, the KDE (Kernel Density Estimate) lines provide a smoothed view of how sii varies.\nSome clusters have distinct peaks (indicating concentration of values), while others have flatter distributions.\n\nRare Values:\n\nValues of sii beyond 2.0 are rare and mostly dominated by a few clusters (e.g., Cluster 7).\n\nCluster Overlaps:\n\nThere is significant overlap between clusters for sii values between 0 and 1. This overlap may indicate that these clusters are not well-separated based on sii.\n\nPotential Insights:\nCluster Characteristics: Certain clusters (e.g., 6 and 7) dominate in specific ranges of sii. This could hint at distinct group behaviors or traits for these clusters.\nPotential Overlap: The overlap between clusters suggests that sii alone might not be sufficient for fully distinguishing between clusters.","metadata":{}},{"cell_type":"code","source":"cluster_colors = {cluster: color for cluster, color in zip(train_data_after_clustering['Cluster'].unique(), custom_colors)}\n\ng = sns.FacetGrid(\n    train_data_after_clustering,\n    col='Cluster',\n    col_wrap=4,\n    height=3,\n    sharex=True,\n    sharey=True,\n)\ng.map(sns.histplot, 'sii', kde=True, bins=30, color=None)\n\n# Set custom colors\nfor ax, cluster in zip(g.axes.flat, train_data_after_clustering['Cluster'].unique()):\n    ax.patches[0].set_facecolor(cluster_colors[cluster])\n    ax.patches[0].set_edgecolor(\"black\")  # Optional for distinction\n\nplt.subplots_adjust(top=0.9)\ng.fig.suptitle(\"Distribution of 'sii' for Each Cluster\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T13:08:09.494000Z","iopub.execute_input":"2024-12-22T13:08:09.494398Z","iopub.status.idle":"2024-12-22T13:08:12.374883Z","shell.execute_reply.started":"2024-12-22T13:08:09.494340Z","shell.execute_reply":"2024-12-22T13:08:12.373834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"columns_to_examine = [\n    'Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-CGAS_Score', 'Physical-BMI', 'Physical-Height',\n    'Physical-Weight', 'Physical-Waist_Circumference', 'Physical-Diastolic_BP', 'Physical-HeartRate',\n    'Physical-Systolic_BP', 'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins',\n    'Fitness_Endurance-Time_Sec', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', \n    'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU', 'FGC-FGC_PU_Zone',\n    'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL',\n    'FGC-FGC_TL_Zone', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR',\n    'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat',\n    'BIA-BIA_Frame_num', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', \n    'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total', 'PAQ_C-PAQ_C_Total', 'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02',\n    'PCIAT-PCIAT_03', 'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06', 'PCIAT-PCIAT_07',\n    'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12',\n    'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15', 'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17',\n    'PCIAT-PCIAT_18', 'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20', 'PCIAT-PCIAT_Total', 'SDS-SDS_Total_Raw',\n    'SDS-SDS_Total_T', 'PreInt_EduHx-computerinternet_hoursday'\n]\n\ncluster_summary = train_data_after_clustering.groupby('Cluster')[columns_to_examine].describe()\n\nprint(cluster_summary)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T13:44:26.480509Z","iopub.execute_input":"2024-12-22T13:44:26.480952Z","iopub.status.idle":"2024-12-22T13:44:27.356722Z","shell.execute_reply.started":"2024-12-22T13:44:26.480921Z","shell.execute_reply":"2024-12-22T13:44:27.355580Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for column in columns_to_examine: \n    plt.figure(figsize=(16, 8))\n    sns.boxplot(data=train_data_after_clustering, x='Cluster', y=column, palette='Set2')\n    plt.title(f'Variation of {column} Across Clusters')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T13:46:46.332020Z","iopub.execute_input":"2024-12-22T13:46:46.332528Z","iopub.status.idle":"2024-12-22T13:47:07.267966Z","shell.execute_reply.started":"2024-12-22T13:46:46.332496Z","shell.execute_reply":"2024-12-22T13:47:07.266839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from math import pi\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\n# Compute mean values of selected features for each cluster\nselected_features = [\n    'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW',\n    'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n    'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW',\n]\ncluster_means = train_data_after_clustering.groupby('Cluster')[selected_features].mean()\n\nnormalized_data = (cluster_means - cluster_means.min()) / (cluster_means.max() - cluster_means.min())\n\n# Radar Chart\ncategories = normalized_data.columns.tolist()\nnum_vars = len(categories)\n\n# Create radar plots for each cluster\nfor cluster in normalized_data.index:\n    values = normalized_data.loc[cluster].tolist()\n    values += values[:1]  # repeat first value to close the radar chart\n\n    angles = [n / float(num_vars) * 2 * pi for n in range(num_vars)]\n    angles += angles[:1]\n\n    plt.figure(figsize=(8, 8))\n    ax = plt.subplot(111, polar=True)\n    plt.xticks(angles[:-1], categories, color='grey', size=8)\n    ax.plot(angles, values, linewidth=2, linestyle='solid', label=f'Cluster {cluster}')\n    ax.fill(angles, values, alpha=0.4)\n    plt.title(f'Radar Chart for Cluster {cluster}', size=15, color='darkblue', y=1.1)\n    plt.legend(loc='upper right', bbox_to_anchor=(0.1, 0.1))\n    plt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}