{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.impute import KNNImputer, SimpleImputer\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nfrom concurrent.futures import ThreadPoolExecutor\nfrom scipy.optimize import minimize\nimport os\nimport pyarrow.parquet as pq\nfrom datetime import timedelta\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.cluster import KMeans\nfrom sklearn.metrics import silhouette_score\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.experimental import enable_iterative_imputer \nfrom sklearn.impute import IterativeImputer\n\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-01T18:50:02.316857Z","iopub.execute_input":"2025-01-01T18:50:02.317277Z","iopub.status.idle":"2025-01-01T18:50:05.080677Z","shell.execute_reply.started":"2025-01-01T18:50:02.317230Z","shell.execute_reply":"2025-01-01T18:50:05.079238Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install kmodes\nfrom kmodes.kprototypes import KPrototypes\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T18:50:05.082225Z","iopub.execute_input":"2025-01-01T18:50:05.082718Z","iopub.status.idle":"2025-01-01T18:50:17.390534Z","shell.execute_reply.started":"2025-01-01T18:50:05.082682Z","shell.execute_reply":"2025-01-01T18:50:17.389216Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T18:50:17.393479Z","iopub.execute_input":"2025-01-01T18:50:17.393866Z","iopub.status.idle":"2025-01-01T18:50:17.483088Z","shell.execute_reply.started":"2025-01-01T18:50:17.393832Z","shell.execute_reply":"2025-01-01T18:50:17.482009Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df=pd.DataFrame(train)\n\nnumerical_columns = train_df.select_dtypes(include=['int64', 'float64']).columns\ncategorical_columns = train_df.select_dtypes(include=['object']).columns\n\nknn_imputer = KNNImputer(n_neighbors=2)\ntrain_df[numerical_columns] = knn_imputer.fit_transform(train_df[numerical_columns])\n\nmost_frequent_imputer = SimpleImputer(strategy='most_frequent')\ntrain_df[categorical_columns] = most_frequent_imputer.fit_transform(train_df[categorical_columns])\n\ntrain = train_df.to_numpy()\n\ncategorical_indices = [train_df.columns.get_loc(col) for col in categorical_columns]\n\nkproto = KPrototypes(n_clusters=4, init='Huang', verbose=2)\n\nkproto.fit(train, categorical=categorical_indices)\n\nlabels = kproto.labels_\n\ntrain_df['Cluster'] = labels\n\nprint(train_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T18:50:17.487965Z","iopub.execute_input":"2025-01-01T18:50:17.488312Z","iopub.status.idle":"2025-01-01T18:52:09.155881Z","shell.execute_reply.started":"2025-01-01T18:50:17.488279Z","shell.execute_reply":"2025-01-01T18:52:09.154531Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"custom_colors = ['#FF0000', '#0000FF', '#00FF00', '#FFA500']\n\nplt.figure(figsize=(10, 6))\nsns.histplot(\n    data=train_df,\n    x='sii',\n    hue='Cluster',\n    kde=True,\n    bins=30,\n    palette=custom_colors  ,\n    alpha=0.5 \n)\nplt.title(\"Distribution of 'sii' Across Clusters\")\nplt.xlabel(\"sii\")\nplt.ylabel(\"Density\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T18:52:09.157325Z","iopub.execute_input":"2025-01-01T18:52:09.157701Z","iopub.status.idle":"2025-01-01T18:52:09.774231Z","shell.execute_reply.started":"2025-01-01T18:52:09.157668Z","shell.execute_reply":"2025-01-01T18:52:09.772911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cluster_colors = {cluster: color for cluster, color in zip(train_df['Cluster'].unique(), custom_colors)}\n\ng = sns.FacetGrid(\n    train_df,\n    col='Cluster',\n    col_wrap=4,\n    height=3,\n    sharex=True,\n    sharey=True,\n)\ng.map(sns.histplot, 'sii', kde=True, bins=30, color=None)\n\n# Set custom colors\nfor ax, cluster in zip(g.axes.flat, train_df['Cluster'].unique()):\n    ax.patches[0].set_facecolor(cluster_colors[cluster])\n    ax.patches[0].set_edgecolor(\"black\")  # Optional for distinction\n\nplt.subplots_adjust(top=0.9)\ng.fig.suptitle(\"Distribution of 'sii' for Each Cluster\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T18:53:58.107642Z","iopub.execute_input":"2025-01-01T18:53:58.108243Z","iopub.status.idle":"2025-01-01T18:53:59.540751Z","shell.execute_reply.started":"2025-01-01T18:53:58.108200Z","shell.execute_reply":"2025-01-01T18:53:59.539254Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"columns_to_examine = [\n    'Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-CGAS_Score', 'Physical-BMI', 'Physical-Height',\n    'Physical-Weight', 'Physical-Waist_Circumference', 'Physical-Diastolic_BP', 'Physical-HeartRate',\n    'Physical-Systolic_BP', 'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins',\n    'Fitness_Endurance-Time_Sec', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', \n    'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU', 'FGC-FGC_PU_Zone',\n    'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL',\n    'FGC-FGC_TL_Zone', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR',\n    'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat',\n    'BIA-BIA_Frame_num', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', \n    'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total', 'PAQ_C-PAQ_C_Total', 'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02',\n    'PCIAT-PCIAT_03', 'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06', 'PCIAT-PCIAT_07',\n    'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12',\n    'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15', 'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17',\n    'PCIAT-PCIAT_18', 'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20', 'PCIAT-PCIAT_Total', 'SDS-SDS_Total_Raw',\n    'SDS-SDS_Total_T', 'PreInt_EduHx-computerinternet_hoursday'\n]\n\ncluster_summary = train_df.groupby('Cluster')[columns_to_examine].describe()\n\nprint(cluster_summary)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T18:54:07.983898Z","iopub.execute_input":"2025-01-01T18:54:07.984331Z","iopub.status.idle":"2025-01-01T18:54:08.458494Z","shell.execute_reply.started":"2025-01-01T18:54:07.984297Z","shell.execute_reply":"2025-01-01T18:54:08.457121Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for column in columns_to_examine: \n    plt.figure(figsize=(16, 8))\n    sns.boxplot(data=train_df, x='Cluster', y=column, palette='Set2')\n    plt.title(f'Variation of {column} Across Clusters')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T18:54:15.895137Z","iopub.execute_input":"2025-01-01T18:54:15.895541Z","iopub.status.idle":"2025-01-01T18:54:36.764976Z","shell.execute_reply.started":"2025-01-01T18:54:15.895499Z","shell.execute_reply":"2025-01-01T18:54:36.763718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from math import pi\n\nselected_features = [\n    'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW',\n    'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n    'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW',\n]\ncluster_means = train_df.groupby('Cluster')[selected_features].mean()\n\nnormalized_data = (cluster_means - cluster_means.min()) / (cluster_means.max() - cluster_means.min())\n\ncategories = normalized_data.columns.tolist()\nnum_vars = len(categories)\n\nfor cluster in normalized_data.index:\n    values = normalized_data.loc[cluster].tolist()\n    values += values[:1]  # repeat first value to close the radar chart\n\n    angles = [n / float(num_vars) * 2 * pi for n in range(num_vars)]\n    angles += angles[:1]\n\n    plt.figure(figsize=(8, 8))\n    ax = plt.subplot(111, polar=True)\n    plt.xticks(angles[:-1], categories, color='grey', size=8)\n    ax.plot(angles, values, linewidth=2, linestyle='solid', label=f'Cluster {cluster}')\n    ax.fill(angles, values, alpha=0.4)\n    plt.title(f'Radar Chart for Cluster {cluster}', size=15, color='darkblue', y=1.1)\n    plt.legend(loc='upper right', bbox_to_anchor=(0.1, 0.1))\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T18:54:38.364606Z","iopub.execute_input":"2025-01-01T18:54:38.365066Z","iopub.status.idle":"2025-01-01T18:54:39.962780Z","shell.execute_reply.started":"2025-01-01T18:54:38.364990Z","shell.execute_reply":"2025-01-01T18:54:39.961185Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pandas.plotting import parallel_coordinates\n\nnormalized_data['Cluster'] = normalized_data.index\nplt.figure(figsize=(12, 6))\nparallel_coordinates(normalized_data, 'Cluster', color=['r', 'g', 'b', 'c', 'm', 'y', 'k'])\nplt.title('Parallel Coordinates Plot')\nplt.xlabel('Features')\nplt.ylabel('Normalized Values')\nplt.xticks(rotation=90)\nplt.grid()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T18:54:47.503525Z","iopub.execute_input":"2025-01-01T18:54:47.503988Z","iopub.status.idle":"2025-01-01T18:54:48.132684Z","shell.execute_reply.started":"2025-01-01T18:54:47.503953Z","shell.execute_reply":"2025-01-01T18:54:48.131407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cluster_means = train_df.groupby('Cluster')[selected_features].mean()\ncluster_means.T.plot(kind='bar', figsize=(14, 8), width=0.8)\nplt.title('Cluster Feature Comparison (Raw Values)')\nplt.ylabel('Raw Value')\nplt.xlabel('Features')\nplt.xticks(rotation=45)\nplt.legend(title='Cluster', bbox_to_anchor=(1.05, 1), loc='upper left')\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T18:54:52.199496Z","iopub.execute_input":"2025-01-01T18:54:52.199980Z","iopub.status.idle":"2025-01-01T18:54:52.759206Z","shell.execute_reply.started":"2025-01-01T18:54:52.199939Z","shell.execute_reply":"2025-01-01T18:54:52.757604Z"}},"outputs":[],"execution_count":null}]}