{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport sklearn\nimport seaborn as sns\nfrom sklearn import datasets\nfrom sklearn.decomposition import PCA\nfrom sklearn import preprocessing\n\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:29.621298Z","iopub.execute_input":"2024-11-28T02:12:29.623154Z","iopub.status.idle":"2024-11-28T02:12:29.630566Z","shell.execute_reply.started":"2024-11-28T02:12:29.623105Z","shell.execute_reply":"2024-11-28T02:12:29.629245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\n\n\ntrain.isna().sum().to_frame().sort_values(0, ascending=False) / len(train) * 100","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:29.632645Z","iopub.execute_input":"2024-11-28T02:12:29.633060Z","iopub.status.idle":"2024-11-28T02:12:29.715176Z","shell.execute_reply.started":"2024-11-28T02:12:29.633013Z","shell.execute_reply":"2024-11-28T02:12:29.714070Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"temp_things = [train.columns, \n               train[\"sii\"].isna().sum() / len(train)    \n    ]\n\nfor thing in temp_things:\n    print(thing)\n    print()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:29.716321Z","iopub.execute_input":"2024-11-28T02:12:29.716619Z","iopub.status.idle":"2024-11-28T02:12:29.723751Z","shell.execute_reply.started":"2024-11-28T02:12:29.716591Z","shell.execute_reply":"2024-11-28T02:12:29.722767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pciat_cols = [col for col in train.columns if col.startswith(\"PCIAT-PCIAT\")]\npciat_cols.append(\"sii\")\n\npciat_train = train[pciat_cols]\n\n# the higher PCIAT scores, the higher sii = more severe, but what are the thresholds between sii scores?\n# plot\nfig, ax = plt.subplots()\npciat_train_notna = pciat_train[[\"PCIAT-PCIAT_Total\", \"sii\"]][pciat_train[[\"PCIAT-PCIAT_Total\", \"sii\"]].notna().all(axis=1)]\nax.scatter(pciat_train_notna[\"PCIAT-PCIAT_Total\"], pciat_train_notna[\"sii\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:29.725832Z","iopub.execute_input":"2024-11-28T02:12:29.726206Z","iopub.status.idle":"2024-11-28T02:12:30.017459Z","shell.execute_reply.started":"2024-11-28T02:12:29.726173Z","shell.execute_reply":"2024-11-28T02:12:30.016255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# okay the graph looks nice, what are the thresholds?\npciat_train_notna.groupby(\"sii\").max()\n\n# for now:\n# sii = 0 <=> PCIAT in [0, 30]\n# sii = 1 <=> PCIAT in [31, 49]\n# sii = 2 <=> PCIAT in [50, 79]\n# sii = 3 <=> PCIAT in [80, 100]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:30.018702Z","iopub.execute_input":"2024-11-28T02:12:30.019079Z","iopub.status.idle":"2024-11-28T02:12:30.034755Z","shell.execute_reply.started":"2024-11-28T02:12:30.019046Z","shell.execute_reply":"2024-11-28T02:12:30.033799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# but what about the NaN values in the responses to 20 questions\n\n# make sure the PCIAT score added up correctly\nrecalculated_pciat = pciat_train.drop(columns=[\"PCIAT-PCIAT_Total\", \"sii\"]).sum(axis=1)\ndiff = pciat_train[\"PCIAT-PCIAT_Total\"] - recalculated_pciat\ndiff.sum()\n# all correct, but still need to investigate the NaN issues","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:30.036156Z","iopub.execute_input":"2024-11-28T02:12:30.036535Z","iopub.status.idle":"2024-11-28T02:12:30.048337Z","shell.execute_reply.started":"2024-11-28T02:12:30.036500Z","shell.execute_reply":"2024-11-28T02:12:30.047325Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# see how many rows of all NaN values\nlen(pciat_train[pciat_train.isna().all(axis=1)]) / len(pciat_train)\n\n# 1224 rows, 30% of data, that's a lot! likely needs an unsupervised learning # TODO","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:30.049584Z","iopub.execute_input":"2024-11-28T02:12:30.049891Z","iopub.status.idle":"2024-11-28T02:12:30.059888Z","shell.execute_reply.started":"2024-11-28T02:12:30.049853Z","shell.execute_reply":"2024-11-28T02:12:30.058744Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# but what about the other rows with partially NaN values\n# because here missing responses do not necessarily mean that 0 score\n# rmb: each question has a max score of 5\n\npciat_train[~pciat_train.isna().all(axis=1)].isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:30.061141Z","iopub.execute_input":"2024-11-28T02:12:30.061463Z","iopub.status.idle":"2024-11-28T02:12:30.073705Z","shell.execute_reply.started":"2024-11-28T02:12:30.061432Z","shell.execute_reply":"2024-11-28T02:12:30.072708Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.set_option('display.max_rows', pciat_train.shape[0])     \npd.set_option('display.max_columns', pciat_train.shape[1]) \n\ntemp = pciat_train[~pciat_train.isna().all(axis=1)][pciat_train.isna().any(axis=1)]\n\n# the second row is weird: all responses are NaN but the total score is still 0 => very biased\n# how to deal with this?\ntemp","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:30.077940Z","iopub.execute_input":"2024-11-28T02:12:30.078375Z","iopub.status.idle":"2024-11-28T02:12:30.186173Z","shell.execute_reply.started":"2024-11-28T02:12:30.078341Z","shell.execute_reply":"2024-11-28T02:12:30.185108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.reset_option(\"display.max_rows\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:30.187523Z","iopub.execute_input":"2024-11-28T02:12:30.187863Z","iopub.status.idle":"2024-11-28T02:12:30.192817Z","shell.execute_reply.started":"2024-11-28T02:12:30.187830Z","shell.execute_reply":"2024-11-28T02:12:30.191751Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# consider two extremes:\n# for a given row with some partial NaN values, if we replace the NaN with the scores of 5 and add them all\n# if the total score now is still within the predefined thresholds, it means the sii still holds\n# if it's larger, then we now just don't know and can't assume anything, it must be NaN for sii score\n# we will deal with NaN sii scores later \n\n# I learned this from https://www.kaggle.com/code/antoninadolgorukova/cmi-piu-features-eda?scriptVersionId=206130660&cellId=32\n\ndef recalculate_sii(row):\n    responses20_cols = [col for col in train.columns if col.startswith(\"PCIAT-PCIAT\") and col != \"PCIAT-PCIAT_Total\"]\n    \n    if pd.isna(row['PCIAT-PCIAT_Total']):\n        return np.nan\n    max_possible = row['PCIAT-PCIAT_Total'] + row[responses20_cols].isna().sum() * 5\n    if row['PCIAT-PCIAT_Total'] <= 30 and max_possible <= 30:\n        return 0\n    elif 31 <= row['PCIAT-PCIAT_Total'] <= 49 and max_possible <= 49:\n        return 1\n    elif 50 <= row['PCIAT-PCIAT_Total'] <= 79 and max_possible <= 79:\n        return 2\n    elif row['PCIAT-PCIAT_Total'] >= 80 and max_possible >= 80:\n        return 3\n    return np.nan\n\ntemp['recalc_sii'] = temp.apply(recalculate_sii, axis=1)\n\nlen(temp), temp['recalc_sii'].isna().sum()\n\n# only 17 rows out of 65 that are now having missing sii scores (not including initially NaN sii rows)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:30.194419Z","iopub.execute_input":"2024-11-28T02:12:30.195336Z","iopub.status.idle":"2024-11-28T02:12:30.254075Z","shell.execute_reply.started":"2024-11-28T02:12:30.195285Z","shell.execute_reply":"2024-11-28T02:12:30.252965Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.reset_option(\"display.max_rows\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:30.256134Z","iopub.execute_input":"2024-11-28T02:12:30.256590Z","iopub.status.idle":"2024-11-28T02:12:30.265663Z","shell.execute_reply.started":"2024-11-28T02:12:30.256540Z","shell.execute_reply":"2024-11-28T02:12:30.264681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train[\"sii\"].isna().sum() / len(train) # after preprocessing ssi, there is a bit more NaN","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:30.267072Z","iopub.execute_input":"2024-11-28T02:12:30.267855Z","iopub.status.idle":"2024-11-28T02:12:30.279541Z","shell.execute_reply.started":"2024-11-28T02:12:30.267806Z","shell.execute_reply":"2024-11-28T02:12:30.278344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:30.281091Z","iopub.execute_input":"2024-11-28T02:12:30.281526Z","iopub.status.idle":"2024-11-28T02:12:30.315956Z","shell.execute_reply.started":"2024-11-28T02:12:30.281480Z","shell.execute_reply":"2024-11-28T02:12:30.314974Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# labels distribution\nfig, ax = plt.subplots()\ncounts = train[\"sii\"].value_counts(dropna=False).sort_index()\nax.bar(counts.index.astype(str), counts.values)\ncounts\n\n# mostly no severity, but still lots of missing sii","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:30.317330Z","iopub.execute_input":"2024-11-28T02:12:30.317661Z","iopub.status.idle":"2024-11-28T02:12:30.488717Z","shell.execute_reply.started":"2024-11-28T02:12:30.317626Z","shell.execute_reply":"2024-11-28T02:12:30.487746Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def transform_tabular(data):\n    # drop the PCIAT-PCIAT columns to avoid potential data leakage\n    # for train mostly\n    responses20_cols = [col for col in train.columns if col.startswith(\"PCIAT\")]\n    data = data.drop(columns=responses20_cols)\n\n    return data\n\ntrain = transform_tabular(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:30.489901Z","iopub.execute_input":"2024-11-28T02:12:30.490340Z","iopub.status.idle":"2024-11-28T02:12:30.498534Z","shell.execute_reply.started":"2024-11-28T02:12:30.490292Z","shell.execute_reply":"2024-11-28T02:12:30.497467Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"###\n# check column types\ncats = [col for col in train.columns if train[col].dtype == \"object\"]\ncons = [col for col in train.columns if train[col].dtype != \"object\"]\ncats, cons","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:30.500075Z","iopub.execute_input":"2024-11-28T02:12:30.500411Z","iopub.status.idle":"2024-11-28T02:12:30.515430Z","shell.execute_reply.started":"2024-11-28T02:12:30.500379Z","shell.execute_reply":"2024-11-28T02:12:30.514432Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# imputation\n\n# first, let's focus on continuous vars\n\n# these columns seem to have no missing data, we can use them as group matching for imputing continuous vars? \n# hmm let's try\ntrain[[\"Basic_Demos-Enroll_Season\", \"Basic_Demos-Age\", \"Basic_Demos-Sex\"]].isna().sum()\n# but I don't use \"Basic_Demos-Enroll_Season\" because I expect every season would observe some \n# similar people within a certain group of age and sex, not including it would help me have a \n# more broader groups  by age and sex","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:30.516681Z","iopub.execute_input":"2024-11-28T02:12:30.517044Z","iopub.status.idle":"2024-11-28T02:12:30.530250Z","shell.execute_reply.started":"2024-11-28T02:12:30.517012Z","shell.execute_reply":"2024-11-28T02:12:30.529241Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# try on Physical-Height\ncols_no_missing = [\"Basic_Demos-Age\", \"Basic_Demos-Sex\"]\ngroup_means_draft = train.groupby(cols_no_missing)[\"Physical-Height\"].agg(\"mean\")\ngroup_means_draft\n\n# it looks like group of 22-yo and male only has one person and that person's data is NaN so I replace\n# it with data from male person but 21-yo (mean)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:30.531527Z","iopub.execute_input":"2024-11-28T02:12:30.531857Z","iopub.status.idle":"2024-11-28T02:12:30.547138Z","shell.execute_reply.started":"2024-11-28T02:12:30.531825Z","shell.execute_reply":"2024-11-28T02:12:30.546133Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calculate_group_means(data, cols_no_missing, cons):\n    \"\"\"\n    Calculate group means for continuous variables based on given grouping columns.\n    \"\"\"\n    group_means_dict = {}\n    too_many_missing = []\n    \n    for col in cons:\n        if col not in cols_no_missing:\n            group_means = data.groupby(cols_no_missing)[col].mean()\n            group_means_dict[col] = group_means\n\n            data[col] = data.groupby(cols_no_missing)[col].transform(\"mean\")\n            \n            if data[col].isna().sum() > 10:\n                too_many_missing.append(col)\n    \n    return group_means_dict, too_many_missing\n\n\ndef impute_remaining_values(data, group_means_dict, selected_cons):\n    \"\"\"\n    Impute remaining missing values in the group means dictionary.\n    \"\"\"\n    for col, group_means in group_means_dict.items():\n        if col in selected_cons:\n            overall_mean = data[col].mean()\n            group_means.fillna(overall_mean, inplace=True)\n\n    return data\n\n\n# calculate group means and identify columns with too many missing values\ncols_no_missing = [\"Basic_Demos-Age\", \"Basic_Demos-Sex\"]\ntransformed_cons_train = train.copy()\ngroup_means_dict, too_many_missing = calculate_group_means(transformed_cons_train, cols_no_missing, cons)\n\n# filter selected continuous variables\nselected_cons = [col for col in cons if col not in too_many_missing]\n\n# impute missing values in the group means dictionary\nimpute_remaining_values(transformed_cons_train, group_means_dict, selected_cons)\n\n\nfor col, group_means in group_means_dict.items():\n    transformed_cons_train[col] = transformed_cons_train.apply(\n        lambda row: group_means.get((row[\"Basic_Demos-Age\"], row[\"Basic_Demos-Sex\"]), row[col])\n        if pd.isna(row[col]) else row[col],\n        axis=1\n    )\n\ntransformed_cons_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:30.548410Z","iopub.execute_input":"2024-11-28T02:12:30.548736Z","iopub.status.idle":"2024-11-28T02:12:32.929527Z","shell.execute_reply.started":"2024-11-28T02:12:30.548705Z","shell.execute_reply":"2024-11-28T02:12:32.928442Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transformed_cons_train[transformed_cons_train[selected_cons].isna().any(axis=1)]\n# this one record is all NaN => but can't drop, because in the hidden test set might have it","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:32.931071Z","iopub.execute_input":"2024-11-28T02:12:32.931813Z","iopub.status.idle":"2024-11-28T02:12:32.951294Z","shell.execute_reply.started":"2024-11-28T02:12:32.931761Z","shell.execute_reply":"2024-11-28T02:12:32.950282Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# transformed_cons_train = transformed_cons_train[transformed_cons_train[\"id\"] != \"3cb2c4da\"]\n\ntransformed_cons_train[\"sii\"] = transformed_cons_train[\"sii\"].apply(lambda x: round(x))\n\ntransformed_cons_train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:32.952679Z","iopub.execute_input":"2024-11-28T02:12:32.953486Z","iopub.status.idle":"2024-11-28T02:12:32.963235Z","shell.execute_reply.started":"2024-11-28T02:12:32.953434Z","shell.execute_reply":"2024-11-28T02:12:32.962149Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"only_transformed_cons = [col for col in selected_cons if transformed_cons_train[col].dtype != \"object\"]\nonly_transformed_cons_train = transformed_cons_train[only_transformed_cons]\n\nonly_transformed_cons_train.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:32.965191Z","iopub.execute_input":"2024-11-28T02:12:32.965703Z","iopub.status.idle":"2024-11-28T02:12:32.987205Z","shell.execute_reply.started":"2024-11-28T02:12:32.965653Z","shell.execute_reply":"2024-11-28T02:12:32.986138Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = only_transformed_cons_train.drop(columns=[\"sii\"])\ny = only_transformed_cons_train[\"sii\"]\nX.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:32.991600Z","iopub.execute_input":"2024-11-28T02:12:32.992329Z","iopub.status.idle":"2024-11-28T02:12:33.000958Z","shell.execute_reply.started":"2024-11-28T02:12:32.992289Z","shell.execute_reply":"2024-11-28T02:12:33.000059Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# (1) attemp PCA on the continous variables (unsupervised, no need sii)\n\n# standardize the cons features\n# X_mean = np.mean(X, axis=0)\n# X_std = np.std(X, axis=0)\n# X_standardized = (X - X_mean) / X_std\n\n# cov_matrix = np.cov(X_standardized, rowvar=False)\n# eigenvalues, eigenvectors = np.linalg.eig(cov_matrix)\n\n# # sort eigenvalues and corresponding eigenvectors\n# sorted_indices = np.argsort(eigenvalues)[::-1]\n# eigenvalues = eigenvalues[sorted_indices]\n# eigenvectors = eigenvectors[:, sorted_indices]\n\n# k = 5\n# principal_components = eigenvectors[:, :k]\n# X_pca = np.dot(X_standardized, principal_components)\n\n# normalize data\nX_scaled = pd.DataFrame(preprocessing.scale(X),columns = X.columns) \n\n# PCA\npca = PCA(n_components=3)\npca.fit_transform(X_scaled)\n\nPCS = ['PC1','PC2', 'PC3'] #, 'PC4', 'PC5']\n\npc_data = pd.DataFrame(pca.components_,columns=X_scaled.columns,\n             index = PCS)\npc_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:33.002326Z","iopub.execute_input":"2024-11-28T02:12:33.002756Z","iopub.status.idle":"2024-11-28T02:12:33.153402Z","shell.execute_reply.started":"2024-11-28T02:12:33.002711Z","shell.execute_reply":"2024-11-28T02:12:33.151163Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pc_data = pc_data.T.reset_index().rename(columns={\"index\": \"Field\"})\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:33.154637Z","iopub.execute_input":"2024-11-28T02:12:33.155042Z","iopub.status.idle":"2024-11-28T02:12:33.169742Z","shell.execute_reply.started":"2024-11-28T02:12:33.155000Z","shell.execute_reply":"2024-11-28T02:12:33.167507Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"info = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv\")[[\"Field\", \"Description\"]]\n\npc_data_info = pd.merge(pc_data, info, on=\"Field\")\npc_data_info[PCS] = pc_data_info[PCS].apply(lambda x: abs(x))\n\npc_data_info","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:33.170956Z","iopub.execute_input":"2024-11-28T02:12:33.171371Z","iopub.status.idle":"2024-11-28T02:12:33.227241Z","shell.execute_reply.started":"2024-11-28T02:12:33.171330Z","shell.execute_reply":"2024-11-28T02:12:33.226005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# top 10 each PC\npc_data_info[[\"Field\", \"PC1\"]].sort_values(\"PC1\", ascending=False)[:10]\n\n# this PC1 gives very close coefficient values, why? likely because these BIA variables are highly correlated => not good\n# we just pick BIA-BIA_LST \n# need to re-check this\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:33.229174Z","iopub.execute_input":"2024-11-28T02:12:33.229626Z","iopub.status.idle":"2024-11-28T02:12:33.243254Z","shell.execute_reply.started":"2024-11-28T02:12:33.229576Z","shell.execute_reply":"2024-11-28T02:12:33.242198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pc_data_info[[\"Field\", \"PC2\"]].sort_values(\"PC2\", ascending=False)[:10]\n\n# this PC2 gives somewhat more \"diverse\" coefficients","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:33.244597Z","iopub.execute_input":"2024-11-28T02:12:33.245096Z","iopub.status.idle":"2024-11-28T02:12:33.261001Z","shell.execute_reply.started":"2024-11-28T02:12:33.245049Z","shell.execute_reply":"2024-11-28T02:12:33.259929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pc_data_info[[\"Field\", \"PC3\"]].sort_values(\"PC3\", ascending=False)[:10]\n# SDS-SDS_Total_T and SDS-SDS_Total_Raw seem to influence the most!","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:33.262343Z","iopub.execute_input":"2024-11-28T02:12:33.262683Z","iopub.status.idle":"2024-11-28T02:12:33.276881Z","shell.execute_reply.started":"2024-11-28T02:12:33.262639Z","shell.execute_reply":"2024-11-28T02:12:33.275788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"picked_cons = [\"BIA-BIA_LST\", \n               \"Physical-Weight\", \"BIA-BIA_BMI\", \"Physical-Waist_Circumference\", \n               \"SDS-SDS_Total_T\", \"SDS-SDS_Total_Raw\"\n              ]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:33.278285Z","iopub.execute_input":"2024-11-28T02:12:33.279162Z","iopub.status.idle":"2024-11-28T02:12:33.285773Z","shell.execute_reply.started":"2024-11-28T02:12:33.279097Z","shell.execute_reply":"2024-11-28T02:12:33.284558Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"filtered_group_means_dict = {key: value for key, value in group_means_dict.items() if key in picked_cons}\nfiltered_group_means_dict","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:33.287343Z","iopub.execute_input":"2024-11-28T02:12:33.287659Z","iopub.status.idle":"2024-11-28T02:12:33.311218Z","shell.execute_reply.started":"2024-11-28T02:12:33.287627Z","shell.execute_reply":"2024-11-28T02:12:33.309971Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = only_transformed_cons_train[picked_cons]\ny = y","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:33.312540Z","iopub.execute_input":"2024-11-28T02:12:33.312885Z","iopub.status.idle":"2024-11-28T02:12:33.319751Z","shell.execute_reply.started":"2024-11-28T02:12:33.312849Z","shell.execute_reply":"2024-11-28T02:12:33.318873Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#  modeling/training\nX","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:33.321091Z","iopub.execute_input":"2024-11-28T02:12:33.321529Z","iopub.status.idle":"2024-11-28T02:12:33.339930Z","shell.execute_reply.started":"2024-11-28T02:12:33.321483Z","shell.execute_reply":"2024-11-28T02:12:33.338766Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ordinal classification using decision tree\n# https://www.tandfonline.com/doi/full/10.1080/24725854.2022.2081745#abstract\n# https://scikit-learn.org/stable/modules/generated/sklearn.tree.DecisionTreeClassifier.html\n\nX_train, X_valid, y_train, y_valid = train_test_split(X, y, test_size=0.2, random_state=123)\n\ndt_model = DecisionTreeClassifier(max_depth=3, random_state=123) \ndt_model.fit(X_train, y_train)\n\ny_pred = dt_model.predict(X_valid)\nprint(classification_report(y_valid, y_pred))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:33.341560Z","iopub.execute_input":"2024-11-28T02:12:33.341993Z","iopub.status.idle":"2024-11-28T02:12:33.372492Z","shell.execute_reply.started":"2024-11-28T02:12:33.341943Z","shell.execute_reply":"2024-11-28T02:12:33.371417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")\n# there are some rows that have almost no data at all!\n\n\nfor col in picked_cons:\n    if col != \"sii\" and col != \"Basic_Demos-Age\" and col != \"Basic_Demos-Sex\":\n        test[col] = test.apply(\n            lambda row: filtered_group_means_dict[col].get((row[\"Basic_Demos-Age\"], row[\"Basic_Demos-Sex\"]), np.nan)\n            if pd.isna(row[col]) else row[col],\n            axis=1\n        )\n\nX_test = test[picked_cons]\nX_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:33.373978Z","iopub.execute_input":"2024-11-28T02:12:33.374332Z","iopub.status.idle":"2024-11-28T02:12:33.403127Z","shell.execute_reply.started":"2024-11-28T02:12:33.374297Z","shell.execute_reply":"2024-11-28T02:12:33.402006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dt_model.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:33.404347Z","iopub.execute_input":"2024-11-28T02:12:33.404701Z","iopub.status.idle":"2024-11-28T02:12:33.412927Z","shell.execute_reply.started":"2024-11-28T02:12:33.404667Z","shell.execute_reply":"2024-11-28T02:12:33.411887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.DataFrame({\n    \"id\": test[\"id\"],\n    \"sii\": dt_model.predict(X_test)\n})\nsubmission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:33.414193Z","iopub.execute_input":"2024-11-28T02:12:33.414524Z","iopub.status.idle":"2024-11-28T02:12:33.430096Z","shell.execute_reply.started":"2024-11-28T02:12:33.414480Z","shell.execute_reply":"2024-11-28T02:12:33.429010Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:33.431391Z","iopub.execute_input":"2024-11-28T02:12:33.431729Z","iopub.status.idle":"2024-11-28T02:12:33.447542Z","shell.execute_reply.started":"2024-11-28T02:12:33.431694Z","shell.execute_reply":"2024-11-28T02:12:33.446316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mine = pd.read_csv(\"/kaggle/working/submission.csv\")\nmine","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:33.449183Z","iopub.execute_input":"2024-11-28T02:12:33.449643Z","iopub.status.idle":"2024-11-28T02:12:33.462903Z","shell.execute_reply.started":"2024-11-28T02:12:33.449594Z","shell.execute_reply":"2024-11-28T02:12:33.461657Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in filtered_group_means_dict.values():\n    print(type(i))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T02:12:33.464367Z","iopub.execute_input":"2024-11-28T02:12:33.464792Z","iopub.status.idle":"2024-11-28T02:12:33.472560Z","shell.execute_reply.started":"2024-11-28T02:12:33.464740Z","shell.execute_reply":"2024-11-28T02:12:33.471559Z"}},"outputs":[],"execution_count":null}]}