{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport sklearn\nimport seaborn as sns\nfrom sklearn import datasets\nfrom sklearn.decomposition import PCA\nfrom sklearn import preprocessing\n\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:26.136420Z","iopub.execute_input":"2024-11-27T08:28:26.136840Z","iopub.status.idle":"2024-11-27T08:28:27.250254Z","shell.execute_reply.started":"2024-11-27T08:28:26.136807Z","shell.execute_reply":"2024-11-27T08:28:27.249324Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Tabular data (HBN Instruments)","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\n\n\ntrain.isna().sum().to_frame().sort_values(0, ascending=False) / len(train) * 100","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:27.252453Z","iopub.execute_input":"2024-11-27T08:28:27.253045Z","iopub.status.idle":"2024-11-27T08:28:27.321672Z","shell.execute_reply.started":"2024-11-27T08:28:27.253000Z","shell.execute_reply":"2024-11-27T08:28:27.320740Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"temp_things = [train.columns, \n               train[\"sii\"].isna().sum() / len(train)    \n    ]\n\nfor thing in temp_things:\n    print(thing)\n    print()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:27.322941Z","iopub.execute_input":"2024-11-27T08:28:27.323254Z","iopub.status.idle":"2024-11-27T08:28:27.329769Z","shell.execute_reply.started":"2024-11-27T08:28:27.323226Z","shell.execute_reply":"2024-11-27T08:28:27.328781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pciat_cols = [col for col in train.columns if col.startswith(\"PCIAT-PCIAT\")]\npciat_cols.append(\"sii\")\n\npciat_train = train[pciat_cols]\n\n# the higher PCIAT scores, the higher sii = more severe, but what are the thresholds between sii scores?\n# plot\nfig, ax = plt.subplots()\npciat_train_notna = pciat_train[[\"PCIAT-PCIAT_Total\", \"sii\"]][pciat_train[[\"PCIAT-PCIAT_Total\", \"sii\"]].notna().all(axis=1)]\nax.scatter(pciat_train_notna[\"PCIAT-PCIAT_Total\"], pciat_train_notna[\"sii\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:27.331671Z","iopub.execute_input":"2024-11-27T08:28:27.331982Z","iopub.status.idle":"2024-11-27T08:28:27.599254Z","shell.execute_reply.started":"2024-11-27T08:28:27.331952Z","shell.execute_reply":"2024-11-27T08:28:27.598304Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# okay the graph looks nice, what are the thresholds?\npciat_train_notna.groupby(\"sii\").max()\n\n# for now:\n# sii = 0 <=> PCIAT in [0, 30]\n# sii = 1 <=> PCIAT in [31, 49]\n# sii = 2 <=> PCIAT in [50, 79]\n# sii = 3 <=> PCIAT in [80, 100]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:27.600510Z","iopub.execute_input":"2024-11-27T08:28:27.600848Z","iopub.status.idle":"2024-11-27T08:28:27.612191Z","shell.execute_reply.started":"2024-11-27T08:28:27.600808Z","shell.execute_reply":"2024-11-27T08:28:27.611084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# but what about the NaN values in the responses to 20 questions\n\n# make sure the PCIAT score added up correctly\nrecalculated_pciat = pciat_train.drop(columns=[\"PCIAT-PCIAT_Total\", \"sii\"]).sum(axis=1)\ndiff = pciat_train[\"PCIAT-PCIAT_Total\"] - recalculated_pciat\ndiff.sum()\n# all correct, but still need to investigate the NaN issues","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:27.613798Z","iopub.execute_input":"2024-11-27T08:28:27.614107Z","iopub.status.idle":"2024-11-27T08:28:27.624635Z","shell.execute_reply.started":"2024-11-27T08:28:27.614078Z","shell.execute_reply":"2024-11-27T08:28:27.623474Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# see how many rows of all NaN values\nlen(pciat_train[pciat_train.isna().all(axis=1)]) / len(pciat_train)\n\n# 1224 rows, 30% of data, that's a lot! likely needs an unsupervised learning # TODO","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:27.625901Z","iopub.execute_input":"2024-11-27T08:28:27.626238Z","iopub.status.idle":"2024-11-27T08:28:27.636581Z","shell.execute_reply.started":"2024-11-27T08:28:27.626207Z","shell.execute_reply":"2024-11-27T08:28:27.635480Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# but what about the other rows with partially NaN values\n# because here missing responses do not necessarily mean that 0 score\n# rmb: each question has a max score of 5\n\npciat_train[~pciat_train.isna().all(axis=1)].isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:27.637803Z","iopub.execute_input":"2024-11-27T08:28:27.638145Z","iopub.status.idle":"2024-11-27T08:28:27.650160Z","shell.execute_reply.started":"2024-11-27T08:28:27.638096Z","shell.execute_reply":"2024-11-27T08:28:27.649285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.set_option('display.max_rows', pciat_train.shape[0])     \npd.set_option('display.max_columns', pciat_train.shape[1]) \n\ntemp = pciat_train[~pciat_train.isna().all(axis=1)][pciat_train.isna().any(axis=1)]\n\n# the second row is weird: all responses are NaN but the total score is still 0 => very biased\n# how to deal with this?\ntemp","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:27.651658Z","iopub.execute_input":"2024-11-27T08:28:27.651977Z","iopub.status.idle":"2024-11-27T08:28:27.761210Z","shell.execute_reply.started":"2024-11-27T08:28:27.651947Z","shell.execute_reply":"2024-11-27T08:28:27.760245Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.reset_option(\"display.max_rows\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:27.764369Z","iopub.execute_input":"2024-11-27T08:28:27.764690Z","iopub.status.idle":"2024-11-27T08:28:27.769840Z","shell.execute_reply.started":"2024-11-27T08:28:27.764660Z","shell.execute_reply":"2024-11-27T08:28:27.768747Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# consider two extremes:\n# for a given row with some partial NaN values, if we replace the NaN with the scores of 5 and add them all\n# if the total score now is still within the predefined thresholds, it means the sii still holds\n# if it's larger, then we now just don't know and can't assume anything, it must be NaN for sii score\n# we will deal with NaN sii scores later \n\n# I learned this from https://www.kaggle.com/code/antoninadolgorukova/cmi-piu-features-eda?scriptVersionId=206130660&cellId=32\n\ndef recalculate_sii(row):\n    responses20_cols = [col for col in train.columns if col.startswith(\"PCIAT-PCIAT\") and col != \"PCIAT-PCIAT_Total\"]\n    \n    if pd.isna(row['PCIAT-PCIAT_Total']):\n        return np.nan\n    max_possible = row['PCIAT-PCIAT_Total'] + row[responses20_cols].isna().sum() * 5\n    if row['PCIAT-PCIAT_Total'] <= 30 and max_possible <= 30:\n        return 0\n    elif 31 <= row['PCIAT-PCIAT_Total'] <= 49 and max_possible <= 49:\n        return 1\n    elif 50 <= row['PCIAT-PCIAT_Total'] <= 79 and max_possible <= 79:\n        return 2\n    elif row['PCIAT-PCIAT_Total'] >= 80 and max_possible >= 80:\n        return 3\n    return np.nan\n\ntemp['recalc_sii'] = temp.apply(recalculate_sii, axis=1)\n\nlen(temp), temp['recalc_sii'].isna().sum()\n\n# only 17 rows out of 65 that are now having missing sii scores (not including initially NaN sii rows)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:27.770902Z","iopub.execute_input":"2024-11-27T08:28:27.771203Z","iopub.status.idle":"2024-11-27T08:28:27.819859Z","shell.execute_reply.started":"2024-11-27T08:28:27.771174Z","shell.execute_reply":"2024-11-27T08:28:27.818880Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.reset_option(\"display.max_rows\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:27.820930Z","iopub.execute_input":"2024-11-27T08:28:27.821215Z","iopub.status.idle":"2024-11-27T08:28:27.831874Z","shell.execute_reply.started":"2024-11-27T08:28:27.821188Z","shell.execute_reply":"2024-11-27T08:28:27.830906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# apply this to the main train\n# run once\n\ntrain['recalc_sii'] = train.apply(recalculate_sii, axis=1)\ntrain = train.drop(columns=['sii'])\ntrain = train.rename(columns={'recalc_sii': 'sii'})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:27.833024Z","iopub.execute_input":"2024-11-27T08:28:27.833405Z","iopub.status.idle":"2024-11-27T08:28:29.245969Z","shell.execute_reply.started":"2024-11-27T08:28:27.833360Z","shell.execute_reply":"2024-11-27T08:28:29.245124Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train[\"sii\"].isna().sum() / len(train) # after preprocessing ssi, there is a bit more NaN","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:29.247054Z","iopub.execute_input":"2024-11-27T08:28:29.247403Z","iopub.status.idle":"2024-11-27T08:28:29.255289Z","shell.execute_reply.started":"2024-11-27T08:28:29.247359Z","shell.execute_reply":"2024-11-27T08:28:29.254077Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:29.256719Z","iopub.execute_input":"2024-11-27T08:28:29.257004Z","iopub.status.idle":"2024-11-27T08:28:29.286442Z","shell.execute_reply.started":"2024-11-27T08:28:29.256976Z","shell.execute_reply":"2024-11-27T08:28:29.285350Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# labels distribution\nfig, ax = plt.subplots()\ncounts = train[\"sii\"].value_counts(dropna=False).sort_index()\nax.bar(counts.index.astype(str), counts.values)\ncounts\n\n# mostly no severity, but still lots of missing sii","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:29.288143Z","iopub.execute_input":"2024-11-27T08:28:29.288582Z","iopub.status.idle":"2024-11-27T08:28:29.517643Z","shell.execute_reply.started":"2024-11-27T08:28:29.288541Z","shell.execute_reply":"2024-11-27T08:28:29.516687Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def transform_tabular(data):\n    # drop the PCIAT-PCIAT columns to avoid potential data leakage\n    # for train mostly\n    responses20_cols = [col for col in train.columns if col.startswith(\"PCIAT\")]\n    data = data.drop(columns=responses20_cols)\n\n    return data\n\ntrain = transform_tabular(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:29.519019Z","iopub.execute_input":"2024-11-27T08:28:29.519347Z","iopub.status.idle":"2024-11-27T08:28:29.526042Z","shell.execute_reply.started":"2024-11-27T08:28:29.519316Z","shell.execute_reply":"2024-11-27T08:28:29.524998Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"###\n# check column types\ncats = [col for col in train.columns if train[col].dtype == \"object\"]\ncons = [col for col in train.columns if train[col].dtype != \"object\"]\ncats, cons","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:29.527318Z","iopub.execute_input":"2024-11-27T08:28:29.527728Z","iopub.status.idle":"2024-11-27T08:28:29.541752Z","shell.execute_reply.started":"2024-11-27T08:28:29.527686Z","shell.execute_reply":"2024-11-27T08:28:29.540706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# imputation\n\n# first, let's focus on continuous vars\n\n# these columns seem to have no missing data, we can use them as group matching for imputing continuous vars? \n# hmm let's try\ntrain[[\"Basic_Demos-Enroll_Season\", \"Basic_Demos-Age\", \"Basic_Demos-Sex\"]].isna().sum()\n# but I don't use \"Basic_Demos-Enroll_Season\" because I expect every season would observe some \n# similar people within a certain group of age and sex, not including it would help me have a \n# more broader groups  by age and sex","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:29.542929Z","iopub.execute_input":"2024-11-27T08:28:29.543333Z","iopub.status.idle":"2024-11-27T08:28:29.555771Z","shell.execute_reply.started":"2024-11-27T08:28:29.543235Z","shell.execute_reply":"2024-11-27T08:28:29.554933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# try on Physical-Height\ncols_no_missing = [\"Basic_Demos-Age\", \"Basic_Demos-Sex\"]\ngroup_means_draft = train.groupby(cols_no_missing)[\"Physical-Height\"].agg(\"mean\")\ngroup_means_draft\n\n# it looks like group of 22-yo and male only has one person and that person's data is NaN so I replace\n# it with data from male person but 21-yo (mean)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:29.557062Z","iopub.execute_input":"2024-11-27T08:28:29.557395Z","iopub.status.idle":"2024-11-27T08:28:29.571852Z","shell.execute_reply.started":"2024-11-27T08:28:29.557353Z","shell.execute_reply":"2024-11-27T08:28:29.570880Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calculate_group_means(data, cols_no_missing, cons):\n    \"\"\"\n    Calculate group means for continuous variables based on given grouping columns.\n    \"\"\"\n    group_means_dict = {}\n    too_many_missing = []\n    \n    for col in cons:\n        if col not in cols_no_missing:\n            group_means = data.groupby(cols_no_missing)[col].mean()\n            group_means_dict[col] = group_means\n\n            data[col] = data.groupby(cols_no_missing)[col].transform(\"mean\")\n            \n            if data[col].isna().sum() > 10:\n                too_many_missing.append(col)\n    \n    return group_means_dict, too_many_missing\n\n\ndef impute_remaining_values(data, group_means_dict, selected_cons):\n    \"\"\"\n    Impute remaining missing values in the group means dictionary.\n    \"\"\"\n    for col, group_means in group_means_dict.items():\n        if col in selected_cons:\n            overall_mean = data[col].mean()\n            group_means.fillna(overall_mean, inplace=True)\n\n    return data\n\n\n# calculate group means and identify columns with too many missing values\ncols_no_missing = [\"Basic_Demos-Age\", \"Basic_Demos-Sex\"]\ntransformed_cons_train = train.copy()\ngroup_means_dict, too_many_missing = calculate_group_means(transformed_cons_train, cols_no_missing, cons)\n\n# filter selected continuous variables\nselected_cons = [col for col in cons if col not in too_many_missing]\n\n# impute missing values in the group means dictionary\nimpute_remaining_values(transformed_cons_train, group_means_dict, selected_cons)\n\n\nfor col, group_means in group_means_dict.items():\n    transformed_cons_train[col] = transformed_cons_train.apply(\n        lambda row: group_means.get((row[\"Basic_Demos-Age\"], row[\"Basic_Demos-Sex\"]), row[col])\n        if pd.isna(row[col]) else row[col],\n        axis=1\n    )\n\ntransformed_cons_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:29.573361Z","iopub.execute_input":"2024-11-27T08:28:29.573718Z","iopub.status.idle":"2024-11-27T08:28:31.892607Z","shell.execute_reply.started":"2024-11-27T08:28:29.573668Z","shell.execute_reply":"2024-11-27T08:28:31.891622Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transformed_cons_train[transformed_cons_train[selected_cons].isna().any(axis=1)]\n# this one record is all NaN => but can't drop, because in the hidden test set might have it","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:31.893970Z","iopub.execute_input":"2024-11-27T08:28:31.894465Z","iopub.status.idle":"2024-11-27T08:28:31.914825Z","shell.execute_reply.started":"2024-11-27T08:28:31.894417Z","shell.execute_reply":"2024-11-27T08:28:31.913795Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# transformed_cons_train = transformed_cons_train[transformed_cons_train[\"id\"] != \"3cb2c4da\"]\n\ntransformed_cons_train[\"sii\"] = transformed_cons_train[\"sii\"].apply(lambda x: round(x))\n\ntransformed_cons_train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:31.916220Z","iopub.execute_input":"2024-11-27T08:28:31.916992Z","iopub.status.idle":"2024-11-27T08:28:31.926103Z","shell.execute_reply.started":"2024-11-27T08:28:31.916947Z","shell.execute_reply":"2024-11-27T08:28:31.925006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# deal with categorical variables later\n# now, let's apply several methods on the continous variables","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:31.927587Z","iopub.execute_input":"2024-11-27T08:28:31.927995Z","iopub.status.idle":"2024-11-27T08:28:31.942047Z","shell.execute_reply.started":"2024-11-27T08:28:31.927951Z","shell.execute_reply":"2024-11-27T08:28:31.941078Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"only_transformed_cons = [col for col in selected_cons if transformed_cons_train[col].dtype != \"object\"]\nonly_transformed_cons_train = transformed_cons_train[only_transformed_cons]\n\nonly_transformed_cons_train.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:31.943110Z","iopub.execute_input":"2024-11-27T08:28:31.943414Z","iopub.status.idle":"2024-11-27T08:28:31.959647Z","shell.execute_reply.started":"2024-11-27T08:28:31.943385Z","shell.execute_reply":"2024-11-27T08:28:31.958484Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = only_transformed_cons_train.drop(columns=[\"sii\"])\ny = only_transformed_cons_train[\"sii\"]\nX.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:31.961223Z","iopub.execute_input":"2024-11-27T08:28:31.961603Z","iopub.status.idle":"2024-11-27T08:28:31.969742Z","shell.execute_reply.started":"2024-11-27T08:28:31.961573Z","shell.execute_reply":"2024-11-27T08:28:31.968738Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# (1) attemp PCA on the continous variables (unsupervised, no need sii)\n\n# standardize the cons features\n# X_mean = np.mean(X, axis=0)\n# X_std = np.std(X, axis=0)\n# X_standardized = (X - X_mean) / X_std\n\n# cov_matrix = np.cov(X_standardized, rowvar=False)\n# eigenvalues, eigenvectors = np.linalg.eig(cov_matrix)\n\n# # sort eigenvalues and corresponding eigenvectors\n# sorted_indices = np.argsort(eigenvalues)[::-1]\n# eigenvalues = eigenvalues[sorted_indices]\n# eigenvectors = eigenvectors[:, sorted_indices]\n\n# k = 5\n# principal_components = eigenvectors[:, :k]\n# X_pca = np.dot(X_standardized, principal_components)\n\n# normalize data\nX_scaled = pd.DataFrame(preprocessing.scale(X),columns = X.columns) \n\n# PCA\npca = PCA(n_components=3)\npca.fit_transform(X_scaled)\n\nPCS = ['PC1','PC2', 'PC3'] #, 'PC4', 'PC5']\n\npc_data = pd.DataFrame(pca.components_,columns=X_scaled.columns,\n             index = PCS)\npc_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:31.971138Z","iopub.execute_input":"2024-11-27T08:28:31.971977Z","iopub.status.idle":"2024-11-27T08:28:32.157541Z","shell.execute_reply.started":"2024-11-27T08:28:31.971925Z","shell.execute_reply":"2024-11-27T08:28:32.156336Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pc_data = pc_data.T.reset_index().rename(columns={\"index\": \"Field\"})\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:32.176690Z","iopub.execute_input":"2024-11-27T08:28:32.177752Z","iopub.status.idle":"2024-11-27T08:28:32.193516Z","shell.execute_reply.started":"2024-11-27T08:28:32.177680Z","shell.execute_reply":"2024-11-27T08:28:32.189928Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"info = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv\")[[\"Field\", \"Description\"]]\n\npc_data_info = pd.merge(pc_data, info, on=\"Field\")\npc_data_info[PCS] = pc_data_info[PCS].apply(lambda x: abs(x))\n\npc_data_info","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:32.197302Z","iopub.execute_input":"2024-11-27T08:28:32.197806Z","iopub.status.idle":"2024-11-27T08:28:32.245678Z","shell.execute_reply.started":"2024-11-27T08:28:32.197729Z","shell.execute_reply":"2024-11-27T08:28:32.245068Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# top 10 each PC\npc_data_info[[\"Field\", \"PC1\"]].sort_values(\"PC1\", ascending=False)[:10]\n\n# this PC1 gives very close coefficient values, why? likely because these BIA variables are highly correlated => not good\n# we just pick BIA-BIA_LST \n# need to re-check this\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:32.246430Z","iopub.execute_input":"2024-11-27T08:28:32.246692Z","iopub.status.idle":"2024-11-27T08:28:32.257967Z","shell.execute_reply.started":"2024-11-27T08:28:32.246666Z","shell.execute_reply":"2024-11-27T08:28:32.256957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pc_data_info[[\"Field\", \"PC2\"]].sort_values(\"PC2\", ascending=False)[:10]\n\n# this PC2 gives somewhat more \"diverse\" coefficients","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:32.259212Z","iopub.execute_input":"2024-11-27T08:28:32.259687Z","iopub.status.idle":"2024-11-27T08:28:32.274804Z","shell.execute_reply.started":"2024-11-27T08:28:32.259621Z","shell.execute_reply":"2024-11-27T08:28:32.273742Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pc_data_info[[\"Field\", \"PC3\"]].sort_values(\"PC3\", ascending=False)[:10]\n# SDS-SDS_Total_T and SDS-SDS_Total_Raw seem to influence the most!","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:32.276141Z","iopub.execute_input":"2024-11-27T08:28:32.276575Z","iopub.status.idle":"2024-11-27T08:28:32.291111Z","shell.execute_reply.started":"2024-11-27T08:28:32.276529Z","shell.execute_reply":"2024-11-27T08:28:32.290051Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"picked_cons = [\"BIA-BIA_LST\", \n               \"Physical-Weight\", \"BIA-BIA_BMI\", \"Physical-Waist_Circumference\", \n               \"SDS-SDS_Total_T\", \"SDS-SDS_Total_Raw\"\n              ]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:32.292461Z","iopub.execute_input":"2024-11-27T08:28:32.293492Z","iopub.status.idle":"2024-11-27T08:28:32.300547Z","shell.execute_reply.started":"2024-11-27T08:28:32.293443Z","shell.execute_reply":"2024-11-27T08:28:32.299516Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"filtered_group_means_dict = {key: value for key, value in group_means_dict.items() if key in picked_cons}\nfiltered_group_means_dict","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:32.301779Z","iopub.execute_input":"2024-11-27T08:28:32.302105Z","iopub.status.idle":"2024-11-27T08:28:32.328389Z","shell.execute_reply.started":"2024-11-27T08:28:32.302075Z","shell.execute_reply":"2024-11-27T08:28:32.327082Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = only_transformed_cons_train[picked_cons]\ny = y","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:32.329900Z","iopub.execute_input":"2024-11-27T08:28:32.330624Z","iopub.status.idle":"2024-11-27T08:28:32.343114Z","shell.execute_reply.started":"2024-11-27T08:28:32.330574Z","shell.execute_reply":"2024-11-27T08:28:32.342016Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#  modeling/training\nX","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:32.344482Z","iopub.execute_input":"2024-11-27T08:28:32.344895Z","iopub.status.idle":"2024-11-27T08:28:32.364438Z","shell.execute_reply.started":"2024-11-27T08:28:32.344864Z","shell.execute_reply":"2024-11-27T08:28:32.363315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ordinal classification using decision tree\n# https://www.tandfonline.com/doi/full/10.1080/24725854.2022.2081745#abstract\n# https://scikit-learn.org/stable/modules/generated/sklearn.tree.DecisionTreeClassifier.html\n\nX_train, X_valid, y_train, y_valid = train_test_split(X, y, test_size=0.2, random_state=123)\n\ndt_model = DecisionTreeClassifier(max_depth=3, random_state=123) \ndt_model.fit(X_train, y_train)\n\ny_pred = dt_model.predict(X_valid)\nprint(classification_report(y_valid, y_pred))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:32.365614Z","iopub.execute_input":"2024-11-27T08:28:32.365944Z","iopub.status.idle":"2024-11-27T08:28:32.390862Z","shell.execute_reply.started":"2024-11-27T08:28:32.365913Z","shell.execute_reply":"2024-11-27T08:28:32.389841Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")\n# there are some rows that have almost no data at all!\n\n\nfor col in picked_cons:\n    if col != \"sii\" and col != \"Basic_Demos-Age\" and col != \"Basic_Demos-Sex\":\n        test[col] = test.apply(\n            lambda row: filtered_group_means_dict[col].get((row[\"Basic_Demos-Age\"], row[\"Basic_Demos-Sex\"]), np.nan)\n            if pd.isna(row[col]) else row[col],\n            axis=1\n        )\n\nX_test = test[picked_cons]\nX_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:32.391998Z","iopub.execute_input":"2024-11-27T08:28:32.392332Z","iopub.status.idle":"2024-11-27T08:28:32.421319Z","shell.execute_reply.started":"2024-11-27T08:28:32.392296Z","shell.execute_reply":"2024-11-27T08:28:32.420235Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dt_model.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:32.422347Z","iopub.execute_input":"2024-11-27T08:28:32.422645Z","iopub.status.idle":"2024-11-27T08:28:32.431103Z","shell.execute_reply.started":"2024-11-27T08:28:32.422616Z","shell.execute_reply":"2024-11-27T08:28:32.429921Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.DataFrame({\n    \"id\": test[\"id\"],\n    \"sii\": dt_model.predict(X_test)\n})\nsubmission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:32.432671Z","iopub.execute_input":"2024-11-27T08:28:32.433684Z","iopub.status.idle":"2024-11-27T08:28:32.448340Z","shell.execute_reply.started":"2024-11-27T08:28:32.433651Z","shell.execute_reply":"2024-11-27T08:28:32.446703Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:32.449837Z","iopub.execute_input":"2024-11-27T08:28:32.450268Z","iopub.status.idle":"2024-11-27T08:28:32.465778Z","shell.execute_reply.started":"2024-11-27T08:28:32.450226Z","shell.execute_reply":"2024-11-27T08:28:32.464636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mine = pd.read_csv(\"/kaggle/working/submission.csv\")\nmine","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:32.467365Z","iopub.execute_input":"2024-11-27T08:28:32.468340Z","iopub.status.idle":"2024-11-27T08:28:32.485955Z","shell.execute_reply.started":"2024-11-27T08:28:32.468291Z","shell.execute_reply":"2024-11-27T08:28:32.483733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# unsupervised learning (clustering) to get the NA values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:32.487528Z","iopub.execute_input":"2024-11-27T08:28:32.488953Z","iopub.status.idle":"2024-11-27T08:28:32.494650Z","shell.execute_reply.started":"2024-11-27T08:28:32.488906Z","shell.execute_reply":"2024-11-27T08:28:32.493111Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in filtered_group_means_dict.values():\n    print(type(i))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T08:28:32.495830Z","iopub.execute_input":"2024-11-27T08:28:32.496245Z","iopub.status.idle":"2024-11-27T08:28:32.505834Z","shell.execute_reply.started":"2024-11-27T08:28:32.496202Z","shell.execute_reply":"2024-11-27T08:28:32.504670Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}