{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n#import os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# configs\npd.set_option('display.max_columns', None) # we want to display all columns in this notebook\npd.set_option('display.max_rows', 100) # increase rows to be displayed\npd.set_option('display.max_colwidth', None) # show full cell contents\n\ndf_train = pd.read_csv('../input/child-mind-institute-problematic-internet-use/train.csv')\ndf_test = pd.read_csv('../input/child-mind-institute-problematic-internet-use/test.csv')\ndf_sub = pd.read_csv('../input/child-mind-institute-problematic-internet-use/sample_submission.csv')\ndf_dict = pd.read_csv('../input/child-mind-institute-problematic-internet-use/data_dictionary.csv')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-12-19T20:01:03.463245Z","iopub.execute_input":"2024-12-19T20:01:03.463630Z","iopub.status.idle":"2024-12-19T20:01:03.523572Z","shell.execute_reply.started":"2024-12-19T20:01:03.463598Z","shell.execute_reply":"2024-12-19T20:01:03.521938Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_features = []\npciat_features = []\n\nfor i in range(1, 81):\n    sample_type = df_dict.iloc[i]['Type']\n    if 'str' in sample_type:\n        #categorical_features.append(i-1)\n        categorical_features.append(df_dict.iloc[i]['Field'])\n\n    sample_field = df_dict.iloc[i]['Field']\n    if 'PCIAT' in sample_field:\n        #categorical_features.append(i-1)\n        pciat_features.append(df_dict.iloc[i]['Field'])\n\n# include PCIAT_Season feature or not\ncategorical_features.remove('PCIAT-Season')\n\n# include PCIAT_Season feature or not\n#pciat_features.remove('PCIAT-Season')\n\n# this make model overfit\n# pciat_features.remove('PCIAT-PCIAT_Total')\nprint(categorical_features)\nprint(pciat_features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T20:01:03.525739Z","iopub.execute_input":"2024-12-19T20:01:03.526202Z","iopub.status.idle":"2024-12-19T20:01:03.542910Z","shell.execute_reply.started":"2024-12-19T20:01:03.526150Z","shell.execute_reply":"2024-12-19T20:01:03.541913Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mostly_null_features = ['Physical-Waist_Circumference', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec', 'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T20:01:03.544743Z","iopub.execute_input":"2024-12-19T20:01:03.545083Z","iopub.status.idle":"2024-12-19T20:01:03.554336Z","shell.execute_reply.started":"2024-12-19T20:01:03.545050Z","shell.execute_reply":"2024-12-19T20:01:03.553362Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import HistGradientBoostingRegressor\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.impute import KNNImputer\n\noptimal_thresholds = [0.5, 1.5, 2.5]\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return quadratic_weighted_kappa(y_true, rounded_p)\n\ndef feature_engineering(df):\n    #Age\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['Physical-Waist_Age'] = df['Basic_Demos-Age'] * df['Physical-Waist_Circumference']\n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Physical-Height_Age'] = df['Basic_Demos-Age'] * df['Physical-Height']\n    df['SDS_InternetHours'] = df['SDS-SDS_Total_T'] * df['PreInt_EduHx-computerinternet_hoursday']\n\n    #SDS\n    df['SDS_BMI'] = df['BIA-BIA_BMI'] * df['SDS-SDS_Total_T']\n    df['CGAS_SDS'] = df['CGAS-CGAS_Score'] * df['SDS-SDS_Total_T']\n    df['CGAS_Endurance_Mins'] = df['CGAS-CGAS_Score'] * df['Fitness_Endurance-Time_Mins']\n    df['SDS_Activity'] = df['BIA-BIA_Activity_Level_num'] * df['SDS-SDS_Total_T']\n\n    df['BMI_Systolic_BP'] = df['BIA-BIA_BMI'] * df['Physical-Systolic_BP']\n    df['Age_Systolic_BP'] = df['Basic_Demos-Age'] * df['Physical-Systolic_BP']\n    df['PreInt_Systolic_BP'] = df['Physical-Systolic_BP'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['PAQ_A_Activity'] = df['BIA-BIA_Activity_Level_num'] * df['PAQ_A-PAQ_A_Total']\n    df['Activity_CU_PU'] = df['BIA-BIA_Activity_Level_num'] * df['FGC-FGC_CU'] * df['FGC-FGC_PU']\n\n    #FGC\n    df['FGC_CU_PU'] = df['FGC-FGC_CU'] * df['FGC-FGC_PU']\n    df['FGC_CU_PU_Age'] = df['FGC-FGC_CU'] * df['FGC-FGC_PU'] * df['Basic_Demos-Age']\n    df['FGC_GSND_GSD'] = df['FGC-FGC_GSND'] * df['FGC-FGC_GSD']\n    df['FGC_GSND_GSD_Age'] = df['FGC-FGC_GSND'] * df['FGC-FGC_GSD'] * df['Basic_Demos-Age']\n    df['CGAS_CU_PU'] = df['CGAS-CGAS_Score'] * df['FGC-FGC_CU'] * df['FGC-FGC_PU']\n    df['PreInt_FGC_CU_PU'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['FGC-FGC_CU'] * df['FGC-FGC_PU']\n    df['Endurance_CU_PU'] = df['Fitness_Endurance-Time_Mins'] * df['FGC-FGC_CU'] * df['FGC-FGC_PU']\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T20:01:03.559389Z","iopub.execute_input":"2024-12-19T20:01:03.559740Z","iopub.status.idle":"2024-12-19T20:01:03.572471Z","shell.execute_reply.started":"2024-12-19T20:01:03.559707Z","shell.execute_reply":"2024-12-19T20:01:03.571357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cleaned_df_train = df_train.dropna(subset=['sii']).drop(columns=['id'] + pciat_features + categorical_features)\ncleaned_df_train = feature_engineering(cleaned_df_train)\n\n#for feature in categorical_features:\n#    cleaned_df_train[feature + '_Encoded'] = cleaned_df_train[feature].astype('category').cat.codes\n#    cleaned_df_train = cleaned_df_train.drop(columns=[feature])\n\nimputer = KNNImputer(n_neighbors=10)\n\nnumeric_cols = cleaned_df_train.select_dtypes(include=['float64', 'int64']).columns\nimputed_data = imputer.fit_transform(cleaned_df_train[numeric_cols])\n\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\n\nfor col in cleaned_df_train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = cleaned_df_train[col]\n\n\ncleaned_df_train = train_imputed\ncleaned_df_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T20:01:03.581064Z","iopub.execute_input":"2024-12-19T20:01:03.581492Z","iopub.status.idle":"2024-12-19T20:01:06.832074Z","shell.execute_reply.started":"2024-12-19T20:01:03.581450Z","shell.execute_reply":"2024-12-19T20:01:06.831024Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#df_train = pd.get_dummies(df_train, columns=categorical_features)\n\ny = cleaned_df_train.sii\nX = cleaned_df_train.drop(columns = ['sii'])\n\n#print(X.info())\n\nX_train, X_val, y_train, y_val = train_test_split(X, y, random_state=0)\nmodel = HistGradientBoostingRegressor().fit(X_train, y_train)\n\n\npred = model.predict(X_val)\nprint(evaluate_predictions(optimal_thresholds, y_val, pred))\n#print(pred)\n#print(y_val)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T20:01:06.834749Z","iopub.execute_input":"2024-12-19T20:01:06.835217Z","iopub.status.idle":"2024-12-19T20:01:07.590562Z","shell.execute_reply.started":"2024-12-19T20:01:06.835167Z","shell.execute_reply":"2024-12-19T20:01:07.589604Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cleaned_df_test = df_test.drop(columns=['id'] + categorical_features)\ncleaned_df_test = feature_engineering(cleaned_df_test)\n\n\n#for feature in categorical_features:\n#    cleaned_df_test[feature + '_Encoded'] = cleaned_df_test[feature].astype('category').cat.codes\n#    cleaned_df_test = cleaned_df_test.drop(columns=[feature])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T20:01:07.591649Z","iopub.execute_input":"2024-12-19T20:01:07.591968Z","iopub.status.idle":"2024-12-19T20:01:07.613787Z","shell.execute_reply.started":"2024-12-19T20:01:07.591933Z","shell.execute_reply":"2024-12-19T20:01:07.612610Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#df_test = pd.get_dummies(df_test, columns=categorical_features)\n\n#print(df_test.info())\n\nX_test = cleaned_df_test\ny_test = df_sub.sii\n\npred = model.predict(X_test)\n\nprint(print(evaluate_predictions(optimal_thresholds, y_test, pred)))\nprint(pred.round().astype(int))\nprint(threshold_Rounder(pred, optimal_thresholds))\nprint(y_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T20:01:07.616453Z","iopub.execute_input":"2024-12-19T20:01:07.617218Z","iopub.status.idle":"2024-12-19T20:01:07.637832Z","shell.execute_reply.started":"2024-12-19T20:01:07.617164Z","shell.execute_reply":"2024-12-19T20:01:07.637018Z"},"_kg_hide-input":false},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save test predictions to file\noutput = pd.DataFrame({'id': df_test.id.values,\n                       'sii': pred.round().astype(int)})\noutput.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T20:01:07.638723Z","iopub.execute_input":"2024-12-19T20:01:07.638996Z","iopub.status.idle":"2024-12-19T20:01:07.649471Z","shell.execute_reply.started":"2024-12-19T20:01:07.638965Z","shell.execute_reply":"2024-12-19T20:01:07.648477Z"}},"outputs":[],"execution_count":null}]}