{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\npd.set_option('display.max_columns',100)\npd.set_option('display.max_rows',100)\npd.set_option('display.max_colwidth', None)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-19T01:24:53.405158Z","iopub.execute_input":"2024-12-19T01:24:53.405677Z","iopub.status.idle":"2024-12-19T01:24:54.670407Z","shell.execute_reply.started":"2024-12-19T01:24:53.405622Z","shell.execute_reply":"2024-12-19T01:24:54.668983Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Loading the data","metadata":{}},{"cell_type":"markdown","source":"Drop all the missing label.","metadata":{}},{"cell_type":"code","source":"data_train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ndata_train.dropna(subset=['sii'], inplace = True)\ndata_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T01:24:57.178827Z","iopub.execute_input":"2024-12-19T01:24:57.179245Z","iopub.status.idle":"2024-12-19T01:24:57.342733Z","shell.execute_reply.started":"2024-12-19T01:24:57.179207Z","shell.execute_reply":"2024-12-19T01:24:57.341524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_train.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T01:24:59.722236Z","iopub.execute_input":"2024-12-19T01:24:59.722641Z","iopub.status.idle":"2024-12-19T01:24:59.739165Z","shell.execute_reply.started":"2024-12-19T01:24:59.722604Z","shell.execute_reply":"2024-12-19T01:24:59.737885Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Seems that almost every column has a missing values, and some of them even have half or more of the data missing. Since it will be hard to predict the remaining missing values if the majority of the data is missing, it is better to just leave them.","metadata":{}},{"cell_type":"code","source":"def select_cols_worthy(data, tolerance = 0.5):\n    num_rows = len(data)\n    selected_cols = []\n    for col in data.columns:\n        if (data[col].isnull().sum() / num_rows) < tolerance:\n            selected_cols = np.append(selected_cols, col)\n    return data[selected_cols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:41:33.755130Z","iopub.execute_input":"2024-12-18T04:41:33.755562Z","iopub.status.idle":"2024-12-18T04:41:33.761804Z","shell.execute_reply.started":"2024-12-18T04:41:33.755526Z","shell.execute_reply":"2024-12-18T04:41:33.760580Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_train = select_cols_worthy(data_train)\ndata_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:41:35.399995Z","iopub.execute_input":"2024-12-18T04:41:35.400443Z","iopub.status.idle":"2024-12-18T04:41:35.482965Z","shell.execute_reply.started":"2024-12-18T04:41:35.400405Z","shell.execute_reply":"2024-12-18T04:41:35.481718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dict = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')\nint_cols = np.array(dict[(dict['Type'] == 'categorical int') | (dict['Type'] == 'int')]['Field'])\nint_cols = np.append(int_cols, 'sii')\nint_cols = [col for col in int_cols if col in data_train.columns]\n#int_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:41:37.321244Z","iopub.execute_input":"2024-12-18T04:41:37.322428Z","iopub.status.idle":"2024-12-18T04:41:37.333572Z","shell.execute_reply.started":"2024-12-18T04:41:37.322378Z","shell.execute_reply":"2024-12-18T04:41:37.332392Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Apparently PCIAT is only exclusive to the train dataset, then it is better to drop them (except *PCIAT-PCIAT_Total*, later we'll see why).","metadata":{}},{"cell_type":"code","source":"data_train_sel = data_train[[col for col in data_train.columns if 'PCIAT' not in col and col != 'id' or 'PCIAT-PCIAT_Total' == col]]\ndata_train_sel.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:41:39.219359Z","iopub.execute_input":"2024-12-18T04:41:39.219742Z","iopub.status.idle":"2024-12-18T04:41:39.269235Z","shell.execute_reply.started":"2024-12-18T04:41:39.219711Z","shell.execute_reply":"2024-12-18T04:41:39.268095Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Now, let's map the categorical values. Most categorical columns already encoded, like *Basic_Demos-Sex*, but apparently for *Season* columns, they are still in a string. The usage of label encoding or just simply convert them into a *categorical* data type is fairly dependent towards preference.","metadata":{}},{"cell_type":"code","source":"map_season = {'Spring': 0, 'Summer':1, 'Fall':2, 'Winter':3, 'None':-1}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:41:41.999497Z","iopub.execute_input":"2024-12-18T04:41:41.999898Z","iopub.status.idle":"2024-12-18T04:41:42.008828Z","shell.execute_reply.started":"2024-12-18T04:41:41.999863Z","shell.execute_reply":"2024-12-18T04:41:42.007597Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def season_encoder(data, map):\n    for col in data.columns:\n        if 'Season' in col:\n            data.loc[:, col] = data[col].fillna('None')\n            data.loc[:, col] = data[col].map(map)\n            data.loc[:, col] = data[col].astype('int64')\n    return data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:41:43.560019Z","iopub.execute_input":"2024-12-18T04:41:43.561511Z","iopub.status.idle":"2024-12-18T04:41:43.567489Z","shell.execute_reply.started":"2024-12-18T04:41:43.561467Z","shell.execute_reply":"2024-12-18T04:41:43.566133Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_train_sel = season_encoder(data_train_sel, map_season)\ndata_train_sel.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:41:45.062222Z","iopub.execute_input":"2024-12-18T04:41:45.062933Z","iopub.status.idle":"2024-12-18T04:41:45.142387Z","shell.execute_reply.started":"2024-12-18T04:41:45.062874Z","shell.execute_reply":"2024-12-18T04:41:45.141276Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Correlation and feature selection","metadata":{}},{"cell_type":"markdown","source":"Apparently, *PCIAT-PCIAT_Total* is highly positively correlated with *sii* (0.9). Meaning that *PCIAT-PCIAT_Total* can describe *sii* alone pretty well. Now the idea is that we can work on 2 different problems:\n* Regression: We use *PCIAT-PCIAT_Total* as our label, then we use a simple classification algorithm like logistic regression to convert it into classification.\n* Classification: We use *sii* as our label.","metadata":{}},{"cell_type":"code","source":"corr_sii_df = data_train_sel.corr()['sii'].drop(['sii', 'PCIAT-PCIAT_Total']).reset_index()\ncorr_sii_df.columns = ['Columns', 'Correlation']\ncorr_sii_df = corr_sii_df.sort_values(ascending = False, by = 'Correlation')\ncorr_sii_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:41:46.962354Z","iopub.execute_input":"2024-12-18T04:41:46.962761Z","iopub.status.idle":"2024-12-18T04:41:46.996571Z","shell.execute_reply.started":"2024-12-18T04:41:46.962725Z","shell.execute_reply":"2024-12-18T04:41:46.995304Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"corr_total_df = data_train_sel.corr()['PCIAT-PCIAT_Total'].drop(['sii','PCIAT-PCIAT_Total']).reset_index()\ncorr_total_df.columns = ['Columns', 'Correlation']\ncorr_total_df = corr_total_df.sort_values(ascending = False, by = 'Correlation')\ncorr_total_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:41:49.983710Z","iopub.execute_input":"2024-12-18T04:41:49.984143Z","iopub.status.idle":"2024-12-18T04:41:50.020540Z","shell.execute_reply.started":"2024-12-18T04:41:49.984103Z","shell.execute_reply":"2024-12-18T04:41:50.019566Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<p>It seems that the correlation for *PCIAT-PCIAT_Total* is strongger. Hence, the decision to use a regression problem.</p>\n<p>Apparently, not every column has a strong correlation. So, it is better to drop some of them that are weakly correlated.</p>","metadata":{}},{"cell_type":"code","source":"sel_cols = corr_total_df[(corr_total_df['Correlation'] > 0.1) | (corr_total_df['Correlation'] < -0.1)]\nsel_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:41:52.981584Z","iopub.execute_input":"2024-12-18T04:41:52.981994Z","iopub.status.idle":"2024-12-18T04:41:52.994682Z","shell.execute_reply.started":"2024-12-18T04:41:52.981957Z","shell.execute_reply":"2024-12-18T04:41:52.993275Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<p>There's also a duplicate within our data. Columns like <i>Physical-BMI</i> and <i>BIA-BIA_BMI</i> tells us the same BMI, so lets drop the one that are least correlated.</p>","metadata":{}},{"cell_type":"code","source":"sel_cols = np.array(sel_cols[~sel_cols['Columns'].isin(['Physical-BMI', 'SDS-SDS_Total_Raw'])]['Columns'])\nsel_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:41:55.365187Z","iopub.execute_input":"2024-12-18T04:41:55.365636Z","iopub.status.idle":"2024-12-18T04:41:55.376525Z","shell.execute_reply.started":"2024-12-18T04:41:55.365596Z","shell.execute_reply":"2024-12-18T04:41:55.375305Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Hence, the final data that's going to be used for training:","metadata":{}},{"cell_type":"code","source":"data_train_sel = data_train_sel[sel_cols]\ndata_train_sel","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:41:57.075587Z","iopub.execute_input":"2024-12-18T04:41:57.076022Z","iopub.status.idle":"2024-12-18T04:41:57.103530Z","shell.execute_reply.started":"2024-12-18T04:41:57.075970Z","shell.execute_reply":"2024-12-18T04:41:57.102376Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Modelling","metadata":{}},{"cell_type":"markdown","source":"Now, lets use bunch of common models to perform parameter tuning and CV, these are the models that's used:\n* XGBoost\n* CatBoost\n* LGBM","metadata":{}},{"cell_type":"code","source":"from catboost import CatBoostRegressor\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import train_test_split, KFold\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import mean_squared_error, accuracy_score\nimport optuna\nimport xgboost as xgb\nimport lightgbm as lgb\nfrom lightgbm import LGBMRegressor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:41:59.313386Z","iopub.execute_input":"2024-12-18T04:41:59.313762Z","iopub.status.idle":"2024-12-18T04:41:59.319159Z","shell.execute_reply.started":"2024-12-18T04:41:59.313732Z","shell.execute_reply":"2024-12-18T04:41:59.317981Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"To fill out the missing values, we can use KNN. This process will be caried out during the CV process to avoid data leakage.","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import KNNImputer\n\ndef missing_imputer(data, n=5):\n    imputer = KNNImputer(n_neighbors=n)\n    data_imputed = pd.DataFrame(imputer.fit_transform(data), columns=data.columns)\n    return data_imputed","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:42:01.112408Z","iopub.execute_input":"2024-12-18T04:42:01.113215Z","iopub.status.idle":"2024-12-18T04:42:01.118568Z","shell.execute_reply.started":"2024-12-18T04:42:01.113171Z","shell.execute_reply":"2024-12-18T04:42:01.117410Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(data_train_sel, data_train['PCIAT-PCIAT_Total'], test_size=0.2, random_state=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:42:03.150620Z","iopub.execute_input":"2024-12-18T04:42:03.151512Z","iopub.status.idle":"2024-12-18T04:42:03.159684Z","shell.execute_reply.started":"2024-12-18T04:42:03.151471Z","shell.execute_reply":"2024-12-18T04:42:03.158536Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def objective(trial):\n#     param = {\n#         \"objective\": \"reg:squarederror\",\n#         \"booster\": \"gbtree\",\n#         \"n_estimators\": trial.suggest_int(\"n_estimators\", 100, 500),\n#         \"learning_rate\": trial.suggest_float(\"learning_rate\", 0.01, 0.3),\n#         \"max_depth\": trial.suggest_int(\"max_depth\", 3, 12),\n#         \"min_child_weight\": trial.suggest_int(\"min_child_weight\", 1, 10),\n#         \"subsample\": trial.suggest_float(\"subsample\", 0.6, 1.0),\n#         \"colsample_bytree\": trial.suggest_float(\"colsample_bytree\", 0.6, 1.0),\n#         \"random_state\": 1\n#     }\n    \n#     kf = KFold(n_splits=5, shuffle=True, random_state=1)\n#     cv = []\n#     for train_index, valid_index in kf.split(X_train):\n#         X_train_cv, X_valid = X_train.iloc[train_index], X_train.iloc[valid_index]\n#         y_train_cv, y_valid = y_train.iloc[train_index], y_train.iloc[valid_index]\n#         model = xgb.XGBRegressor(**param, eval_metric='rmse')\n#         #\n#         X_train_cv = missing_imputer(X_train_cv)\n#         model.fit(X_train_cv, y_train_cv, verbose=False)\n#         #\n#         X_valid = missing_imputer(X_valid)\n#         preds = model.predict(X_valid)\n#         error = np.sqrt(mean_squared_error(y_valid, preds))\n#         cv.append(error)\n#     return np.mean(cv)\n\n# study = optuna.create_study(direction=\"minimize\")\n# study.optimize(objective, n_trials=50)\n\n# print(\"Best Parameters:\", study.best_params)\n# print(\"Best RMSLE:\", study.best_value)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T03:57:09.546344Z","iopub.execute_input":"2024-12-18T03:57:09.546741Z","iopub.status.idle":"2024-12-18T03:57:09.555625Z","shell.execute_reply.started":"2024-12-18T03:57:09.546710Z","shell.execute_reply":"2024-12-18T03:57:09.554421Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def objective(trial):\n#     params = {\n#         \"iterations\": trial.suggest_int(\"iterations\", 500, 2000),\n#         \"depth\": trial.suggest_int(\"depth\", 4, 10),\n#         \"learning_rate\": trial.suggest_float(\"learning_rate\", 0.01, 0.3, log=True),\n#         \"l2_leaf_reg\": trial.suggest_float(\"l2_leaf_reg\", 1e-3, 10.0, log=True),\n#         \"border_count\": trial.suggest_int(\"border_count\", 32, 255),\n#         \"bagging_temperature\": trial.suggest_float(\"bagging_temperature\", 0, 1.0),\n#         \"random_strength\": trial.suggest_float(\"random_strength\", 1e-9, 10, log=True),\n#     }\n#     skf = KFold(n_splits=5, shuffle=True, random_state=1)\n#     cv_scores = []\n#     for train_index, valid_index in skf.split(X_train):\n#         X_train_cv, X_valid = X_train.iloc[train_index], X_train.iloc[valid_index]\n#         y_train_cv, y_valid = y_train.iloc[train_index], y_train.iloc[valid_index]\n#         #\n#         X_train_cv = missing_imputer(X_train_cv)\n#         model = CatBoostRegressor(\n#             **params,\n#             verbose=False,\n#             random_seed=1)\n#         model.fit(X_train_cv, y_train_cv, early_stopping_rounds=50)\n#         preds = model.predict(X_valid)\n#         acc = np.sqrt(mean_squared_error(y_valid, preds))\n#         cv_scores.append(acc)\n#     return np.mean(cv_scores)\n\n# study = optuna.create_study(direction='minimize')\n# study.optimize(objective, n_trials=50)\n\n# print(\"Best hyperparameters:\", study.best_params)\n# print(\"Best RMSE:\", study.best_value)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T03:57:00.804962Z","iopub.execute_input":"2024-12-18T03:57:00.805439Z","iopub.status.idle":"2024-12-18T03:57:00.813926Z","shell.execute_reply.started":"2024-12-18T03:57:00.805399Z","shell.execute_reply":"2024-12-18T03:57:00.812675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def objective(trial):\n#     param = {\n#         'objective': 'regression',\n#         'metric': 'rmse',\n#         'boosting_type': 'gbdt',\n#         'max_depth': trial.suggest_int('max_depth', 3, 12),\n#         'num_leaves': trial.suggest_int('num_leaves', 20, 300),\n#         'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.3, log=True),\n#         'feature_fraction': trial.suggest_float('feature_fraction', 0.6, 1.0),\n#         'bagging_fraction': trial.suggest_float('bagging_fraction', 0.6, 1.0),\n#         'bagging_freq': trial.suggest_int('bagging_freq', 1, 10),\n#         'min_child_samples': trial.suggest_int('min_child_samples', 5, 50),\n#         'lambda_l1': trial.suggest_float('lambda_l1', 1e-8, 10.0, log=True),\n#         'lambda_l2': trial.suggest_float('lambda_l2', 1e-8, 10.0, log=True),\n#         'verbosity': -1\n#     }\n#     skf = KFold(n_splits=5, shuffle=True, random_state=1)\n#     cv_scores = []\n\n#     for train_idx, valid_idx in skf.split(X_train):\n#         X_train_cv, X_valid = X_train.iloc[train_idx], X_train.iloc[valid_idx]\n#         y_train_cv, y_valid = y_train.iloc[train_idx], y_train.iloc[valid_idx]\n#         #\n#         X_train_cv = missing_imputer(X_train_cv)\n#         model = LGBMRegressor(**param, random_state=1)\n#         model.fit(X_train_cv, y_train_cv)\n#         #\n#         X_valid = missing_imputer(X_valid)\n#         preds = model.predict(X_valid)\n#         acc = np.sqrt(mean_squared_error(y_valid, preds))\n#         cv_scores.append(acc)\n#     return np.mean(cv_scores)\n\n# study = optuna.create_study(direction=\"minimize\")\n# study.optimize(objective, n_trials=50)\n\n# print(\"Best parameters:\", study.best_params)\n# print(\"Best RMSE:\", study.best_value)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T03:56:39.010498Z","iopub.execute_input":"2024-12-18T03:56:39.011097Z","iopub.status.idle":"2024-12-18T03:56:39.018758Z","shell.execute_reply.started":"2024-12-18T03:56:39.011042Z","shell.execute_reply":"2024-12-18T03:56:39.017386Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# XGBOOST\n# Best Parameters: {'n_estimators': 274, 'learning_rate': 0.010807680637319593, 'max_depth': 3, 'min_child_weight': 9, 'subsample': 0.7958738757851406, 'colsample_bytree': 0.8842172644172359}\n# Best RMSLE: 17.364805242584783\n# CATBOOST\n# Best hyperparameters: {'iterations': 504, 'depth': 4, 'learning_rate': 0.012521795209001835, 'l2_leaf_reg': 5.127774664179875, 'border_count': 152, 'bagging_temperature': 0.40204136942117574, 'random_strength': 0.0015185324497143386}\n# Best CV Accuracy: 17.825797955517093\n# LGBM\n# Best parameters: {'max_depth': 3, 'num_leaves': 281, 'learning_rate': 0.06414058861691865, 'feature_fraction': 0.8246221558659342, 'bagging_fraction': 0.8182976094382267, 'bagging_freq': 5, 'min_child_samples': 16, 'lambda_l1': 0.0006651353305645184, 'lambda_l2': 1.1911661122214907}\n# Best RMSE: 17.348440954021935","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mse_models_df = pd.DataFrame({'Models':['XGBoost', 'CatBoost', 'LGBM'], 'RMSE':[17.364805242584783, 17.825797955517093, 17.348440954021935]})\nmse_models_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:42:09.908022Z","iopub.execute_input":"2024-12-18T04:42:09.908522Z","iopub.status.idle":"2024-12-18T04:42:09.921890Z","shell.execute_reply.started":"2024-12-18T04:42:09.908481Z","shell.execute_reply":"2024-12-18T04:42:09.920811Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Altough, most of them has the same error, LGBM has achieved the highest accuracy. Hence, we will proceed with LGBM.","metadata":{}},{"cell_type":"code","source":"best_param = {'max_depth': 3, 'num_leaves': 281, 'learning_rate': 0.06414058861691865, 'feature_fraction': 0.8246221558659342, 'bagging_fraction': 0.8182976094382267, 'bagging_freq': 5, 'min_child_samples': 16, 'lambda_l1': 0.0006651353305645184, 'lambda_l2': 1.1911661122214907}\nX_train = missing_imputer(X_train)\nmodel = LGBMRegressor(**best_param, random_state=1, verbosity=-1)\nmodel.fit(X_train, y_train)\n#\nX_test = missing_imputer(X_test)\npreds = model.predict(X_test)\nerror = np.sqrt(mean_squared_error(y_test, preds))\nprint(f\"LGBM RMSE: {error}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:42:17.029973Z","iopub.execute_input":"2024-12-18T04:42:17.030444Z","iopub.status.idle":"2024-12-18T04:42:17.597729Z","shell.execute_reply.started":"2024-12-18T04:42:17.030403Z","shell.execute_reply":"2024-12-18T04:42:17.594204Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Logistic Regression for classification","metadata":{}},{"cell_type":"markdown","source":"To test whether logistic regression is suitable enough to predict *sii*, let's perform a simple CV with it.","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import cohen_kappa_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:42:21.506566Z","iopub.execute_input":"2024-12-18T04:42:21.506976Z","iopub.status.idle":"2024-12-18T04:42:21.512167Z","shell.execute_reply.started":"2024-12-18T04:42:21.506941Z","shell.execute_reply":"2024-12-18T04:42:21.510952Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_train['PCIAT-PCIAT_Total']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:42:23.477611Z","iopub.execute_input":"2024-12-18T04:42:23.478018Z","iopub.status.idle":"2024-12-18T04:42:23.490498Z","shell.execute_reply.started":"2024-12-18T04:42:23.477981Z","shell.execute_reply":"2024-12-18T04:42:23.489426Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train_log, X_test_log, y_train_log, y_test_log = train_test_split(data_train['PCIAT-PCIAT_Total'], data_train['sii'], test_size=0.2, random_state=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:42:26.914261Z","iopub.execute_input":"2024-12-18T04:42:26.915512Z","iopub.status.idle":"2024-12-18T04:42:26.922578Z","shell.execute_reply.started":"2024-12-18T04:42:26.915467Z","shell.execute_reply":"2024-12-18T04:42:26.921492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train_log = X_train_log.values.reshape(-1, 1)\nX_test_log = X_test_log.values.reshape(-1, 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:49:07.294438Z","iopub.execute_input":"2024-12-18T04:49:07.294876Z","iopub.status.idle":"2024-12-18T04:49:07.300800Z","shell.execute_reply.started":"2024-12-18T04:49:07.294839Z","shell.execute_reply":"2024-12-18T04:49:07.299577Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"kf = KFold(n_splits=5, shuffle=True, random_state=1)\n# Define the Logistic Regression model\nkappa_scores = []\nfor train_idx, valid_idx in kf.split(X_train_log):\n    X_train_cv, X_valid = X_train_log[train_idx], X_train_log[valid_idx]\n    y_train_cv, y_valid = y_train_log.iloc[train_idx], y_train_log.iloc[valid_idx]\n\n    model = LogisticRegression(max_iter=1000, random_state=1)\n    model.fit(X_train_cv, y_train_cv)\n    y_pred = model.predict(X_valid)\n    \n    kappa = cohen_kappa_score(y_valid, y_pred)\n    kappa_scores.append(kappa)\n\nprint(f\"Cohen's Kappa for each fold: {kappa_scores}\")\nprint(f\"Mean Cohen's Kappa: {np.mean(kappa_scores)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:49:44.893699Z","iopub.execute_input":"2024-12-18T04:49:44.894097Z","iopub.status.idle":"2024-12-18T04:49:45.142426Z","shell.execute_reply.started":"2024-12-18T04:49:44.894061Z","shell.execute_reply":"2024-12-18T04:49:45.141297Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = LogisticRegression(max_iter=1000, random_state=1)\nmodel.fit(X_train_log, y_train_log)\ny_pred = model.predict(X_test_log)\nkappa = cohen_kappa_score(y_test_log, y_pred)\nprint(f\"Logistic Regression's Kappa: {kappa}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:52:51.806704Z","iopub.execute_input":"2024-12-18T04:52:51.807702Z","iopub.status.idle":"2024-12-18T04:52:51.864558Z","shell.execute_reply.started":"2024-12-18T04:52:51.807660Z","shell.execute_reply":"2024-12-18T04:52:51.863378Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<p>Logistic regression is able to predict with 100% accuracy.<p>\n<p>now lets use logisitic regression to predict <i>sii</i></p>","metadata":{}},{"cell_type":"code","source":"y_pred_log = model.predict(preds.reshape(-1, 1))\nkappa = cohen_kappa_score(y_test_log, y_pred_log)\nacc = accuracy_score(y_test_log, y_pred_log)\nprint(f\"Logistic Regression + XGBoost Kappa: {kappa}\")\nprint(f\"Logistic Regression + XGBoost Accuracy: {acc}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:59:12.670634Z","iopub.execute_input":"2024-12-18T04:59:12.671097Z","iopub.status.idle":"2024-12-18T04:59:12.683151Z","shell.execute_reply.started":"2024-12-18T04:59:12.671059Z","shell.execute_reply":"2024-12-18T04:59:12.681953Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Wrap it up","metadata":{}},{"cell_type":"markdown","source":"Finally, let's train the whole data and predict the test dataset.","metadata":{}},{"cell_type":"code","source":"best_param = {'max_depth': 3, 'num_leaves': 281, 'learning_rate': 0.06414058861691865, 'feature_fraction': 0.8246221558659342, 'bagging_fraction': 0.8182976094382267, 'bagging_freq': 5, 'min_child_samples': 16, 'lambda_l1': 0.0006651353305645184, 'lambda_l2': 1.1911661122214907}\ndata_train_sel = missing_imputer(data_train_sel)\nmodel_xgb = LGBMRegressor(**best_param, random_state=1, verbosity=-1)\nmodel_xgb.fit(data_train_sel, data_train['PCIAT-PCIAT_Total'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T05:01:51.904456Z","iopub.execute_input":"2024-12-18T05:01:51.904871Z","iopub.status.idle":"2024-12-18T05:01:52.651776Z","shell.execute_reply.started":"2024-12-18T05:01:51.904836Z","shell.execute_reply":"2024-12-18T05:01:52.650609Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n#data_test = select_cols_worthy(data_test)\ndata_test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T05:15:27.181624Z","iopub.execute_input":"2024-12-18T05:15:27.182030Z","iopub.status.idle":"2024-12-18T05:15:27.235925Z","shell.execute_reply.started":"2024-12-18T05:15:27.181994Z","shell.execute_reply":"2024-12-18T05:15:27.234809Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_test = season_encoder(data_test, map_season)\ndata_test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T05:15:29.278440Z","iopub.execute_input":"2024-12-18T05:15:29.279267Z","iopub.status.idle":"2024-12-18T05:15:29.346981Z","shell.execute_reply.started":"2024-12-18T05:15:29.279224Z","shell.execute_reply":"2024-12-18T05:15:29.345797Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_test_sel = data_test[sel_cols]\n#\ndata_test_sel = missing_imputer(data_test_sel)\ndata_test_sel.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T05:15:31.406485Z","iopub.execute_input":"2024-12-18T05:15:31.407340Z","iopub.status.idle":"2024-12-18T05:15:31.440029Z","shell.execute_reply.started":"2024-12-18T05:15:31.407281Z","shell.execute_reply":"2024-12-18T05:15:31.438873Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred_xgb = model_xgb.predict(data_test_sel)\ny_pred_xgb = y_pred_xgb.reshape(-1, 1)\ny_pred_xgb","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T05:15:37.339404Z","iopub.execute_input":"2024-12-18T05:15:37.339787Z","iopub.status.idle":"2024-12-18T05:15:37.351144Z","shell.execute_reply.started":"2024-12-18T05:15:37.339753Z","shell.execute_reply":"2024-12-18T05:15:37.349987Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model_log = LogisticRegression(max_iter=1000, random_state=1)\nmodel_log.fit(data_train['PCIAT-PCIAT_Total'].values.reshape(-1, 1), data_train['sii'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T05:10:56.230121Z","iopub.execute_input":"2024-12-18T05:10:56.231241Z","iopub.status.idle":"2024-12-18T05:10:56.324586Z","shell.execute_reply.started":"2024-12-18T05:10:56.231200Z","shell.execute_reply":"2024-12-18T05:10:56.322043Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred_log = model_log.predict(y_pred_xgb)\ny_pred_log","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T05:15:44.927969Z","iopub.execute_input":"2024-12-18T05:15:44.928579Z","iopub.status.idle":"2024-12-18T05:15:44.936577Z","shell.execute_reply.started":"2024-12-18T05:15:44.928540Z","shell.execute_reply":"2024-12-18T05:15:44.935267Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_res = pd.DataFrame({'id':data_test['id'], 'sii':y_pred_log})\ndf_res","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T05:15:52.792549Z","iopub.execute_input":"2024-12-18T05:15:52.792936Z","iopub.status.idle":"2024-12-18T05:15:52.805474Z","shell.execute_reply.started":"2024-12-18T05:15:52.792904Z","shell.execute_reply":"2024-12-18T05:15:52.804366Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_res.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T05:18:47.917687Z","iopub.execute_input":"2024-12-18T05:18:47.918173Z","iopub.status.idle":"2024-12-18T05:18:47.925430Z","shell.execute_reply.started":"2024-12-18T05:18:47.918134Z","shell.execute_reply":"2024-12-18T05:18:47.924266Z"}},"outputs":[],"execution_count":null}]}