{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Import thư viện cần thiết\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import roc_curve, auc, classification_report, accuracy_score, precision_score, recall_score, f1_score, confusion_matrix, ConfusionMatrixDisplay\nfrom sklearn.preprocessing import StandardScaler\nfrom imblearn.over_sampling import SMOTE\nimport matplotlib.pyplot as plt\nimport lightgbm as lgb\nfrom lightgbm import LGBMClassifier\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nimport optuna\nfrom sklearn.ensemble import VotingClassifier","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T14:46:51.014376Z","iopub.execute_input":"2024-12-19T14:46:51.014725Z","iopub.status.idle":"2024-12-19T14:46:51.020195Z","shell.execute_reply.started":"2024-12-19T14:46:51.014693Z","shell.execute_reply":"2024-12-19T14:46:51.019371Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Đọc dữ liệu\ntrain_file = '/kaggle/input/child-mind-institute-problematic-internet-use/train.csv'\ntest_file = '/kaggle/input/child-mind-institute-problematic-internet-use/test.csv'\n\ndf = pd.read_csv(train_file)\ndf_test = pd.read_csv(test_file)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T14:46:51.050202Z","iopub.execute_input":"2024-12-19T14:46:51.050433Z","iopub.status.idle":"2024-12-19T14:46:51.094397Z","shell.execute_reply.started":"2024-12-19T14:46:51.050410Z","shell.execute_reply":"2024-12-19T14:46:51.093722Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Xử lý cột ID\nid_column = df_test['id']\ndf_test = df_test.drop(columns=['id'])\ndf = df.drop(columns=['id'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T14:46:51.095741Z","iopub.execute_input":"2024-12-19T14:46:51.096025Z","iopub.status.idle":"2024-12-19T14:46:51.101841Z","shell.execute_reply.started":"2024-12-19T14:46:51.095999Z","shell.execute_reply":"2024-12-19T14:46:51.101070Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Xử lý cột mục tiêu\ntarget = df.pop('sii')\ndf = df[target.notna()]\ntarget = target[target.notna()]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T14:46:51.102890Z","iopub.execute_input":"2024-12-19T14:46:51.103254Z","iopub.status.idle":"2024-12-19T14:46:51.114676Z","shell.execute_reply.started":"2024-12-19T14:46:51.103226Z","shell.execute_reply":"2024-12-19T14:46:51.114023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# encode object type\nfor column in df.columns:\n    if df[column].dtype == object:\n        df[column], _ = pd.factorize(df[column])\nfor column in df_test.columns:\n    if df_test[column].dtype == object:\n        df_test[column], _ = pd.factorize(df_test[column])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T14:46:51.115652Z","iopub.execute_input":"2024-12-19T14:46:51.115888Z","iopub.status.idle":"2024-12-19T14:46:51.135972Z","shell.execute_reply.started":"2024-12-19T14:46:51.115864Z","shell.execute_reply":"2024-12-19T14:46:51.135131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Điền giá trị thiếu\ndf.fillna(df.median(), inplace=True)\ndf_test.fillna(df_test.median(), inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T14:46:51.137689Z","iopub.execute_input":"2024-12-19T14:46:51.137927Z","iopub.status.idle":"2024-12-19T14:46:51.195137Z","shell.execute_reply.started":"2024-12-19T14:46:51.137903Z","shell.execute_reply":"2024-12-19T14:46:51.194284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Đồng bộ hóa cột giữa df và df_test\n#common_columns = df.columns.intersection(df_test.columns)\ncommon_columns = ['SDS-SDS_Total_Raw', 'Basic_Demos-Age', 'SDS-SDS_Total_T', 'Physical-Height',\n                  'Physical-Weight', 'PreInt_EduHx-computerinternet_hoursday', 'Physical-BMI',\n                  'Physical-HeartRate', 'Physical-Systolic_BP', 'CGAS-CGAS_Score',\n                  'Physical-Diastolic_BP', 'PAQ_C-PAQ_C_Total', 'FGC-FGC_CU',\n                  'BIA-BIA_LDM', 'BIA-BIA_DEE', 'BIA-BIA_ICW', 'BIA-BIA_FFMI',\n                  'BIA-BIA_BMC', 'BIA-BIA_LST', 'BIA-BIA_ECW', 'BIA-BIA_SMM',\n                  'FGC-FGC_SRR', 'BIA-BIA_FFM', 'BIA-BIA_BMR', 'BIA-BIA_Fat',\n                  'FGC-FGC_SRL', 'BIA-BIA_FMI', 'BIA-BIA_TBW', 'FGC-FGC_TL',\n                  'CGAS-Season', 'FGC-FGC_PU', 'FGC-Season', 'BIA-BIA_BMI',\n                  'FGC-FGC_GSD', 'FGC-FGC_GSND', 'SDS-Season', 'Physical-Season',\n                  'PAQ_C-Season', 'PreInt_EduHx-Season', 'Basic_Demos-Enroll_Season',\n                  'Fitness_Endurance-Time_Sec', 'Fitness_Endurance-Season']\ndf = df[common_columns]\ndf_test = df_test[common_columns]\ndf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T14:46:51.196061Z","iopub.execute_input":"2024-12-19T14:46:51.196294Z","iopub.status.idle":"2024-12-19T14:46:51.228354Z","shell.execute_reply.started":"2024-12-19T14:46:51.196270Z","shell.execute_reply":"2024-12-19T14:46:51.227559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Chuẩn hóa dữ liệu\nscaler = StandardScaler()\nX = scaler.fit_transform(df)\nX_test = scaler.transform(df_test)\ny = target.values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T14:46:51.229339Z","iopub.execute_input":"2024-12-19T14:46:51.229579Z","iopub.status.idle":"2024-12-19T14:46:51.241098Z","shell.execute_reply.started":"2024-12-19T14:46:51.229550Z","shell.execute_reply":"2024-12-19T14:46:51.240403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cân bằng dữ liệu với SMOTE\nsmote = SMOTE(k_neighbors=5)\nX, y = smote.fit_resample(X, y)\nX, y","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T14:46:51.242218Z","iopub.execute_input":"2024-12-19T14:46:51.242463Z","iopub.status.idle":"2024-12-19T14:46:51.263034Z","shell.execute_reply.started":"2024-12-19T14:46:51.242439Z","shell.execute_reply":"2024-12-19T14:46:51.261405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#optuna: params tunning\n#def objective(\n#     trial: optuna.Trial,\n# ) -> float:\n#     \"\"\"Objective function for optuna optimisation.\"\"\"\n#     params = {\n#         \"boosting_type\": \"gbdt\",\n#         \"learning_rate\": trial.suggest_float(\"learning_rate\", 0.1, 0.3, step=0.01),\n#         \"min_data_in_leaf\": trial.suggest_int(\"min_data_in_leaf\", 5, 45, step=10),\n#         \"max_depth\": 10,\n#         \"max_leave\": 100,\n#         \"bagging_fraction\" : trial.suggest_float(\"bagging_fraction\", 0.5, 0.9, step=0.1),\n#         \"bagginf_freq\" : trial.suggest_int(\"bagging_freq\", 1, 3, step=1),\n#         \"feature_fraction\": trial.suggest_float(\"feature_fraction\", 0.5, 0.9, step=0.1),\n#        \"verbose\" : -1,\n#     }\n    \n     #kfold\n#     kappas = []\n#     kf = StratifiedKFold(n_splits=5, shuffle=True)\n#     for fold, (train_idx, val_idx) in enumerate(kf.split(X, y)):\n#         X_train, X_val = X[train_idx], X[val_idx]\n#         y_train, y_val = y[train_idx], y[val_idx]\n\n#         train_data = lgb.Dataset(X_train, y_train)\n#         val_data = lgb.Dataset(X_val, y_val)\n#         model_lgb = lgb.train(params, train_data)\n\n    \n#         y_pred = np.round(model_lgb.predict(X_val), 0)\n#         kappa = cohen_kappa_score(y_val, y_pred, weights='quadratic')\n#         kappas.append(kappa)\n#     return np.mean(kappas)\n\n     #single fit\n     # classifier = LGBMClassifier(**params)\n     # classifier.fit(X_train, y_train)\n     # y_pred = classifier.predict(X_val)\n     # return cohen_kappa_score(y_val, y_pred, weights='quadratic')\n#objective_func = lambda trial: objective(\n#     trial,\n#)\n\n # Run the optimisation\n#study = optuna.create_study(direction='maximize')\n#study.optimize(objective_func, n_trials=50)\n\n # Get the best hyperparameters\n#print(study.best_params)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T14:46:51.265042Z","iopub.execute_input":"2024-12-19T14:46:51.265290Z","iopub.status.idle":"2024-12-19T14:46:51.270562Z","shell.execute_reply.started":"2024-12-19T14:46:51.265263Z","shell.execute_reply":"2024-12-19T14:46:51.268813Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"params = {\n    \"boosting_type\" : \"gbdt\", 'verbose' : -1, 'max_bin': 255,\n    'learning_rate': 0.27, 'min_data_in_leaf': 5, 'bagging_fraction': 0.7,\n    'bagging_freq': 3, 'feature_fraction': 0.7\n}\n#LGBM default 0.8778305601985833 base\n#model = LGBMClassifier(verbose=-1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T14:46:51.271814Z","iopub.execute_input":"2024-12-19T14:46:51.272180Z","iopub.status.idle":"2024-12-19T14:46:51.281941Z","shell.execute_reply.started":"2024-12-19T14:46:51.272144Z","shell.execute_reply":"2024-12-19T14:46:51.281320Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# #8:2 split training\n# X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2)\n\n# #lgb train\n# train_data = lgb.Dataset(X_train, y_train)\n# val_data = lgb.Dataset(X_val, y_val)\n# model_lgb = lgb.train(params, train_data)\n\n# #Skit fit\n# # model_lgb = LGBMClassifier(**params)\n# # model_lgb.fit(X_train, y_train)\n\n# y_pred = np.round(model_lgb.predict(X_val), 0)\n# kappa = cohen_kappa_score(y_val, y_pred, weights='quadratic')\n# print(kappa)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T14:46:51.283050Z","iopub.execute_input":"2024-12-19T14:46:51.283569Z","iopub.status.idle":"2024-12-19T14:46:51.294605Z","shell.execute_reply.started":"2024-12-19T14:46:51.283524Z","shell.execute_reply":"2024-12-19T14:46:51.293987Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#kfold train\nmodel_lgb = LGBMClassifier(**params)\n\nkappas = []\nkf = StratifiedKFold(n_splits=5, shuffle=True)\nfor fold, (train_idx, val_idx) in enumerate(kf.split(X, y)):\n    X_train, X_val = X[train_idx], X[val_idx]\n    y_train, y_val = y[train_idx], y[val_idx]\n\n    # train_data = lgb.Dataset(X_train, y_train)\n    # val_data = lgb.Dataset(X_val, y_val)\n    # model_lgb = lgb.train(params, train_data)\n\n    # y_pred = np.round(model_lgb.predict(X_val), 0)\n    # y_pred = np.where(y_pred == -0. , 0. , y_pred)\n\n    model_lgb.fit(X_train, y_train)\n    y_pred = model_lgb.predict(X_val)\n    # print(y_pred)\n    kappa = cohen_kappa_score(y_val, y_pred, weights='quadratic')\n    kappas.append(kappa)\n\nprint (np.mean(kappas))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T14:46:51.295318Z","iopub.execute_input":"2024-12-19T14:46:51.295519Z","iopub.status.idle":"2024-12-19T14:46:56.283844Z","shell.execute_reply.started":"2024-12-19T14:46:51.295497Z","shell.execute_reply":"2024-12-19T14:46:56.282943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Huấn luyện mô hình Random Forest\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2)\nrf_model = RandomForestClassifier(random_state=42, n_estimators=100)\nrf_model.fit(X_train, y_train)\ny_pred_rf = rf_model.predict(X_val)\nkappa = cohen_kappa_score(y_val, y_pred_rf, weights='quadratic')\nprint(kappa)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T14:46:56.285054Z","iopub.execute_input":"2024-12-19T14:46:56.285367Z","iopub.status.idle":"2024-12-19T14:46:58.196727Z","shell.execute_reply.started":"2024-12-19T14:46:56.285338Z","shell.execute_reply":"2024-12-19T14:46:58.195835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#RF Kfold\n#for fold, (train_idx, val_idx) in enumerate(kf.split(X, y)):\n#    X_train, X_val = X[train_idx], X[val_idx]\n#    y_train, y_val = y[train_idx], y[val_idx]\n\n    # train_data = lgb.Dataset(X_train, y_train)\n    # val_data = lgb.Dataset(X_val, y_val)\n    # model_lgb = lgb.train(params, train_data)\n\n    # y_pred = np.round(model_lgb.predict(X_val), 0)\n    # y_pred = np.where(y_pred == -0. , 0. , y_pred)\n\n#    rf_model.fit(X_train, y_train)\n#    y_pred = rf_model.predict(X_val)\n    # print(y_pred)\n#    kappa = cohen_kappa_score(y_val, y_pred, weights='quadratic')\n#    kappas.append(kappa)\n\n#print (np.mean(kappas))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T14:46:58.197706Z","iopub.execute_input":"2024-12-19T14:46:58.197989Z","iopub.status.idle":"2024-12-19T14:46:58.201789Z","shell.execute_reply.started":"2024-12-19T14:46:58.197962Z","shell.execute_reply":"2024-12-19T14:46:58.200997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def objective_weights(\n#     trial: optuna.Trial,\n# ) -> float:\n#     params = {\n#         \"weight_lgb\" : trial.suggest_float(\"weight_lgb\", 1.0, 2.0, step=0.1),\n#         \"weight_rf\" : trial.suggest_float(\"weight_rf\", 1.0, 2.0, step=0.1)\n#     }\n#     weight_lgb = params[\"weight_lgb\"]\n#     weight_rf = params[\"weight_rf\"]\n\n#     voting_clf = VotingClassifier(\n#         estimators=[\n#         ('lightgbm', model_lgb),\n#         ('random_forest', rf_model)\n#         ],weights=[weight_lgb, weight_rf]\n#     )\n\n#     voting_clf.fit(X_train, y_train)\n#     y_pred_ens = voting_clf.predict(X_val)\n#     kappa = cohen_kappa_score(y_val, y_pred_ens, weights='quadratic')\n#     return kappa\n\n# objective_func = lambda trial: objective_weights(\n#     trial,\n# )\n\n# study = optuna.create_study(direction='maximize')\n# study.optimize(objective_func, n_trials=50)\n\n# print(study.best_params)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T14:46:58.202843Z","iopub.execute_input":"2024-12-19T14:46:58.203155Z","iopub.status.idle":"2024-12-19T14:46:58.217608Z","shell.execute_reply.started":"2024-12-19T14:46:58.203128Z","shell.execute_reply":"2024-12-19T14:46:58.216935Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2)\nvoting_clf = VotingClassifier(\n    estimators=[\n    ('lightgbm', model_lgb),\n    ('random_forest', rf_model)\n    ],weights=[1.0, 1.0]\n)\n\nvoting_clf.fit(X_train, y_train)\ny_pred_ens = voting_clf.predict(X_val)\nkappa = cohen_kappa_score(y_val, y_pred_ens, weights='quadratic')\nprint(kappa)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T14:46:58.218397Z","iopub.execute_input":"2024-12-19T14:46:58.218692Z","iopub.status.idle":"2024-12-19T14:47:01.085706Z","shell.execute_reply.started":"2024-12-19T14:46:58.218667Z","shell.execute_reply":"2024-12-19T14:47:01.084811Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = pd.read_csv(test_file)\ntest","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T14:47:01.086877Z","iopub.execute_input":"2024-12-19T14:47:01.087249Z","iopub.status.idle":"2024-12-19T14:47:01.120434Z","shell.execute_reply.started":"2024-12-19T14:47:01.087210Z","shell.execute_reply":"2024-12-19T14:47:01.119612Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T14:47:01.121596Z","iopub.execute_input":"2024-12-19T14:47:01.121935Z","iopub.status.idle":"2024-12-19T14:47:01.159212Z","shell.execute_reply.started":"2024-12-19T14:47:01.121899Z","shell.execute_reply":"2024-12-19T14:47:01.158366Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# y_pred_test = np.round(model_lgb.predict(X_test), 0)\n# y_pred_test = np.where(y_pred_test == -0. , 0. , y_pred_test)\n# y_pred_test\n#votingclf\ny_pred_test = voting_clf.predict(X_test)\ny_pred_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T14:47:01.160199Z","iopub.execute_input":"2024-12-19T14:47:01.160467Z","iopub.status.idle":"2024-12-19T14:47:01.174675Z","shell.execute_reply.started":"2024-12-19T14:47:01.160438Z","shell.execute_reply":"2024-12-19T14:47:01.173811Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#lgbmClf\n# y_pred_test = voting_clf.predict(X_test)\n# y_pred_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T14:47:01.177107Z","iopub.execute_input":"2024-12-19T14:47:01.177386Z","iopub.status.idle":"2024-12-19T14:47:01.181550Z","shell.execute_reply.started":"2024-12-19T14:47:01.177360Z","shell.execute_reply":"2024-12-19T14:47:01.180791Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test['sii'] = y_pred_test\ntest","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T14:47:01.182502Z","iopub.execute_input":"2024-12-19T14:47:01.183201Z","iopub.status.idle":"2024-12-19T14:47:01.219806Z","shell.execute_reply.started":"2024-12-19T14:47:01.183172Z","shell.execute_reply":"2024-12-19T14:47:01.218921Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = test[['id', 'sii']]\nprint(submission)\nsubmission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T14:47:01.220900Z","iopub.execute_input":"2024-12-19T14:47:01.221276Z","iopub.status.idle":"2024-12-19T14:47:01.229588Z","shell.execute_reply.started":"2024-12-19T14:47:01.221235Z","shell.execute_reply":"2024-12-19T14:47:01.228610Z"}},"outputs":[],"execution_count":null}]}