{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ## This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-12-13T15:30:59.739136Z","iopub.execute_input":"2024-12-13T15:30:59.740105Z","iopub.status.idle":"2024-12-13T15:30:59.744648Z","shell.execute_reply.started":"2024-12-13T15:30:59.740066Z","shell.execute_reply":"2024-12-13T15:30:59.743641Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T15:30:59.752280Z","iopub.execute_input":"2024-12-13T15:30:59.752926Z","iopub.status.idle":"2024-12-13T15:30:59.764563Z","shell.execute_reply.started":"2024-12-13T15:30:59.752839Z","shell.execute_reply":"2024-12-13T15:30:59.763481Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T15:30:59.766392Z","iopub.execute_input":"2024-12-13T15:30:59.766732Z","iopub.status.idle":"2024-12-13T15:30:59.829730Z","shell.execute_reply.started":"2024-12-13T15:30:59.766699Z","shell.execute_reply":"2024-12-13T15:30:59.828704Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.columns\n\n\n\n# Tạo ánh xạ giá trị\nseason_mapping = {\n    'Spring': 1,\n    'Summer': 2,\n    'Fall': 3,\n    'Winter': 4\n}\n\n# Tìm các cột chứa \"season\" (không phân biệt chữ hoa thường)\nseason_columns = [col for col in train.columns if 'season' in col.lower()]\n\n# Ánh xạ giá trị trong các cột này\nfor col in season_columns:\n    train[col] = train[col].map(season_mapping)\n\nprint(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T15:30:59.831175Z","iopub.execute_input":"2024-12-13T15:30:59.831520Z","iopub.status.idle":"2024-12-13T15:30:59.865220Z","shell.execute_reply.started":"2024-12-13T15:30:59.831485Z","shell.execute_reply":"2024-12-13T15:30:59.864007Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Bo het cot co Season\ntrain.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T15:30:59.866685Z","iopub.execute_input":"2024-12-13T15:30:59.867148Z","iopub.status.idle":"2024-12-13T15:30:59.875886Z","shell.execute_reply.started":"2024-12-13T15:30:59.867100Z","shell.execute_reply":"2024-12-13T15:30:59.874779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"correlation_matrix = train.corr(numeric_only=True)\n\n# Lấy cột 'sii' từ ma trận tương quan\nsii_correlation = correlation_matrix['sii']\npd.set_option('display.max_rows', None)  # Hiển thị tất cả các dòng\n\nprint(sii_correlation)\n\n# Lọc các giá trị có hệ số tương quan lớn hơn 0.04\ntrain = train[sii_correlation[sii_correlation > 0.15].index.tolist()]\npd.reset_option('display.max_rows')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T15:30:59.878793Z","iopub.execute_input":"2024-12-13T15:30:59.879274Z","iopub.status.idle":"2024-12-13T15:30:59.968263Z","shell.execute_reply.started":"2024-12-13T15:30:59.879236Z","shell.execute_reply":"2024-12-13T15:30:59.967130Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"So feature sau khi loc theo correlation: {len(train.columns)}\")\ntrain.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T15:30:59.969799Z","iopub.execute_input":"2024-12-13T15:30:59.970539Z","iopub.status.idle":"2024-12-13T15:30:59.979350Z","shell.execute_reply.started":"2024-12-13T15:30:59.970490Z","shell.execute_reply":"2024-12-13T15:30:59.978116Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Loại bỏ giá trị null \nnull_ratio = train.isnull().mean()\n# null_ratio = null_ratio.sort_values(ascending=False) Không cần sort chỗ này\nfiltered_columns = null_ratio[null_ratio <= 0.6].index\ntrain = train[filtered_columns]\nprint(f\"So features sau khi loc {len(train.columns)}\")\nfiltered_columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T15:30:59.980725Z","iopub.execute_input":"2024-12-13T15:30:59.981143Z","iopub.status.idle":"2024-12-13T15:30:59.996144Z","shell.execute_reply.started":"2024-12-13T15:30:59.981077Z","shell.execute_reply":"2024-12-13T15:30:59.994998Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train_data_filtered  = train[filtered_columns]\n# train_data_filtered = train_data_filtered.dropna(subset=['sii'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T15:30:59.997565Z","iopub.execute_input":"2024-12-13T15:30:59.998040Z","iopub.status.idle":"2024-12-13T15:31:00.007719Z","shell.execute_reply.started":"2024-12-13T15:30:59.997993Z","shell.execute_reply":"2024-12-13T15:31:00.006501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\n\n# Thay thế giá trị Null với giá trị trung bình (cho dữ liệu số)\nimputer = SimpleImputer(strategy='mean')\nimputed_data = imputer.fit_transform(train)\ntrain = pd.DataFrame(imputed_data, columns=train.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T15:31:00.009251Z","iopub.execute_input":"2024-12-13T15:31:00.009698Z","iopub.status.idle":"2024-12-13T15:31:00.029429Z","shell.execute_reply.started":"2024-12-13T15:31:00.009650Z","shell.execute_reply":"2024-12-13T15:31:00.028380Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T15:31:00.031494Z","iopub.execute_input":"2024-12-13T15:31:00.031828Z","iopub.status.idle":"2024-12-13T15:31:00.054567Z","shell.execute_reply.started":"2024-12-13T15:31:00.031794Z","shell.execute_reply":"2024-12-13T15:31:00.053391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Loại bỏ các cột không cần thiết (PCIAT) trong dữ liệu\nfeature_columns = [col for col in train if col.split('-')[0] != 'PCIAT' and col != 'sii']\nlabels = train['sii'].astype('int32')\ndf = train[feature_columns]\n\nprint(len(feature_columns))\ndf.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T15:31:00.056577Z","iopub.execute_input":"2024-12-13T15:31:00.056993Z","iopub.status.idle":"2024-12-13T15:31:00.082416Z","shell.execute_reply.started":"2024-12-13T15:31:00.056950Z","shell.execute_reply":"2024-12-13T15:31:00.081039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.datasets import make_regression\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error, cohen_kappa_score\nfrom sklearn.model_selection import GridSearchCV\nimport xgboost as xgb\nfrom sklearn.metrics import make_scorer\n\n# Tạo dữ liệu mẫu\nX = df\ny = labels\n\n# Chia dữ liệu thành tập huấn luyện và kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Chuyển đổi dữ liệu thành định dạng DMatrix của XGBoost\ndtrain = xgb.DMatrix(X_train, label=y_train)\ndtest = xgb.DMatrix(X_test, label=y_test)\n\nparams = {\n    \"objective\": \"multi:softmax\",  # Phân loại đa lớp\n    \"num_class\": 4,                # Số lớp (ví dụ có 3 lớp)\n    \"max_depth\": 6,\n    \"eta\": 0.1,\n    \"lambda\": 10,\n    \"eval_metric\": \"merror\"        # Metric đánh giá cho phân loại đa lớp (accuracy)\n}\n\n# Huấn luyện mô hình\nwatchlist = [(dtrain, \"train\"), (dtest, \"eval\")]\nmodel = xgb.train(params, dtrain, num_boost_round=100, evals=watchlist, early_stopping_rounds=10)\n\n# Dự đoán và đánh giá\npredictions = model.predict(dtest)\n# Chuyển predictions về dạng integer (vì \"multi:softmax\" dự đoán giá trị nguyên)\npredictions = predictions.astype(int)\n\n# Tính Cohen's Kappa\nkappa = cohen_kappa_score(y_test, predictions)\nprint(f\"Cohen's Kappa trên tập kiểm tra: {kappa:.4f}\")\n\n# Xuất mô hình thành file nếu cần\nmodel.save_model(\"xgboost_model_with_l2.json\")\n\n# Tìm kiếm tham số tối ưu bằng Grid Search cho XGBoost\nxgb_model = xgb.XGBClassifier(objective=\"multi:softmax\", num_class=4, eval_metric=\"merror\")\n\nparam_grid = {\n    'max_depth': [ 6, 7, 8],\n    'eta': [0.01, 0.1, 0.2],\n    'lambda': [1, 5, 10]\n}\n\n# Tạo scorer sử dụng Cohen's Kappa\nkappa_scorer = make_scorer(cohen_kappa_score)\n\n# Tìm kiếm tham số tối ưu sử dụng Kappa\ngrid_search = GridSearchCV(estimator=xgb_model, param_grid=param_grid, scoring=kappa_scorer, cv=5, verbose=1)\ngrid_search.fit(X_train, y_train)\n\n# In kết quả tốt nhất\nprint(\"Best parameters found: \", grid_search.best_params_)\nprint(\"Best Kappa score: \", grid_search.best_score_)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T15:31:00.084179Z","iopub.execute_input":"2024-12-13T15:31:00.084647Z","iopub.status.idle":"2024-12-13T15:32:10.440654Z","shell.execute_reply.started":"2024-12-13T15:31:00.084594Z","shell.execute_reply":"2024-12-13T15:32:10.439437Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Bien doi du lieu test.csv\n\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n\nfrom sklearn.impute import SimpleImputer\n\n\n# Bỏ cột giá trị string nhưng tập test yêu cầu đầu ra phải có id nên nhớ lưu id lại\nid_columns = test['id']\n#loc \n\n\n# Tìm các cột chứa \"season\" (không phân biệt chữ hoa thường)\nseason_columns = [col for col in test.columns if 'season' in col.lower()]\n\n# Ánh xạ giá trị trong các cột này\nfor col in season_columns:\n    test[col] = test[col].map(season_mapping)\nprint(test.shape)\n#\n\n\ntest = test[feature_columns]\n\n\n# Thay thế giá trị Null với giá trị trung bình (cho dữ liệu số)\nimputer = SimpleImputer(strategy='mean')\nimputed_data = imputer.fit_transform(test)\ntest = pd.DataFrame(imputed_data, columns=test.columns)\n\n# Chuyển từ DataFrame sang DMatrix cho XGBoost\n\n\n# Dự đoán kết quả bằng mô hình XGBoost đã huấn luyện\ny_pre = grid_search.predict(test)\nresult = pd.DataFrame({\n    'id': id_columns,\n    'sii': y_pre\n})\nresult.to_csv('submission.csv',sep = ',', index=False)\n\n\nresult\n# Chỗ này hệ số tương quan của tập test sẽ khác tập train nên các feature bị cắt có thể khác nhau.\n# Chô này phải lấy các feature đã được chọn lúc train lấy feature_columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T15:34:12.455389Z","iopub.execute_input":"2024-12-13T15:34:12.456647Z","iopub.status.idle":"2024-12-13T15:34:12.515224Z","shell.execute_reply.started":"2024-12-13T15:34:12.456592Z","shell.execute_reply":"2024-12-13T15:34:12.513948Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}