{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ## This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-12-06T17:01:57.183163Z","iopub.execute_input":"2024-12-06T17:01:57.183620Z","iopub.status.idle":"2024-12-06T17:01:57.189370Z","shell.execute_reply.started":"2024-12-06T17:01:57.183581Z","shell.execute_reply":"2024-12-06T17:01:57.188272Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T17:01:57.194298Z","iopub.execute_input":"2024-12-06T17:01:57.194845Z","iopub.status.idle":"2024-12-06T17:01:57.209211Z","shell.execute_reply.started":"2024-12-06T17:01:57.194795Z","shell.execute_reply":"2024-12-06T17:01:57.207992Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T17:01:57.211502Z","iopub.execute_input":"2024-12-06T17:01:57.212407Z","iopub.status.idle":"2024-12-06T17:01:57.281014Z","shell.execute_reply.started":"2024-12-06T17:01:57.212353Z","shell.execute_reply":"2024-12-06T17:01:57.279743Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T17:01:57.282814Z","iopub.execute_input":"2024-12-06T17:01:57.283502Z","iopub.status.idle":"2024-12-06T17:01:57.291497Z","shell.execute_reply.started":"2024-12-06T17:01:57.283451Z","shell.execute_reply":"2024-12-06T17:01:57.290201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Bo het cot co Season\ntrain = train.loc[: ,~train.columns.str.contains('season', case=False)]\ntrain.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T17:01:57.293761Z","iopub.execute_input":"2024-12-06T17:01:57.294152Z","iopub.status.idle":"2024-12-06T17:01:57.307062Z","shell.execute_reply.started":"2024-12-06T17:01:57.294118Z","shell.execute_reply":"2024-12-06T17:01:57.305982Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"correlation_matrix = train.corr(numeric_only=True)\npd.set_option('display.max_rows', None)  # Không giới hạn số dòng hiển thị\n\n# Lấy cột 'sii' từ ma trận tương quan\nsii_correlation = correlation_matrix['sii']\n\n# Lọc các giá trị có hệ số tương quan lớn hơn 0.05\ntrain = train[sii_correlation[sii_correlation > 0.05].index.tolist()]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T17:01:57.308483Z","iopub.execute_input":"2024-12-06T17:01:57.308951Z","iopub.status.idle":"2024-12-06T17:01:57.377060Z","shell.execute_reply.started":"2024-12-06T17:01:57.308900Z","shell.execute_reply":"2024-12-06T17:01:57.376001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"So feature sau khi loc theo correlation: {len(train.columns)}\")\ntrain.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T17:01:57.378450Z","iopub.execute_input":"2024-12-06T17:01:57.378935Z","iopub.status.idle":"2024-12-06T17:01:57.387582Z","shell.execute_reply.started":"2024-12-06T17:01:57.378886Z","shell.execute_reply":"2024-12-06T17:01:57.386573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Loại bỏ giá trị null \nnull_ratio = train.isnull().mean()\n# null_ratio = null_ratio.sort_values(ascending=False) Không cần sort chỗ này\nfiltered_columns = null_ratio[null_ratio <= 0.6].index\ntrain = train[filtered_columns]\nprint(f\"So features sau khi loc {len(train.columns)}\")\nfiltered_columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T17:01:57.388979Z","iopub.execute_input":"2024-12-06T17:01:57.389335Z","iopub.status.idle":"2024-12-06T17:01:57.404324Z","shell.execute_reply.started":"2024-12-06T17:01:57.389300Z","shell.execute_reply":"2024-12-06T17:01:57.403073Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train_data_filtered  = train[filtered_columns]\n# train_data_filtered = train_data_filtered.dropna(subset=['sii'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T17:01:57.405607Z","iopub.execute_input":"2024-12-06T17:01:57.405925Z","iopub.status.idle":"2024-12-06T17:01:57.416754Z","shell.execute_reply.started":"2024-12-06T17:01:57.405894Z","shell.execute_reply":"2024-12-06T17:01:57.415596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\n\n# Thay thế giá trị Null với giá trị trung bình (cho dữ liệu số)\nimputer = SimpleImputer(strategy='mean')\nimputed_data = imputer.fit_transform(train)\ntrain = pd.DataFrame(imputed_data, columns=train.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T17:01:57.419414Z","iopub.execute_input":"2024-12-06T17:01:57.419862Z","iopub.status.idle":"2024-12-06T17:01:57.441258Z","shell.execute_reply.started":"2024-12-06T17:01:57.419809Z","shell.execute_reply":"2024-12-06T17:01:57.439961Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T17:01:57.442397Z","iopub.execute_input":"2024-12-06T17:01:57.442732Z","iopub.status.idle":"2024-12-06T17:01:57.464808Z","shell.execute_reply.started":"2024-12-06T17:01:57.442700Z","shell.execute_reply":"2024-12-06T17:01:57.463757Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Loại bỏ các cột không cần thiết (PCIAT) trong dữ liệu\nfeature_columns = [col for col in train if col.split('-')[0] != 'PCIAT' and col != 'sii']\nlabels = train['sii'].astype('int32')\ndf = train[feature_columns]\ndf.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T17:01:57.465948Z","iopub.execute_input":"2024-12-06T17:01:57.466256Z","iopub.status.idle":"2024-12-06T17:01:57.492348Z","shell.execute_reply.started":"2024-12-06T17:01:57.466225Z","shell.execute_reply":"2024-12-06T17:01:57.490826Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score, classification_report\nfrom sklearn.model_selection import train_test_split\n\n# Chia dữ liệu thành đặc trưng (X) và nhãn (y)\nX = df\ny = labels\n\n# Chia dữ liệu thành tập huấn luyện và kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, train_size=0.8, random_state=42)\n\n# Khởi tạo mô hình Random Forest\nmodel = RandomForestClassifier(n_estimators=100, max_depth=10, random_state=42)\n\n# Huấn luyện mô hình với dữ liệu huấn luyện\nmodel.fit(X_train, y_train)\n\n# Dự đoán trên tập kiểm tra\ny_pred = model.predict(X_test)\n\n# # Đánh giá mô hình\naccuracy = accuracy_score(y_test, y_pred)\nprint(f\"Accuracy: {accuracy:.4f}\")\n\n# # In báo cáo phân loại\n# print(classification_report(y_test, y_pred))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T17:01:57.493997Z","iopub.execute_input":"2024-12-06T17:01:57.494357Z","iopub.status.idle":"2024-12-06T17:01:58.112636Z","shell.execute_reply.started":"2024-12-06T17:01:57.494323Z","shell.execute_reply":"2024-12-06T17:01:58.111432Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Bien doi du lieu test.csv\n# Bỏ cột giá trị string nhưng tập test yêu cầu đầu ra phải có id nên nhớ lưu id lại\nid_columns = test['id']\ntest = train.loc[: ,~train.columns.str.contains('season', case=False)]\ntest = test[feature_columns]\n\nmodel.predict(test)\n# Chỗ này hệ số tương quan của tập test sẽ khác tập train nên các feature bị cắt có thể khác nhau.\n# Chô này phải lấy các feature đã được chọn lúc train lấy feature_columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T17:01:58.113768Z","iopub.execute_input":"2024-12-06T17:01:58.114074Z","iopub.status.idle":"2024-12-06T17:01:58.181004Z","shell.execute_reply.started":"2024-12-06T17:01:58.114043Z","shell.execute_reply":"2024-12-06T17:01:58.179872Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}