{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ## This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-12-06T06:54:51.284045Z","iopub.execute_input":"2024-12-06T06:54:51.284640Z","iopub.status.idle":"2024-12-06T06:54:51.290162Z","shell.execute_reply.started":"2024-12-06T06:54:51.284583Z","shell.execute_reply":"2024-12-06T06:54:51.289009Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T06:54:51.292282Z","iopub.execute_input":"2024-12-06T06:54:51.292728Z","iopub.status.idle":"2024-12-06T06:54:51.305349Z","shell.execute_reply.started":"2024-12-06T06:54:51.292652Z","shell.execute_reply":"2024-12-06T06:54:51.304131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T06:54:51.306466Z","iopub.execute_input":"2024-12-06T06:54:51.307004Z","iopub.status.idle":"2024-12-06T06:54:51.363410Z","shell.execute_reply.started":"2024-12-06T06:54:51.306961Z","shell.execute_reply":"2024-12-06T06:54:51.362161Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_nas(df: pd.DataFrame):\n    # Calculate the missing and available data ratios\n    na_df = (df.isnull().sum() / len(df)) * 100  # Missing ratio in percentage\n    na_df = na_df.sort_values(ascending=False)  # Sort values\n    \n    available_df = 100 - na_df  # Available ratio in percentage\n    \n    # Create a horizontal stacked bar chart\n    plot_width, plot_height = (16, 18)\n    plt.rcParams['figure.figsize'] = (plot_width, plot_height)\n    \n    fig, ax = plt.subplots()\n    bar_height = 0.4  # Set bar height to make them thinner\n    y_pos = np.arange(len(na_df))  # Positions for bars\n    \n    ax.barh(y_pos, na_df, color='salmon', label='Missing Ratio (%)', height=bar_height)\n    ax.barh(y_pos, available_df, left=na_df, color='lightgreen', label='Available Ratio (%)', height=bar_height)\n    \n    ax.set_yticks(y_pos)\n    ax.set_yticklabels(na_df.index)\n    ax.set_xlabel('Percentage')\n    ax.set_title('Missing and Available Data Ratios')\n    ax.legend()\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T06:54:51.365467Z","iopub.execute_input":"2024-12-06T06:54:51.365809Z","iopub.status.idle":"2024-12-06T06:54:51.373691Z","shell.execute_reply.started":"2024-12-06T06:54:51.365777Z","shell.execute_reply":"2024-12-06T06:54:51.372585Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" plot_nas(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T06:54:51.375453Z","iopub.execute_input":"2024-12-06T06:54:51.375813Z","iopub.status.idle":"2024-12-06T06:54:52.697031Z","shell.execute_reply.started":"2024-12-06T06:54:51.375778Z","shell.execute_reply":"2024-12-06T06:54:52.695796Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nprint(\"Features:\", train.columns.tolist())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T06:54:52.698439Z","iopub.execute_input":"2024-12-06T06:54:52.698816Z","iopub.status.idle":"2024-12-06T06:54:52.704824Z","shell.execute_reply.started":"2024-12-06T06:54:52.698780Z","shell.execute_reply":"2024-12-06T06:54:52.703689Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ntrain_columns = [  'Basic_Demos-Age',\n                 'Basic_Demos-Sex',  'CGAS-CGAS_Score',\n                  'Physical-BMI', 'Physical-Height', \n                 'Physical-Weight', 'Physical-Waist_Circumference', 'Physical-Diastolic_BP', \n                 'Physical-HeartRate', 'Physical-Systolic_BP',\n                 'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins',\n                 'Fitness_Endurance-Time_Sec',  'FGC-FGC_CU',\n                 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', \n                 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone',\n                 'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                 'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE',\n                 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW',  \n                 'PAQ_A-PAQ_A_Total',  'PAQ_C-PAQ_C_Total', 'PCIAT-PCIAT_01',\n                 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03', 'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06', \n                 'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11', \n                 'PCIAT-PCIAT_12', 'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15', 'PCIAT-PCIAT_16', \n                 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18', 'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20', 'PCIAT-PCIAT_Total', \n                 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T',  \n                 'PreInt_EduHx-computerinternet_hoursday', 'sii']\nprint(len(train_columns))\nselected_train = train[train_columns]\ncorrelation_matrix = selected_train.corr()\npd.set_option('display.max_rows', None)  # Không giới hạn số dòng hiển thị\n\n# Lấy cột 'sii' từ ma trận tương quan\nsii_correlation = correlation_matrix['sii']\n\n# Lọc các giá trị có hệ số tương quan lớn hơn 0.05\nsii_correlation_filtered = sii_correlation[sii_correlation > 0.05]\nfiltered_features = sii_correlation_filtered.index.tolist()\n\n\n# In ra các kết quả\nprint(sii_correlation_filtered)\nprint(len(sii_correlation_filtered))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T06:54:52.707183Z","iopub.execute_input":"2024-12-06T06:54:52.707597Z","iopub.status.idle":"2024-12-06T06:54:52.782394Z","shell.execute_reply.started":"2024-12-06T06:54:52.707544Z","shell.execute_reply":"2024-12-06T06:54:52.781254Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = train[filtered_features]\npd.reset_option('display.max_rows')\n\nplot_nas(train_data)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T06:54:52.783518Z","iopub.execute_input":"2024-12-06T06:54:52.783815Z","iopub.status.idle":"2024-12-06T06:54:53.686921Z","shell.execute_reply.started":"2024-12-06T06:54:52.783786Z","shell.execute_reply":"2024-12-06T06:54:53.685710Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"null_ratio = train_data.isnull().mean()\nnull_ratio = null_ratio.sort_values(ascending=False)\nfiltered_columns = null_ratio[null_ratio <= 0.6].index\n\nprint(null_ratio)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T06:54:53.688374Z","iopub.execute_input":"2024-12-06T06:54:53.688737Z","iopub.status.idle":"2024-12-06T06:54:53.699421Z","shell.execute_reply.started":"2024-12-06T06:54:53.688703Z","shell.execute_reply":"2024-12-06T06:54:53.698332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ntrain_data_filtered  = train[filtered_columns]\n\ntrain_data_filtered = train_data_filtered.dropna(subset=['sii'])\n\nplot_nas(train_data_filtered)\nprint(train_data_filtered)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T06:54:53.700778Z","iopub.execute_input":"2024-12-06T06:54:53.701343Z","iopub.status.idle":"2024-12-06T06:54:54.448239Z","shell.execute_reply.started":"2024-12-06T06:54:53.701295Z","shell.execute_reply":"2024-12-06T06:54:54.446912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\n\n# Thay thế giá trị Null với giá trị trung bình (cho dữ liệu số)\nimputer = SimpleImputer(strategy='mean')\ntrain_data_cleaned = pd.DataFrame(imputer.fit_transform(train_data_filtered), \n                                  columns=train_data_filtered.columns)\nplot_nas(train_data_cleaned)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T06:54:54.450010Z","iopub.execute_input":"2024-12-06T06:54:54.450382Z","iopub.status.idle":"2024-12-06T06:54:55.315363Z","shell.execute_reply.started":"2024-12-06T06:54:54.450346Z","shell.execute_reply":"2024-12-06T06:54:55.314252Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Kiểm tra kích thước của tập dữ liệu (số mẫu, số cột)\nprint(f\"Number of rows and columns: {train_data_cleaned.shape}\")\n\n# Hoặc kiểm tra số lượng mẫu (số hàng) và số lượng đặc trưng (số cột)\nnum_rows, num_columns = train_data_cleaned.shape\nprint(f\"Number of rows: {num_rows}\")\nprint(f\"Number of columns: {num_columns}\")\n\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T06:54:55.316754Z","iopub.execute_input":"2024-12-06T06:54:55.317043Z","iopub.status.idle":"2024-12-06T06:54:55.323231Z","shell.execute_reply.started":"2024-12-06T06:54:55.317015Z","shell.execute_reply":"2024-12-06T06:54:55.322197Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\nfrom sklearn.metrics import accuracy_score, classification_report\nfrom sklearn.model_selection import train_test_split\n\ndf = train_data_cleaned\ndf = df[[col for col in df.columns if  'PCIAT' not in col]]\n\n\nX = df.drop('sii', axis=1)\ny = df['sii']\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=42)\n# Khởi tạo mô hình Decision Tree\nmodel = DecisionTreeClassifier(max_depth=10, random_state=42)\n\n# Huấn luyện mô hình với dữ liệu huấn luyện\nmodel.fit(X_train, y_train)\n\n# Dự đoán trên tập kiểm tra\ny_pred = model.predict(X_test)\n\n# Đánh giá mô hình\naccuracy = accuracy_score(y_test, y_pred)\nprint(f\"Accuracy: {accuracy:.4f}\")\n\n# In báo cáo phân loại\nprint(classification_report(y_test, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T07:08:17.835760Z","iopub.execute_input":"2024-12-06T07:08:17.836192Z","iopub.status.idle":"2024-12-06T07:08:17.882584Z","shell.execute_reply.started":"2024-12-06T07:08:17.836150Z","shell.execute_reply":"2024-12-06T07:08:17.881402Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score, classification_report\nfrom sklearn.model_selection import train_test_split\n\n# Loại bỏ các cột không cần thiết (PCIAT) trong dữ liệu\ndf = train_data_cleaned\ndf = df[[col for col in df.columns if 'PCIAT' not in col]]\n\n\n# Chia dữ liệu thành đặc trưng (X) và nhãn (y)\nX = df.drop('sii', axis=1)\ny = df['sii']\n\n# Chia dữ liệu thành tập huấn luyện và kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Khởi tạo mô hình Random Forest\nmodel = RandomForestClassifier(n_estimators=100, max_depth=10, random_state=42, class_weight='balanced')\n\n# Huấn luyện mô hình với dữ liệu huấn luyện\nmodel.fit(X_train, y_train)\n\n# Dự đoán trên tập kiểm tra\ny_pred = model.predict(X_test)\n\n# Đánh giá mô hình\naccuracy = accuracy_score(y_test, y_pred)\nprint(f\"Accuracy: {accuracy:.4f}\")\n\n# In báo cáo phân loại\nprint(classification_report(y_test, y_pred))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T07:09:12.279437Z","iopub.execute_input":"2024-12-06T07:09:12.279837Z","iopub.status.idle":"2024-12-06T07:09:12.852151Z","shell.execute_reply.started":"2024-12-06T07:09:12.279801Z","shell.execute_reply":"2024-12-06T07:09:12.851028Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.svm import SVC\n# Khởi tạo SVM với kernel RBF\nmodel = SVC(kernel='rbf', class_weight='balanced', random_state=42)\n\n# Huấn luyện và đánh giá\nmodel.fit(X_train, y_train)\ny_pred = model.predict(X_test)\n\naccuracy = accuracy_score(y_test, y_pred)\nprint(f\"Accuracy: {accuracy:.4f}\")\nprint(classification_report(y_test, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T07:10:45.113149Z","iopub.execute_input":"2024-12-06T07:10:45.113641Z","iopub.status.idle":"2024-12-06T07:10:45.496752Z","shell.execute_reply.started":"2024-12-06T07:10:45.113604Z","shell.execute_reply":"2024-12-06T07:10:45.495487Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nsns.histplot(df['BIA-BIA_FMI'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T07:20:21.329442Z","iopub.execute_input":"2024-12-06T07:20:21.329909Z","iopub.status.idle":"2024-12-06T07:20:23.393893Z","shell.execute_reply.started":"2024-12-06T07:20:21.329870Z","shell.execute_reply":"2024-12-06T07:20:23.392625Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\n# Khởi tạo Logistic Regression\nmodel = LogisticRegression(max_iter=1000, class_weight='balanced', random_state=42)\n\n# Huấn luyện và đánh giá\nmodel.fit(X_train, y_train)\ny_pred = model.predict(X_test)\n\naccuracy = accuracy_score(y_test, y_pred)\nprint(f\"Accuracy: {accuracy:.4f}\")\nprint(classification_report(y_test, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T07:11:14.402991Z","iopub.execute_input":"2024-12-06T07:11:14.403420Z","iopub.status.idle":"2024-12-06T07:11:15.086670Z","shell.execute_reply.started":"2024-12-06T07:11:14.403383Z","shell.execute_reply":"2024-12-06T07:11:15.085612Z"}},"outputs":[],"execution_count":null}]}