{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:21.683206Z","iopub.execute_input":"2024-12-21T09:15:21.683625Z","iopub.status.idle":"2024-12-21T09:15:22.088720Z","shell.execute_reply.started":"2024-12-21T09:15:21.683583Z","shell.execute_reply":"2024-12-21T09:15:22.087601Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_train = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\ndata_test = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")\ndata_dictionary = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:22.089866Z","iopub.execute_input":"2024-12-21T09:15:22.090431Z","iopub.status.idle":"2024-12-21T09:15:22.176936Z","shell.execute_reply.started":"2024-12-21T09:15:22.090390Z","shell.execute_reply":"2024-12-21T09:15:22.175609Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:22.178035Z","iopub.execute_input":"2024-12-21T09:15:22.178428Z","iopub.status.idle":"2024-12-21T09:15:22.185148Z","shell.execute_reply.started":"2024-12-21T09:15:22.178396Z","shell.execute_reply":"2024-12-21T09:15:22.183965Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_train.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:22.187322Z","iopub.execute_input":"2024-12-21T09:15:22.187647Z","iopub.status.idle":"2024-12-21T09:15:22.205661Z","shell.execute_reply.started":"2024-12-21T09:15:22.187617Z","shell.execute_reply":"2024-12-21T09:15:22.204481Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:22.207745Z","iopub.execute_input":"2024-12-21T09:15:22.208353Z","iopub.status.idle":"2024-12-21T09:15:22.284760Z","shell.execute_reply.started":"2024-12-21T09:15:22.208311Z","shell.execute_reply":"2024-12-21T09:15:22.280026Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:22.285618Z","iopub.execute_input":"2024-12-21T09:15:22.285957Z","iopub.status.idle":"2024-12-21T09:15:22.345930Z","shell.execute_reply.started":"2024-12-21T09:15:22.285928Z","shell.execute_reply":"2024-12-21T09:15:22.343655Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nimport polars as pl\n\nfiltered_data = data_train[data_train['sii'].notna()]\nprint(filtered_data['sii'].count())\nprint(filtered_data)\n\nsns.countplot(x='sii', data=filtered_data)\nplt.title('Phân phối cột mục tiêu (sii)')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:22.347341Z","iopub.execute_input":"2024-12-21T09:15:22.347686Z","iopub.status.idle":"2024-12-21T09:15:24.383009Z","shell.execute_reply.started":"2024-12-21T09:15:22.347656Z","shell.execute_reply":"2024-12-21T09:15:24.381901Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('Columns missing in test:')\nprint([f for f in data_train.columns if f not in data_test.columns])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:24.384171Z","iopub.execute_input":"2024-12-21T09:15:24.384845Z","iopub.status.idle":"2024-12-21T09:15:24.390292Z","shell.execute_reply.started":"2024-12-21T09:15:24.384803Z","shell.execute_reply":"2024-12-21T09:15:24.389208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_values = filtered_data.isnull().sum()\nprint(\"Số lượng giá trị thiếu:\\n\", missing_values[missing_values >= 0])\nmissing_percent = (missing_values / len(data_train)) * 100\n\nmissing_data = pd.DataFrame({\n    'Số lượng thiếu': missing_values,\n    'Phần trăm (%)': missing_percent\n})\n\nmissing_data = missing_data[missing_data['Số lượng thiếu'] >= 0]\n\nmissing_data_sort = missing_data.sort_values(by='Phần trăm (%)', ascending=True)\n\nif missing_data_sort.empty:\n    print(\"Không có cột nào có giá trị thiếu.\")\nelse:\n    plt.figure(figsize=(10, len(missing_data) * 0.2))\n    sns.barplot(\n        y=missing_data_sort.index, \n        x=missing_data_sort['Phần trăm (%)'], \n        palette=\"viridis\"\n    )\n    plt.xticks(rotation=45, ha='right')\n    plt.title(\"Tỷ lệ phần trăm giá trị thiếu theo cột\", fontsize=14)\n    plt.ylabel(\"Feature\", fontsize=12)\n    plt.xlabel(\"Phần trăm (%)\", fontsize=12)\n\n    plt.gca().yaxis.set_tick_params(labelsize=10, pad=5)\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:24.391321Z","iopub.execute_input":"2024-12-21T09:15:24.391682Z","iopub.status.idle":"2024-12-21T09:15:25.696420Z","shell.execute_reply.started":"2024-12-21T09:15:24.391645Z","shell.execute_reply":"2024-12-21T09:15:25.695084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"columns_to_drop = missing_data[missing_data['Phần trăm (%)'] > 30].index\n\nprint(\"Các feature có tỷ lệ giá trị thiếu lớn hơn 30%:\")\nprint(columns_to_drop)\n\ndata_train_cleaned = data_train.drop(columns=columns_to_drop)\n\ndata_train_cleaned.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:25.697703Z","iopub.execute_input":"2024-12-21T09:15:25.698132Z","iopub.status.idle":"2024-12-21T09:15:25.729321Z","shell.execute_reply.started":"2024-12-21T09:15:25.698093Z","shell.execute_reply":"2024-12-21T09:15:25.728171Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(data_train[['Basic_Demos-Age', 'Basic_Demos-Sex']].describe())\nsns.histplot(data_train['Basic_Demos-Age'], kde=True)\nplt.title('Age Distribution')\nplt.show()\nsns.countplot(x='Basic_Demos-Sex', data=data_train)\nplt.title('Sex Distribution')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:25.730427Z","iopub.execute_input":"2024-12-21T09:15:25.730745Z","iopub.status.idle":"2024-12-21T09:15:26.321295Z","shell.execute_reply.started":"2024-12-21T09:15:25.730719Z","shell.execute_reply":"2024-12-21T09:15:26.320015Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import KNNImputer\n\nknn_imputer = KNNImputer(n_neighbors=5)\n\nnum_cols = data_train_cleaned.select_dtypes(include=['float64', 'int64']).columns\n\ndata_train_cleaned[num_cols] = knn_imputer.fit_transform(data_train_cleaned[num_cols])\n\nprint(data_train_cleaned.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:26.322305Z","iopub.execute_input":"2024-12-21T09:15:26.322640Z","iopub.status.idle":"2024-12-21T09:15:33.196617Z","shell.execute_reply.started":"2024-12-21T09:15:26.322609Z","shell.execute_reply":"2024-12-21T09:15:33.195492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\n\ncat_cols = data_train_cleaned.select_dtypes(include=['object']).columns\ncat_imputer = SimpleImputer(strategy='most_frequent')\n\ndata_train_cleaned[cat_cols] = cat_imputer.fit_transform(data_train_cleaned[cat_cols])\n\nprint(data_train_cleaned.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:33.200315Z","iopub.execute_input":"2024-12-21T09:15:33.200641Z","iopub.status.idle":"2024-12-21T09:15:33.227683Z","shell.execute_reply.started":"2024-12-21T09:15:33.200613Z","shell.execute_reply":"2024-12-21T09:15:33.226425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_values_train_after = data_train_cleaned.isnull().sum()\nprint(\"Missing values after imputation in train dataset:\\n\", missing_values_train_after)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:33.229444Z","iopub.execute_input":"2024-12-21T09:15:33.229803Z","iopub.status.idle":"2024-12-21T09:15:33.246684Z","shell.execute_reply.started":"2024-12-21T09:15:33.229774Z","shell.execute_reply":"2024-12-21T09:15:33.245585Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_cols = ['Basic_Demos-Age', 'Physical-BMI', 'PCIAT-PCIAT_Total']\nfor col in num_cols:\n    sns.boxplot(x='sii', y=col, data=data_train_cleaned)\n    plt.title(f'Mối quan hệ giữa {col} và sii')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:33.247792Z","iopub.execute_input":"2024-12-21T09:15:33.248087Z","iopub.status.idle":"2024-12-21T09:15:34.270294Z","shell.execute_reply.started":"2024-12-21T09:15:33.248063Z","shell.execute_reply":"2024-12-21T09:15:34.269201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_cols = ['Basic_Demos-Sex', 'Basic_Demos-Enroll_Season', 'PCIAT-Season']\nfor col in cat_cols:\n    sns.countplot(x=col, hue='sii', data=data_train_cleaned)\n    plt.title(f'Mối quan hệ giữa {col} và sii')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:34.271479Z","iopub.execute_input":"2024-12-21T09:15:34.271876Z","iopub.status.idle":"2024-12-21T09:15:35.411963Z","shell.execute_reply.started":"2024-12-21T09:15:34.271836Z","shell.execute_reply":"2024-12-21T09:15:35.410865Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"season_mapping = {'Spring': 1, 'Summer': 2, 'Fall': 3, 'Winter': 4}\nseason_cols = [col for col in data_train_cleaned.columns if 'Season' in col]\nfor col in season_cols:\n    data_train_cleaned[col] = data_train_cleaned[col].map(season_mapping)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:35.412941Z","iopub.execute_input":"2024-12-21T09:15:35.413286Z","iopub.status.idle":"2024-12-21T09:15:35.427673Z","shell.execute_reply.started":"2024-12-21T09:15:35.413248Z","shell.execute_reply":"2024-12-21T09:15:35.426584Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_train_no_id = data_train_cleaned.drop(columns = ['id'], errors = 'ignore')\ndata_train_no_id.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:35.428822Z","iopub.execute_input":"2024-12-21T09:15:35.429275Z","iopub.status.idle":"2024-12-21T09:15:35.470395Z","shell.execute_reply.started":"2024-12-21T09:15:35.429237Z","shell.execute_reply":"2024-12-21T09:15:35.469196Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pciat_columns = [col for col in data_train_cleaned.columns if 'PCIAT-PCIAT' in col and col != 'PCIAT-PCIAT_Total']\ncorr_with_total = data_train_cleaned[pciat_columns].corrwith(data_train_cleaned['PCIAT-PCIAT_Total'])\nprint(\"Mối tương quan với PCIAT_PCIAT_TOTAL:\")\nprint(corr_with_total)\n\ndata_train_no_id.drop(columns= pciat_columns, inplace=True)\ndata_train_cleaned.drop(columns= pciat_columns, inplace=True)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:35.471319Z","iopub.execute_input":"2024-12-21T09:15:35.471650Z","iopub.status.idle":"2024-12-21T09:15:35.496966Z","shell.execute_reply.started":"2024-12-21T09:15:35.471621Z","shell.execute_reply":"2024-12-21T09:15:35.495492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"corr_matrix = data_train_no_id.corr()\n\nplt.figure(figsize=(30, 30))\nsns.heatmap(corr_matrix, annot=True, cmap='coolwarm', linewidths=0.5, fmt='.2f', vmin=-1, vmax=1)\nplt.title('Heatmap of Correlation Matrix')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:35.498218Z","iopub.execute_input":"2024-12-21T09:15:35.498626Z","iopub.status.idle":"2024-12-21T09:15:40.935943Z","shell.execute_reply.started":"2024-12-21T09:15:35.498598Z","shell.execute_reply":"2024-12-21T09:15:40.934675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"threshold = 0.8\n\nto_drop = set()\nfor i in range(len(corr_matrix.columns)):\n    for j in range(i):\n        if abs(corr_matrix.iloc[i, j]) > threshold:\n            colname = corr_matrix.columns[i]\n            to_drop.add(colname)\n\nto_drop.discard('sii')\n\ndata_train_cleaned = data_train_cleaned.drop(columns=to_drop)\n\nprint(f\"Những cột đã bị loại bỏ: {to_drop}\")\nprint(data_train_cleaned.shape)  \n\ndata_train_cleaned = data_train_cleaned.drop(columns=['PCIAT-Season', 'PCIAT-PCIAT_Total'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:40.937213Z","iopub.execute_input":"2024-12-21T09:15:40.937575Z","iopub.status.idle":"2024-12-21T09:15:40.981304Z","shell.execute_reply.started":"2024-12-21T09:15:40.937544Z","shell.execute_reply":"2024-12-21T09:15:40.980086Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:40.982367Z","iopub.execute_input":"2024-12-21T09:15:40.982710Z","iopub.status.idle":"2024-12-21T09:15:41.046310Z","shell.execute_reply.started":"2024-12-21T09:15:40.982673Z","shell.execute_reply":"2024-12-21T09:15:41.045315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:41.047361Z","iopub.execute_input":"2024-12-21T09:15:41.047696Z","iopub.status.idle":"2024-12-21T09:15:41.082054Z","shell.execute_reply.started":"2024-12-21T09:15:41.047668Z","shell.execute_reply":"2024-12-21T09:15:41.080690Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_train_cleaned.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:41.083130Z","iopub.execute_input":"2024-12-21T09:15:41.083474Z","iopub.status.idle":"2024-12-21T09:15:41.107045Z","shell.execute_reply.started":"2024-12-21T09:15:41.083419Z","shell.execute_reply":"2024-12-21T09:15:41.105968Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def filter_test_data(train_data, test_data, target_column):\n    train_columns = [col for col in train_data.columns if col != target_column]\n    \n    test_data_filtered = test_data[train_columns]\n    \n    return test_data_filtered\n\ntest = filter_test_data (data_train_cleaned, test, 'sii')\nseason_mapping = {'Spring': 1, 'Summer': 2, 'Fall': 3, 'Winter': 4}\nseason_cols = [col for col in test.columns if 'Season' in col]\nfor col in season_cols:\n    test[col] = test[col].map(season_mapping)\ntest.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:41.108252Z","iopub.execute_input":"2024-12-21T09:15:41.108633Z","iopub.status.idle":"2024-12-21T09:15:41.139515Z","shell.execute_reply.started":"2024-12-21T09:15:41.108594Z","shell.execute_reply":"2024-12-21T09:15:41.138333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import cohen_kappa_score\nfrom sklearn.linear_model import LinearRegression\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.impute import SimpleImputer\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    rounded_y_true = threshold_Rounder(y_true, [0.5, 1.5, 2.5])\n    rounded_y_pred = threshold_Rounder(y_pred, [0.5, 1.5, 2.5]) \n    return cohen_kappa_score(rounded_y_true, rounded_y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef TrainML(train_data, test_data, target_column, random_state=42):\n    X = train_data.drop(columns=[target_column, 'id'])\n    y = train_data[target_column]\n    test_data_dropped = test_data.drop(columns=['id'])\n    \n    test_ids = test_data['id']\n    \n    imputer = SimpleImputer(strategy='mean')\n    X = imputer.fit_transform(X)\n    test_data_dropped = imputer.transform(test_data_dropped)\n    \n    # Sử dụng train_test_split\n    X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=random_state)\n\n    # Khởi tạo mô hình Linear Regression\n    model = LinearRegression()\n\n    model.fit(X_train, y_train)\n    y_val_pred = model.predict(X_val)\n\n    # Áp dụng threshold\n    oof_tuned = threshold_Rounder(y_val_pred, [0.5, 1.5, 2.5])\n    tKappa = quadratic_weighted_kappa(y_val, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {tKappa:.3f}\")\n\n    # Dự đoán tập test\n    test_preds = model.predict(test_data_dropped)\n    tpTuned = threshold_Rounder(test_preds, [0.5, 1.5, 2.5])\n\n    submission = pd.DataFrame({\n        'id': test_ids,\n        target_column: tpTuned\n    })\n\n    return submission\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:41.140635Z","iopub.execute_input":"2024-12-21T09:15:41.140958Z","iopub.status.idle":"2024-12-21T09:15:41.166425Z","shell.execute_reply.started":"2024-12-21T09:15:41.140930Z","shell.execute_reply":"2024-12-21T09:15:41.165325Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = TrainML(data_train_cleaned, test, target_column=\"sii\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:41.167544Z","iopub.execute_input":"2024-12-21T09:15:41.167941Z","iopub.status.idle":"2024-12-21T09:15:41.244205Z","shell.execute_reply.started":"2024-12-21T09:15:41.167900Z","shell.execute_reply":"2024-12-21T09:15:41.240395Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:41.245039Z","iopub.execute_input":"2024-12-21T09:15:41.245398Z","iopub.status.idle":"2024-12-21T09:15:41.288241Z","shell.execute_reply.started":"2024-12-21T09:15:41.245366Z","shell.execute_reply":"2024-12-21T09:15:41.286409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:15:41.290776Z","iopub.execute_input":"2024-12-21T09:15:41.294756Z","iopub.status.idle":"2024-12-21T09:15:41.316203Z","shell.execute_reply.started":"2024-12-21T09:15:41.294708Z","shell.execute_reply":"2024-12-21T09:15:41.313252Z"}},"outputs":[],"execution_count":null}]}