{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-27T11:53:21.737906Z","iopub.execute_input":"2024-12-27T11:53:21.738197Z","iopub.status.idle":"2024-12-27T11:53:27.387998Z","shell.execute_reply.started":"2024-12-27T11:53:21.738168Z","shell.execute_reply":"2024-12-27T11:53:27.387018Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dict_data = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv\")\ntrain_data = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T11:53:41.655222Z","iopub.execute_input":"2024-12-27T11:53:41.655671Z","iopub.status.idle":"2024-12-27T11:53:41.735426Z","shell.execute_reply.started":"2024-12-27T11:53:41.655634Z","shell.execute_reply":"2024-12-27T11:53:41.734371Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# remove rows which don't have sii value\ntrain_data = train_data.dropna(subset=['sii'])\ntrain_data  = train_data.drop(columns = \"id\")\ntrain_data  = train_data.drop(columns = \"sii\")\n\ncolumns_to_drop = [\n    col for col in train_data.columns\n    if \"PCIAT-PCIAT\" in col and col != \"PCIAT-PCIAT_Total\"\n]\ntrain_data = train_data.drop(columns=columns_to_drop)\n\ncolumns_to_on_hot_encoding = [col for col in train_data.columns if \"Season\" in col]\ntrain_data = pd.get_dummies(train_data, columns=columns_to_on_hot_encoding, dummy_na=False, dtype=int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T11:53:44.137639Z","iopub.execute_input":"2024-12-27T11:53:44.137999Z","iopub.status.idle":"2024-12-27T11:53:44.186348Z","shell.execute_reply.started":"2024-12-27T11:53:44.137966Z","shell.execute_reply":"2024-12-27T11:53:44.185163Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from imblearn.combine import SMOTEENN","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:01:52.498812Z","iopub.execute_input":"2024-12-27T09:01:52.499131Z","iopub.status.idle":"2024-12-27T09:01:52.898218Z","shell.execute_reply.started":"2024-12-27T09:01:52.499106Z","shell.execute_reply":"2024-12-27T09:01:52.897469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split, ShuffleSplit, KFold \nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import ConfusionMatrixDisplay, classification_report, confusion_matrix, f1_score, precision_recall_curve\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.impute import KNNImputer\nfrom sklearn.metrics import mean_squared_error\n\n\nknn_imputer = KNNImputer(n_neighbors=5)\nX = train_data.drop(columns=[ \"PCIAT-PCIAT_Total\"])\ntarget = train_data[\"PCIAT-PCIAT_Total\"]\nX_imputed = knn_imputer.fit_transform(X)\n\nle = LabelEncoder()\nY = le.fit_transform(target)\nX_train, X_test, Y_train, Y_test = train_test_split(X, Y, test_size=0.1, random_state=52)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T11:53:53.279408Z","iopub.execute_input":"2024-12-27T11:53:53.279808Z","iopub.status.idle":"2024-12-27T11:53:57.474455Z","shell.execute_reply.started":"2024-12-27T11:53:53.279772Z","shell.execute_reply":"2024-12-27T11:53:57.473308Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from catboost import CatBoostRegressor\n\nCBR = CatBoostRegressor(verbose=0 ,\n    iterations=2000,            # 增加迭代次數\n    learning_rate=0.01,         # 降低學習率\n    depth=8,                    # 增加樹的深度\n    l2_leaf_reg=10              # 增加正則化\n)\nCBR.fit(X_train, Y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T11:54:02.746518Z","iopub.execute_input":"2024-12-27T11:54:02.747038Z","iopub.status.idle":"2024-12-27T11:54:28.936238Z","shell.execute_reply.started":"2024-12-27T11:54:02.747004Z","shell.execute_reply":"2024-12-27T11:54:28.935190Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score\n\n# 使用交叉驗證評估模型\ncv_scores = cross_val_score(CBR, X, Y, cv=5, scoring='neg_mean_squared_error')\nprint(f'Cross-validated MSE: {-cv_scores.mean()}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T11:54:57.070462Z","iopub.execute_input":"2024-12-27T11:54:57.070850Z","iopub.status.idle":"2024-12-27T11:56:56.175124Z","shell.execute_reply.started":"2024-12-27T11:54:57.070818Z","shell.execute_reply":"2024-12-27T11:56:56.174072Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")\nidx = test_data[\"id\"]\ntest_data = test_data.drop(columns = \"id\")\ntrain_cols = train_data.columns\n\ncolumns_to_on_hot_encoding = [col for col in test_data.columns if \"Season\" in col]\ntest_data = pd.get_dummies(test_data, columns=columns_to_on_hot_encoding, dummy_na=False, dtype=int)\ntest_data = test_data.reindex(columns=train_cols, fill_value=0)\npredictions = CBR.predict(test_data) \npredictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T11:57:31.307834Z","iopub.execute_input":"2024-12-27T11:57:31.308209Z","iopub.status.idle":"2024-12-27T11:57:31.344326Z","shell.execute_reply.started":"2024-12-27T11:57:31.308176Z","shell.execute_reply":"2024-12-27T11:57:31.343162Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"converted_arr = np.digitize(predictions, bins=[30, 49, 79, 100])\nconverted_arr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T11:57:33.528512Z","iopub.execute_input":"2024-12-27T11:57:33.528837Z","iopub.status.idle":"2024-12-27T11:57:33.536167Z","shell.execute_reply.started":"2024-12-27T11:57:33.528812Z","shell.execute_reply":"2024-12-27T11:57:33.535079Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Submission = pd.DataFrame({\n    'id': idx,\n    'sii': converted_arr\n})\nSubmission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T11:57:35.227761Z","iopub.execute_input":"2024-12-27T11:57:35.228230Z","iopub.status.idle":"2024-12-27T11:57:35.238320Z","shell.execute_reply.started":"2024-12-27T11:57:35.228187Z","shell.execute_reply":"2024-12-27T11:57:35.236752Z"}},"outputs":[],"execution_count":null}]}