{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:41:11.677436Z","iopub.execute_input":"2024-12-23T07:41:11.677851Z","iopub.status.idle":"2024-12-23T07:41:13.113717Z","shell.execute_reply.started":"2024-12-23T07:41:11.677821Z","shell.execute_reply":"2024-12-23T07:41:13.112650Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:41:13.115541Z","iopub.execute_input":"2024-12-23T07:41:13.116017Z","iopub.status.idle":"2024-12-23T07:41:13.121608Z","shell.execute_reply.started":"2024-12-23T07:41:13.115977Z","shell.execute_reply":"2024-12-23T07:41:13.120264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:41:13.123314Z","iopub.execute_input":"2024-12-23T07:41:13.123716Z","iopub.status.idle":"2024-12-23T07:41:13.196853Z","shell.execute_reply.started":"2024-12-23T07:41:13.123675Z","shell.execute_reply":"2024-12-23T07:41:13.195579Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom tqdm import tqdm\nfrom concurrent.futures import ThreadPoolExecutor\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"Stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    \n    return df\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:41:51.849668Z","iopub.execute_input":"2024-12-23T07:41:51.850024Z","iopub.status.idle":"2024-12-23T07:43:28.186678Z","shell.execute_reply.started":"2024-12-23T07:41:51.849996Z","shell.execute_reply":"2024-12-23T07:43:28.185466Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', \n          'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('object')\n    return df\n        \ntrain = update(train)\ntest = update(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:43:42.840208Z","iopub.execute_input":"2024-12-23T07:43:42.840563Z","iopub.status.idle":"2024-12-23T07:43:42.868832Z","shell.execute_reply.started":"2024-12-23T07:43:42.840536Z","shell.execute_reply":"2024-12-23T07:43:42.867101Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 缺失值處理\n- 針對連續型的資料特徵 ➡️ 使用中位數填補","metadata":{}},{"cell_type":"code","source":"num_imputer = SimpleImputer(strategy='median')\nnumerical_features = train.select_dtypes(include=['float64', 'int64']).columns\nmissing_in_test = [col for col in numerical_features if col not in test.columns]\nfor col in missing_in_test:\n    test[col] = np.nan\n\ntrain[numerical_features] = num_imputer.fit_transform(train[numerical_features])\ntest[numerical_features] = num_imputer.transform(test[numerical_features])\nprint(\"Continuous numerical missing values filled with median.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:43:54.977795Z","iopub.execute_input":"2024-12-23T07:43:54.978285Z","iopub.status.idle":"2024-12-23T07:43:55.086538Z","shell.execute_reply.started":"2024-12-23T07:43:54.978246Z","shell.execute_reply":"2024-12-23T07:43:55.085220Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- 針對類別型的資料特徵 ➡️ 使用多數填補","metadata":{}},{"cell_type":"code","source":"cat_imputer = SimpleImputer(strategy='most_frequent')\ncategorical_features = train.select_dtypes(include=['object']).columns\nmissing_in_test = [col for col in categorical_features if col not in test.columns]\nfor col in missing_in_test:\n    test[col] = np.nan\n\nif categorical_features.all() :\n    train[categorical_features] = cat_imputer.fit_transform(train[categorical_features])\n    test[categorical_features] = cat_imputer.transform(test[categorical_features])\n\nprint(\"Categorical missing values filled with mode.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:43:58.948627Z","iopub.execute_input":"2024-12-23T07:43:58.949271Z","iopub.status.idle":"2024-12-23T07:43:58.980140Z","shell.execute_reply.started":"2024-12-23T07:43:58.949218Z","shell.execute_reply":"2024-12-23T07:43:58.978772Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 將類別型特徵進行編碼","metadata":{}},{"cell_type":"code","source":"le = LabelEncoder()\nfor col in categorical_features:\n    if col == \"id\":\n        continue\n    train[col] = le.fit_transform(train[col])\n    test[col] = le.transform(test[col])\n\nprint(\"類別型特徵已進行 Label Encoding\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:44:02.855806Z","iopub.execute_input":"2024-12-23T07:44:02.856201Z","iopub.status.idle":"2024-12-23T07:44:02.873605Z","shell.execute_reply.started":"2024-12-23T07:44:02.856170Z","shell.execute_reply":"2024-12-23T07:44:02.872103Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 將資料進行標準化","metadata":{}},{"cell_type":"code","source":"numerical_features = [col for col in numerical_features if col not in ['sii', 'id']]\nscaler = StandardScaler()\ntrain[numerical_features] = scaler.fit_transform(train[numerical_features])\ntest[numerical_features] = scaler.transform(test[numerical_features])\n\nprint(\"數值特徵已進行標準化\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:44:06.546943Z","iopub.execute_input":"2024-12-23T07:44:06.547442Z","iopub.status.idle":"2024-12-23T07:44:06.617189Z","shell.execute_reply.started":"2024-12-23T07:44:06.547395Z","shell.execute_reply":"2024-12-23T07:44:06.615702Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 特徵選擇","metadata":{}},{"cell_type":"code","source":"from sklearn.feature_selection import SelectKBest, f_regression\n\n# 移除 `sii` 和 `id`\nX_train = train.drop(columns=['sii'])\ny_train = train['sii']\nX_test = test\n\n\n# 確保 `train` 和 `test` 特徵名稱一致\nX_test = X_test[X_train.columns]\n\n# 特徵選擇\nselector = SelectKBest(score_func=f_regression, k=120)\nX_train_selected = selector.fit_transform(X_train, y_train)\n\n# 獲取選擇的特徵名稱\nselected_features = X_train.columns[selector.get_support()]\nprint(\"選擇的 120 個最佳特徵：\", selected_features)\n\n# 在 `test` 中應用相同的特徵選擇\nX_test_selected = X_test[selected_features].to_numpy()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:44:09.518826Z","iopub.execute_input":"2024-12-23T07:44:09.519206Z","iopub.status.idle":"2024-12-23T07:44:09.615729Z","shell.execute_reply.started":"2024-12-23T07:44:09.519176Z","shell.execute_reply":"2024-12-23T07:44:09.613174Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 模型訓練","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier, ExtraTreesClassifier\nfrom sklearn.model_selection import KFold, cross_val_score, StratifiedKFold\nfrom sklearn.metrics import classification_report\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom imblearn.over_sampling import SMOTE\n\n\n# 訓練分類模型\nX = train[selected_features]\ny = train['sii']\n\n# 應用 SMOTE\nsmote = SMOTE(random_state=42)\nX_train_resampled, y_train_resampled = smote.fit_resample(X, y)\n\nmodel = ExtraTreesClassifier(n_estimators=300, random_state=42, max_depth=5, class_weight= {0: 4, 1: 5, 2: 5, 3: 3}, min_samples_split=5, min_samples_leaf=4)\n\n# 交叉驗證\nkf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\ncv_scores = cross_val_score(model, X_train_resampled, y_train_resampled, cv=kf, scoring='accuracy')\n\nprint(\"交叉驗證平均準確率：\", cv_scores.mean())\n\n# 模型訓練與預測\nmodel.fit(X_train_resampled, y_train_resampled)\ny_pred = model.predict(test[selected_features])\n\n# 評估模型效能\nprint(\"模型評估報告：\")\nprint(classification_report(y_train_resampled, model.predict(X_train_resampled)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:44:14.275333Z","iopub.execute_input":"2024-12-23T07:44:14.275699Z","iopub.status.idle":"2024-12-23T07:44:22.203789Z","shell.execute_reply.started":"2024-12-23T07:44:14.275668Z","shell.execute_reply":"2024-12-23T07:44:22.202600Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Confusion Matrix","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\n\ny_train_pred = model.predict(X_train_resampled)\n\ncm = confusion_matrix(y_train_resampled, y_train_pred)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=model.classes_)\ndisp.plot(cmap='Blues')\nplt.title('Confusion Matrix')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:44:27.404881Z","iopub.execute_input":"2024-12-23T07:44:27.405408Z","iopub.status.idle":"2024-12-23T07:44:27.940190Z","shell.execute_reply.started":"2024-12-23T07:44:27.405376Z","shell.execute_reply":"2024-12-23T07:44:27.938591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_id = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")\nsubmission = pd.DataFrame({'id': test_id['id'], 'sii': y_pred})\nsubmission.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:49:55.628296Z","iopub.execute_input":"2024-12-23T07:49:55.628802Z","iopub.status.idle":"2024-12-23T07:49:55.641949Z","shell.execute_reply.started":"2024-12-23T07:49:55.628768Z","shell.execute_reply":"2024-12-23T07:49:55.640676Z"}},"outputs":[],"execution_count":null}]}