{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\nfrom concurrent.futures import ThreadPoolExecutor\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split, StratifiedKFold, cross_val_score\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.compose import ColumnTransformer\nfrom xgboost import XGBClassifier\nfrom sklearn.svm import LinearSVC, SVC\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.ensemble import RandomForestClassifier, ExtraTreesClassifier, AdaBoostClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-13T01:26:45.105047Z","iopub.execute_input":"2025-03-13T01:26:45.105344Z","iopub.status.idle":"2025-03-13T01:26:45.110125Z","shell.execute_reply.started":"2025-03-13T01:26:45.105317Z","shell.execute_reply":"2025-03-13T01:26:45.109402Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_file(filename, dirname):\n    \"\"\"Process a single parquet file and extract time-based features\"\"\"\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    \n    # Drop 'step' column if it exists\n    if 'step' in df.columns:\n        df.drop('step', axis=1, inplace=True)\n    \n    # Convert time_of_day to hours\n    df[\"hours\"] = df[\"time_of_day\"] // (3_600 * 1_000_000_000)\n    \n    # Define time periods\n    night = ((df[\"hours\"] >= 22) | (df[\"hours\"] <= 5))\n    day = ((df[\"hours\"] <= 20) & (df[\"hours\"] >= 7))\n    \n    # Initialize features dictionary\n    features = {}\n    \n    # Basic activity features\n    features['non_wear_mean'] = df[\"non-wear_flag\"].mean()\n    features['active_enmo_sum'] = df[\"enmo\"][df[\"enmo\"] >= 0.05].sum()\n    \n    # Process each column for different time periods\n    for col in ['enmo', 'anglez', 'light', 'battery_voltage']:\n        # Full day statistics\n        features[f\"{col}_mean\"] = df[col].mean()\n        features[f\"{col}_std\"] = df[col].std()\n        features[f\"{col}_max\"] = df[col].max()\n        features[f\"{col}_min\"] = df[col].min()\n        features[f\"{col}_diff_mean\"] = df[col].diff().mean()\n        features[f\"{col}_diff_std\"] = df[col].diff().std()\n        \n        # Night time statistics\n        night_data = df.loc[night, col]\n        features[f\"{col}_night_mean\"] = night_data.mean()\n        features[f\"{col}_night_std\"] = night_data.std()\n        features[f\"{col}_night_max\"] = night_data.max()\n        features[f\"{col}_night_min\"] = night_data.min()\n        \n        # Day time statistics\n        day_data = df.loc[day, col]\n        features[f\"{col}_day_mean\"] = day_data.mean()\n        features[f\"{col}_day_std\"] = day_data.std()\n        features[f\"{col}_day_max\"] = day_data.max()\n        features[f\"{col}_day_min\"] = day_data.min()\n    \n    return features, filename.split('=')[1]\n\ndef load_data_parquet(dirname) -> pd.DataFrame:\n    \"\"\"Load and process time series data from directory in parallel\"\"\"\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    features_list, indexes = zip(*results)\n    \n    # Create DataFrame with extracted features and IDs\n    df = pd.DataFrame(features_list)\n    df['id'] = indexes\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-13T01:26:45.113806Z","iopub.execute_input":"2025-03-13T01:26:45.114063Z","iopub.status.idle":"2025-03-13T01:26:45.132966Z","shell.execute_reply.started":"2025-03-13T01:26:45.114044Z","shell.execute_reply":"2025-03-13T01:26:45.132260Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday','sii']\n\n# chia thành 3 nhóm features chính (Bộ dữ liệu khách quan csv)\ndemographicFeatures = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'sii']\n\nphycsicsFeatures = ['CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW','sii' ]\n\nbehaviorFeatures = ['PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                    'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday','sii']\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ntrain_ts = load_data_parquet('/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet')\ntest_ts = load_data_parquet(r'/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet')\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\n\nbehaviorFeatures += time_series_cols\n\ntrain_df = train[behaviorFeatures]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-13T01:26:45.146726Z","iopub.execute_input":"2025-03-13T01:26:45.147094Z","iopub.status.idle":"2025-03-13T01:27:22.732279Z","shell.execute_reply.started":"2025-03-13T01:26:45.147067Z","shell.execute_reply":"2025-03-13T01:27:22.731521Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Loại bỏ các feature liên quan đến \"season\"\nfiltered_features = [feature for feature in behaviorFeatures if feature not in cat_c and feature != 'sii']\n\n# Loại bỏ các hàng có giá trị NaN trong y\ntrain_df = train_df[behaviorFeatures].dropna(subset=['sii'])\n\n# Chuẩn bị dữ liệu X và y\nX = train_df[filtered_features]\ny = train_df['sii']\n\n# Định nghĩa pipeline xử lý dữ liệu số\nnum_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='mean')),\n    ('scaler', StandardScaler())\n])\n\n# Định nghĩa ColumnTransformer để áp dụng pipeline cho các cột số\npreprocessor = ColumnTransformer(transformers=[\n    ('num', num_transformer, filtered_features)\n])\n\n# Fit và transform X\npreprocessor.fit(X)\nX_transformed = pd.DataFrame(preprocessor.transform(X), columns=filtered_features)\n\n# Kiểm tra các dòng đầu tiên của dữ liệu đã transform\nprint(\"Transformed X DataFrame:\")\nprint(X_transformed.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-13T01:27:22.733355Z","iopub.execute_input":"2025-03-13T01:27:22.733532Z","iopub.status.idle":"2025-03-13T01:27:22.798211Z","shell.execute_reply.started":"2025-03-13T01:27:22.733516Z","shell.execute_reply":"2025-03-13T01:27:22.797176Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_val, y_train, y_val = train_test_split(X_transformed, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-13T01:27:22.799483Z","iopub.execute_input":"2025-03-13T01:27:22.799680Z","iopub.status.idle":"2025-03-13T01:27:22.806344Z","shell.execute_reply.started":"2025-03-13T01:27:22.799663Z","shell.execute_reply":"2025-03-13T01:27:22.805439Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"XGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': 2023,\n    'tree_method': 'gpu_hist',\n}\nRF_Params = {\n    'n_estimators': 200,         # Increased from 100 to match better performing RandomForest_Default\n    'max_depth': 15,             # Added specific depth to control complexity\n    'min_samples_split': 5,      # Increased to reduce overfitting\n    'min_samples_leaf': 2,       # Increased to ensure more robust leaf nodes\n    'max_features': 'sqrt',      # Keep sqrt as it works well for classification\n    'bootstrap': True,           # Keep bootstrapping enabled\n    'random_state': 2023,        # Keep same random state for reproducibility\n    'n_jobs': -1,                # Keep using all cores\n    'class_weight': 'balanced',  # Keep balanced class weights\n    'criterion': 'gini'          # Keep gini criterion\n}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-13T01:27:22.807263Z","iopub.execute_input":"2025-03-13T01:27:22.807535Z","iopub.status.idle":"2025-03-13T01:27:22.821965Z","shell.execute_reply.started":"2025-03-13T01:27:22.807507Z","shell.execute_reply":"2025-03-13T01:27:22.821095Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import cohen_kappa_score, confusion_matrix, classification_report\nfrom sklearn.model_selection import cross_val_score, StratifiedKFold\nimport warnings\n\n# Ensemble model imports\nfrom sklearn.ensemble import (\n    RandomForestClassifier,\n    GradientBoostingClassifier, \n    AdaBoostClassifier,\n    ExtraTreesClassifier,\n    BaggingClassifier,\n    StackingClassifier,\n    VotingClassifier\n)\nfrom xgboost import XGBClassifier\nfrom catboost import CatBoostClassifier\nfrom lightgbm import LGBMClassifier\n\nwarnings.filterwarnings(\"ignore\")\n\n# Set a random seed for reproducibility\nseed = 2023\nnp.random.seed(seed)\n\n# Base models for stacking/voting\nbase_rf = RandomForestClassifier(n_estimators=100, random_state=seed)\nbase_gb = GradientBoostingClassifier(n_estimators=100, random_state=seed)\nbase_xgb = XGBClassifier(n_estimators=100, random_state=seed, eval_metric='mlogloss')\n\n# Create ensemble models\nmodels = [\n    RandomForestClassifier(n_estimators=100, random_state=seed),\n    GradientBoostingClassifier(n_estimators=100, random_state=seed),\n    ExtraTreesClassifier(n_estimators=100, random_state=seed),\n    AdaBoostClassifier(random_state=seed),\n    BaggingClassifier(random_state=seed),\n    XGBClassifier(n_estimators=100, eval_metric='mlogloss', random_state=seed),\n    LGBMClassifier(n_estimators=100, random_state=seed),\n    CatBoostClassifier(n_estimators=100, random_state=seed, verbose=0),\n    VotingClassifier(estimators=[\n        ('rf', base_rf),\n        ('gb', base_gb),\n        ('xgb', base_xgb)\n    ], voting='soft'),\n    StackingClassifier(estimators=[\n        ('rf', base_rf),\n        ('gb', base_gb),\n        ('xgb', base_xgb)\n    ], final_estimator=GradientBoostingClassifier(random_state=seed))\n]\n\nmodel_names = [\n    'RandomForest',\n    'GradientBoosting',\n    'ExtraTrees',\n    'AdaBoost',\n    'Bagging',\n    'XGBoost',\n    'LightGBM',\n    'CatBoost',\n    'VotingEnsemble',\n    'StackingEnsemble'\n]\n\n# Function to generate baseline results0\ndef generate_baseline_results(models, model_names, X, y, metrics='kappa', cv=5, plot_results=False):\n    \"\"\"\n    Evaluate multiple ensemble models with cross-validation and optionally plot results.\n    \n    Parameters:\n    -----------\n    models: list\n        List of initialized model objects\n    model_names: list\n        List of model names (should match length of models)\n    X: DataFrame or array\n        Feature matrix\n    y: Series or array\n        Target variable\n    metrics: str\n        Scoring metric to use\n    cv: int\n        Number of cross-validation folds\n    plot_results: bool\n        Whether to display a boxplot of results\n        \n    Returns:\n    --------\n    DataFrame with mean and std dev of model performance\n    \"\"\"\n    # Define k-fold\n    kfold = StratifiedKFold(n_splits=cv, shuffle=True, random_state=42)\n    entries = []\n    \n    # Loop through each model\n    for model, model_name in zip(models, model_names):\n        print(f\"Training: {model_name}\")\n        try:\n            scores = cross_val_score(model, X, y, scoring=metrics, cv=kfold)\n            # Save results for all models\n            entries.extend([(model_name, fold_idx, score) for fold_idx, score in enumerate(scores)])\n        except Exception as e:\n            print(f\"Error with {model_name}: {e}\")\n    \n    # Create DataFrame\n    cv_df = pd.DataFrame(entries, columns=['model_name', 'fold_id', 'cohen_kappa_score'])\n    \n    # Optional: Plot results if specified\n    if plot_results and len(cv_df) > 0:\n        plt.figure(figsize=(14, 6))\n        sns.boxplot(x='model_name', y='cohen_kappa_score', data=cv_df, color='lightblue', showmeans=True)\n        plt.title(\"Ensemble Models Performance using 5-fold Cross-Validation\", fontsize=14)\n        plt.xlabel(\"Model\", fontsize=12)\n        plt.ylabel(\"Accuracy Score\", fontsize=12)\n        plt.xticks(rotation=45, ha='right')\n        plt.grid(axis='y', linestyle='--', alpha=0.7)\n        plt.tight_layout()\n        plt.show()\n    \n    # Summary result\n    if len(cv_df) > 0:\n        mean = cv_df.groupby('model_name')['cohen_kappa_score'].mean()\n        std = cv_df.groupby('model_name')['cohen_kappa_score'].std()\n\n        baseline_results = pd.concat([mean, std], axis=1)\n        baseline_results.columns = ['Mean', 'Standard Deviation']\n\n        # Sort results\n        baseline_results.sort_values(by='Mean', ascending=False, inplace=True)\n        return baseline_results\n    else:\n        return pd.DataFrame(columns=['Mean', 'Standard Deviation'])\n\n# Run evaluation and display results\nprint(\"\\n===== Ensemble Models Evaluation =====\\n\")\ncv_results = generate_baseline_results(models, model_names, X_transformed, y, metrics='kappa', cv=5, plot_results=True)\n\n# Print full results\nprint(\"\\nEnsemble Model Performance Summary:\")\nprint(cv_results)\n\n# Visualize top performers with a bar plot\nplt.figure(figsize=(14, 6))\ntop_results = cv_results.head(5)\nbar = plt.bar(\n    top_results.index,\n    top_results['Mean'],\n    yerr=top_results['Standard Deviation'],\n    capsize=10,\n    color='skyblue',\n    edgecolor='navy'\n)\n\n# Add values on top of bars\nfor i, b in enumerate(bar):\n    plt.text(\n        b.get_x() + b.get_width()/2,\n        b.get_height() + 0.005,\n        f\"{top_results['Mean'].iloc[i]:.4f}\",\n        ha='center',\n        fontweight='bold'\n    )\n\nplt.title('Top 5 Ensemble Models by Accuracy', fontsize=14)\nplt.xlabel('Model', fontsize=12)\nplt.ylabel('Mean Accuracy Score', fontsize=12)\nplt.ylim(top=1.0)\nplt.grid(axis='y', linestyle='--', alpha=0.7)\nplt.xticks(rotation=45, ha='right')\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-13T01:27:22.822690Z","iopub.execute_input":"2025-03-13T01:27:22.822941Z","iopub.status.idle":"2025-03-13T01:27:26.167465Z","shell.execute_reply.started":"2025-03-13T01:27:22.822922Z","shell.execute_reply":"2025-03-13T01:27:26.166370Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n# Preprocess the test data\nX_test = test[filtered_features]\nX_test = pd.DataFrame(preprocessor.transform(X_test), columns=filtered_features)\n\n# Use the trained model to make predictions\nbest_model =  GradientBoostingClassifier(n_estimators=100, random_state=seed)\nbest_model.fit(X_train, y_train)\ny_test_pred = best_model.predict(X_test)\n\n# Create a submission DataFrame\nsubmission = pd.DataFrame({\n    'id': test['id'],\n    'sii': y_test_pred\n})\n\n# Save the submission DataFrame to a CSV file\nsubmission.to_csv('submission.csv', index=False)\n\n\n\nprint(\"Submission file created successfully.\")  # In ra thông báo khi hoàn thành","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-13T01:27:26.168194Z","iopub.execute_input":"2025-03-13T01:27:26.168603Z","iopub.status.idle":"2025-03-13T01:27:32.398412Z","shell.execute_reply.started":"2025-03-13T01:27:26.168585Z","shell.execute_reply":"2025-03-13T01:27:32.397554Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submiss = pd.read_csv('/kaggle/working/submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-13T01:27:32.399082Z","iopub.execute_input":"2025-03-13T01:27:32.399286Z","iopub.status.idle":"2025-03-13T01:27:32.406506Z","shell.execute_reply.started":"2025-03-13T01:27:32.399261Z","shell.execute_reply":"2025-03-13T01:27:32.405097Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}