{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\n\nfrom sklearn.preprocessing import MinMaxScaler, StandardScaler, RobustScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, TensorDataset\n\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.decomposition import PCA\nimport pandas as pd\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nSEED = 42\nn_splits = 5","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-10-23T06:26:05.500675Z","iopub.execute_input":"2024-10-23T06:26:05.502837Z","iopub.status.idle":"2024-10-23T06:26:05.525710Z","shell.execute_reply.started":"2024-10-23T06:26:05.502742Z","shell.execute_reply":"2024-10-23T06:26:05.524318Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-23T06:26:05.530133Z","iopub.execute_input":"2024-10-23T06:26:05.530636Z","iopub.status.idle":"2024-10-23T06:26:05.542309Z","shell.execute_reply.started":"2024-10-23T06:26:05.530580Z","shell.execute_reply":"2024-10-23T06:26:05.540787Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tp_rounded = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n    return tp_rounded\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-23T06:26:05.544370Z","iopub.execute_input":"2024-10-23T06:26:05.544930Z","iopub.status.idle":"2024-10-23T06:27:43.150649Z","shell.execute_reply.started":"2024-10-23T06:26:05.544841Z","shell.execute_reply":"2024-10-23T06:27:43.149471Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-23T06:27:43.152120Z","iopub.execute_input":"2024-10-23T06:27:43.152511Z","iopub.status.idle":"2024-10-23T06:27:43.316317Z","shell.execute_reply.started":"2024-10-23T06:27:43.152469Z","shell.execute_reply":"2024-10-23T06:27:43.315140Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-23T06:27:43.320562Z","iopub.execute_input":"2024-10-23T06:27:43.321142Z","iopub.status.idle":"2024-10-23T06:27:43.541580Z","shell.execute_reply.started":"2024-10-23T06:27:43.321067Z","shell.execute_reply":"2024-10-23T06:27:43.539972Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nfrom sklearn.linear_model import LinearRegression\n\ndef remove_outliers(X, y, num_outliers):\n    sorted_indices = np.argsort(y)\n    return X.iloc[sorted_indices[num_outliers:-num_outliers]], y.iloc[sorted_indices[num_outliers:-num_outliers]]\n\ndef linear_regression_with_outlier_removal(train, test, weight_col, bmi_col, sii_col, num_outliers=300):\n    \"\"\"\n    指定された列名に基づいて、外れ値を削除し、線形回帰を行い、残差を計算する関数。\n\n    Parameters:\n    - train: トレーニングデータのDataFrame\n    - test: テストデータのDataFrame\n    - weight_col: 体重データの列名\n    - bmi_col: BMIデータの列名\n    - sii_col: SIIデータの列名\n    - num_outliers: 削除する外れ値の数（デフォルトは300）\n    \"\"\"\n    # 散布図を描画\n    plt.scatter(train[weight_col], train[bmi_col], c=train[sii_col], cmap='viridis', s=1)\n    plt.colorbar(label='Target Value')\n\n    unique_sii_values = train[sii_col].unique()\n\n    for sii_value in unique_sii_values:\n        mask = train[sii_col] == sii_value\n        X = train.loc[mask, [weight_col]]\n        y = train.loc[mask, bmi_col]\n        \n        # NaNを含むデータを除外\n        non_nan_mask = X.notna().all(axis=1) & y.notna()\n        X_non_nan = X[non_nan_mask]\n        y_non_nan = y[non_nan_mask]\n        \n        # 外れ値を削除\n        if len(y_non_nan) > 2 * num_outliers:  # データ数が十分にある場合のみ削除\n            X_non_nan, y_non_nan = remove_outliers(X_non_nan, y_non_nan, num_outliers)\n        \n        # 線形回帰モデルをフィッティング\n        model = LinearRegression()\n        model.fit(X_non_nan, y_non_nan)\n        \n        # 回帰直線を作成\n        x_range = np.linspace(X_non_nan.min(), X_non_nan.max(), 100)\n        y_range = model.predict(x_range)\n        \n        # 各sii値に対して異なる色の直線をプロット\n        plt.plot(x_range, y_range, label=f'sii={sii_value}')\n        \n        # 残差を計算し、新しい特徴量として保存\n        train_y_pred = model.predict(train[weight_col].dropna().values.reshape(-1, 1))\n        test_y_pred = model.predict(test[weight_col].dropna().values.reshape(-1, 1))\n        \n        # 残差の計算後、NaNがある場所にNaNを戻す\n        train_residual = np.full(len(train), np.nan)\n        train_residual[train[weight_col].notna()] = train[bmi_col].loc[train[weight_col].notna()] - train_y_pred\n\n        test_residual = np.full(len(test), np.nan)\n        test_residual[test[weight_col].notna()] = test[bmi_col].loc[test[weight_col].notna()] - test_y_pred\n        \n        train[f\"{weight_col}_{bmi_col}_res_{sii_value}\"] = train_residual\n        test[f\"{weight_col}_{bmi_col}_res_{sii_value}\"] = test_residual\n\n    # グラフのラベルを追加\n    plt.xlabel(weight_col)\n    plt.ylabel(bmi_col)\n    plt.legend()\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-23T06:27:43.543200Z","iopub.execute_input":"2024-10-23T06:27:43.543597Z","iopub.status.idle":"2024-10-23T06:27:43.559462Z","shell.execute_reply.started":"2024-10-23T06:27:43.543557Z","shell.execute_reply":"2024-10-23T06:27:43.558034Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"linear_regression_with_outlier_removal(train, test, \"Physical-Weight\", \"Physical-BMI\", \"sii\", num_outliers=300)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-23T06:27:43.560982Z","iopub.execute_input":"2024-10-23T06:27:43.561385Z","iopub.status.idle":"2024-10-23T06:27:44.042058Z","shell.execute_reply.started":"2024-10-23T06:27:43.561333Z","shell.execute_reply":"2024-10-23T06:27:44.040862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# linear_regression_with_outlier_removal(train, test, \"Physical-Weight\", \"Basic_Demos-Age\", \"sii\", num_outliers=300)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-23T06:27:44.043536Z","iopub.execute_input":"2024-10-23T06:27:44.043948Z","iopub.status.idle":"2024-10-23T06:27:44.049509Z","shell.execute_reply.started":"2024-10-23T06:27:44.043899Z","shell.execute_reply":"2024-10-23T06:27:44.048248Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# linear_regression_with_outlier_removal(train, test, \"Physical-Height\", \"Basic_Demos-Age\", \"sii\", num_outliers=300)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-23T06:27:44.051218Z","iopub.execute_input":"2024-10-23T06:27:44.051811Z","iopub.status.idle":"2024-10-23T06:27:44.060542Z","shell.execute_reply.started":"2024-10-23T06:27:44.051735Z","shell.execute_reply":"2024-10-23T06:27:44.059117Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#linear_regression_with_outlier_removal(train, test, \"Physical-BMI\", \"Physical-Systolic_BP\", \"sii\", num_outliers=300)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-23T06:27:44.062550Z","iopub.execute_input":"2024-10-23T06:27:44.063076Z","iopub.status.idle":"2024-10-23T06:27:44.073490Z","shell.execute_reply.started":"2024-10-23T06:27:44.063029Z","shell.execute_reply":"2024-10-23T06:27:44.071757Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# linear_regression_with_outlier_removal(train, test, \"Physical-Systolic_BP\", \"Physical-Diastolic_BP\", \"sii\", num_outliers=300)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-23T06:27:44.075105Z","iopub.execute_input":"2024-10-23T06:27:44.075492Z","iopub.status.idle":"2024-10-23T06:27:44.086619Z","shell.execute_reply.started":"2024-10-23T06:27:44.075453Z","shell.execute_reply":"2024-10-23T06:27:44.085327Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# linear_regression_with_outlier_removal(train, test, \"Physical-Systolic_BP\", \"Physical-HeartRate\", \"sii\", num_outliers=300)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-23T06:27:44.088256Z","iopub.execute_input":"2024-10-23T06:27:44.088777Z","iopub.status.idle":"2024-10-23T06:27:44.100818Z","shell.execute_reply.started":"2024-10-23T06:27:44.088717Z","shell.execute_reply":"2024-10-23T06:27:44.099340Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"linear_regression_with_outlier_removal(train, test, \"Physical-Diastolic_BP\", \"Physical-HeartRate\", \"sii\", num_outliers=300)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-23T06:27:44.102803Z","iopub.execute_input":"2024-10-23T06:27:44.103419Z","iopub.status.idle":"2024-10-23T06:27:44.596208Z","shell.execute_reply.started":"2024-10-23T06:27:44.103357Z","shell.execute_reply":"2024-10-23T06:27:44.594972Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-23T06:27:44.599613Z","iopub.execute_input":"2024-10-23T06:27:44.600026Z","iopub.status.idle":"2024-10-23T06:27:44.771379Z","shell.execute_reply.started":"2024-10-23T06:27:44.599985Z","shell.execute_reply":"2024-10-23T06:27:44.770159Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Params = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01  # Increased from 2.68e-06\n}\n\n\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED,\n    'tree_method': 'exact'\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10  # Increase this value\n}\n\n# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\n# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model)\n])\n# Train the ensemble model\nSubmission4 = TrainML(voting_model, test)\n\n# Save submission\n#Submission2.to_csv('submission.csv', index=False)\nSubmission4","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-23T06:27:44.772867Z","iopub.execute_input":"2024-10-23T06:27:44.773275Z","iopub.status.idle":"2024-10-23T06:28:34.931477Z","shell.execute_reply.started":"2024-10-23T06:27:44.773235Z","shell.execute_reply":"2024-10-23T06:28:34.930391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Submission4 = pd.DataFrame({\n    'id': sample['id'],\n    'sii': Submission4\n})\n\nSubmission4.to_csv('submission.csv', index=False)\nSubmission4","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-23T06:28:34.933166Z","iopub.execute_input":"2024-10-23T06:28:34.933662Z","iopub.status.idle":"2024-10-23T06:28:34.951207Z","shell.execute_reply.started":"2024-10-23T06:28:34.933606Z","shell.execute_reply":"2024-10-23T06:28:34.949955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}