{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":9629349,"sourceType":"datasetVersion","datasetId":5878331}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Thanks for some work earlier**\n\n* https://www.kaggle.com/code/ambrosm/piu-eda-which-makes-sense\n* https://www.kaggle.com/code/abdmental01/cmi-best-single-model\n* https://www.kaggle.com/code/lennarthaupts/cmi-detecting-problematic-digital-behavior\n\nBased on the previous work to build timing features, this Notebook splits into building timing features on the day and night and quarterly dimensions","metadata":{}},{"cell_type":"code","source":"import os\nimport random\nfrom sklearn.base import clone\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nfrom IPython.display import clear_output\nimport pandas as pd\nimport numpy as np\nimport lightgbm\nfrom scipy.optimize import minimize\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score, ConfusionMatrixDisplay\nfrom catboost import CatBoostRegressor, Pool, CatBoostClassifier\nfrom lightgbm import LGBMRegressor, LGBMClassifier, Booster\nimport warnings\nfrom colorama import Fore, Style  # output style lib\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-23T12:05:40.359421Z","iopub.execute_input":"2024-10-23T12:05:40.360032Z","iopub.status.idle":"2024-10-23T12:05:40.370249Z","shell.execute_reply.started":"2024-10-23T12:05:40.359968Z","shell.execute_reply":"2024-10-23T12:05:40.368750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)","metadata":{"execution":{"iopub.status.busy":"2024-10-23T12:05:40.373252Z","iopub.execute_input":"2024-10-23T12:05:40.373778Z","iopub.status.idle":"2024-10-23T12:05:40.387515Z","shell.execute_reply.started":"2024-10-23T12:05:40.373721Z","shell.execute_reply":"2024-10-23T12:05:40.386268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load datasets\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-23T12:05:40.389257Z","iopub.execute_input":"2024-10-23T12:05:40.389746Z","iopub.status.idle":"2024-10-23T12:05:40.464750Z","shell.execute_reply.started":"2024-10-23T12:05:40.389690Z","shell.execute_reply":"2024-10-23T12:05:40.463681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"agg_funcs = {\n    'mean': 'mean',\n    'std': 'std',\n    'min': 'min',\n    'max': 'max',\n    lambda x: x.quantile(0.25): 'Q1',\n    lambda x: x.quantile(0.50): 'Q2',\n    lambda x: x.quantile(0.75): 'Q3'\n}\n\n\ndef process_file(filename, dirname):\n    ts = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    ts = ts[ts['non-wear_flag'] == 0]\n\n    ts['id'] = filename.split('=')[1]\n    num_cols = ['X', 'Y', 'Z', 'enmo', 'anglez', 'light']\n    for col in num_cols:\n        ts[col] = ts[col].abs()\n\n    ts[\"hours\"] = ts[\"time_of_day\"] // (3_600 * 1_000_000_000)\n    ts = ts[((ts[\"hours\"] >= 22) | (ts[\"hours\"] <= 5)) | ((ts[\"hours\"] >= 7) & (ts[\"hours\"] <= 20))]\n    ts['day_night'] = np.where((ts[\"hours\"] >= 22) | (ts[\"hours\"] <= 5), 'night', 'day')\n\n    # Aggregate values are calculated separately for each quarter\n    quarter_data = []\n    for q in range(1, 5):  # Go through 1 to 4 quarters\n        for dn in ['day', 'night']:\n            quarter_stats = ts[(ts['quarter'] == q) & (ts['day_night'] == dn)].groupby('id').agg({\n                'X': agg_funcs.keys(),\n                'Y': agg_funcs.keys(),\n                'Z': agg_funcs.keys(),\n                'enmo': agg_funcs.keys(),\n                'anglez': agg_funcs.keys(),\n                'light': agg_funcs.keys()\n            })\n\n            # Simplify column names and add quarterly identifiers\n            quarter_stats.columns = [f'Q{q}_{dn}_{col[0]}_{col[1]}' for col in quarter_stats.columns]\n            quarter_data.append(quarter_stats)\n\n    return pd.concat(quarter_data, axis=1).reset_index()\n\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    # ts_feats, indexes = zip(*results)\n    df = pd.concat(results)\n    # df['id'] = indexes\n    return df\n\n\n# Load time series data\n## It will take about 10 minutes to build time series statistical features with training data, \n## which can be saved after the first build, \n## and the constructed features can be directly loaded in later runs\n\n# train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntrain_ts = pd.read_csv('/kaggle/input/cmi-train-time-series/train_ts.csv')\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\ntrain_ts","metadata":{"execution":{"iopub.status.busy":"2024-10-23T12:05:40.466111Z","iopub.execute_input":"2024-10-23T12:05:40.466504Z","iopub.status.idle":"2024-10-23T12:05:41.606401Z","shell.execute_reply.started":"2024-10-23T12:05:40.466461Z","shell.execute_reply":"2024-10-23T12:05:41.603084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"drop_cols = ['PCIAT-Season', 'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03', 'PCIAT-PCIAT_04',\n             'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06', 'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09',\n             'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12', 'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14',\n             'PCIAT-PCIAT_15', 'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18', 'PCIAT-PCIAT_19',\n             'PCIAT-PCIAT_20', 'PCIAT-PCIAT_Total']","metadata":{"execution":{"iopub.status.busy":"2024-10-23T12:05:41.610279Z","iopub.execute_input":"2024-10-23T12:05:41.610886Z","iopub.status.idle":"2024-10-23T12:05:41.618988Z","shell.execute_reply.started":"2024-10-23T12:05:41.610821Z","shell.execute_reply":"2024-10-23T12:05:41.617066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop(drop_cols, axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-23T12:05:41.620718Z","iopub.execute_input":"2024-10-23T12:05:41.621281Z","iopub.status.idle":"2024-10-23T12:05:41.635379Z","shell.execute_reply.started":"2024-10-23T12:05:41.621220Z","shell.execute_reply":"2024-10-23T12:05:41.633920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.merge(train_ts, how='left', on='id')\ntest = test.merge(test_ts, how='left', on='id')\n\ntrain.drop('id', inplace=True, axis=1)\ntest.drop('id', inplace=True, axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-23T12:05:41.637266Z","iopub.execute_input":"2024-10-23T12:05:41.637819Z","iopub.status.idle":"2024-10-23T12:05:41.679919Z","shell.execute_reply.started":"2024-10-23T12:05:41.637741Z","shell.execute_reply":"2024-10-23T12:05:41.678287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.dropna(subset=['sii'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-23T12:05:41.682296Z","iopub.execute_input":"2024-10-23T12:05:41.682879Z","iopub.status.idle":"2024-10-23T12:05:41.697108Z","shell.execute_reply.started":"2024-10-23T12:05:41.682814Z","shell.execute_reply":"2024-10-23T12:05:41.695395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols = []\nfor col in train.columns:\n    dtype = train[col].dtype\n    if pd.api.types.is_object_dtype(dtype):\n        cat_cols.append(col)\n\ncat_cols","metadata":{"execution":{"iopub.status.busy":"2024-10-23T12:05:41.699097Z","iopub.execute_input":"2024-10-23T12:05:41.699644Z","iopub.status.idle":"2024-10-23T12:05:41.742558Z","shell.execute_reply.started":"2024-10-23T12:05:41.699585Z","shell.execute_reply":"2024-10-23T12:05:41.741176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seed = 42\nn_splits = 5\n\nnp.random.seed(42)\n\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\ndef train_model(model_class, data):\n    data = data.copy()\n    X = data.drop('sii', axis=1)\n    y = data['sii']\n    \n    X[cat_cols] = X[cat_cols].fillna('missing')\n    \n    for col in cat_cols:\n        X[col] = X[col].astype('category')\n        mapping = create_mapping(col, X)\n        X[col] = X[col].replace(mapping).astype(int)\n\n    train_scores = []\n    valid_scores = []\n\n    skf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=seed)\n    oof_non_rounded = np.zeros(len(y), dtype=float)\n    oof_rounded = np.zeros(len(y), dtype=int)\n\n    models = []\n\n    for fold, (train_idx, valid_idx) in enumerate(tqdm(skf.split(X, y), total=n_splits)):\n        X_train, X_valid = X.iloc[train_idx], X.iloc[valid_idx]\n        y_train, y_valid = y.iloc[train_idx], y.iloc[valid_idx]\n\n        model = clone(model_class)\n\n        if isinstance(model_class, (CatBoostClassifier, CatBoostRegressor)):\n            train_pool = Pool(X_train, y_train, cat_features=cat_cols)\n            val_pool = Pool(X_valid, y_valid, cat_features=cat_cols)\n            model.fit(train_pool, eval_set=val_pool, verbose=500)\n\n        elif isinstance(model_class, (LGBMRegressor, LGBMClassifier)):\n            model.fit(X_train, y_train, eval_set=[(X_valid, y_valid)], categorical_feature=cat_cols)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_valid)\n\n        oof_non_rounded[valid_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[valid_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_valid, y_val_pred_rounded)\n\n        train_scores.append(train_kappa)\n        valid_scores.append(val_kappa)\n        models.append(model)\n\n        print(f\"Fold {fold + 1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_scores):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(valid_scores):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions, x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded),\n                              method='Nelder-Mead')  # Nelder-Mead | # Powell\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    opt_weight = KappaOPtimizer.x\n    oof_tuned = threshold_Rounder(oof_non_rounded, opt_weight)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    return models, opt_weight, np.mean(train_scores), np.mean(valid_scores)\n\n\nlgb_params = {'n_estimators': 700, 'learning_rate': 0.02, 'max_depth': 14, 'num_leaves': 250, 'min_data_in_leaf': 55,\n              'feature_fraction': 0.65, 'bagging_fraction': 0.4, 'bagging_freq': 10, 'lambda_l1': 1.5101399570185197,\n              'lambda_l2': 5.410621798275615e-06, 'min_child_samples': 35, 'subsample': 0.95, 'colsample_bytree': 1.0,\n              'reg_alpha': 0.025255054819997888, 'min_split_gain': 0.8056566656167757,\n              'min_child_weight': 0.023203626491117978, 'iterations': 305, 'depth': 10,\n              'l2_leaf_reg': 2.654917385797282e-07, 'random_state': seed, 'verbose': -1}\n\nlgb_model = LGBMRegressor(**lgb_params)\nmodels, opt_weight, train_score, valid_score = train_model(lgb_model, train)","metadata":{"execution":{"iopub.status.busy":"2024-10-23T12:06:15.582219Z","iopub.execute_input":"2024-10-23T12:06:15.582831Z","iopub.status.idle":"2024-10-23T12:06:25.394510Z","shell.execute_reply.started":"2024-10-23T12:06:15.582745Z","shell.execute_reply":"2024-10-23T12:06:25.393126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame({'feat': train.columns.drop('sii'), 'imp': models[0].feature_importances_}).sort_values('imp',\n                                                                                                     ascending=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-23T12:05:51.648034Z","iopub.execute_input":"2024-10-23T12:05:51.648447Z","iopub.status.idle":"2024-10-23T12:05:51.666236Z","shell.execute_reply.started":"2024-10-23T12:05:51.648402Z","shell.execute_reply":"2024-10-23T12:05:51.664954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def inference(models, data, opt_weight):\n    data[cat_cols] = data[cat_cols].fillna(\"missing\")\n\n    if isinstance(models[0], (CatBoostClassifier, CatBoostRegressor)):\n        pool = Pool(data, cat_features=cat_cols)\n        preds = np.mean([model.predict(pool) for model in models], axis=0)\n\n    elif isinstance(models[0], (LGBMRegressor, LGBMClassifier)):\n        for col in cat_cols:\n            data[col] = data[col].astype('category')\n            mapping = create_mapping(col, data)\n            data[col] = data[col].replace(mapping).astype(int)\n        preds = np.mean([model.predict(data) for model in models], axis=0)\n    \n    preds_tuned = threshold_Rounder(preds, opt_weight)\n    \n    return preds_tuned\n\n\npreds = inference(models, test, opt_weight)\npreds\n","metadata":{"execution":{"iopub.status.busy":"2024-10-23T12:05:51.668045Z","iopub.execute_input":"2024-10-23T12:05:51.668455Z","iopub.status.idle":"2024-10-23T12:05:51.740583Z","shell.execute_reply.started":"2024-10-23T12:05:51.668413Z","shell.execute_reply":"2024-10-23T12:05:51.739402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': preds\n    })\nsubmission","metadata":{"execution":{"iopub.status.busy":"2024-10-23T12:05:51.742080Z","iopub.execute_input":"2024-10-23T12:05:51.742528Z","iopub.status.idle":"2024-10-23T12:05:51.757000Z","shell.execute_reply.started":"2024-10-23T12:05:51.742486Z","shell.execute_reply":"2024-10-23T12:05:51.755671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-23T12:05:51.761126Z","iopub.execute_input":"2024-10-23T12:05:51.761567Z","iopub.status.idle":"2024-10-23T12:05:51.771545Z","shell.execute_reply.started":"2024-10-23T12:05:51.761522Z","shell.execute_reply":"2024-10-23T12:05:51.770175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}