{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport polars as pl\nimport lightgbm as lgb\nfrom glob import glob\nfrom tqdm.auto import tqdm\nfrom matplotlib import pyplot as plt\nfrom sklearn.preprocessing import StandardScaler, PowerTransformer\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-12-20T08:20:59.269194Z","iopub.execute_input":"2024-12-20T08:20:59.269678Z","iopub.status.idle":"2024-12-20T08:20:59.279687Z","shell.execute_reply.started":"2024-12-20T08:20:59.269636Z","shell.execute_reply":"2024-12-20T08:20:59.278218Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# suppress lightgbm output with custom objective\n# https://www.kaggle.com/competitions/learning-agency-lab-automated-essay-scoring-2/discussion/502231\n\nimport logging\n\nclass CustomLogger:\n    def __init__(self):\n        self.logger = logging.getLogger(\"lightgbm_custom\")\n        self.logger.setLevel(logging.ERROR)\n\n    def info(self, message):\n        self.logger.info(message)\n\n    def warning(self, message):\n        pass\n\n    def error(self, message):\n        self.logger.error(message)\n\nlgb.register_logger(CustomLogger())","metadata":{"execution":{"iopub.status.busy":"2024-12-20T08:20:59.282364Z","iopub.execute_input":"2024-12-20T08:20:59.282866Z","iopub.status.idle":"2024-12-20T08:20:59.300508Z","shell.execute_reply.started":"2024-12-20T08:20:59.282815Z","shell.execute_reply":"2024-12-20T08:20:59.298919Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"INPUT_DIR = \"/kaggle/input/child-mind-institute-problematic-internet-use/\"","metadata":{"execution":{"iopub.status.busy":"2024-12-20T08:20:59.303055Z","iopub.execute_input":"2024-12-20T08:20:59.303674Z","iopub.status.idle":"2024-12-20T08:20:59.315712Z","shell.execute_reply.started":"2024-12-20T08:20:59.303615Z","shell.execute_reply":"2024-12-20T08:20:59.314371Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Preprocess","metadata":{}},{"cell_type":"code","source":"def preprocess(df):\n    df_new = df.copy()\n    df_new[\"Physical-BMI\"] = df_new[\"Physical-BMI\"].replace(0, np.nan)\n    df_new[\"Physical-Weight\"] = df_new[\"Physical-Weight\"].replace(0, np.nan)\n    df_new[\"Fitness_Endurance-Time_Sec\"] += df_new[\"Fitness_Endurance-Time_Mins\"] * 60\n    df_new = df_new.drop([\"Fitness_Endurance-Max_Stage\", \"Fitness_Endurance-Time_Mins\", \"SDS-SDS_Total_T\"], axis=1)\n    \n    return df_new","metadata":{"execution":{"iopub.status.busy":"2024-12-20T08:20:59.317364Z","iopub.execute_input":"2024-12-20T08:20:59.317871Z","iopub.status.idle":"2024-12-20T08:20:59.330644Z","shell.execute_reply.started":"2024-12-20T08:20:59.317818Z","shell.execute_reply":"2024-12-20T08:20:59.329411Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(INPUT_DIR + \"train.csv\")\ndf_test = pd.read_csv(INPUT_DIR + \"test.csv\")\n\nIS_SUBMIT = len(df_test) != 20","metadata":{"execution":{"iopub.status.busy":"2024-12-20T08:20:59.334104Z","iopub.execute_input":"2024-12-20T08:20:59.334644Z","iopub.status.idle":"2024-12-20T08:20:59.429843Z","shell.execute_reply.started":"2024-12-20T08:20:59.334573Z","shell.execute_reply":"2024-12-20T08:20:59.428778Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = preprocess(df)\ndf_test = preprocess(df_test)","metadata":{"execution":{"iopub.status.busy":"2024-12-20T08:20:59.431225Z","iopub.execute_input":"2024-12-20T08:20:59.431682Z","iopub.status.idle":"2024-12-20T08:20:59.449012Z","shell.execute_reply.started":"2024-12-20T08:20:59.431635Z","shell.execute_reply":"2024-12-20T08:20:59.447729Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Missing Value Imputation (IterativeImputer)","metadata":{}},{"cell_type":"code","source":"drop_cols = list(set(df.columns) - set(df_test.columns))","metadata":{"execution":{"iopub.status.busy":"2024-12-20T08:20:59.450572Z","iopub.execute_input":"2024-12-20T08:20:59.451072Z","iopub.status.idle":"2024-12-20T08:20:59.457788Z","shell.execute_reply.started":"2024-12-20T08:20:59.451016Z","shell.execute_reply":"2024-12-20T08:20:59.456232Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_all = pd.concat([df, df_test])\n\nX = df_all.drop([\"id\", \"sii\"] + drop_cols, axis=1)\ncat_cols = X.columns[X.dtypes==\"object\"].tolist()\nX = X.drop(cat_cols, axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-12-20T08:20:59.459605Z","iopub.execute_input":"2024-12-20T08:20:59.460106Z","iopub.status.idle":"2024-12-20T08:20:59.482484Z","shell.execute_reply.started":"2024-12-20T08:20:59.460053Z","shell.execute_reply":"2024-12-20T08:20:59.481273Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scaler = StandardScaler().set_output(transform=\"pandas\")\nscaler.fit_transform(X.iloc[:len(df)])\nX = scaler.transform(X)\n\nimputer = IterativeImputer(random_state=0, verbose=1).set_output(transform=\"pandas\")\nimputer.fit(X.iloc[:len(df)])\nimputer.set_params(verbose=0)\n\nX = imputer.transform(X)\n\n# drop colums not to impute\nX = X.loc[:, ~X.columns.str.contains(\"BIA-BIA\")]\nX = X.loc[:, ~X.columns.str.contains(\"FGC-FGC_S\")]\nX = X.drop([\"CGAS-CGAS_Score\", \"SDS-SDS_Total_Raw\"], axis=1)\n\nfor col in X.columns:\n    df_all[col] = X[col].values","metadata":{"execution":{"iopub.status.busy":"2024-12-20T08:20:59.484139Z","iopub.execute_input":"2024-12-20T08:20:59.484621Z","iopub.status.idle":"2024-12-20T08:21:33.508669Z","shell.execute_reply.started":"2024-12-20T08:20:59.484571Z","shell.execute_reply":"2024-12-20T08:21:33.507062Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature from Parquet Files","metadata":{}},{"cell_type":"code","source":"files = glob(INPUT_DIR + \"series_train.parquet/*\")\nif IS_SUBMIT:\n    files += glob(INPUT_DIR + \"series_test.parquet/*\")\n\nlist_df = []\nfor file in tqdm(files):\n    df_series = (\n        pl.read_parquet(file)\n        .with_columns(\n            (\n                (pl.col(\"relative_date_PCIAT\") - pl.col(\"relative_date_PCIAT\").min())*24\n                + (pl.col(\"time_of_day\") // int(1e9)) / 3600\n            ).floor().cast(int).alias(\"total_hours\")\n        )\n        .filter(pl.col(\"non-wear_flag\") != 1)\n        .filter(pl.col(\"step\").count().over(\"total_hours\") == 12 * 60)\n        .group_by(\"total_hours\").agg(\n            pl.col(\"enmo\").std().alias(\"enmo_std\"),\n            pl.col(\"anglez\").std().alias(\"anglez_std\"),\n            pl.col(\"light\").std().alias(\"light_std\")\n        )\n        .with_columns(\n            (pl.col(\"total_hours\") % 24).alias(\"hour\"),\n            pl.lit(file.split(\"/\")[-1][3:]).alias(\"id\")\n        )\n    )\n    list_df.append(df_series.to_pandas())\n\ndf_series = pd.concat(list_df)\ndf_series[\"enmo_std\"] = np.log(df_series[\"enmo_std\"] + 0.01)\ndf_series[\"anglez_std\"] = np.log(df_series[\"anglez_std\"] + 1)\ndf_series[\"light_std\"] = np.log(df_series[\"light_std\"] + 0.01)\n\ndf_agg = df_series.groupby(\"id\")[[\"enmo_std\", \"anglez_std\", \"light_std\"]].agg([\"mean\", \"std\"]).reset_index()\ndf_agg.columns = [cols[0] + \"_\" + cols[1] if cols[1] != \"\" else cols[0] for cols in df_agg.columns]\ndf_agg","metadata":{"execution":{"iopub.status.busy":"2024-12-20T08:21:33.510381Z","iopub.execute_input":"2024-12-20T08:21:33.510863Z","iopub.status.idle":"2024-12-20T08:22:28.137102Z","shell.execute_reply.started":"2024-12-20T08:21:33.510811Z","shell.execute_reply":"2024-12-20T08:22:28.135757Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_all = df_all.merge(df_agg, how=\"left\", on=\"id\")\ndf_all[cat_cols] = df_all[cat_cols].astype(\"category\")\n\ndf = df_all.iloc[:len(df)]\ndf_test = df_all.iloc[len(df):]\n    \ndf = df.dropna(subset=[\"sii\"]).reset_index(drop=True)\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2024-12-20T08:22:28.140587Z","iopub.execute_input":"2024-12-20T08:22:28.141043Z","iopub.status.idle":"2024-12-20T08:22:28.180675Z","shell.execute_reply.started":"2024-12-20T08:22:28.141002Z","shell.execute_reply":"2024-12-20T08:22:28.179432Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = df.drop([\"id\", \"sii\"] + drop_cols, axis=1)\ny = df[\"sii\"].astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-12-20T08:22:28.181933Z","iopub.execute_input":"2024-12-20T08:22:28.182326Z","iopub.status.idle":"2024-12-20T08:22:28.190203Z","shell.execute_reply.started":"2024-12-20T08:22:28.182281Z","shell.execute_reply":"2024-12-20T08:22:28.189025Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Custom Functions","metadata":{}},{"cell_type":"markdown","source":"### References\n\nMy Public Notebook  \nhttps://www.kaggle.com/code/rsakata/cmi-piu-optimize-qwk-by-lgb\n\nDiscussion Post (from [@chumajin](https://www.kaggle.com/chumajin))  \nhttps://www.kaggle.com/competitions/child-mind-institute-problematic-internet-use/discussion/535052\n\nMy original blog post (Japanese)  \nhttps://zenn.dev/jackthekaggler/articles/cf988ca341e34ed83034","metadata":{}},{"cell_type":"code","source":"a = y.mean()\nb = y.var(ddof=0)\ny_min = y.min()\ny_max = y.max()","metadata":{"execution":{"iopub.status.busy":"2024-12-20T08:22:28.191596Z","iopub.execute_input":"2024-12-20T08:22:28.191986Z","iopub.status.idle":"2024-12-20T08:22:28.207667Z","shell.execute_reply.started":"2024-12-20T08:22:28.191924Z","shell.execute_reply":"2024-12-20T08:22:28.206436Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# This function can calculate the original Quadratic Weighted Kappa and also supports continuous predictions.\ndef generalized_qwk(labels, preds):\n    preds = preds.clip(y_min, y_max)\n    qwk = 1 - ((labels - preds)**2).sum() / ((preds - a)**2 + b).sum()\n    \n    return qwk\n\n\ndef qwk_metric(preds, data):\n    y_true = data.get_label().round()\n    y_pred = preds\n    qwk = generalized_qwk(y_true, y_pred)\n    \n    return 'QWK', qwk, True\n\n\ndef qwk_obj(preds, dtrain):\n    labels = dtrain.get_label()\n    preds = preds.clip(y_min, y_max)\n    f = 1/2*np.sum((preds-labels)**2)\n    g = 1/2*np.sum((preds-a)**2+b)\n    df = preds - labels\n    dg = preds - a\n    grad = (df/g - f*dg/g**2) * len(labels)\n    hess = np.ones(len(labels))\n    \n    return grad, hess","metadata":{"execution":{"iopub.status.busy":"2024-12-20T08:22:28.209032Z","iopub.execute_input":"2024-12-20T08:22:28.209389Z","iopub.status.idle":"2024-12-20T08:22:28.227838Z","shell.execute_reply.started":"2024-12-20T08:22:28.209356Z","shell.execute_reply":"2024-12-20T08:22:28.226429Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Nested CV (10 x 10)","metadata":{}},{"cell_type":"code","source":"params = {\n    \"objective\": qwk_obj,\n    \"metric\": \"None\",\n    \"learning_rate\": 0.01,\n    \"num_leaves\": 8,\n    \"feature_fraction_bynode\": 0.6,\n    \"min_data_in_leaf\": 200,\n}","metadata":{"execution":{"iopub.status.busy":"2024-12-20T08:22:28.229226Z","iopub.execute_input":"2024-12-20T08:22:28.229607Z","iopub.status.idle":"2024-12-20T08:22:28.249975Z","shell.execute_reply.started":"2024-12-20T08:22:28.229572Z","shell.execute_reply":"2024-12-20T08:22:28.248551Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"n_splits = 10\ninit_score = a","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T08:22:28.251259Z","iopub.execute_input":"2024-12-20T08:22:28.252313Z","iopub.status.idle":"2024-12-20T08:22:28.268576Z","shell.execute_reply.started":"2024-12-20T08:22:28.252232Z","shell.execute_reply":"2024-12-20T08:22:28.267248Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"list_target_and_preds_inner_oof = []\npreds_outer_oof = np.zeros(len(df))\npreds_test = np.zeros(len(df_test))\ndfs_importance = []\nskf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=10)\nfor fold, (idx_train, idx_test) in enumerate(skf.split(X, y)):\n    X_train, y_train = X.iloc[idx_train], y.iloc[idx_train]\n    X_test, y_test = X.iloc[idx_test], y.iloc[idx_test]\n    \n    folds = [(idx_train, idx_valid) for idx_train, idx_valid in skf.split(X_train, y_train)]\n    models = lgb.cv(\n        params=params,\n        train_set=lgb.Dataset(X_train, y_train, init_score=[init_score]*len(X_train)),\n        num_boost_round=10000,\n        folds=folds,\n        feval=qwk_metric,\n        callbacks=[\n            lgb.early_stopping(stopping_rounds=100),\n        ],\n        return_cvbooster=True\n    )[\"cvbooster\"].boosters\n\n    for fold, model in enumerate(models):\n        df_importance = pd.DataFrame(model.feature_importance(importance_type=\"gain\"), index=X_train.columns).reset_index()\n        df_importance.columns = [\"feature\", \"importance\"]\n        df_importance[\"fold\"] = fold\n        dfs_importance.append(df_importance)\n\n    preds_inner_oof = np.zeros(len(X_train))\n    for model, (idx_train, idx_valid) in zip(models, folds):\n        preds_inner_oof[idx_valid] = model.predict(X_train.iloc[idx_valid]) + init_score\n    list_target_and_preds_inner_oof.append(np.stack([np.array(y_train), preds_inner_oof]).T)\n\n    preds = np.zeros(len(X_test))\n    for model in models:\n        preds += (model.predict(X_test) + init_score) / n_splits\n        preds_test += (model.predict(df_test[X.columns]) + init_score) / n_splits / n_splits\n    preds_outer_oof[idx_test] = preds\n\ndf_importance = pd.concat(dfs_importance, ignore_index=True)\ndf_importance = df_importance.groupby(\"feature\")[\"importance\"].agg([\"mean\", \"std\"])\ndf_importance.to_csv(\"importance.csv\")\ndf_importance = df_importance.reset_index().sort_values(\"mean\").iloc[-30:]\nfig, ax = plt.subplots()\nax.barh(\n    y=range(len(df_importance)),\n    width=df_importance[\"mean\"],\n    xerr=df_importance[\"std\"],\n    tick_label=df_importance[\"feature\"]\n)\nplt.show()\nplt.close()","metadata":{"execution":{"iopub.status.busy":"2024-12-20T08:22:28.270337Z","iopub.execute_input":"2024-12-20T08:22:28.270805Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Optimize thresholds","metadata":{}},{"cell_type":"code","source":"def apply_thresholds(preds, thresholds):\n    preds_new = np.zeros(len(preds), dtype=int)\n    for threshold in thresholds:\n        preds_new += preds > threshold\n    \n    return preds_new","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# iterative grid search \ndef optimize_thresholds(y, preds):\n    thresholds = np.array([0.5, 1.5, 2.5])\n    best_qwk = -np.inf\n    while True:\n        update_flag = False\n        for i in range(len(thresholds)):\n            best_threshold = thresholds[i]\n            thresholds[i] = 0\n            for j in range(300):\n                thresholds[i] += 0.01\n                qwk_tmp = cohen_kappa_score(y, apply_thresholds(preds, thresholds), weights=\"quadratic\")\n                if qwk_tmp > best_qwk:\n                    update_flag = True\n                    best_qwk = qwk_tmp\n                    best_threshold = thresholds[i]\n                    print(\"updated thresholds:\", thresholds, \"QWK:\", best_qwk)\n            thresholds[i] = best_threshold\n        if not update_flag:\n            break\n        thresholds = np.sort(thresholds)\n    \n    return thresholds","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preds_all = np.concatenate(list_target_and_preds_inner_oof)\nthresholds = optimize_thresholds(preds_all[:, 0], preds_all[:, 1])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"qwks_inner = []\nfor fold in range(len(list_target_and_preds_inner_oof)):\n    labels = list_target_and_preds_inner_oof[fold][:, 0].astype(int)\n    preds = list_target_and_preds_inner_oof[fold][:, 1]\n    preds = apply_thresholds(preds, thresholds)\n    qwks_inner.append(cohen_kappa_score(labels, preds, weights=\"quadratic\"))\n    print(f\"fold: {fold}\\tQWK-inner: {qwks_inner[-1]:.4f}\")\n\npreds_outer_oof = preds_outer_oof.clip(y_min, y_max)\npreds_outer_oof = apply_thresholds(preds_outer_oof, thresholds)\nqwk_outer = cohen_kappa_score(y, preds_outer_oof, weights=\"quadratic\")\nprint(f\"QWK-inner: {np.mean(qwks_inner):.4f} ± {np.std(qwks_inner):.4f}\\tQWK-outer: {qwk_outer:.4f}\")\nprint()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preds_test = preds_test.clip(y_min, y_max)\npreds_test = apply_thresholds(preds_test, thresholds)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_submission = pd.read_csv(INPUT_DIR + \"sample_submission.csv\")\ndf_submission[\"sii\"] = preds_test\ndf_submission.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}