{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":7074842,"sourceType":"datasetVersion","datasetId":4074593},{"sourceId":7570020,"sourceType":"datasetVersion","datasetId":4021289},{"sourceId":203746103,"sourceType":"kernelVersion"}],"dockerImageVersionId":30805,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install --no-index -U --find-links=/kaggle/input/tensorflow-2-15/tensorflow tensorflow==2.15.0\n!pip install --no-index -U --find-links=/kaggle/input/deeptables-v0-2-5/deeptables-0.2.5 deeptables==0.2.5\n!pip install --no-index -U --find-links=/kaggle/input/fix-deeptables/deeptables-0.2.6 deeptables==0.2.6","metadata":{"trusted":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-12-13T14:49:01.645449Z","iopub.execute_input":"2024-12-13T14:49:01.645710Z","iopub.status.idle":"2024-12-13T14:50:20.492703Z","shell.execute_reply.started":"2024-12-13T14:49:01.645683Z","shell.execute_reply":"2024-12-13T14:50:20.491542Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport math\nimport random\nimport warnings\nfrom pathlib import Path\nimport polars.selectors as cs\nimport matplotlib.pyplot as plt\nfrom colorama import Fore, Style\nfrom numpy.typing import ArrayLike\nfrom scipy.optimize import minimize\nfrom sklearn.base import BaseEstimator\nfrom polars.testing import assert_frame_equal\nfrom sklearn.metrics import cohen_kappa_score\nimport numpy as np, pandas as pd, polars as pl\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\n\nimport tensorflow as tf, deeptables as dt\nfrom tensorflow.keras import backend as K\nfrom tensorflow.keras.utils import plot_model\nfrom tensorflow.keras.optimizers.legacy import Adam\nfrom deeptables.models import DeepTable, ModelConfig, deepnets\n\nwarnings.filterwarnings(\"ignore\")\nprint('TensorFlow version:',tf.__version__+',',\n      'GPU =',tf.test.is_gpu_available())\nprint('DeepTables version:',dt.__version__)","metadata":{"trusted":true,"_kg_hide-output":false,"execution":{"iopub.status.busy":"2024-12-13T14:50:20.495056Z","iopub.execute_input":"2024-12-13T14:50:20.495813Z","iopub.status.idle":"2024-12-13T14:50:32.512402Z","shell.execute_reply.started":"2024-12-13T14:50:20.495770Z","shell.execute_reply":"2024-12-13T14:50:32.511508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def seed_everything(seed):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\nseed_everything(seed=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T14:50:32.513610Z","iopub.execute_input":"2024-12-13T14:50:32.514553Z","iopub.status.idle":"2024-12-13T14:50:32.519765Z","shell.execute_reply.started":"2024-12-13T14:50:32.514490Z","shell.execute_reply":"2024-12-13T14:50:32.518782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load data\nDATA_DIR = Path(\"/kaggle/input/child-mind-institute-problematic-internet-use\")\ntrain = pl.read_csv(DATA_DIR / \"train.csv\")\ntest = pl.read_csv(DATA_DIR / \"test.csv\")\nprint(\"Init train shape:\", train.shape)\nprint(\"Init test shape:\", test.shape,\"\\n\")\n\nTARGET_COLS = [\"PCIAT-PCIAT_01\", \"PCIAT-PCIAT_02\", \"PCIAT-PCIAT_03\",\n               \"PCIAT-PCIAT_04\", \"PCIAT-PCIAT_05\", \"PCIAT-PCIAT_06\",\n               \"PCIAT-PCIAT_07\", \"PCIAT-PCIAT_08\", \"PCIAT-PCIAT_09\",\n               \"PCIAT-PCIAT_10\", \"PCIAT-PCIAT_11\", \"PCIAT-PCIAT_12\",\n               \"PCIAT-PCIAT_13\", \"PCIAT-PCIAT_14\", \"PCIAT-PCIAT_15\",\n               \"PCIAT-PCIAT_16\", \"PCIAT-PCIAT_17\", \"PCIAT-PCIAT_18\",\n               \"PCIAT-PCIAT_19\", \"PCIAT-PCIAT_20\",\n               \"PCIAT-PCIAT_Total\", \"sii\"]\n\ntrain_test = pl.concat([train, test], how=\"diagonal\")\nassert_frame_equal(train, train_test[: train.height].select(train.columns))\nassert_frame_equal(test, train_test[train.height :].select(test.columns))\n\n# Cast string columns to categorical\ntrain_test = train_test.with_columns(cs.string().cast(pl.Categorical))\ntrain = train_test[: train.height]\ntest = train_test[train.height :]\n\n# Ignore rows with null values in TARGET_COLS\ntrain_without_null = train_test.drop_nulls(subset=TARGET_COLS)\nX = train_without_null.select(pl.col(\"*\").exclude([\"id\", \"PCIAT-Season\"] + TARGET_COLS))\nX_test = test.select(pl.col(\"*\").exclude([\"id\", \"PCIAT-Season\"] + TARGET_COLS))\n\n# Exclude columns with more than threshold ratio NaNs\nthreshold = 0.30\nlow_nan_cols = [col for col in X.columns if X[col].is_null().sum() / X.height < threshold]\nX = X.select(low_nan_cols)\nX_test = X_test.select(low_nan_cols)\n\ny = train_without_null.select(TARGET_COLS)\ny_sii = y.get_column(\"sii\").to_numpy()  # Ground truth\nprint(\"Final train shape:\", X.shape)\nprint(\"Final test shape:\", X_test.shape,\"\\n\")\n\ncat_features = X.select(cs.categorical()).columns\nnum_features = [i for i in X.columns if i not in cat_features]\nprint(\"len(cat_features):\", len(cat_features))\nprint(\"len(num_features):\", len(num_features))\n\n# Preprocessing numerical\nimputer = IterativeImputer(max_iter=20, random_state=0)\nX[num_features] = imputer.fit_transform(X[num_features])\nX_test[num_features] = imputer.transform(X_test[num_features])\nscaler = MinMaxScaler()\nX[num_features] = scaler.fit_transform(X[num_features])\nX_test[num_features] = scaler.transform(X_test[num_features])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T14:50:32.522085Z","iopub.execute_input":"2024-12-13T14:50:32.522631Z","iopub.status.idle":"2024-12-13T14:50:32.745208Z","shell.execute_reply.started":"2024-12-13T14:50:32.522602Z","shell.execute_reply":"2024-12-13T14:50:32.744286Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T14:50:32.746163Z","iopub.execute_input":"2024-12-13T14:50:32.746426Z","iopub.status.idle":"2024-12-13T14:50:32.751577Z","shell.execute_reply.started":"2024-12-13T14:50:32.746402Z","shell.execute_reply":"2024-12-13T14:50:32.750826Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"LR_START = 1e-3\nLR_MAX = 5e-3\nLR_MIN = 1e-3\nLR_RAMPUP_EPOCHS = 0\nLR_SUSTAIN_EPOCHS = 0\nEPOCHS = 10\n\ndef lrfn(epoch):\n    if epoch < LR_RAMPUP_EPOCHS:\n        lr = (LR_MAX - LR_START) / LR_RAMPUP_EPOCHS * epoch + LR_START\n    elif epoch < LR_RAMPUP_EPOCHS + LR_SUSTAIN_EPOCHS:\n        lr = LR_MAX\n    else:\n        decay_total_epochs = EPOCHS - LR_RAMPUP_EPOCHS - LR_SUSTAIN_EPOCHS - 1\n        decay_epoch_index = epoch - LR_RAMPUP_EPOCHS - LR_SUSTAIN_EPOCHS\n        phase = math.pi * decay_epoch_index / decay_total_epochs\n        cosine_decay = 0.5 * (1 + math.cos(phase))\n        lr = (LR_MAX - LR_MIN) * cosine_decay + LR_MIN    \n    return lr\n\nrng = [i for i in range(EPOCHS)]\nlr_y = [lrfn(x) for x in rng]\nplt.figure(figsize=(10, 4))\nplt.plot(rng, lr_y, '-o')\nplt.xlabel('Epoch'); plt.ylabel('LR')\nprint(\"Learning rate schedule: {:.3g} to {:.3g} to {:.3g}\". \\\n      format(lr_y[0], max(lr_y), lr_y[-1]))\nLR_Scheduler = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose=True)","metadata":{"trusted":true,"_kg_hide-output":false,"execution":{"iopub.status.busy":"2024-12-13T14:50:32.752844Z","iopub.execute_input":"2024-12-13T14:50:32.753531Z","iopub.status.idle":"2024-12-13T14:50:33.031858Z","shell.execute_reply.started":"2024-12-13T14:50:32.753480Z","shell.execute_reply":"2024-12-13T14:50:33.031028Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CFG:\n    folds = 5\n    epochs = 10\n    batch_size = 32\n    LR_Scheduler = [LR_Scheduler]\n    optimizer = Adam(learning_rate=1e-3)\n\n    conf = ModelConfig(auto_imputation=True,\n                       auto_discrete=False,\n                       categorical_columns='auto',\n                       fixed_embedding_dim=True,\n                       embeddings_output_dim=4,\n                       embedding_dropout=0.3,\n                       nets=deepnets.DeepFM,\n                       dnn_params={\n                           'hidden_units': ((512, 0.5, True),\n                                            (256, 0.5, True),\n                                            (128, 0.3, True),\n                                            (64, 0.2, True)),\n                           'dnn_activation': 'relu',\n                           },\n                       stacking_op='concat',\n                       output_use_bias=False,\n                       optimizer=optimizer,\n                       task='regression',\n                       loss='auto',\n                       metrics=[\"RootMeanSquaredError\"],\n                       earlystopping_patience=1,\n                       )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T14:50:33.032905Z","iopub.execute_input":"2024-12-13T14:50:33.033160Z","iopub.status.idle":"2024-12-13T14:50:33.039644Z","shell.execute_reply.started":"2024-12-13T14:50:33.033135Z","shell.execute_reply":"2024-12-13T14:50:33.038774Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n# Cross-validation\nskf = StratifiedKFold(n_splits=CFG.folds, shuffle=True, random_state=42)\noof_non_rounded = np.zeros(len(y_sii), dtype=float) \nvalid_scores = []\nmodels = []\nprint(\"Data shape:\", X.shape,\"\\n\")\nfor fi, (train_idx, valid_idx) in enumerate(skf.split(X, y_sii)):\n    print(\"#\"*25)\n    print(f\"### Fold {fi+1}/{CFG.folds} ...\")\n    print(\"#\"*25)\n\n    X_train, y_train = X[train_idx], y_sii[train_idx]\n    X_valid, y_valid = X[valid_idx], y_sii[valid_idx]\n\n    K.clear_session()\n    model = DeepTable(config=CFG.conf)\n    model.fit(X_train.to_pandas(), y_train,\n              validation_data=(X_valid.to_pandas(), y_valid),\n              callbacks=CFG.LR_Scheduler,\n              batch_size=CFG.batch_size, epochs=CFG.epochs, verbose=2)\n    models.append(model)\n\n    # Save model\n    os.makedirs(f\"/kaggle/working/models/fold{fi}\", exist_ok=True)\n    os.makedirs(f\"/tmp/workdir/kaggle/working/models/fold{fi}\", exist_ok=True)\n    model.save(f'/kaggle/working/models/fold{fi}')\n    os.system(f'cp -r /kaggle/working/models/fold{fi}/* /tmp/workdir/kaggle/working/models/fold{fi}/')\n    # Avoid some errors\n    with K.name_scope(CFG.optimizer.__class__.__name__):\n        for j, var in enumerate(CFG.optimizer.weights):\n            name = 'variable{}'.format(j)\n            CFG.optimizer.weights[j] = tf.Variable(var, name=name)\n    CFG.conf = CFG.conf._replace(optimizer=CFG.optimizer)\n\n    y_valid_pred = model.predict(X_valid.to_pandas(), verbose=1, batch_size=512).flatten()\n    oof_non_rounded[valid_idx] = y_valid_pred\n    y_valid_pred_rounded = y_valid_pred.round(0).astype(int)\n\n    valid_kappa = quadratic_weighted_kappa(y_valid, y_valid_pred_rounded)\n    valid_scores.append(valid_kappa)\n    print(f\"\\n{Fore.GREEN}{Style.BRIGHT}Fold {fi+1} | valid qwk: {valid_kappa:.4f}\\n\")\nprint(f\"\\n{Fore.BLUE}{Style.BRIGHT}OOF mean valid qwk ---> {np.mean(valid_scores):.4f}\\n\")\n\nkappa_optimizer = minimize(\n    evaluate_predictions,\n    x0=[0.5, 1.5, 2.5],\n    args=(y_sii, oof_non_rounded),\n    method=\"Nelder-Mead\",\n)\nassert kappa_optimizer.success, \"Optimization did not converge.\"\nbest_thresholds = kappa_optimizer.x\nprint(\"Best thresholds:\", best_thresholds)\noof_tuned = threshold_rounder(oof_non_rounded, best_thresholds)\ntuned_kappa = quadratic_weighted_kappa(y_sii, oof_tuned)\nprint(f\"\\n----> || Optimized qwk: {Fore.BLUE}{Style.BRIGHT}{tuned_kappa:.4f}\\n\")\n\n# Global maximum search as introduced in the notebook by the link below\n# https://www.kaggle.com/code/vitalykudelya/cmi-issues-with-the-metric-and-baseline#Issues-with-the-Quadratic-Weighted-Kappa-metric\nnew_best_thresholds = best_thresholds.copy()\nfor t_idx in range(3):\n    df_lst = []\n    for t in np.arange(0.0, 3.0, 0.001):\n        thresholds = best_thresholds.copy()\n        thresholds[t_idx] = t\n        score = -evaluate_predictions(thresholds, y_sii, oof_non_rounded)\n        df_lst.append({f't_{t_idx}': t, f'score_{t_idx}': score})\n    df = pd.DataFrame(df_lst)\n    new_best_thresholds[t_idx] = df.iloc[df[f'score_{t_idx}'].idxmax()][0]\nprint(\"New best thresholds:\", new_best_thresholds)\nnew_oof_tuned = threshold_rounder(oof_non_rounded, new_best_thresholds)\nnew_tuned_kappa = quadratic_weighted_kappa(y_sii, new_oof_tuned)\nprint(f\"\\n----> || New optimized qwk: {Fore.MAGENTA}{Style.BRIGHT}{new_tuned_kappa:.4f}\\n\")\n\nplot_model(model.get_model().model)","metadata":{"trusted":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-12-13T14:50:33.040873Z","iopub.execute_input":"2024-12-13T14:50:33.041117Z","iopub.status.idle":"2024-12-13T14:51:19.508354Z","shell.execute_reply.started":"2024-12-13T14:50:33.041094Z","shell.execute_reply":"2024-12-13T14:51:19.507483Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load models\ndef load_model(paths):\n    models = []\n    for fold in sorted(os.listdir(paths)):\n        path = os.path.join(paths, fold)\n        for file in os.listdir(path):\n            if file.endswith('.h5'):\n                models.append(DeepTable.load(path, file))\n    return models\n\nmodels = load_model(\"/kaggle/working/models\")\nprint('\\nmodels:', models)","metadata":{"trusted":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-12-13T14:51:19.509457Z","iopub.execute_input":"2024-12-13T14:51:19.509735Z","iopub.status.idle":"2024-12-13T14:51:21.676661Z","shell.execute_reply.started":"2024-12-13T14:51:19.509710Z","shell.execute_reply":"2024-12-13T14:51:21.675813Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Inference\nclass AvgModel:\n    def __init__(self, models: list[BaseEstimator]):\n        self.models = models\n    def predict(self, X: ArrayLike):\n        preds = []\n        for model in self.models:\n            pred = model.predict(X, verbose=1, batch_size=512).flatten()\n            preds.append(pred)\n        return np.mean(preds, axis=0)\n\navg_model = AvgModel(models)\ntest_pred = avg_model.predict(X_test.to_pandas())\ntest_pred_rounded = threshold_rounder(test_pred, new_best_thresholds)\n\ntest.select(\"id\").with_columns(\n    pl.Series(\"sii\", pl.Series(\"sii\", test_pred_rounded)),\n).write_csv(\"submission.csv\")\npd.read_csv(\"submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T14:51:21.679505Z","iopub.execute_input":"2024-12-13T14:51:21.679957Z","iopub.status.idle":"2024-12-13T14:51:22.646320Z","shell.execute_reply.started":"2024-12-13T14:51:21.679927Z","shell.execute_reply":"2024-12-13T14:51:22.645570Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}