{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## The QWK objective is taken from [this](https://www.kaggle.com/code/ravi20076/cmi2024-baseline-v2) notebook by [Ravi Ramakrishnan](https://www.kaggle.com/ravi20076)\n\nThe rest is my initial approach to this competition :)\n\nI like all in one approach (train, debug, submit in one notebook) where I logg everything so after a week and XXX experiments I can still keep track of what is going on in my experiments.\n\nYou can set `IS_SUBMIT=False` and disable internet access to submit from this notebook\n\nLGB parameters are not yet optimized <- that is the next step\n\nFeedback is a gift :heart:","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nfrom pathlib import Path\nimport random\nfrom typing import Dict, List, Any\nimport glob\nfrom fastcore.utils import partialler # https://fastpages.fast.ai/fastcore/\nfrom dataclasses import dataclass, field, asdict\n\nimport lightgbm as lgb\nimport optuna\nimport warnings\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score, ConfusionMatrixDisplay\nfrom sklearn.ensemble import IsolationForest\nfrom sklearn.neighbors import LocalOutlierFactor\nfrom wandb.integration.lightgbm import wandb_callback, log_summary\nfrom scipy.stats.mstats import winsorize\nfrom scipy.optimize import minimize\n\nwarnings.simplefilter('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-05T08:07:51.394662Z","iopub.execute_input":"2024-10-05T08:07:51.395590Z","iopub.status.idle":"2024-10-05T08:07:51.402861Z","shell.execute_reply.started":"2024-10-05T08:07:51.395545Z","shell.execute_reply":"2024-10-05T08:07:51.401705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DEBUG = False\nIS_SUBMIT = True\n\n@dataclass\nclass DatasetConfig:\n    path: Path = Path('/kaggle/input/child-mind-institute-problematic-internet-use')\n    missing_target_strategy: str = 'drop'\n    y_min = 0.\n    y_max = 3.\n    a = 0.5804093567251462\n    b = 0.5944115283335043\n    drop_corelated: bool = False\n    add_missin_count_per_row: bool = True\n#     add_outliers_info: Any = field(default_factory=IsolationForest)\n    add_outliers_info: Any = field(default_factory=partialler(LocalOutlierFactor, novelty=True))\n#     fill_strategy_for_outliers = -10_000.0 # outlier detector doesn't work with nan so I want to kinda put them in one cluster far away from everything else\n    fill_strategy_for_outliers = 'mean'\n\nlgb_config_factory = lambda : {\n        \"objective\": 'qwk_obj',\n        \"metric\": \"None\",\n        \"verbosity\": -1,\n        \"learning_rate\": 0.01,\n        \"num_leaves\": 24,\n#         'max_depth': 5,\n        \"feature_fraction\": 0.5,\n        \"num_boost_round\": 10000\n}\n\nlgb_callbacs_factory = lambda : [\n        lgb.early_stopping(stopping_rounds=100, verbose=True),\n        lgb.log_evaluation(100),\n        wandb_callback()\n    ]\n\n@dataclass\nclass Config:\n    dataset: DatasetConfig = field(default_factory=DatasetConfig)\n    lgb_config: Dict[str, Any] = field(default_factory=lgb_config_factory)\n    lgb_callbacs: List[Any] = field(default_factory=lgb_callbacs_factory)\n    init_score = 2.\n    seed: int = 0\n    n_folds: int = 5\n    debug: bool = DEBUG\n        \nconfig = Config()\nasdict(config)","metadata":{"execution":{"iopub.status.busy":"2024-10-05T08:07:51.404925Z","iopub.execute_input":"2024-10-05T08:07:51.406112Z","iopub.status.idle":"2024-10-05T08:07:51.423802Z","shell.execute_reply.started":"2024-10-05T08:07:51.406063Z","shell.execute_reply":"2024-10-05T08:07:51.422860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if DEBUG or IS_SUBMIT:\n    os.environ['WANDB_MODE'] = 'offline' \nelse:\n    os.environ['WANDB_MODE'] = 'online'\n\nimport datetime\nimport wandb\n\nNAME = datetime.datetime.now().strftime(\"%Y-%m-%d %H:%M:%S\")\n    \nif not IS_SUBMIT:\n    from kaggle_secrets import UserSecretsClient\n    user_secrets = UserSecretsClient()\n    secret_value_0 = user_secrets.get_secret(\"wandb\")\n\n    !wandb login $secret_value_0\n\nwandb.init(\n    project = 'CMI-PIU',\n    config = asdict(config),\n    name = NAME,\n    tags = [],\n    anonymous=\"never\" if not IS_SUBMIT else \"allow\"\n)","metadata":{"execution":{"iopub.status.busy":"2024-10-05T08:07:51.425026Z","iopub.execute_input":"2024-10-05T08:07:51.425368Z","iopub.status.idle":"2024-10-05T08:07:55.744775Z","shell.execute_reply.started":"2024-10-05T08:07:51.425323Z","shell.execute_reply":"2024-10-05T08:07:55.743736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(config.dataset.path / 'train.csv')\ntest_df = pd.read_csv(config.dataset.path / 'test.csv')\ndrop_cols = list(set(df.columns) - set(test_df.columns)) + ['id']\ndrop_cols = [col for col in drop_cols if col != 'sii']\n\nprint(f\"Drop columns: {drop_cols}\")\n\nif config.dataset.missing_target_strategy == 'drop':\n    df = df.dropna(subset=['sii']).reset_index(drop=True)\n    \ndf = df.drop(drop_cols, axis=1).reset_index(drop=True)\n\nif config.dataset.add_missin_count_per_row:\n    df['n_missing'] = df.isna().sum(1).values\n    test_df['n_missing'] = test_df.isna().sum(1).values\n\n    \ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-05T08:07:55.747268Z","iopub.execute_input":"2024-10-05T08:07:55.747625Z","iopub.status.idle":"2024-10-05T08:07:55.844821Z","shell.execute_reply.started":"2024-10-05T08:07:55.747590Z","shell.execute_reply":"2024-10-05T08:07:55.843798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dd_df = pd.read_csv(config.dataset.path / 'data_dictionary.csv')\ndd_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-05T08:07:55.846006Z","iopub.execute_input":"2024-10-05T08:07:55.846324Z","iopub.status.idle":"2024-10-05T08:07:55.861215Z","shell.execute_reply.started":"2024-10-05T08:07:55.846271Z","shell.execute_reply":"2024-10-05T08:07:55.860327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dd_df['Type'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-10-05T08:07:55.862288Z","iopub.execute_input":"2024-10-05T08:07:55.863559Z","iopub.status.idle":"2024-10-05T08:07:55.871544Z","shell.execute_reply.started":"2024-10-05T08:07:55.863523Z","shell.execute_reply":"2024-10-05T08:07:55.870260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols = dd_df[(dd_df['Type'] == 'categorical int') | (dd_df['Type'] == 'str')].loc[1:, 'Field'].tolist()\ncat_cols = list(set(cat_cols) - set(drop_cols))\nif cat_cols:\n    df[cat_cols] = df[cat_cols].astype(\"category\")\n    test_df[cat_cols] = test_df[cat_cols].astype(\"category\")\n    \ncont_cols = list(set(df.columns) - set(cat_cols) - {'sii'})\n\n\nif config.dataset.add_outliers_info is not None:\n    tmp_df = df[cont_cols].copy()\n    \n    if config.dataset.fill_strategy_for_outliers == 'mean':\n        fill_value = tmp_df.mean()\n    if isinstance(config.dataset.fill_strategy_for_outliers, float):\n        fill_value = config.dataset.fill_strategy_for_outliers\n        \n    detector = config.dataset.add_outliers_info.fit(tmp_df.fillna(fill_value).values)\n        \n    df['is_outlier'] = detector.predict(tmp_df.fillna(fill_value).values)\n    test_df['is_outlier'] = detector.predict(test_df[cont_cols].fillna(fill_value).values)\n    \n    cont_cols.append('is_outlier')","metadata":{"execution":{"iopub.status.busy":"2024-10-05T08:07:55.873040Z","iopub.execute_input":"2024-10-05T08:07:55.873468Z","iopub.status.idle":"2024-10-05T08:07:56.208425Z","shell.execute_reply.started":"2024-10-05T08:07:55.873413Z","shell.execute_reply":"2024-10-05T08:07:56.207373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom scipy.stats import chi2_contingency\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import mutual_info_score\n\n# sns.pairplot(Xy_to_plot, y_vars=['y'])\n\ndef cramers_v(x, y):\n    contingency_table = pd.crosstab(x, y)\n    chi2, p, dof, expected = chi2_contingency(contingency_table)\n    n = contingency_table.values.sum()\n    phi2 = chi2 / n\n    r, k = contingency_table.shape\n    phi2corr = max(0, phi2 - ((k-1)*(r-1))/(n-1))\n    r_corr = r - ((r-1)**2)/(n-1)\n    k_corr = k - ((k-1)**2)/(n-1)\n    return np.sqrt(phi2corr / min((k_corr-1), (r_corr-1)))\n\ndef compute_cramers_v_matrix(df, categorical_columns):\n    n = len(categorical_columns)\n    cramers_v_matrix = pd.DataFrame(np.zeros((n, n)), index=categorical_columns, columns=categorical_columns)\n    \n    for col1 in categorical_columns:\n        for col2 in categorical_columns:\n            if col1 == col2:\n                cramers_v_matrix.loc[col1, col2] = 1.0\n            elif cramers_v_matrix.loc[col1, col2] == 0:\n                v = cramers_v(df[col1], df[col2])\n                cramers_v_matrix.loc[col1, col2] = v\n                cramers_v_matrix.loc[col2, col1] = v\n    return cramers_v_matrix\n\n\ncorrelation_cat = compute_cramers_v_matrix(df, cat_cols)\ncorrelation_cont_pearson = df[cont_cols].corr()","metadata":{"execution":{"iopub.status.busy":"2024-10-05T08:07:56.209642Z","iopub.execute_input":"2024-10-05T08:07:56.209964Z","iopub.status.idle":"2024-10-05T08:07:58.302872Z","shell.execute_reply.started":"2024-10-05T08:07:56.209928Z","shell.execute_reply":"2024-10-05T08:07:58.301989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 10))\nsns.set_palette(\"Blues\")\nsns.heatmap(correlation_cont_pearson)","metadata":{"execution":{"iopub.status.busy":"2024-10-05T08:07:58.306936Z","iopub.execute_input":"2024-10-05T08:07:58.307355Z","iopub.status.idle":"2024-10-05T08:07:59.328569Z","shell.execute_reply.started":"2024-10-05T08:07:58.307307Z","shell.execute_reply":"2024-10-05T08:07:59.327570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 10))\nsns.set_palette(\"Blues\")\nsns.heatmap(correlation_cat)","metadata":{"execution":{"iopub.status.busy":"2024-10-05T08:07:59.330024Z","iopub.execute_input":"2024-10-05T08:07:59.330442Z","iopub.status.idle":"2024-10-05T08:08:00.056110Z","shell.execute_reply.started":"2024-10-05T08:07:59.330397Z","shell.execute_reply":"2024-10-05T08:08:00.055088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pick_to_drop(df, treshold=0.9):\n    to_drop = set()\n    \n    for x in range(len(df)):\n        for y in range(len(df)):\n            if x == y:\n                continue\n            \n            if df.iloc[x, y] > treshold:\n                if df.columns[y] not in to_drop:\n                    to_drop.add(df.columns[x])\n                    \n    return to_drop\n\ncat_correlated_to_drop = pick_to_drop(correlation_cat)\ncont_correlated_to_drop = pick_to_drop(correlation_cont_pearson)\n\ncorrelated_to_drop = list(cat_correlated_to_drop.union(cont_correlated_to_drop))\n\nif not IS_SUBMIT:\n    wandb.summary['correlated_to_drop'] = correlated_to_drop\n\nif config.dataset.drop_corelated:\n    df = df.drop(correlated_to_drop, axis=1)\n    test_df = test_df.drop(correlated_to_drop, axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-05T08:08:00.057329Z","iopub.execute_input":"2024-10-05T08:08:00.057653Z","iopub.status.idle":"2024-10-05T08:08:00.116977Z","shell.execute_reply.started":"2024-10-05T08:08:00.057619Z","shell.execute_reply":"2024-10-05T08:08:00.116068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skf = StratifiedKFold(config.n_folds, shuffle=True, random_state=config.seed)\nfolds = [(idx_train, idx_valid) for idx_train, idx_valid in skf.split(df, df['sii'])]","metadata":{"execution":{"iopub.status.busy":"2024-10-05T08:08:00.118278Z","iopub.execute_input":"2024-10-05T08:08:00.118636Z","iopub.status.idle":"2024-10-05T08:08:00.130482Z","shell.execute_reply.started":"2024-10-05T08:08:00.118602Z","shell.execute_reply":"2024-10-05T08:08:00.129605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def quadratic_weighted_kappa(preds, data):\n    y_true = data.get_label()\n    y_pred = preds.clip(config.dataset.y_min, config.dataset.y_max).round()\n    qwk = cohen_kappa_score(y_true, y_pred, weights=\"quadratic\")\n    return 'QWK', qwk, True\n\na, b = config.dataset.a, config.dataset.b\n\ndef qwk_obj(preds, dtrain):\n    labels = dtrain.get_label()\n    preds = preds.clip(config.dataset.y_min, config.dataset.y_max)\n    f = 1/2 * np.sum((preds - labels)**2)\n    g = 1/2 * np.sum((preds - a)**2 + b)\n    df = preds - labels\n    dg = preds - a\n    grad = (df/g - f*dg/g**2)*len(labels)\n    hess = np.ones(len(labels))\n    return grad, hess","metadata":{"execution":{"iopub.status.busy":"2024-10-05T08:08:00.131760Z","iopub.execute_input":"2024-10-05T08:08:00.132263Z","iopub.status.idle":"2024-10-05T08:08:00.140637Z","shell.execute_reply.started":"2024-10-05T08:08:00.132215Z","shell.execute_reply":"2024-10-05T08:08:00.139685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb_params = config.lgb_config.copy()\n\nif lgb_params['objective'] == 'qwk_obj':\n    lgb_params['objective'] = qwk_obj\n\nmodels = lgb.cv(\n    params=lgb_params,\n    train_set=lgb.Dataset(df.drop(['sii'], axis=1), df['sii'], init_score=[config.init_score]*len(df)),\n    folds=folds,\n    feval=quadratic_weighted_kappa,\n    callbacks=config.lgb_callbacs,\n    return_cvbooster=True,\n)[\"cvbooster\"].boosters\n\npreds_oof = np.zeros(len(df))\nfor fold, (model, (idx_train, idx_valid)) in enumerate(zip(models, folds)):\n    preds_oof[idx_valid] = model.predict(df.drop(['sii'], axis=1).iloc[idx_valid]) + config.init_score\n    model.save_model(f'best_model_fold_{fold}.txt', num_iteration=model.best_iteration)","metadata":{"execution":{"iopub.status.busy":"2024-10-05T08:08:00.141839Z","iopub.execute_input":"2024-10-05T08:08:00.142164Z","iopub.status.idle":"2024-10-05T08:08:27.218761Z","shell.execute_reply.started":"2024-10-05T08:08:00.142120Z","shell.execute_reply":"2024-10-05T08:08:27.217726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_oof_dummy_round = preds_oof.clip(config.dataset.y_min, config.dataset.y_max).round()\nqwk_dummy_round = cohen_kappa_score(df['sii'], preds_oof_dummy_round, weights=\"quadratic\")\nprint(\"QWK:\", qwk_dummy_round)\nwandb.log({'oof QWK dummy round': qwk_dummy_round})","metadata":{"execution":{"iopub.status.busy":"2024-10-05T08:08:27.220293Z","iopub.execute_input":"2024-10-05T08:08:27.221011Z","iopub.status.idle":"2024-10-05T08:08:27.233276Z","shell.execute_reply.started":"2024-10-05T08:08:27.220962Z","shell.execute_reply":"2024-10-05T08:08:27.232340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots()\nConfusionMatrixDisplay.from_predictions(df['sii'], preds_oof_dummy_round, ax=ax)\nax.set_title('Confusion matrix without tuned thresholds')\n\nwandb.log({\"Confusion matrix without tuned thresholds\": wandb.Image(fig)})","metadata":{"execution":{"iopub.status.busy":"2024-10-05T08:08:27.234621Z","iopub.execute_input":"2024-10-05T08:08:27.235510Z","iopub.status.idle":"2024-10-05T08:08:27.732558Z","shell.execute_reply.started":"2024-10-05T08:08:27.235460Z","shell.execute_reply":"2024-10-05T08:08:27.731731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def round_with_thresholds(raw_preds, thresholds):\n    \"\"\"Round the raw predictions using specified thresholds\n    \n    Parameters\n    ----------\n    raw_preds: raw predictions of the regressor, array of n_samples float values\n    thresholds: 3-element float array\n    \n    Returns\n    -------\n    rounded_preds: rounded predictions, array of n_samples int values in range 0..3\n    \"\"\"\n    return np.where(raw_preds < thresholds[0], 0,\n                    np.where(raw_preds < thresholds[1], 1,\n                             np.where(raw_preds < thresholds[2], 2, 3)))\n\n\ndef fun(thresholds, y_true, raw_preds):\n    \"\"\"Function to be minimized: negative quadratic kappa score\n    \n    Parameters:\n    thresholds: ndarray of shape (3, )\n    y_true: ndarray of shape (n_samples, )\n    raw_preds: ndarray of shape (n_samples, )\n    \n    Returns:\n    negative quadratic kappa score for the predictions rounded at the specified thresholds\n    \"\"\"\n    rounded_preds = round_with_thresholds(raw_preds, thresholds)\n    return - cohen_kappa_score(y_true, rounded_preds, weights='quadratic')\n\n# Determine the thresholds which give the highest quadratic kappa score\nres = minimize(fun, x0=[0.5, 1.5, 2.5], args=(df['sii'].values, preds_oof), method='Powell')\nassert res.success\noof_tuned = round_with_thresholds(preds_oof, res.x)\nprint(f\"# Optimized thresholds: {res.x.round(2)}\")\n# print(f\"# Score with default rounding:     {cohen_kappa_score(df['sii'], oof, weights='quadratic'):.3f}\")\nqwk_optimized_round = cohen_kappa_score(df['sii'], oof_tuned, weights='quadratic')\nprint(f\"# Score with optimized thresholds: {qwk_optimized_round:.3f}\")\nwandb.log({'oof QWK optimized round': qwk_optimized_round})\n\nfig, ax = plt.subplots()\nConfusionMatrixDisplay.from_predictions(df['sii'], oof_tuned, ax=ax)\nax.set_title('Confusion matrix with tuned thresholds')\n\nwandb.log({\"Confusion matrix with tuned thresholds\": wandb.Image(fig)})","metadata":{"execution":{"iopub.status.busy":"2024-10-05T08:08:27.733733Z","iopub.execute_input":"2024-10-05T08:08:27.734060Z","iopub.status.idle":"2024-10-05T08:08:28.724350Z","shell.execute_reply.started":"2024-10-05T08:08:27.734024Z","shell.execute_reply":"2024-10-05T08:08:28.723368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"names = models[0].feature_name()\ntable = wandb.Table(data=[[\n        name, *fold_values\n    ] for (name, fold_values) in zip(names, np.vstack([model.feature_importance('split') for model in models]).T)], \n    columns=['name', *[f'importance_fold_{f}' for f in range(config.n_folds)]])\nwandb.log({\n            \"feature-importance-split\": table\n}\n    )\n\nnames = models[0].feature_name()\ntable = wandb.Table(data=[[\n        name, *fold_values\n    ] for (name, fold_values) in zip(names, np.vstack([model.feature_importance('gain') for model in models]).T)], \n    columns=['name', *[f'importance_fold_{f}' for f in range(config.n_folds)]])\nwandb.log({\n            \"feature-importance-gain\": table\n}\n    )","metadata":{"execution":{"iopub.status.busy":"2024-10-05T08:08:28.725888Z","iopub.execute_input":"2024-10-05T08:08:28.726595Z","iopub.status.idle":"2024-10-05T08:08:29.085678Z","shell.execute_reply.started":"2024-10-05T08:08:28.726544Z","shell.execute_reply":"2024-10-05T08:08:29.084851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wandb.finish()","metadata":{"execution":{"iopub.status.busy":"2024-10-05T08:08:29.086800Z","iopub.execute_input":"2024-10-05T08:08:29.087090Z","iopub.status.idle":"2024-10-05T08:08:34.561011Z","shell.execute_reply.started":"2024-10-05T08:08:29.087058Z","shell.execute_reply":"2024-10-05T08:08:34.560072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss = pd.read_csv(config.dataset.path / 'sample_submission.csv')\nss.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-05T08:08:34.562174Z","iopub.execute_input":"2024-10-05T08:08:34.562524Z","iopub.status.idle":"2024-10-05T08:08:34.574509Z","shell.execute_reply.started":"2024-10-05T08:08:34.562485Z","shell.execute_reply":"2024-10-05T08:08:34.573497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ids = test_df.pop('id')","metadata":{"execution":{"iopub.status.busy":"2024-10-05T08:08:34.575593Z","iopub.execute_input":"2024-10-05T08:08:34.575877Z","iopub.status.idle":"2024-10-05T08:08:34.581326Z","shell.execute_reply.started":"2024-10-05T08:08:34.575845Z","shell.execute_reply":"2024-10-05T08:08:34.580373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = []\nfor model in models:\n    preds.append(model.predict(test_df) + config.init_score)\n                 \npreds = np.mean(preds, 0)\npreds = round_with_thresholds(preds, res.x).astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-10-05T08:08:34.582721Z","iopub.execute_input":"2024-10-05T08:08:34.583042Z","iopub.status.idle":"2024-10-05T08:08:34.654872Z","shell.execute_reply.started":"2024-10-05T08:08:34.583008Z","shell.execute_reply":"2024-10-05T08:08:34.653857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss['id'] = test_ids\nss['sii'] = preds","metadata":{"execution":{"iopub.status.busy":"2024-10-05T08:08:34.656119Z","iopub.execute_input":"2024-10-05T08:08:34.656477Z","iopub.status.idle":"2024-10-05T08:08:34.660922Z","shell.execute_reply.started":"2024-10-05T08:08:34.656440Z","shell.execute_reply":"2024-10-05T08:08:34.660001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-05T08:08:34.664983Z","iopub.execute_input":"2024-10-05T08:08:34.665389Z","iopub.status.idle":"2024-10-05T08:08:34.671057Z","shell.execute_reply.started":"2024-10-05T08:08:34.665346Z","shell.execute_reply":"2024-10-05T08:08:34.670105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss","metadata":{"execution":{"iopub.status.busy":"2024-10-05T08:08:34.672166Z","iopub.execute_input":"2024-10-05T08:08:34.672488Z","iopub.status.idle":"2024-10-05T08:08:34.684766Z","shell.execute_reply.started":"2024-10-05T08:08:34.672455Z","shell.execute_reply":"2024-10-05T08:08:34.683952Z"},"trusted":true},"execution_count":null,"outputs":[]}]}