{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"}],"dockerImageVersionId":30648,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import warnings\n# warnings.simplefilter(\"ignore\", UserWarning)\nwarnings.simplefilter(\"ignore\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-02-08T10:36:53.625621Z","iopub.execute_input":"2024-02-08T10:36:53.626310Z","iopub.status.idle":"2024-02-08T10:36:53.663070Z","shell.execute_reply.started":"2024-02-08T10:36:53.626262Z","shell.execute_reply":"2024-02-08T10:36:53.661954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os, glob\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom pathlib import Path\n\nPATH_DATASET = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nPATH_PARQUETS = PATH_DATASET / \"parquet_files\"\nPARQUETS_TRAIN = PATH_PARQUETS / \"train\"\nPARQUETS_TEST = PATH_PARQUETS / \"test\"\npd.set_option('display.max_columns', None)\npd.set_option('display.max_rows', None)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-02-08T10:36:53.664799Z","iopub.execute_input":"2024-02-08T10:36:53.665642Z","iopub.status.idle":"2024-02-08T10:36:56.143305Z","shell.execute_reply.started":"2024-02-08T10:36:53.665600Z","shell.execute_reply":"2024-02-08T10:36:56.142455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Explore the training data\n\n**Borrowed from the competition describtion:**\n\nTable Description\nThis dataset contains a large number of tables as a result of utilizing diverse data sources and the varying levels of data aggregation used while preparing the dataset. Note: All files listed below are found in both .csv and .parquet formats.\n\n### Depth values\n\n- **depth=0** - These are static features directly tied to a specific case_id.\n- **depth=1** - Each case_id has an associated historical record, indexed by num_group1.\n- **depth=2** - Each case_id has an associated historical record, indexed by both num_group1 and num_group2.\n\nYou can read more about Credit bureau (CB) here https://en.wikipedia.org/wiki/Credit_bureau.","metadata":{}},{"cell_type":"code","source":"!cat /kaggle/input/home-credit-credit-risk-model-stability/feature_definitions.csv","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-02-08T10:36:56.144398Z","iopub.execute_input":"2024-02-08T10:36:56.145048Z","iopub.status.idle":"2024-02-08T10:36:57.207807Z","shell.execute_reply.started":"2024-02-08T10:36:56.145018Z","shell.execute_reply":"2024-02-08T10:36:57.206342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def _short_array(arr):\n    if len(arr) <= 5:\n        return repr(arr)\n    return f\"[{', '.join(map(str, arr[:2]))}, ..., {', '.join(map(str, arr[-2:]))}]\"\n\n# taking inpiration with column names from https://www.kaggle.com/code/greysky/home-credit-baseline\ndef convert_dtypes(df):\n    cols = []\n    for col, dt in dict(df.dtypes).items():\n        if col.startswith(\"for\"):\n            df[col] = df[col].fillna(0).astype(\"int16\")\n        elif \"num\" in col or \"cnt\" in col:\n            df[col] = df[col].fillna(0).astype(\"int32\")\n        elif col.startswith(\"pct\"):\n            df[col] = df[col].astype(\"float16\")\n        elif col[-1] in (\"A\", \"P\"):\n            df[col] = df[col].astype(\"float32\")\n        elif col[-1] in (\"D\", ):\n            df[col] = pd.to_datetime(df[col])\n        elif col[-1] in (\"M\", \"L\"):\n            if col[-1] == \"L\" and dt.name.startswith(\"int\"):\n                df[col] = df[col].astype(\"int32\")\n            elif col[-1] == \"L\" and dt.name.startswith(\"float\"):\n                df[col] = df[col].astype(\"float32\")\n            else:\n                uq = list(df[col].unique())\n                print(f'{col} -> #{len(uq)} -> {_short_array(uq)}')\n                df[col] = df[col].astype(\"category\")\n        if col[-1] in (\"A\", \"P\", \"M\", \"L\"):\n            cols.append(col)\n    return cols","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-02-08T10:36:57.210786Z","iopub.execute_input":"2024-02-08T10:36:57.211124Z","iopub.status.idle":"2024-02-08T10:36:57.222498Z","shell.execute_reply.started":"2024-02-08T10:36:57.211092Z","shell.execute_reply":"2024-02-08T10:36:57.221383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Base tables\n\nBase tables store the basic information about the observation and case_id. This is a unique identification of every observation and you need to use it to join the other tables to base tables.\n\ntrain_base.csv","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(PATH_DATASET / \"csv_files\" / \"train\" / \"train_base.csv\")\nprint(f\"size: {len(df_train)}\")\ndisplay(df_train.head())","metadata":{"execution":{"iopub.status.busy":"2024-02-08T10:36:57.223892Z","iopub.execute_input":"2024-02-08T10:36:57.224949Z","iopub.status.idle":"2024-02-08T10:36:58.418426Z","shell.execute_reply.started":"2024-02-08T10:36:57.224920Z","shell.execute_reply":"2024-02-08T10:36:58.417078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[\"date_decision\"] = pd.to_datetime(df_train[\"date_decision\"]).dt.date\n# delete redundat cols\ndel df_train[\"MONTH\"], df_train[\"WEEK_NUM\"]","metadata":{"execution":{"iopub.status.busy":"2024-02-08T10:36:58.419662Z","iopub.execute_input":"2024-02-08T10:36:58.419994Z","iopub.status.idle":"2024-02-08T10:36:59.020158Z","shell.execute_reply.started":"2024-02-08T10:36:58.419967Z","shell.execute_reply":"2024-02-08T10:36:59.019082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def merge_parquets(df, name, prefix=\"train\", folder=PATH_PARQUETS):\n    df_ = pd.concat(\n        [pd.read_parquet(p) for p in glob.glob(str(folder / prefix / f\"{prefix}_{name}*.parquet\"))],\n    )\n    if \"num_group1\" in df_.columns:\n        del df_[\"num_group1\"]\n    df_.drop_duplicates(inplace=True)\n    print(f\"{name} size: {len(df_)} with features: {len(df_.columns)}\")\n    display(df_.head())\n    if len(df_) > len(df_[\"case_id\"].unique()):\n        print(f\"`{name}` requires SPECIAL treatment as single case_id has nultiple entries\")\n        return df\n    cols = convert_dtypes(df_) + [\"case_id\"]\n    df = df.merge(df_[cols], how=\"left\", on=\"case_id\")\n    print(f\"fused size: {len(df)}\")\n    return df","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-02-08T10:36:59.021429Z","iopub.execute_input":"2024-02-08T10:36:59.021733Z","iopub.status.idle":"2024-02-08T10:36:59.029460Z","shell.execute_reply.started":"2024-02-08T10:36:59.021706Z","shell.execute_reply":"2024-02-08T10:36:59.028291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Depth 0\n\n### static_0 | Properties: depth=0, internal data source\n\n- train_static_0_0.csv\n- train_static_0_1.csv\n\n### static_cb_0 | Properties: depth=0, external data source\n\n- train_static_cb_0.csv","metadata":{}},{"cell_type":"code","source":"df_train = merge_parquets(df_train, \"static_0\")","metadata":{"execution":{"iopub.status.busy":"2024-02-08T10:36:59.030743Z","iopub.execute_input":"2024-02-08T10:36:59.031050Z","iopub.status.idle":"2024-02-08T10:37:30.674509Z","shell.execute_reply.started":"2024-02-08T10:36:59.031026Z","shell.execute_reply":"2024-02-08T10:37:30.673343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = merge_parquets(df_train, \"static_cb_0\")","metadata":{"execution":{"iopub.status.busy":"2024-02-08T10:37:30.675859Z","iopub.execute_input":"2024-02-08T10:37:30.676165Z","iopub.status.idle":"2024-02-08T10:37:37.536861Z","shell.execute_reply.started":"2024-02-08T10:37:30.676139Z","shell.execute_reply":"2024-02-08T10:37:37.535544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Depth 1\n\n### applprev_1 | Properties: depth=1, internal data source\n\n- train_applprev_1_0.csv\n- train_applprev_1_1.csv\n\n### other_1 | Properties: depth=1, internal data source\n\n- train_other_1.csv\n\n### tax_registry_a_1 | Properties: depth=1, external data source, Tax registry provider A\n\n- train_tax_registry_a_1.csv\n\n### tax_registry_b_1 | Properties: depth=1, external data source, Tax registry provider B\n\n- train_tax_registry_b_1.csv\n\n### tax_registry_c_1 | Properties: depth=1, external data source, Tax registry provider C\n\n- train_tax_registry_c_1.csv\n\n### credit_bureau_a_1 | Properties: depth=1, external data source, Credit bureau provider A\n\n- train_credit_bureau_a_1_0.csv\n- train_credit_bureau_a_1_1.csv\n- train_credit_bureau_a_1_2.csv\n- train_credit_bureau_a_1_3.csv\n\n### credit_bureau_b_1 | Properties: depth=1, external data source, Credit bureau provider B\n\n- train_credit_bureau_b_1.csv\n\n### deposit_1 | Properties: depth=1, internal data source\n\n- train_deposit_1.csv\n\n### person_1 | Properties: depth=1, internal data source\n\n- train_person_1.csv\n\n### debitcard_1 | Properties: depth=1, internal data source\n\n- train_debitcard_1.csv","metadata":{}},{"cell_type":"code","source":"# TODO","metadata":{"execution":{"iopub.status.busy":"2024-02-08T10:37:37.541380Z","iopub.execute_input":"2024-02-08T10:37:37.541701Z","iopub.status.idle":"2024-02-08T10:37:37.545863Z","shell.execute_reply.started":"2024-02-08T10:37:37.541672Z","shell.execute_reply":"2024-02-08T10:37:37.544830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Depth 2\n\n### applprev_2 | Properties: depth=2, internal data source\n\n- train_applprev_2.csv\n\n### person_2 | Properties: depth=2, internal data source\n\n- train_person_2.csv\n\n### redit_bureau_a_2 | Properties: depth=2, external data source, Credit bureau provider A\n\n- train_credit_bureau_a_2_0.csv\n- train_credit_bureau_a_2_1.csv\n- train_credit_bureau_a_2_2.csv\n- train_credit_bureau_a_2_3.csv\n- train_credit_bureau_a_2_4.csv\n- train_credit_bureau_a_2_5.csv\n- train_credit_bureau_a_2_6.csv\n- train_credit_bureau_a_2_7.csv\n- train_credit_bureau_a_2_8.csv\n- train_credit_bureau_a_2_9.csv\n- train_credit_bureau_a_2_10.csv\n\n### credit_bureau_b_2 | Properties: depth=2, external data source, Credit bureau provider B\n\n- train_credit_bureau_b_2.csv","metadata":{}},{"cell_type":"code","source":"# TODO","metadata":{"execution":{"iopub.status.busy":"2024-02-08T10:37:37.547145Z","iopub.execute_input":"2024-02-08T10:37:37.547801Z","iopub.status.idle":"2024-02-08T10:37:37.556571Z","shell.execute_reply.started":"2024-02-08T10:37:37.547736Z","shell.execute_reply":"2024-02-08T10:37:37.555057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Brows the data","metadata":{}},{"cell_type":"code","source":"print(f\"data size: {len(df_train)}\")\nprint(f\"unique: {len(df_train['case_id'].unique())}\")","metadata":{"execution":{"iopub.status.busy":"2024-02-08T10:37:37.558000Z","iopub.execute_input":"2024-02-08T10:37:37.558670Z","iopub.status.idle":"2024-02-08T10:37:37.605013Z","shell.execute_reply.started":"2024-02-08T10:37:37.558631Z","shell.execute_reply":"2024-02-08T10:37:37.603935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(df_train.head().T)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-02-08T10:37:37.606044Z","iopub.execute_input":"2024-02-08T10:37:37.606350Z","iopub.status.idle":"2024-02-08T10:37:37.650839Z","shell.execute_reply.started":"2024-02-08T10:37:37.606324Z","shell.execute_reply":"2024-02-08T10:37:37.649826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.groupby('date_decision')['target'].mean().plot(\n    figsize=(14, 2), grid=True,\n    xlabel=\"date of decision\",\n    ylabel=\"day mean / proxi ratio\",\n)","metadata":{"execution":{"iopub.status.busy":"2024-02-08T10:37:37.652135Z","iopub.execute_input":"2024-02-08T10:37:37.652703Z","iopub.status.idle":"2024-02-08T10:37:38.170913Z","shell.execute_reply.started":"2024-02-08T10:37:37.652673Z","shell.execute_reply":"2024-02-08T10:37:38.169802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(df_train.dtypes)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-02-08T10:37:38.172334Z","iopub.execute_input":"2024-02-08T10:37:38.172650Z","iopub.status.idle":"2024-02-08T10:37:38.184744Z","shell.execute_reply.started":"2024-02-08T10:37:38.172622Z","shell.execute_reply":"2024-02-08T10:37:38.183769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# convert_dtypes(df_train)\n# display(df_train.head().T)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-02-08T10:37:38.186048Z","iopub.execute_input":"2024-02-08T10:37:38.186348Z","iopub.status.idle":"2024-02-08T10:37:38.192073Z","shell.execute_reply.started":"2024-02-08T10:37:38.186321Z","shell.execute_reply":"2024-02-08T10:37:38.191118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train LBGM model","metadata":{}},{"cell_type":"code","source":"CATEGORY_COLUMNS = {}\nfor col, dt in dict(df_train.dtypes).items():\n    if str(dt) != \"category\":\n        continue\n    uq = list(df_train[col].unique())\n    CATEGORY_COLUMNS[col] = uq\n    print(f\"{col} -> {uq}\")","metadata":{"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col, vals in CATEGORY_COLUMNS.items():\n    lut = dict(zip(vals, range(len(vals))))\n    df_train[col] = df_train[col].map(lut)\ndisplay(df_train.head().T)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-02-08T10:37:38.462401Z","iopub.execute_input":"2024-02-08T10:37:38.462678Z","iopub.status.idle":"2024-02-08T10:37:38.824594Z","shell.execute_reply.started":"2024-02-08T10:37:38.462653Z","shell.execute_reply":"2024-02-08T10:37:38.823564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.replace([np.inf, -np.inf], np.nan, inplace=True)\nTRAIN_COLUMNS = [c for c in df_train.columns if c not in (\"case_id\", \"date_decision\", \"target\")]\nNUM_ITERATIONS = 350","metadata":{"execution":{"iopub.status.busy":"2024-02-08T10:37:38.825600Z","iopub.execute_input":"2024-02-08T10:37:38.825917Z","iopub.status.idle":"2024-02-08T10:37:39.759101Z","shell.execute_reply.started":"2024-02-08T10:37:38.825890Z","shell.execute_reply":"2024-02-08T10:37:39.758180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\nimport optuna\nfrom sklearn.model_selection import cross_val_score\n\ndef lgb_objective(trial):\n    params = {\n        # \"device\": \"gpu\",\n        \"n_jobs\": -1,\n        'n_iter': int(NUM_ITERATIONS / 5),\n        'verbosity': -1,\n        #\"boosting_type\": \"gbdt\",\n        #\"feature_name\": TRAIN_COLUMNS,\n        #\"categorical_feature\": list(CATEGORY_COLUMNS.keys()),\n        #\"max_bin\": 255,\n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.6, 1.0),\n        'colsample_bynode': trial.suggest_float('colsample_bynode', 0.6, 1.0),\n        'n_estimators': trial.suggest_int('n_estimators', 50, 800),\n        'max_depth': trial.suggest_int('max_depth', 3, 20),\n        'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.1, log=True),\n        'lambda_l1': trial.suggest_float('lambda_l1', 1e-2, 10.0),\n        'lambda_l2': trial.suggest_float('lambda_l2', 1e-2, 10.0),\n        'num_leaves': trial.suggest_int('num_leaves', 16, 512),\n        'min_data_in_leaf': trial.suggest_int('min_data_in_leaf', 4, 512),\n    }\n    \n    model  = lgb.LGBMClassifier(**params)\n    scores = cross_val_score(\n        model, df_train[TRAIN_COLUMNS].values, df_train['target'].values, cv=3, scoring='roc_auc'\n    )\n        \n    return np.mean(scores)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-02-08T10:37:39.760622Z","iopub.execute_input":"2024-02-08T10:37:39.760967Z","iopub.status.idle":"2024-02-08T10:37:42.451468Z","shell.execute_reply.started":"2024-02-08T10:37:39.760940Z","shell.execute_reply":"2024-02-08T10:37:42.450223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study = optuna.create_study(direction='maximize', study_name='Classifier')\nstudy.optimize(lgb_objective, n_trials=100, show_progress_bar=True)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-02-08T10:37:42.452742Z","iopub.execute_input":"2024-02-08T10:37:42.453106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Show trials","metadata":{}},{"cell_type":"code","source":"from optuna.visualization import plot_optimization_history\n\nplot_optimization_history(study).show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from optuna.visualization import plot_param_importances\n\nplot_param_importances(study).show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from optuna.visualization import plot_parallel_coordinate\n\nplot_parallel_coordinate(study).show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### REfit on whole dataset","metadata":{}},{"cell_type":"code","source":"from pprint import pprint\n\n# pprint(study.trials)\nbest_params = study.best_params\nbest_params.update({\n#     \"device\": \"gpu\",\n    \"n_jobs\": -1,\n    \"n_iter\": NUM_ITERATIONS,\n    #\"categorical_feature\": list(CATEGORY_COLUMNS.keys()),\n})\npprint(best_params)","metadata":{"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = lgb.LGBMClassifier(**best_params)\nmodel.fit(df_train[TRAIN_COLUMNS].values, df_train['target'].values)\ndel df_train","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading test data","metadata":{}},{"cell_type":"markdown","source":"## base data","metadata":{"execution":{"iopub.status.busy":"2024-02-07T11:36:19.157007Z","iopub.execute_input":"2024-02-07T11:36:19.157411Z","iopub.status.idle":"2024-02-07T11:36:19.166433Z","shell.execute_reply.started":"2024-02-07T11:36:19.157376Z","shell.execute_reply":"2024-02-07T11:36:19.165721Z"}}},{"cell_type":"code","source":"df_test = pd.read_csv(PATH_DATASET / \"csv_files\" / \"test\" / \"test_base.csv\")\nprint(f\"size: {len(df_test)}\")\ndisplay(df_test.head())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test[\"date_decision\"] = pd.to_datetime(df_test[\"date_decision\"]).dt.date\n# delete redundat cols\ndel df_test[\"MONTH\"], df_test[\"WEEK_NUM\"]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Depth 0","metadata":{}},{"cell_type":"code","source":"for name in [\"static_0\", \"static_cb_0\"]:\n    df_test = merge_parquets(df_test, name, \"test\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Overview","metadata":{}},{"cell_type":"code","source":"display(df_test.head().T)","metadata":{"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(df_test.dtypes)","metadata":{"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# convert_dtypes(df_test)\n# display(df_test.head().T)","metadata":{"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict and submit","metadata":{}},{"cell_type":"code","source":"!head /kaggle/input/home-credit-credit-risk-model-stability/sample_submission.csv","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col, vals in CATEGORY_COLUMNS.items():\n    lut = dict(zip(vals, range(len(vals))))\n    df_test[col] = df_test[col].map(lut)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.replace([np.inf, -np.inf], np.nan, inplace=True)\npreds_proba = model.predict_proba(df_test[TRAIN_COLUMNS].values)\n\ndf_test[\"score\"] = np.clip(preds_proba[:, 1], 0, 1)\ndisplay(df_test[[\"case_id\", \"date_decision\", \"score\"]].head(10).T)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_test[\"case_id\"] = df_test[\"case_id\"].astype(\"int32\")\n# df_test[\"score\"] = df_test[\"score\"].astype(\"float32\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test[[\"case_id\", \"score\"]].to_csv(\"submission.csv\", float_format='%.3f', index=False)\n\n!head submission.csv","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}