{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os,glob\nimport pandas as pd\nimport numpy as np\nfrom pathlib import Path\n\nPATH_DATASET = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nPATH_PARQUETS = PATH_DATASET / \"parquet_files\"\nPATH_TRAIN = PATH_PARQUETS / \"train\"\nPATH_TEST = PATH_PARQUETS / \"test\"\npd.set_option('display.max_columns',None)\npd.set_option('display.max_rows',None)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-06T15:17:04.662692Z","iopub.execute_input":"2024-02-06T15:17:04.663065Z","iopub.status.idle":"2024-02-06T15:17:05.763528Z","shell.execute_reply.started":"2024-02-06T15:17:04.663036Z","shell.execute_reply":"2024-02-06T15:17:05.762593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_parquet(PATH_TRAIN / \"train_base.parquet\")\nprint(f\"size: {len(df_train)}\")\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-06T15:17:05.766015Z","iopub.execute_input":"2024-02-06T15:17:05.766947Z","iopub.status.idle":"2024-02-06T15:17:06.208886Z","shell.execute_reply.started":"2024-02-06T15:17:05.766907Z","shell.execute_reply":"2024-02-06T15:17:06.207707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['date_decision'] = pd.to_datetime(df_train['date_decision']).dt.to_pydatetime\ndel df_train['MONTH'],df_train['WEEK_NUM']","metadata":{"execution":{"iopub.status.busy":"2024-02-06T15:17:06.210994Z","iopub.execute_input":"2024-02-06T15:17:06.211387Z","iopub.status.idle":"2024-02-06T15:17:06.423524Z","shell.execute_reply.started":"2024-02-06T15:17:06.211349Z","shell.execute_reply":"2024-02-06T15:17:06.422378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dft_static = pd.concat([pd.read_parquet(p) for p in glob.glob(str(PATH_TRAIN / \"train_static_0_*\"))],)\ndf_train = df_train.merge(dft_static,how = \"left\",on = \"case_id\")","metadata":{"execution":{"iopub.status.busy":"2024-02-06T15:17:06.427601Z","iopub.execute_input":"2024-02-06T15:17:06.428346Z","iopub.status.idle":"2024-02-06T15:17:27.967231Z","shell.execute_reply.started":"2024-02-06T15:17:06.428303Z","shell.execute_reply":"2024-02-06T15:17:27.966158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dft_static = pd.read_parquet(PATH_TRAIN / \"train_static_cb_0.parquet\")\ndf_train = df_train.merge(dft_static, how=\"left\", on=\"case_id\")","metadata":{"execution":{"iopub.status.busy":"2024-02-06T15:17:27.971288Z","iopub.execute_input":"2024-02-06T15:17:27.971608Z","iopub.status.idle":"2024-02-06T15:17:35.818455Z","shell.execute_reply.started":"2024-02-06T15:17:27.971582Z","shell.execute_reply":"2024-02-06T15:17:35.817568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(df_train))","metadata":{"execution":{"iopub.status.busy":"2024-02-06T15:17:35.819793Z","iopub.execute_input":"2024-02-06T15:17:35.820115Z","iopub.status.idle":"2024-02-06T15:17:35.826397Z","shell.execute_reply.started":"2024-02-06T15:17:35.820087Z","shell.execute_reply":"2024-02-06T15:17:35.825383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col,dtype in dict(df_train.dtypes).items():\n    if str(dtype).startswith(\"int\"):\n        df_train[col] = df_train[col].astype(\"int32\")\n    elif str(dtype).startswith(\"float\"):\n        df_train[col] = df_train[col].astype(\"float16\")\n    else:\n        print(f'{col}->{df_train[col].nunique()}')\n        df_train[col] = df_train[col].astype(\"category\")\n","metadata":{"execution":{"iopub.status.busy":"2024-02-06T15:17:35.827399Z","iopub.execute_input":"2024-02-06T15:17:35.827714Z","iopub.status.idle":"2024-02-06T15:17:45.254094Z","shell.execute_reply.started":"2024-02-06T15:17:35.827684Z","shell.execute_reply":"2024-02-06T15:17:45.253078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ndf_train.replace([np.inf,-np.inf],np.nan,inplace =True)\ntrain_cols = [c for c in df_train.columns if c not in (\"case_id\",'date_decision','target')]\nX_train,X_valid,y_train,y_valid = train_test_split(df_train[train_cols],df_train['target'],test_size = 0.2)","metadata":{"execution":{"iopub.status.busy":"2024-02-06T15:17:45.255350Z","iopub.execute_input":"2024-02-06T15:17:45.255702Z","iopub.status.idle":"2024-02-06T15:17:55.531712Z","shell.execute_reply.started":"2024-02-06T15:17:45.255672Z","shell.execute_reply":"2024-02-06T15:17:55.530594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import xgboost as xgb","metadata":{"execution":{"iopub.status.busy":"2024-02-06T15:17:55.533183Z","iopub.execute_input":"2024-02-06T15:17:55.533696Z","iopub.status.idle":"2024-02-06T15:17:55.790121Z","shell.execute_reply.started":"2024-02-06T15:17:55.533660Z","shell.execute_reply":"2024-02-06T15:17:55.788984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = xgb.XGBClassifier(\n    device=\"cuda\",\n    objective='binary:logistic',\n    tree_method=\"hist\",\n    enable_categorical=True,\n    eval_metric='auc',\n    subsample=1,\n    colsample_bytree=1,\n    min_child_weight=1,\n    max_depth=20,\n    n_estimators=800,\n    random_state=42,\n)","metadata":{"execution":{"iopub.status.busy":"2024-02-06T15:17:55.791577Z","iopub.execute_input":"2024-02-06T15:17:55.791948Z","iopub.status.idle":"2024-02-06T15:17:55.797164Z","shell.execute_reply.started":"2024-02-06T15:17:55.791916Z","shell.execute_reply":"2024-02-06T15:17:55.796076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(\n    X_train, y_train,\n    eval_set=[(X_valid, y_valid)],\n    early_stopping_rounds=100,\n    verbose=True,\n)","metadata":{"execution":{"iopub.status.busy":"2024-02-06T15:17:55.798716Z","iopub.execute_input":"2024-02-06T15:17:55.799080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_parquet(PATH_TEST / \"test_base.parquet\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test[\"date_decision\"] = pd.to_datetime(df_test[\"date_decision\"]).dt.date\ndel df_test[\"MONTH\"], df_test[\"WEEK_NUM\"]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dft_static = pd.concat(\n    [pd.read_parquet(p) for p in glob.glob(str(PATH_TEST / \"test_static_0_*\"))],\n)\ndf_test = df_test.merge(dft_static, how=\"left\", on=\"case_id\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dft_static = pd.read_parquet(PATH_TEST / \"test_static_cb_0.parquet\")\ndf_test = df_test.merge(dft_static, how=\"left\", on=\"case_id\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col, dtype in dict(df_test.dtypes).items():\n    if str(dtype).startswith(\"int\"):\n        df_test[col] = df_test[col].astype(\"int32\")\n    elif str(dtype).startswith(\"float\"):\n        df_test[col] = df_test[col].astype(\"float16\")\n    else:\n        print(f'{col} -> {df_test[col].nunique()}')\n        df_test[col] = df_test[col].astype(\"category\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.replace([np.inf, -np.inf], np.nan, inplace=True)\npreds_proba = model.predict_proba(df_test[train_cols])\ndf_test[\"score\"] = list(round(pred[1], 3) for pred in preds_proba)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test[[\"case_id\", \"score\"]].to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}