{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":177043296,"sourceType":"kernelVersion"}],"dockerImageVersionId":30699,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Dependencies","metadata":{"_uuid":"68270060-52db-44d3-82b1-ff34f13db263","_cell_guid":"87ba732c-852a-4a26-a87a-b435bec6bb4e","trusted":true}},{"cell_type":"code","source":"import os\nimport gc\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport joblib\nimport lightgbm as lgb\n\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.ensemble import VotingClassifier\nfrom sklearn.preprocessing import LabelEncoder\n\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)","metadata":{"_uuid":"6b236753-e70c-475d-b368-b52a90dfa2df","_cell_guid":"c5edd542-72a4-4d44-9635-1546291f7544","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-05-11T17:50:31.356009Z","iopub.execute_input":"2024-05-11T17:50:31.356778Z","iopub.status.idle":"2024-05-11T17:50:37.143103Z","shell.execute_reply.started":"2024-05-11T17:50:31.356745Z","shell.execute_reply":"2024-05-11T17:50:37.142163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import home_credit_baseline_0 as data_nb","metadata":{"_uuid":"2f1e4c29-6a6a-40fc-b8e4-859fbb496e57","_cell_guid":"bac32ea7-9d33-40a6-91ce-ad2148d1652a","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-05-11T17:50:44.864916Z","iopub.execute_input":"2024-05-11T17:50:44.865490Z","iopub.status.idle":"2024-05-11T17:50:44.888927Z","shell.execute_reply.started":"2024-05-11T17:50:44.865448Z","shell.execute_reply":"2024-05-11T17:50:44.887952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Input","metadata":{"_uuid":"9a2320f2-0f93-4e65-be61-6f74f45b1376","_cell_guid":"1732e135-b5ce-4df8-b7fe-819c40f29db7","trusted":true}},{"cell_type":"code","source":"f_list = [\n    # depth = 0\n    \"_static_cb_0.parquet\",\n    \"_static_0_*.parquet\",\n    # depth = 1\n    \"_applprev_1_*.parquet\",\n    \"_other_1.parquet\",\n    \"_person_1.parquet\",\n    \"_deposit_1.parquet\",\n    \"_debitcard_1.parquet\",\n    \"_tax_registry_a_1.parquet\",\n    \"_tax_registry_b_1.parquet\",\n    \"_tax_registry_c_1.parquet\",\n    \"_credit_bureau_a_1_*.parquet\",\n    \"_credit_bureau_b_1.parquet\",\n    # depth = 2\n    \"_credit_bureau_a_2_*.parquet\",\n    \"_credit_bureau_b_2.parquet\",\n]","metadata":{"_uuid":"9d1cbee9-1c1f-4017-a48a-f245d584c47e","_cell_guid":"782d48b7-63f6-490b-a301-9216124f62f4","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-05-11T17:50:50.313032Z","iopub.execute_input":"2024-05-11T17:50:50.313804Z","iopub.status.idle":"2024-05-11T17:50:50.318584Z","shell.execute_reply.started":"2024-05-11T17:50:50.313771Z","shell.execute_reply":"2024-05-11T17:50:50.317650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_df = data_nb.prepare_df(f_list, data_nb.CFG.train_dir, data_nb.Aggregator())\ngc.collect()\ncat_cols = list(train_df.select_dtypes(\"category\").columns)\ndisplay(train_df)","metadata":{"_uuid":"8561bd5e-c627-4e37-a0b2-336b9ec380e0","_cell_guid":"b34fc3a1-d2aa-463d-b85f-94287dddd730","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-05-11T17:50:52.406352Z","iopub.execute_input":"2024-05-11T17:50:52.407238Z","iopub.status.idle":"2024-05-11T17:55:50.235440Z","shell.execute_reply.started":"2024-05-11T17:50:52.407201Z","shell.execute_reply":"2024-05-11T17:55:50.234535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = data_nb.prepare_df(f_list, data_nb.CFG.test_dir, data_nb.Aggregator(), mode = \"test\", \n                     cat_cols = cat_cols, train_cols = train_df.columns)\ndisplay(test_df)","metadata":{"_uuid":"137af0c9-c227-4ee0-bf25-d0e69814e7c4","_cell_guid":"49e4164d-843e-43e7-96be-1ccf6548dba3","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-05-11T17:56:01.587607Z","iopub.execute_input":"2024-05-11T17:56:01.587948Z","iopub.status.idle":"2024-05-11T17:56:03.370507Z","shell.execute_reply.started":"2024-05-11T17:56:01.587923Z","shell.execute_reply":"2024-05-11T17:56:03.369590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = data_nb.reduce_mem_usage(train_df)\ntest_df = data_nb.reduce_mem_usage(test_df)\n\nprint(\"train data shape:\\t\", train_df.shape)\nprint(\"test data shape:\\t\", test_df.shape)","metadata":{"_uuid":"187a8f41-33f0-49e4-bdab-55c22bea2c7f","_cell_guid":"aa6b4625-94db-46ec-9b73-27a3948eadba","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-05-11T17:56:05.778030Z","iopub.execute_input":"2024-05-11T17:56:05.778842Z","iopub.status.idle":"2024-05-11T17:56:11.245999Z","shell.execute_reply.started":"2024-05-11T17:56:05.778804Z","shell.execute_reply":"2024-05-11T17:56:11.244982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Drop columns","metadata":{"_uuid":"6b8d3d68-dfb8-4cc1-a72a-09abd1e33826","_cell_guid":"c11910fe-e5fe-4144-87a0-9f6028c37ef0","trusted":true}},{"cell_type":"code","source":"drop_cols = []","metadata":{"_uuid":"228fd6c4-f006-4b2f-a8ff-1efbf66c0126","_cell_guid":"bc6e2e30-1faa-42a2-a7da-160340b5ec29","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-05-11T17:56:16.768783Z","iopub.execute_input":"2024-05-11T17:56:16.769581Z","iopub.status.idle":"2024-05-11T17:56:16.773389Z","shell.execute_reply.started":"2024-05-11T17:56:16.769547Z","shell.execute_reply":"2024-05-11T17:56:16.772388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{"_uuid":"b9f3281d-0f62-4476-a501-aa05f45eee4b","_cell_guid":"3bc234a8-e52b-4ba3-a913-31e047efb420","trusted":true}},{"cell_type":"code","source":"def gini_stability(base, score_col=\"score\", w_fallingrate=88.0, w_resstd=-0.5):\n    gini_in_time = base.loc[:, [\"WEEK_NUM\", \"target\", score_col]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", score_col]]\\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[score_col])-1).tolist()\n    \n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std","metadata":{"_uuid":"c9f9a204-f52b-45fe-999f-277d2738019e","_cell_guid":"8da0d0e5-bc82-46df-a945-e29220d77b54","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-05-11T17:56:18.997520Z","iopub.execute_input":"2024-05-11T17:56:18.998250Z","iopub.status.idle":"2024-05-11T17:56:19.005364Z","shell.execute_reply.started":"2024-05-11T17:56:18.998219Z","shell.execute_reply":"2024-05-11T17:56:19.004450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nX = train_df.drop(columns = [\"target\", \"case_id\", \"WEEK_NUM\"])\ny = train_df[\"target\"]\nweeks = train_df[\"WEEK_NUM\"]\n\nsgkf = StratifiedGroupKFold(n_splits=5, shuffle=False)\n\n# params = {\n#     \"boosting_type\": \"gbdt\",\n#     \"objective\": \"binary\",\n#     \"metric\": \"auc\",\n#     \"max_depth\": 10,  \n#     \"max_bin\"\n#     \"learning_rate\": 0.05,\n#     \"n_estimators\": 2000,  \n#     \"colsample_bytree\": 0.8,\n#     \"colsample_bynode\": 0.8,\n#     \"verbose\": -1,\n#     \"random_state\": 42,\n#     \"reg_alpha\": 0.1,\n#     \"reg_lambda\": 10,\n#     \"extra_trees\":True,\n#     'num_leaves': 64,\n#     \"device\": \"gpu\",  # Change device to CPU\n#     \"verbose\": -1,\n# }\nparams = {\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"binary\",\n    \"metric\": \"auc\",\n    \"max_depth\": 8,\n    \"min_data_in_bin\": 256, \n    \"learning_rate\": 0.05,\n    \"n_estimators\": 1000,\n    \"colsample_bytree\": 0.8, \n    \"colsample_bynode\": 0.8,\n    \"verbose\": -1,\n    \"random_state\": 42,\n    \"device\": \"gpu\",\n}\n\n\nfitted_models = []\noof_pred = np.zeros(X.shape[0])\n\nfor train_idx, valid_idx in sgkf.split(X, y, groups = weeks):\n    X_train, y_train = X.iloc[train_idx], y.iloc[train_idx]\n    X_valid, y_valid = X.iloc[valid_idx], y.iloc[valid_idx]\n    \n    model = lgb.LGBMClassifier(**params)\n    model.fit(X_train, y_train,\n              eval_set = [(X_valid, y_valid)],\n              callbacks = [lgb.log_evaluation(100), lgb.early_stopping(100)])\n    fitted_models.append(model)\n    \n    y_pred_valid = model.predict_proba(X_valid)[:, 1]\n    oof_pred[valid_idx] = y_pred_valid\n    gc.collect()","metadata":{"_uuid":"ca8014ca-d33e-45dd-b5b0-24645546c26d","_cell_guid":"5200da93-0ee8-41b7-8a66-7c664ba7a43c","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-05-11T17:56:21.533081Z","iopub.execute_input":"2024-05-11T17:56:21.533699Z","iopub.status.idle":"2024-05-11T18:16:21.925491Z","shell.execute_reply.started":"2024-05-11T17:56:21.533669Z","shell.execute_reply":"2024-05-11T18:16:21.924556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"roc_auc_oof = roc_auc_score(y, oof_pred)\nprint(\"CV roc_auc_oof: \", roc_auc_oof)","metadata":{"_uuid":"1e557025-5292-4cd4-9fe4-6771f7154c13","_cell_guid":"28bc2a65-b839-44ca-9de0-06e0fb30d058","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-05-11T18:19:36.073920Z","iopub.execute_input":"2024-05-11T18:19:36.074318Z","iopub.status.idle":"2024-05-11T18:19:36.692813Z","shell.execute_reply.started":"2024-05-11T18:19:36.074287Z","shell.execute_reply":"2024-05-11T18:19:36.691861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_df = train_df[[\"WEEK_NUM\", \"target\"]].copy()\noof_df[\"pred_oof\"] = oof_pred\ngini_score = gini_stability(oof_df, score_col=\"pred_oof\")\nprint(\"gini_score:\\t\", gini_score)","metadata":{"_uuid":"03e62d7d-43cb-46f7-a537-96a37c667e6a","_cell_guid":"e42f9615-8466-4d20-a6f5-22b1f9e0055f","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-05-11T18:19:38.354076Z","iopub.execute_input":"2024-05-11T18:19:38.354453Z","iopub.status.idle":"2024-05-11T18:19:39.071442Z","shell.execute_reply.started":"2024-05-11T18:19:38.354422Z","shell.execute_reply":"2024-05-11T18:19:39.070447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_models_dict = [(str(i), model) for i, model in enumerate(fitted_models)]\nmodel = VotingClassifier(\n    estimators=oof_models_dict,\n    voting='soft',\n)\n\nmodel.estimators_ = fitted_models\nmodel.le_ = LabelEncoder().fit(y)\nmodel.classes_ = model.le_.classes_","metadata":{"_uuid":"fc50af12-545b-465e-a15f-54e2266b8aa7","_cell_guid":"8c5d816f-cd4a-438c-8292-59aeeb80e449","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-05-11T18:19:41.123157Z","iopub.execute_input":"2024-05-11T18:19:41.123545Z","iopub.status.idle":"2024-05-11T18:19:41.149827Z","shell.execute_reply.started":"2024-05-11T18:19:41.123513Z","shell.execute_reply":"2024-05-11T18:19:41.148926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"joblib.dump(model, \"oof_model_1.pkl\")","metadata":{"_uuid":"4a1d4721-aa2a-4f60-b4f4-10f538bf3c75","_cell_guid":"d30b67d5-2571-4f1b-abe4-587d9434c97f","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-05-11T18:19:44.095475Z","iopub.execute_input":"2024-05-11T18:19:44.095828Z","iopub.status.idle":"2024-05-11T18:19:44.736458Z","shell.execute_reply.started":"2024-05-11T18:19:44.095800Z","shell.execute_reply":"2024-05-11T18:19:44.735525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"joblib.dump((train_df.columns, cat_cols, drop_cols), \"train_cat_columns.pkl\")","metadata":{"_uuid":"1949f62d-2255-4c72-82d7-abc84fc28e0b","_cell_guid":"7b0f88b0-aa45-4132-b3ef-1d923abef17f","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-05-11T18:19:46.177467Z","iopub.execute_input":"2024-05-11T18:19:46.178145Z","iopub.status.idle":"2024-05-11T18:19:46.187979Z","shell.execute_reply.started":"2024-05-11T18:19:46.178109Z","shell.execute_reply":"2024-05-11T18:19:46.187216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"joblib.dump(oof_pred, \"oof_pred.pkl\")","metadata":{"_uuid":"89dceb27-e99a-4fcf-889c-c94fedc5a166","_cell_guid":"5c765727-f2f8-49bc-a003-a95674c9eba2","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-05-11T18:19:48.447606Z","iopub.execute_input":"2024-05-11T18:19:48.448471Z","iopub.status.idle":"2024-05-11T18:19:48.467348Z","shell.execute_reply.started":"2024-05-11T18:19:48.448434Z","shell.execute_reply":"2024-05-11T18:19:48.466535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# del X, train_df, oof_pred, oof_df\n# gc.collect()","metadata":{"_uuid":"ec73bb63-4ae3-433c-bb9d-da4ef7274dba","_cell_guid":"f25f0096-1b00-4317-a8fd-048c386903f6","collapsed":false,"execution":{"iopub.status.busy":"2024-05-11T15:31:05.196004Z","iopub.execute_input":"2024-05-11T15:31:05.196342Z","iopub.status.idle":"2024-05-11T15:31:05.202245Z","shell.execute_reply.started":"2024-05-11T15:31:05.196317Z","shell.execute_reply":"2024-05-11T15:31:05.201315Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prediction","metadata":{"_uuid":"80565065-68cc-4950-a0d7-6b32ba5d3d24","_cell_guid":"6a2f8aeb-b77b-4beb-b5e5-f0a07fa11690","trusted":true}},{"cell_type":"code","source":"def predict_proba_in_batches(model, data, batch_size=100000):\n    num_samples = len(data)\n    num_batches = int(np.ceil(num_samples / batch_size))\n    probabilities = np.zeros((num_samples,))\n\n    for batch_idx in range(num_batches):\n        print(f\"Processing batch: {batch_idx+1}/{num_batches}\")\n        start_idx = batch_idx * batch_size\n        end_idx = min((batch_idx + 1) * batch_size, num_samples)\n        X_batch = data.iloc[start_idx:end_idx]\n        batch_probs = model.predict_proba(X_batch)[:, 1]\n        probabilities[start_idx:end_idx] = batch_probs\n        gc.collect()\n\n    return probabilities","metadata":{"_uuid":"40a345c3-b303-4a17-9635-2abe4ebe777c","_cell_guid":"dc881a55-3752-46d6-9ec9-f8bf271e560d","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-05-11T18:19:54.378486Z","iopub.execute_input":"2024-05-11T18:19:54.379384Z","iopub.status.idle":"2024-05-11T18:19:54.386601Z","shell.execute_reply.started":"2024-05-11T18:19:54.379346Z","shell.execute_reply":"2024-05-11T18:19:54.385554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = test_df.drop(columns=[\"WEEK_NUM\"] + drop_cols)\nX_test = X_test.set_index(\"case_id\")\nprint(\"X_test shape: \", X_test.shape)\ny_pred = pd.Series(predict_proba_in_batches(model, X_test), index = X_test.index)\ny_pred[:10]","metadata":{"execution":{"iopub.status.busy":"2024-05-11T18:20:34.803715Z","iopub.execute_input":"2024-05-11T18:20:34.804381Z","iopub.status.idle":"2024-05-11T18:20:35.442409Z","shell.execute_reply.started":"2024-05-11T18:20:34.804348Z","shell.execute_reply":"2024-05-11T18:20:35.441429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subm_df = pd.read_csv(data_nb.CFG.root_dir / \"sample_submission.csv\")\nsubm_df = subm_df.set_index(\"case_id\")\nsubm_df[\"score\"] = y_pred\n\nprint(\"Check null: \", subm_df[\"score\"].isnull().any())\nsubm_df","metadata":{"_uuid":"c2784c82-df8d-46b6-8845-bbf47de322e1","_cell_guid":"39de5733-0c76-436e-88cc-fc0ce2996d29","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-05-11T18:20:37.830440Z","iopub.execute_input":"2024-05-11T18:20:37.831217Z","iopub.status.idle":"2024-05-11T18:20:37.847109Z","shell.execute_reply.started":"2024-05-11T18:20:37.831171Z","shell.execute_reply":"2024-05-11T18:20:37.846075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{"_uuid":"79a5de7b-c793-483b-bbd1-8240e3c3fac9","_cell_guid":"b5aebf0d-3a14-4983-8027-adba8a340665","trusted":true}},{"cell_type":"code","source":"subm_df.to_csv(\"submission.csv\")","metadata":{"_uuid":"5be280d0-4ea4-4a8e-9cc2-1e344ffb2826","_cell_guid":"fef16f36-a2ab-4164-ae64-2e19e26c47a6","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-05-11T18:20:43.495080Z","iopub.execute_input":"2024-05-11T18:20:43.495939Z","iopub.status.idle":"2024-05-11T18:20:43.503567Z","shell.execute_reply.started":"2024-05-11T18:20:43.495900Z","shell.execute_reply":"2024-05-11T18:20:43.502656Z"},"trusted":true},"execution_count":null,"outputs":[]}]}