{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"},{"sourceId":161936385,"sourceType":"kernelVersion"}],"dockerImageVersionId":30646,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install --force-reinstall scikit-learn --no-index --find-links=file:///kaggle/input/scikit-learn-1-4-0/ ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-07T13:57:25.919700Z","iopub.execute_input":"2024-02-07T13:57:25.920396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os, glob\nimport gc\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom pathlib import Path\nfrom typing import Literal\nimport polars as pl\nimport polars.selectors as cs\nfrom sklearn.model_selection import train_test_split, cross_validate, StratifiedGroupKFold\nfrom sklearn.metrics import roc_auc_score\n\nimport lightgbm as lgb","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if os.path.exists('/kaggle'):\n    PATH_DATASET = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nelse:\n    PATH_DATASET = Path(\"home-credit-credit-risk-model-stability\")\nPATH_PARQUETS = PATH_DATASET / \"parquet_files\"\nPATH_TRAIN = PATH_PARQUETS / \"train\"\nPATH_TEST = PATH_PARQUETS / \"test\"\n\npd.set_option('display.max_columns', 1000)\npd.set_option('display.max_rows', 1000)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Read and merge data\n","metadata":{}},{"cell_type":"markdown","source":"# Read and merge data\n\n1) Cast datetime features\n\n2) Cast categorical features\n\n3) Drop datetime features and features with too many categories\n\n4) Merge by case_id with base dataframe","metadata":{}},{"cell_type":"code","source":"class DatasetConstructor:\n    def __init__(self, mode: Literal['train', 'test']):\n        self.mode = mode\n        self.path = PATH_PARQUETS / mode\n\n    @staticmethod\n    def reduce_memory_usage_pl(df):\n        \"\"\" Reduce memory usage by polars dataframe {df} with name {name} by changing its data types.\n            Original pandas version of this function: https://www.kaggle.com/code/arjanso/reducing-dataframe-memory-size-by-65 \"\"\"\n        print(f\"Memory usage of dataframe is {round(df.estimated_size('mb'), 2)} MB\")\n        Numeric_Int_types = [pl.Int8,pl.Int16,pl.Int32,pl.Int64]\n        Numeric_Float_types = [pl.Float32,pl.Float64]    \n        for col in df.columns:\n            try:\n                col_type = df[col].dtype\n                if col_type == pl.Categorical:\n                    continue\n                c_min = df[col].min()\n                c_max = df[col].max()\n                if col_type in Numeric_Int_types:\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df = df.with_columns(df[col].cast(pl.Int8))\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df = df.with_columns(df[col].cast(pl.Int16))\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df = df.with_columns(df[col].cast(pl.Int32))\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df = df.with_columns(df[col].cast(pl.Int64))\n                elif col_type in Numeric_Float_types:\n                    if c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df = df.with_columns(df[col].cast(pl.Float32))\n                    else:\n                        pass\n                # elif col_type == pl.Utf8:\n                #     df = df.with_columns(df[col].cast(pl.Categorical))\n                else:\n                    pass\n            except:\n                pass\n        print(f\"Memory usage of dataframe became {round(df.estimated_size('mb'), 2)} MB\")\n        return df\n\n    @staticmethod\n    def detect_datetime_cols(df):\n        return df.select_dtypes(object).apply(lambda x: pd.to_datetime(x, errors='ignore'), axis=0).select_dtypes(np.datetime64).columns.tolist()\n                        \n    def _to_pandas(self, df):\n        df = df.to_pandas().set_index('case_id')\n        df = df.replace([np.inf, -np.inf], np.nan)\n        return df\n\n    def merge_static(self, df):\n        df_static = (\n            pl.concat([pl.scan_parquet(p, low_memory=True) for p in glob.glob(str(self.path / f\"{self.mode}_static_0_*\"))],how=\"vertical_relaxed\",)\n            .with_columns(\n                [\n                    (pl.col(col).cast(pl.String).str.to_date(strict=False)) \n                    for col in [\n                        'datefirstoffer_1144D', \n                        'datelastinstal40dpd_247D',\n                        'datelastunpaid_3546854D', \n                        'dtlastpmtallstes_4499206D',\n                        'firstclxcampaign_1125D', \n                        'firstdatedue_489D', \n                        'lastactivateddate_801D',\n                       'lastapplicationdate_877D', \n                        'lastapprdate_640D', \n                        'lastdelinqdate_224D',\n                       'lastrejectdate_50D', \n                        'lastrepayingdate_696D',\n                       'maxdpdinstldate_3546855D', \n                        'payvacationpostpone_4187118D',\n                       'validfrom_1069D'\n                    ]\n                ] + [\n                    (pl.col(col).cast(pl.String).cast(pl.Categorical))\n                    for col in [\n                        'bankacctype_710L', 'cardtype_51L', 'credtype_322L',\n                       'disbursementtype_67L', 'equalitydataagreement_891L',\n                       'equalityempfrom_62L', 'inittransactioncode_186L',\n                       'isbidproductrequest_292L', 'isdebitcard_729L',\n                       'lastapprcommoditycat_1041M', 'lastapprcommoditytypec_5251766M',\n                       'lastcancelreason_561M', 'lastrejectcommoditycat_161M',\n                       'lastrejectcommodtypec_5251769M', 'lastrejectreason_759M',\n                       'lastrejectreasonclient_4145040M', 'lastst_736L', 'opencred_647L',\n                       'paytype1st_925L', 'paytype_783L', 'previouscontdistrict_112M',\n                       'twobodfilling_608L', 'typesuite_864L'\n                    ]\n                ]\n            )\n        )\n        return df.join(df_static, how=\"left\", on=\"case_id\")\n        \n    def merge_static_cb(self, df):\n        df_static_cb = (\n            pl.scan_parquet(self.path / f\"{self.mode}_static_cb_0.parquet\", low_memory=True)\n            .with_columns(\n                [\n                    (pl.col(col).cast(pl.String).str.to_date(strict=False)) \n                    for col in [\n                        'assignmentdate_238D', \n                        'assignmentdate_4527235D',\n                        'assignmentdate_4955616D', \n                        'birthdate_574D', \n                        'dateofbirth_337D',\n                        'dateofbirth_342D', \n                        'responsedate_1012D', \n                        'responsedate_4527233D',\n                        'responsedate_4917613D'\n                    ] \n                ] + [\n                    (pl.col(col).cast(pl.String).cast(pl.Categorical))\n                    for col in [\n                        'description_5085714M', 'education_1103M', 'education_88M',\n                       'maritalst_385M', 'maritalst_893M', 'requesttype_4525192L',\n                       'riskassesment_302T'\n                    ]\n                ]\n            )\n        )\n        return df.join(df_static_cb, how=\"left\", on=\"case_id\")\n \n    def load(self):\n        df = pl.scan_parquet(self.path / f\"{self.mode}_base.parquet\", low_memory=True).with_columns(\n            pl.col(\"date_decision\").str.to_date()\n        )\n        # Depth=0\n        df = self.merge_static(df)\n        df = self.merge_static_cb(df)\n        \n        df =(\n            df\n            .with_columns(\n                pl.col(pl.Float64).cast(pl.Float32),\n                pl.col(pl.Int64).cast(pl.Int32),\n            )\n        )\n        df = df.select(~cs.date())\n        \n        # Drop categorical large-dimension columns\n        df = df.drop([\n            'lastapprcommoditytypec_5251766M',\n             'previouscontdistrict_112M',\n             'district_544M',\n             'profession_152M',\n             'name_4527232M',\n             'name_4917606M',\n             'employername_160M',\n             'classificationofcontr_400M',\n             'financialinstitution_382M',\n             'contaddr_district_15M',\n             'contaddr_zipcode_807M',\n             'empladdr_district_926M',\n             'empladdr_zipcode_114M',\n             'registaddr_district_1083M',\n             'registaddr_zipcode_184M',\n             'addres_district_368M',\n             'addres_zip_823M'])\n        df = df.collect()\n        df = self.reduce_memory_usage_pl(df)\n        df = self._to_pandas(df)\n        return df","metadata":{"execution":{"iopub.status.busy":"2024-02-07T13:59:31.023784Z","iopub.execute_input":"2024-02-07T13:59:31.024200Z","iopub.status.idle":"2024-02-07T13:59:31.061670Z","shell.execute_reply.started":"2024-02-07T13:59:31.024167Z","shell.execute_reply":"2024-02-07T13:59:31.060349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_constructor = DatasetConstructor('train')\ndf_train = train_constructor.load()\ndf_train.info()","metadata":{"execution":{"iopub.status.busy":"2024-02-07T13:59:38.688620Z","iopub.execute_input":"2024-02-07T13:59:38.689085Z","iopub.status.idle":"2024-02-07T13:59:52.357304Z","shell.execute_reply.started":"2024-02-07T13:59:38.689050Z","shell.execute_reply":"2024-02-07T13:59:52.355993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train simple LGBM","metadata":{"execution":{"iopub.status.busy":"2024-02-07T13:59:52.358959Z","iopub.execute_input":"2024-02-07T13:59:52.359298Z","iopub.status.idle":"2024-02-07T13:59:52.365236Z","shell.execute_reply.started":"2024-02-07T13:59:52.359271Z","shell.execute_reply":"2024-02-07T13:59:52.363792Z"}}},{"cell_type":"code","source":"X, y = df_train.drop(columns='target'), df_train['target']\nX.shape","metadata":{"execution":{"iopub.status.busy":"2024-02-07T13:59:59.830475Z","iopub.execute_input":"2024-02-07T13:59:59.830943Z","iopub.status.idle":"2024-02-07T14:00:00.195952Z","shell.execute_reply.started":"2024-02-07T13:59:59.830907Z","shell.execute_reply":"2024-02-07T14:00:00.194231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = X.drop(columns=['WEEK_NUM'] ) ","metadata":{"execution":{"iopub.status.busy":"2024-02-07T14:00:28.014749Z","iopub.execute_input":"2024-02-07T14:00:28.015175Z","iopub.status.idle":"2024-02-07T14:00:28.363078Z","shell.execute_reply.started":"2024-02-07T14:00:28.015145Z","shell.execute_reply":"2024-02-07T14:00:28.361873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nparams = {\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"binary\",\n    \"metric\": \"auc\",\n    \"max_depth\": 8,\n    \"num_leaves\": 32,\n    \"min_data_in_leaf\": 10,\n    \"learning_rate\": 0.05,\n    \"feature_fraction\": 0.8,\n    \"bagging_fraction\": 0.8,\n    \"bagging_freq\": 5,\n    \"n_estimators\": 100,\n    'min_data_in_bin':1,\n    'max_bin': 64,\n    \"verbose\": -1,\n    \"random_state\": 42, \n    'n_jobs': -1\n}\ncv = StratifiedGroupKFold(n_splits=5, shuffle=False)\ncv_results = cross_validate(\n    lgb.LGBMClassifier(**params), \n    X, y, \n    groups=df_train['WEEK_NUM'], \n    scoring='roc_auc', \n    cv=cv,\n    verbose=3, \n    return_estimator=True, \n    return_indices=True\n)\nprint(f\"AUC: {cv_results['test_score'].mean():.3f}\", f\"+-{cv_results['test_score'].std():.3f}\")","metadata":{"execution":{"iopub.status.busy":"2024-02-07T14:00:34.127124Z","iopub.execute_input":"2024-02-07T14:00:34.127525Z","iopub.status.idle":"2024-02-07T14:12:08.625099Z","shell.execute_reply.started":"2024-02-07T14:00:34.127494Z","shell.execute_reply":"2024-02-07T14:12:08.622283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stability_results = []\nfor fold, (idx, model) in enumerate(zip(cv_results['indices']['test'], cv_results['estimator'])):\n    df_res = pd.DataFrame()\n    \n    df_res['WEEK_NUM'] = df_train['WEEK_NUM'].iloc[idx].values\n    df_res['target'] = df_train['target'].iloc[idx].values\n    df_res['score'] = model.predict_proba(X.iloc[idx])[:, 1]\n    df_res['fold'] = fold\n    stability_results.append(df_res)\n    \ndf_stability_results = pd.concat(stability_results)\ndf_stability_results[df_stability_results['target'] == 0]['score'].plot(kind='hist', alpha=0.5, bins=100, label='target=0')\ndf_stability_results[df_stability_results['target'] == 1]['score'].plot(kind='hist', secondary_y=True, alpha=0.5, bins=100, label='target=1')\nplt.legend();","metadata":{"execution":{"iopub.status.busy":"2024-02-07T14:12:08.631848Z","iopub.execute_input":"2024-02-07T14:12:08.632552Z","iopub.status.idle":"2024-02-07T14:12:35.542434Z","shell.execute_reply.started":"2024-02-07T14:12:08.632494Z","shell.execute_reply":"2024-02-07T14:12:35.540980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.boxenplot(df_stability_results, y='score', x='target');","metadata":{"execution":{"iopub.status.busy":"2024-02-07T14:12:35.544323Z","iopub.execute_input":"2024-02-07T14:12:35.544705Z","iopub.status.idle":"2024-02-07T14:12:36.519159Z","shell.execute_reply.started":"2024-02-07T14:12:35.544675Z","shell.execute_reply":"2024-02-07T14:12:36.517389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def gini_stability(base, w_fallingrate=88.0, w_resstd=-0.5):\n    gini_in_time = base.loc[:, [\"WEEK_NUM\", \"target\", \"score\"]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", \"score\"]]\\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[\"score\"])-1).tolist()\n    \n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std\ndf_stability_results.groupby('fold').apply(gini_stability, include_groups=False)","metadata":{"execution":{"iopub.status.busy":"2024-02-07T14:12:36.534641Z","iopub.execute_input":"2024-02-07T14:12:36.535135Z","iopub.status.idle":"2024-02-07T14:12:37.509767Z","shell.execute_reply.started":"2024-02-07T14:12:36.535091Z","shell.execute_reply":"2024-02-07T14:12:37.508345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gini_stability(df_stability_results)","metadata":{"execution":{"iopub.status.busy":"2024-02-07T14:12:37.512014Z","iopub.execute_input":"2024-02-07T14:12:37.512505Z","iopub.status.idle":"2024-02-07T14:12:38.377215Z","shell.execute_reply.started":"2024-02-07T14:12:37.512462Z","shell.execute_reply":"2024-02-07T14:12:38.375303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = cv_results['estimator']","metadata":{"execution":{"iopub.status.busy":"2024-02-07T14:12:38.379049Z","iopub.execute_input":"2024-02-07T14:12:38.379486Z","iopub.status.idle":"2024-02-07T14:12:38.385660Z","shell.execute_reply.started":"2024-02-07T14:12:38.379453Z","shell.execute_reply":"2024-02-07T14:12:38.384369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df_train, X, y\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-02-07T14:12:38.387472Z","iopub.execute_input":"2024-02-07T14:12:38.388084Z","iopub.status.idle":"2024-02-07T14:12:38.605230Z","shell.execute_reply.started":"2024-02-07T14:12:38.388040Z","shell.execute_reply":"2024-02-07T14:12:38.603379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading test data","metadata":{}},{"cell_type":"code","source":"test_constructor = DatasetConstructor('test')\ndf_test=test_constructor.load()","metadata":{"execution":{"iopub.status.busy":"2024-02-07T14:12:38.607146Z","iopub.execute_input":"2024-02-07T14:12:38.607653Z","iopub.status.idle":"2024-02-07T14:12:38.795002Z","shell.execute_reply.started":"2024-02-07T14:12:38.607605Z","shell.execute_reply":"2024-02-07T14:12:38.793834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict and submit","metadata":{}},{"cell_type":"code","source":"test = df_test[models[0].feature_name_]\npreds_proba = [model.predict_proba(test)[:, 1] for model in models]\ndf_test[\"score\"] = np.average(preds_proba, axis=0)\ndf_test[[\"score\"]].to_csv(\"submission.csv\")\n\n!head submission.csv","metadata":{"execution":{"iopub.status.busy":"2024-02-07T14:13:20.611269Z","iopub.execute_input":"2024-02-07T14:13:20.611672Z","iopub.status.idle":"2024-02-07T14:13:21.954605Z","shell.execute_reply.started":"2024-02-07T14:13:20.611643Z","shell.execute_reply":"2024-02-07T14:13:21.952651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}