{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# data reading","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-03-17T12:39:33.022962Z","iopub.execute_input":"2024-03-17T12:39:33.023481Z","iopub.status.idle":"2024-03-17T12:39:33.030372Z","shell.execute_reply.started":"2024-03-17T12:39:33.023437Z","shell.execute_reply":"2024-03-17T12:39:33.028945Z"}}},{"cell_type":"code","source":"import polars as pl\nimport numpy as np\nimport pandas as pd\nimport lightgbm as lgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score \n\ndataPath = \"/kaggle/input/home-credit-credit-risk-model-stability/\"","metadata":{"execution":{"iopub.status.busy":"2024-03-17T16:50:18.729680Z","iopub.execute_input":"2024-03-17T16:50:18.730222Z","iopub.status.idle":"2024-03-17T16:50:18.737974Z","shell.execute_reply.started":"2024-03-17T16:50:18.730174Z","shell.execute_reply":"2024-03-17T16:50:18.736530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_table_dtypes(df: pl.DataFrame) -> pl.DataFrame:\n    # implement here all desired dtypes for tables\n    # the following is just an example\n    for col in df.columns:\n        # last letter of column name will help you determine the type\n        if col[-1] in (\"P\", \"A\"):\n            df = df.with_columns(pl.col(col).cast(pl.Float64).alias(col))\n\n    return df\n\ndef convert_strings(df: pd.DataFrame) -> pd.DataFrame:\n    for col in df.columns:  \n        if df[col].dtype.name in ['object', 'string']:\n            df[col] = df[col].astype(\"string\").astype('category')\n            current_categories = df[col].cat.categories\n            new_categories = current_categories.to_list() + [\"Unknown\"]\n            new_dtype = pd.CategoricalDtype(categories=new_categories, ordered=True)\n            df[col] = df[col].astype(new_dtype)\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-03-17T16:50:19.326730Z","iopub.execute_input":"2024-03-17T16:50:19.327121Z","iopub.status.idle":"2024-03-17T16:50:19.336680Z","shell.execute_reply.started":"2024-03-17T16:50:19.327092Z","shell.execute_reply":"2024-03-17T16:50:19.335539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_basetable = pl.read_csv(dataPath + \"csv_files/train/train_base.csv\")\ntrain_static = pl.concat(\n    [\n        pl.read_csv(dataPath + \"csv_files/train/train_static_0_0.csv\").pipe(set_table_dtypes),\n        pl.read_csv(dataPath + \"csv_files/train/train_static_0_1.csv\").pipe(set_table_dtypes),\n    ],\n    how=\"vertical_relaxed\",\n)\ntrain_static_cb = pl.read_csv(dataPath + \"csv_files/train/train_static_cb_0.csv\").pipe(set_table_dtypes)\ntrain_person_1 = pl.read_csv(dataPath + \"csv_files/train/train_person_1.csv\").pipe(set_table_dtypes) \ntrain_credit_bureau_b_2 = pl.read_csv(dataPath + \"csv_files/train/train_credit_bureau_b_2.csv\").pipe(set_table_dtypes) ","metadata":{"execution":{"iopub.status.busy":"2024-03-17T16:50:19.794965Z","iopub.execute_input":"2024-03-17T16:50:19.796393Z","iopub.status.idle":"2024-03-17T16:50:37.294465Z","shell.execute_reply.started":"2024-03-17T16:50:19.796352Z","shell.execute_reply":"2024-03-17T16:50:37.293565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_basetable = pl.read_csv(dataPath + \"csv_files/test/test_base.csv\")\ntest_static = pl.concat(\n    [\n        pl.read_csv(dataPath + \"csv_files/test/test_static_0_0.csv\").pipe(set_table_dtypes),\n        pl.read_csv(dataPath + \"csv_files/test/test_static_0_1.csv\").pipe(set_table_dtypes),\n        pl.read_csv(dataPath + \"csv_files/test/test_static_0_2.csv\").pipe(set_table_dtypes),\n    ],\n    how=\"vertical_relaxed\",\n)\ntest_static_cb = pl.read_csv(dataPath + \"csv_files/test/test_static_cb_0.csv\").pipe(set_table_dtypes)\ntest_person_1 = pl.read_csv(dataPath + \"csv_files/test/test_person_1.csv\").pipe(set_table_dtypes) \ntest_credit_bureau_b_2 = pl.read_csv(dataPath + \"csv_files/test/test_credit_bureau_b_2.csv\").pipe(set_table_dtypes) ","metadata":{"execution":{"iopub.status.busy":"2024-03-17T16:50:37.299287Z","iopub.execute_input":"2024-03-17T16:50:37.300461Z","iopub.status.idle":"2024-03-17T16:50:37.349921Z","shell.execute_reply.started":"2024-03-17T16:50:37.300419Z","shell.execute_reply":"2024-03-17T16:50:37.348952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# feature engineeding with pandas did not work out, memory error, thus i use polars library applying on the author of cometition for data reading and loading, but my data enginnering is built different. I believe that he intentionally made mistakes in his starter notebook: meaning the aggregation for the num_group1 cases. It is said in the competition rules and overview and data decription that the num_group1 ==0 or num_group2==0 means the loaner, thus we do not have to aggregate the num_groupN, we just need to stay a filter of num_groupN value equak to zero, as it is the actual data of the person applying to the loan, the applicant. The aggregation of max,sum and et cetera is not right to be used for group_num1 as it is historic data and person does not have this features at current moment and we lose information by aggragating, finding max,sum and et cetera. For credit scroing model to be inpretable the this aggrageted features would be wrong( the person applying in this case_id sometime before had this features, but we lose information about the current state of his).","metadata":{}},{"cell_type":"code","source":"# def feature_engineering_for_num1_and_merge(df, name, prefix=\"train\", folder=PATH_PARQUETS):\n#     df_ = pd.concat(\n#         [pd.read_parquet(p) for p in glob.glob(str(folder / prefix / f\"{prefix}_{name}*.parquet\"))],\n#     )\n#     df_.drop_duplicates(inplace=True)\n#     df_ = df_[df_[\"num_group1\"]==0] # filtration made for the client to be used in training(explanation in markdown)\n#     if \"num_group1\" in df_.columns:\n#         del df_[\"num_group1\"]\n#     cols = convert_dtypes(df_) + [\"case_id\"]\n#     df_ = df_[cols]\n\n#     df = df.merge(df_, how=\"left\", on=\"case_id\")\n#     print(f\"fused size: {len(df)}\")\n#     return df","metadata":{"execution":{"iopub.status.busy":"2024-03-17T16:50:37.351853Z","iopub.execute_input":"2024-03-17T16:50:37.352715Z","iopub.status.idle":"2024-03-17T16:50:37.358457Z","shell.execute_reply.started":"2024-03-17T16:50:37.352673Z","shell.execute_reply":"2024-03-17T16:50:37.357346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_train = feature_engineering_for_num1_and_merge(df_train, \"applprev_1\")\n# df_train = feature_engineering_for_num1_and_merge(df_train, \"other_1\")\n# df_train = feature_engineering_for_num1_and_merge(df_train, \"deposit_1\")\n# df_train = feature_engineering_for_num1_and_merge(df_train, \"debitcard_1\")\n# df_train = feature_engineering_for_num1_and_merge(df_train, \"person_1\")\n# df_train = feature_engineering_for_num1_and_merge(df_train, \"credit_bureau_a_1\")\n# df_train = feature_engineering_for_num1_and_merge(df_train, \"tax_registry_a_1\")\n# df_train = feature_engineering_for_num1_and_merge(df_train, \"tax_registry_b_1\")\n# df_train = feature_engineering_for_num1_and_merge(df_train, \"tax_registry_c_1\")\n","metadata":{"execution":{"iopub.status.busy":"2024-03-17T16:50:37.360831Z","iopub.execute_input":"2024-03-17T16:50:37.361473Z","iopub.status.idle":"2024-03-17T16:50:37.368444Z","shell.execute_reply.started":"2024-03-17T16:50:37.361444Z","shell.execute_reply":"2024-03-17T16:50:37.367532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We need to use aggregation functions in tables with depth > 1, so tables that contain num_group1 column or \n# also num_group2 column.\ntrain_person_1_feats_1 = train_person_1.group_by(\"case_id\").agg(\n    pl.col(\"mainoccupationinc_384A\").max().alias(\"mainoccupationinc_384A_max\"),\n    (pl.col(\"incometype_1044T\") == \"SELFEMPLOYED\").max().alias(\"mainoccupationinc_384A_any_selfemployed\")\n)\n\n# Here num_group1=0 has special meaning, it is the person who applied for the loan.\ntrain_person_1_feats_2 = train_person_1.select([\"case_id\", \"num_group1\", \"housetype_905L\"]).filter(\n    pl.col(\"num_group1\") == 0\n).drop(\"num_group1\").rename({\"housetype_905L\": \"person_housetype\"})\n\n# Here we have num_goup1 and num_group2, so we need to aggregate again.\ntrain_credit_bureau_b_2_feats = train_credit_bureau_b_2.group_by(\"case_id\").agg(\n    pl.col(\"pmts_pmtsoverdue_635A\").max().alias(\"pmts_pmtsoverdue_635A_max\"),\n    (pl.col(\"pmts_dpdvalue_108P\") > 31).max().alias(\"pmts_dpdvalue_108P_over31\")\n)\n\n# We will process in this examples only A-type and M-type columns, so we need to select them.\nselected_static_cols = []\nfor col in train_static.columns:\n    if col[-1] in (\"A\", \"M\"):\n        selected_static_cols.append(col)\nprint(selected_static_cols)\n\nselected_static_cb_cols = []\nfor col in train_static_cb.columns:\n    if col[-1] in (\"A\", \"M\"):\n        selected_static_cb_cols.append(col)\nprint(selected_static_cb_cols)\n\n# Join all tables together.\ndata = train_basetable.join(\n    train_static.select([\"case_id\"]+selected_static_cols), how=\"left\", on=\"case_id\"\n).join(\n    train_static_cb.select([\"case_id\"]+selected_static_cb_cols), how=\"left\", on=\"case_id\"\n).join(\n    train_person_1_feats_1, how=\"left\", on=\"case_id\"\n).join(\n    train_person_1_feats_2, how=\"left\", on=\"case_id\"\n).join(\n    train_credit_bureau_b_2_feats, how=\"left\", on=\"case_id\"\n)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T16:50:37.369828Z","iopub.execute_input":"2024-03-17T16:50:37.370412Z","iopub.status.idle":"2024-03-17T16:50:39.997348Z","shell.execute_reply.started":"2024-03-17T16:50:37.370382Z","shell.execute_reply":"2024-03-17T16:50:39.996368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_person_1_feats_1 = test_person_1.group_by(\"case_id\").agg(\n    pl.col(\"mainoccupationinc_384A\").max().alias(\"mainoccupationinc_384A_max\"),\n    (pl.col(\"incometype_1044T\") == \"SELFEMPLOYED\").max().alias(\"mainoccupationinc_384A_any_selfemployed\")\n)\n\ntest_person_1_feats_2 = test_person_1.select([\"case_id\", \"num_group1\", \"housetype_905L\"]).filter(\n    pl.col(\"num_group1\") == 0\n).drop(\"num_group1\").rename({\"housetype_905L\": \"person_housetype\"})\n\ntest_credit_bureau_b_2_feats = test_credit_bureau_b_2.group_by(\"case_id\").agg(\n    pl.col(\"pmts_pmtsoverdue_635A\").max().alias(\"pmts_pmtsoverdue_635A_max\"),\n    (pl.col(\"pmts_dpdvalue_108P\") > 31).max().alias(\"pmts_dpdvalue_108P_over31\")\n)\n\ndata_submission = test_basetable.join(\n    test_static.select([\"case_id\"]+selected_static_cols), how=\"left\", on=\"case_id\"\n).join(\n    test_static_cb.select([\"case_id\"]+selected_static_cb_cols), how=\"left\", on=\"case_id\"\n).join(\n    test_person_1_feats_1, how=\"left\", on=\"case_id\"\n).join(\n    test_person_1_feats_2, how=\"left\", on=\"case_id\"\n).join(\n    test_credit_bureau_b_2_feats, how=\"left\", on=\"case_id\"\n)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T16:50:39.998755Z","iopub.execute_input":"2024-03-17T16:50:39.999307Z","iopub.status.idle":"2024-03-17T16:50:40.016422Z","shell.execute_reply.started":"2024-03-17T16:50:39.999276Z","shell.execute_reply":"2024-03-17T16:50:40.015551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"case_ids = data[\"case_id\"].unique().shuffle(seed=1)\ncase_ids_train, case_ids_test = train_test_split(case_ids, train_size=0.6, random_state=1)\ncase_ids_valid, case_ids_test = train_test_split(case_ids_test, train_size=0.5, random_state=1)\n\ncols_pred = []\nfor col in data.columns:\n    if col[-1].isupper() and col[:-1].islower():\n        cols_pred.append(col)\n\nprint(cols_pred)\n\ndef from_polars_to_pandas(case_ids: pl.DataFrame) -> pl.DataFrame:\n    return (\n        data.filter(pl.col(\"case_id\").is_in(case_ids))[[\"case_id\", \"WEEK_NUM\", \"target\"]].to_pandas(),\n        data.filter(pl.col(\"case_id\").is_in(case_ids))[cols_pred].to_pandas(),\n        data.filter(pl.col(\"case_id\").is_in(case_ids))[\"target\"].to_pandas()\n    )\n\nbase_train, X_train, y_train = from_polars_to_pandas(case_ids_train)\nbase_valid, X_valid, y_valid = from_polars_to_pandas(case_ids_valid)\nbase_test, X_test, y_test = from_polars_to_pandas(case_ids_test)\n\nfor df in [X_train, X_valid, X_test]:\n    df = convert_strings(df)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T16:50:40.017783Z","iopub.execute_input":"2024-03-17T16:50:40.018347Z","iopub.status.idle":"2024-03-17T16:50:48.389905Z","shell.execute_reply.started":"2024-03-17T16:50:40.018318Z","shell.execute_reply":"2024-03-17T16:50:48.388834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Train: {X_train.shape}\")\nprint(f\"Valid: {X_valid.shape}\")\nprint(f\"Valid: {X_test.shape}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-03-17T16:50:48.391348Z","iopub.execute_input":"2024-03-17T16:50:48.391895Z","iopub.status.idle":"2024-03-17T16:50:48.398427Z","shell.execute_reply.started":"2024-03-17T16:50:48.391865Z","shell.execute_reply":"2024-03-17T16:50:48.397006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# applying feature generation","metadata":{}},{"cell_type":"code","source":"# feature enginering for the resulting dataframes\ndef additional_feature_engineering(df: pd.DataFrame):\n    #ratio of initial transaction amount to the total debt\n    df['ratio_curr_debt_to_all_debt'] = df['inittransactionamount_650A'] / df['totaldebt_9A']\n    \n    # ratio of curr sum of loans to the total debt\n    df['ratio_curr_debt_to_all_debt'] = df['currdebtcredtyperange_828A'] / df['totaldebt_9A']\n    \n    # ratio of finished sum of loans to the total debt\n    df['ratio_settled_left'] = df['totalsettled_863A'] / df['totaldebt_9A']\n    \n    # quantity of loans bigger then median statistic of this column\n    df['number_credamount_above_median'] = (df['credamount_770A'] > df['credamount_770A'].median()).astype(int)\n    \n    # quantity of loans bigger then mean statistic of this column\n    df['number_credamount_above_mean'] = (df['credamount_770A'] >df['credamount_770A'].mean()).astype(int)\n\n    # log_annuity\n    df['log_annuity'] = np.log1p(df['annuity_780A'])\n    \n    # average_number_of_views divided by all debt numbers\n    df['ratio_avgviews_avgpayments'] = df['avgpmtlast12m_4525200A'] / df['avginstallast24m_3658937A']\n    \n    # average_number_of_debts divided by all debt quatity\n    df['ratio_avg_debts'] = df['avgpmtlast12m_4525200A'] / df['totaldebt_9A']\n    \n    # ration of credit price to all debt quantity\n    df['ratio_price_to_debt'] = df['price_1097A'] / df['totaldebt_9A']\n\n    # standard deviation of payments \n    df['std_paym'] = df[['avgpmtlast12m_4525200A', 'maxpmtlast3m_4525190A']].std(axis=1)\n    \n    \n    return df\n\n","metadata":{"execution":{"iopub.status.busy":"2024-03-17T16:50:48.400078Z","iopub.execute_input":"2024-03-17T16:50:48.400521Z","iopub.status.idle":"2024-03-17T16:50:48.411207Z","shell.execute_reply.started":"2024-03-17T16:50:48.400490Z","shell.execute_reply":"2024-03-17T16:50:48.410098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = additional_feature_engineering(X_train)\nX_valid = additional_feature_engineering(X_valid)\nX_test = additional_feature_engineering(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T16:50:48.414939Z","iopub.execute_input":"2024-03-17T16:50:48.415310Z","iopub.status.idle":"2024-03-17T16:50:49.159236Z","shell.execute_reply.started":"2024-03-17T16:50:48.415280Z","shell.execute_reply":"2024-03-17T16:50:49.157971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# I believe that the amount of features to be stable is not to be anotmous. The whole dataset is big enough, I used only some basic tables,but I believe that I can have the good metric( auc - main stability indicator) if the number of predicators is much smaller(. Thus, by feature enginnering I also understand the feature selection - the reduced amount of features, with insignificant decrease of the quality of the model. For this I will use approximate Shap values, which is the most robust approach existing right now. Additionaly, the credit scoring in the fintech industry is very imporant approach and  bussines necessities are mainly apply on the interpretability of the model predictors, no talking about the possibility of using 50+ for prom version is very low, thus I do FeatureSelector from catboost. I believe 40 features is enough to catch all the needful information in the data.","metadata":{}},{"cell_type":"code","source":"from catboost import CatBoostClassifier, EShapCalcType, EFeaturesSelectionAlgorithm, Pool\n\ncat_cols = ['lastapprcommoditycat_1041M',\n            'lastapprcommoditytypec_5251766M',\n            'lastcancelreason_561M',\n            'lastrejectcommoditycat_161M',\n            'lastrejectcommodtypec_5251769M',\n            'lastrejectreason_759M',\n            'lastrejectreasonclient_4145040M',\n            'previouscontdistrict_112M'\n           ]","metadata":{"execution":{"iopub.status.busy":"2024-03-17T16:50:49.161225Z","iopub.execute_input":"2024-03-17T16:50:49.161692Z","iopub.status.idle":"2024-03-17T16:50:49.168642Z","shell.execute_reply.started":"2024-03-17T16:50:49.161647Z","shell.execute_reply":"2024-03-17T16:50:49.167378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# I do not believe soc demographical features to be valuable, they are dublicating , let's call it expert feature selection technique( in company I work at, the recomendational systems models(scoring are quite a part of it are usually not that valuable in most of models, ecpeccialy education adn marital status)","metadata":{}},{"cell_type":"code","source":"X_train.drop(columns = ['description_5085714M','education_1103M',\n             'education_88M',\n             'maritalst_385M',\n             'maritalst_893M'],inplace =True)\nX_valid.drop(columns = ['description_5085714M','education_1103M',\n             'education_88M',\n             'maritalst_385M',\n             'maritalst_893M'],inplace =True)\nX_test.drop(columns = ['description_5085714M','education_1103M',\n             'education_88M',\n             'maritalst_385M',\n             'maritalst_893M'],inplace =True)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T16:50:49.169713Z","iopub.execute_input":"2024-03-17T16:50:49.170028Z","iopub.status.idle":"2024-03-17T16:50:49.445265Z","shell.execute_reply.started":"2024-03-17T16:50:49.170002Z","shell.execute_reply":"2024-03-17T16:50:49.444273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_pool = Pool(X_train, y_train, feature_names=list(X_train.columns), cat_features=cat_cols)\nX_valid_pool = Pool(X_valid, y_valid, feature_names=list(X_train.columns), cat_features=cat_cols)\nX_test_pool = Pool(X_test, y_test, feature_names=list(X_train.columns), cat_features=cat_cols)\ndef select_features(algorithm: EFeaturesSelectionAlgorithm, steps: int = 1): \n    print('Algorithm:', algorithm)\n    model = CatBoostClassifier(iterations=100, random_seed=0)\n    print('model')\n    summary = model.select_features(\n        X_train_pool,\n        eval_set=X_valid_pool,\n        features_for_select=list(range(X_train_pool.num_col())),\n        num_features_to_select=40,\n        steps=steps,\n        algorithm=algorithm,\n        shap_calc_type=EShapCalcType.Approximate,\n        train_final_model=True,\n        logging_level=None,\n        plot=False\n    )\n    print('Selected features:', summary['selected_features_names'])\n    return summary\nselect_features(algorithm=EFeaturesSelectionAlgorithm.RecursiveByShapValues,steps = 1)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T16:50:49.446593Z","iopub.execute_input":"2024-03-17T16:50:49.446922Z","iopub.status.idle":"2024-03-17T16:57:57.951599Z","shell.execute_reply.started":"2024-03-17T16:50:49.446895Z","shell.execute_reply":"2024-03-17T16:57:57.950433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selected_features = ['amtinstpaidbefduel24m_4187115A', 'annuity_780A', 'avginstallast24m_3658937A', 'avgoutstandbalancel6m_4187114A', 'avgpmtlast12m_4525200A', 'credamount_770A', 'currdebt_22A', 'currdebtcredtyperange_828A', 'disbursedcredamount_1113A', 'downpmt_116A', 'inittransactionamount_650A', 'lastapprcommoditycat_1041M', 'lastapprcommoditytypec_5251766M', 'lastapprcredamount_781A', 'lastcancelreason_561M', 'lastrejectcommoditycat_161M', 'lastrejectcredamount_222A', 'lastrejectreason_759M', 'lastrejectreasonclient_4145040M', 'maininc_215A', 'maxannuity_159A', 'maxannuity_4075009A', 'maxdebt4_972A', 'maxlnamtstart6m_4525199A', 'maxoutstandbalancel12m_4187113A', 'maxpmtlast3m_4525190A', 'price_1097A', 'sumoutstandtotal_3546847A', 'totaldebt_9A', 'totalsettled_863A', 'pmtaverage_3A', 'pmtaverage_4527227A', 'pmtaverage_4955615A', 'pmtssum_45A', 'ratio_curr_debt_to_all_debt', 'log_annuity', 'ratio_avgviews_avgpayments', 'ratio_avg_debts', 'ratio_price_to_debt', 'std_paym']","metadata":{"execution":{"iopub.status.busy":"2024-03-17T16:57:57.953008Z","iopub.execute_input":"2024-03-17T16:57:57.953328Z","iopub.status.idle":"2024-03-17T16:57:57.960052Z","shell.execute_reply.started":"2024-03-17T16:57:57.953301Z","shell.execute_reply":"2024-03-17T16:57:57.959064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# In selected features we see our new generated features, the features where were smaller infromation are lost, we will not use them. It is a good result, even though we may not improve the score of the model, I showed that my data generated features have sense and they have useful information.","metadata":{}},{"cell_type":"code","source":"X_train= X_train[selected_features]\nX_test = X_test[selected_features]\nX_valid = X_valid[selected_features]","metadata":{"execution":{"iopub.status.busy":"2024-03-17T16:57:57.961285Z","iopub.execute_input":"2024-03-17T16:57:57.961623Z","iopub.status.idle":"2024-03-17T16:57:58.181029Z","shell.execute_reply.started":"2024-03-17T16:57:57.961594Z","shell.execute_reply":"2024-03-17T16:57:58.179722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Hyperparameter tuninig, the optuna did not showed good resultss(deleted from logs, but the score was close to just random,as the dataset is not huge enough parametres should be choosed accurately, so we will slightly try to improve the existing solution of competition maker by adding some values), try to improve the starter collab balues by adding grid. I choosed lgmb as model, while catboost for additional feature engineering because the inference time showed that LGBM works a litlle bit faster(at my job), and we do not have a huge amount of difficult categorical data, then catboost will be more efficient","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import RandomizedSearchCV\nfrom lightgbm import LGBMClassifier as lgb\n\nparam_dist = {\n    'boosting_type': ['gbdt'],  \n    'objective': ['binary'],    \n    'metric': ['auc'],         \n    'max_depth': [3, 7],        \n    'num_leaves': [31, 40],     \n    'learning_rate': [0.05, 0.001],\n    'feature_fraction': [0.9],  \n    'bagging_fraction': [0.8],  \n    'bagging_freq': [5],\n    'n_estimators': [1000],    \n    'verbose': [-1],           \n}\n\nlgb_estimator = lgb(random_state=0)\n\nopt = RandomizedSearchCV(\n    estimator=lgb_estimator,\n    param_distributions=param_dist,\n    n_iter=3, \n    scoring='roc_auc',\n    cv=3,\n    verbose=3,\n    random_state=0\n)\n\nopt.fit(X_train, y_train)\n\nprint(opt.best_params_,opt.best_score_)\nbest_model = opt.best_estimator_","metadata":{"execution":{"iopub.status.busy":"2024-03-17T16:57:58.182567Z","iopub.execute_input":"2024-03-17T16:57:58.183019Z","iopub.status.idle":"2024-03-17T17:36:02.186832Z","shell.execute_reply.started":"2024-03-17T16:57:58.182988Z","shell.execute_reply":"2024-03-17T17:36:02.185203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for base, X in [(base_train, X_train), (base_valid, X_valid), (base_test, X_test)]:\n    y_pred = best_model.predict_proba(X)[:, 1]  \n    base[\"score\"] = y_pred\n\nprint(f'The AUC score on the train set is: {roc_auc_score(base_train[\"target\"], base_train[\"score\"])}') \nprint(f'The AUC score on the valid set is: {roc_auc_score(base_valid[\"target\"], base_valid[\"score\"])}') \nprint(f'The AUC score on the test set is: {roc_auc_score(base_test[\"target\"], base_test[\"score\"])}') ","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:36:02.188575Z","iopub.execute_input":"2024-03-17T17:36:02.189005Z","iopub.status.idle":"2024-03-17T17:37:20.450489Z","shell.execute_reply.started":"2024-03-17T17:36:02.188969Z","shell.execute_reply":"2024-03-17T17:37:20.449491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def gini_stability(base, w_fallingrate=88.0, w_resstd=-0.5):\n    gini_in_time = base.loc[:, [\"WEEK_NUM\", \"target\", \"score\"]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", \"score\"]]\\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[\"score\"])-1).tolist()\n    \n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std\n\nstability_score_train = gini_stability(base_train)\nstability_score_valid = gini_stability(base_valid)\nstability_score_test = gini_stability(base_test)\n\nprint(f'The stability score on the train set is: {stability_score_train}') \nprint(f'The stability score on the valid set is: {stability_score_valid}') \nprint(f'The stability score on the test set is: {stability_score_test}')  ","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:37:20.452103Z","iopub.execute_input":"2024-03-17T17:37:20.452800Z","iopub.status.idle":"2024-03-17T17:37:21.657513Z","shell.execute_reply.started":"2024-03-17T17:37:20.452731Z","shell.execute_reply":"2024-03-17T17:37:21.656429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_submission = data_submission[cols_pred].to_pandas()\nX_submission = convert_strings(X_submission)\ncategorical_cols = X_train.select_dtypes(include=['category']).columns\n\nfor col in categorical_cols:\n    train_categories = set(X_train[col].cat.categories)\n    submission_categories = set(X_submission[col].cat.categories)\n    new_categories = submission_categories - train_categories\n    X_submission.loc[X_submission[col].isin(new_categories), col] = \"Unknown\"\n    new_dtype = pd.CategoricalDtype(categories=train_categories, ordered=True)\n    X_submission[col] = X_submission[col].astype(new_dtype)\n\n    \nX_submission = additional_feature_engineering(X_submission)\ny_submission_pred = best_model.predict_proba(X_submission[selected_features])[:, 1]","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:40:57.363713Z","iopub.execute_input":"2024-03-17T17:40:57.364155Z","iopub.status.idle":"2024-03-17T17:40:57.428948Z","shell.execute_reply.started":"2024-03-17T17:40:57.364106Z","shell.execute_reply":"2024-03-17T17:40:57.427861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\n    \"case_id\": data_submission[\"case_id\"].to_numpy(),\n    \"score\": y_submission_pred\n}).set_index('case_id')\nsubmission.to_csv(\"./submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:40:59.874837Z","iopub.execute_input":"2024-03-17T17:40:59.875318Z","iopub.status.idle":"2024-03-17T17:40:59.887649Z","shell.execute_reply.started":"2024-03-17T17:40:59.875280Z","shell.execute_reply":"2024-03-17T17:40:59.886249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Additional. Even though it is not mentioned in task, I think it will be good to use calibration for received scores. The real scores are not probabilitites and I believe, that we should use it as in credit scoring modelds - the outputs are probabilities. Most of boostings are not calibrated, so we need to perform it, to use this model in production. I used to choose an Isotonic Regression. However, I did not want to submit these scores as my resluting ones, I will provide a code how it can be done. The calibration should be fiited on the most recent possible data, in our case its so_called validation hold out set. Then we change the scores of the test set, the code below shows it.","metadata":{}},{"cell_type":"code","source":"y_submission_pred","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:41:21.623612Z","iopub.execute_input":"2024-03-17T17:41:21.624581Z","iopub.status.idle":"2024-03-17T17:41:21.632711Z","shell.execute_reply.started":"2024-03-17T17:41:21.624540Z","shell.execute_reply":"2024-03-17T17:41:21.631515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.isotonic import IsotonicRegression\niso_reg = IsotonicRegression(y_min = 0, y_max = 1, out_of_bounds = 'clip').fit(base_valid[\"score\"], y_valid)\nproba_test_isoreg = iso_reg.predict(y_submission_pred)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:42:49.693553Z","iopub.execute_input":"2024-03-17T17:42:49.693963Z","iopub.status.idle":"2024-03-17T17:42:49.772649Z","shell.execute_reply.started":"2024-03-17T17:42:49.693934Z","shell.execute_reply":"2024-03-17T17:42:49.771641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"proba_test_isoreg","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:42:51.845127Z","iopub.execute_input":"2024-03-17T17:42:51.845535Z","iopub.status.idle":"2024-03-17T17:42:51.852987Z","shell.execute_reply.started":"2024-03-17T17:42:51.845505Z","shell.execute_reply":"2024-03-17T17:42:51.851627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Overall, I generated some valuable, information gaining features in two ways: Creating new ones, applying feature selection by Shapley values in order to show that they are valuable(most of them stayed) - I replicated the production type solution for productizing model , then I optimized hyperparametres with RandomSearchCV, in the end I showed how the calibration for real probability creating should be done.\n# All steps of task are done, have a good time taking this as a baseline solution framework to work with. The additional tables of t=2 can be added, another features can be generated, the trials of tuning can be increased, but the all main steps are like tike this and will be replicated.","metadata":{}}]}