{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"raw","source":"import matplotlib\nmatplotlib.__version__","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport hvplot.pandas\nfrom tqdm import tqdm\nfrom typing import Tuple\nimport seaborn as sns\nfrom category_encoders.target_encoder import TargetEncoder\nfrom sklearn.model_selection import StratifiedKFold, train_test_split, GridSearchCV\nfrom sklearn.metrics import make_scorer\nfrom sklearn.base import TransformerMixin\nfrom dataclasses import dataclass\nimport warnings\nimport gc\nfrom sklearn.metrics import auc, roc_auc_score\nwarnings.filterwarnings(\"ignore\")\nsns.set_style('darkgrid')\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-09-02T15:53:38.664716Z","iopub.execute_input":"2022-09-02T15:53:38.665128Z","iopub.status.idle":"2022-09-02T15:53:38.676985Z","shell.execute_reply.started":"2022-09-02T15:53:38.6651Z","shell.execute_reply":"2022-09-02T15:53:38.675531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Read in chunks off \ntrain_data = pd.concat(tqdm(pd.read_csv(\"../input/amex-default-prediction/train_data.csv\",engine=\"c\",chunksize=int(1e6), index_col=0)))\n# test_data = pd.concat(pd.read_csv(\"test_data.csv\",engine=\"c\",chunksize=int(1e6), index_col=0))","metadata":{"execution":{"iopub.status.busy":"2022-09-02T15:53:38.678963Z","iopub.execute_input":"2022-09-02T15:53:38.679303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = pd.read_csv(\"../input/amex-default-prediction/train_labels.csv\",engine=\"c\",index_col=0)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels.value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train = train_data.join(train_labels,how = \"left\").copy()\ntrain = pd.merge(train_data, train_labels, left_index=True, right_index=True)\ndel train_data","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.target.value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Check `customer IDs` and `Target`","metadata":{}},{"cell_type":"code","source":"check_targets_per_client = train.reset_index().groupby(\"customer_ID\").aggregate({\"target\": [np.mean,len]})[\"target\"]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check_targets_per_client.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check for same customer id if labels differ\ncheck_targets_per_client[\"mean\"].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking payment frequenc\ncheck_targets_per_client[(check_targets_per_client[\"len\"]  < 13) & \n                         (check_targets_per_client[\"mean\"] == 0)]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check_targets_per_client.value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels.value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Done precautionary step","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Reduce Train Data Size","metadata":{}},{"cell_type":"code","source":"from functions import reduce_mem_usage","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = reduce_mem_usage(train)\ntrain.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train.to_csv(\"model_data\")\n# train = pd.concat(pd.read_csv(\"model_data.csv\",engine=\"c\",chunksize=int(1e6), index_col=0))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Check Categorical Variables","metadata":{}},{"cell_type":"code","source":"# Which object dtype variables are useful and can be converted to categorical variables\nfor i in train.columns[train.dtypes == object].values:\n    print(i, train[i].unique())","metadata":{"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.S_2 = pd.to_datetime(train.S_2, dayfirst = True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dates_avialable = train.S_2.apply(lambda x: x.strftime(\"%Y-%m\")).unique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dates_avialable","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']].nunique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[['D_68']].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Checking most recent data points per Client","metadata":{}},{"cell_type":"code","source":"# Last payment date for each client\nmax_dates_per_client = train.reset_index().groupby(\"customer_ID\").aggregate({\"S_2\": [max]}).copy()\nmax_dates_per_client.columns = max_dates_per_client.columns.droplevel(0)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_dates_per_client.value_counts().hvplot(kind = \"bar\", ylabel = \"Count\", title = \"Customers Payment Last Date Count\", rot = 45)#.opts(xformatter='%y')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get back only last datepayment date per","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"index_needed = pd.MultiIndex.from_arrays([max_dates_per_client.index.values, max_dates_per_client.values.ravel()])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"last_payment = train.reset_index().set_index([\"customer_ID\", \"S_2\"]).loc[index_needed, :].copy()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"last_payment = last_payment.reset_index(1).rename(columns = {\"level_1\":\"S_2\"})\nlast_payment.index.name = \"customer_ID\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"last_payment.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* By now. I have two data sets that I will experiment on building the model:\n    + `train`: the original dataset\n    - `last_payment`: only last payment per customer ","metadata":{}},{"cell_type":"markdown","source":"### Descriptive Stats","metadata":{}},{"cell_type":"code","source":"# Numerical Variables\ndata_count = train.describe().transpose()\ndata_count","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Choose Columns to Drop","metadata":{}},{"cell_type":"code","source":"drop_list = []\n# Get back numeric variables with zero count which are numerics that are uncountable\nfor i in data_count.index:\n    if data_count.loc[i, 'count'] == 0:\n        drop_list.append(i)\n# Store in our dropping list variables with more than 90% null values-Drop the whole variable\nfor i in train.columns.values:\n    if (train[i].isnull().sum() /len(train)) > 0.95:\n        drop_list.append(i)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"drop_list","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Delinquency_variables = train[[i for i in train.columns if i.startswith(\"D\")] +[\"target\"]].copy()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ratios_D = []\nnull_amounts = []\nfor i in tqdm(Delinquency_variables.columns):\n    temp = Delinquency_variables[~Delinquency_variables[i].isna()][\"target\"].value_counts()\n    null = Delinquency_variables[i].isnull().sum() /len(Delinquency_variables)\n    null_amounts.append(null)\n    zero, one = temp[0], temp[1]\n    zero_one = zero / one\n    ratios_D.append(zero_one)\ndel temp","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Delinquency_variables.target.value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"4153582/1377869","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Analyze from the Delinquency columns which one's have high imbalance between target variables' output before filling the null values","metadata":{}},{"cell_type":"code","source":"pd.DataFrame(zip(ratios_D, null_amounts), index=Delinquency_variables.columns, columns=[\"Ratio(0/1)\",\"Null%\"]).hvplot(kind = \"line\", width = 1000, rot = 50)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- We find that when the ration of 0 / 1 is high in a delinquency column (more non defaults to defaults) the % Null is lesser","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"drop_list.append(\"D_76\") #decided to drop this as had very high 0 to 1 ratio compared to the others","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Check rowwise NANs per customer","metadata":{"tags":[]}},{"cell_type":"code","source":"plt.figure(figsize= (20,8))\nsns.heatmap(train.isnull(),yticklabels=False,cbar=False,cmap='viridis');\nplt.savefig(\"Null_Values.png\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nulll_rows_per_client = pd.DataFrame(np.array([(train.isnull().sum(axis = 1) / len(train.columns)).values, train.target.values]).T, \n                                     index = [train.index, train.S_2]\n                                     ,columns= [\"Null(%) per Client\", \"target\"])\nnulll_rows_per_client.describe().transpose().round(2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nulll_rows_per_client[\"Null(%) per Client\"].plot(kind = \"box\", figsize = (20,8));","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- The number of clients with at least half of there rows is empty is","metadata":{}},{"cell_type":"code","source":"test = []\nfor i in np.arange(0.2,0.55, 0.05):\n    i = round(i,3)\n    length = len(nulll_rows_per_client[~(nulll_rows_per_client[\"Null(%) per Client\"] < i)])\n    target = nulll_rows_per_client[~(nulll_rows_per_client[\"Null(%) per Client\"] > i)][\"target\"].value_counts()\n    ratio = target[0] / target[1]\n    test.append([i,length,ratio])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(nulll_rows_per_client[~(nulll_rows_per_client[\"Null(%) per Client\"] <0.45)])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.DataFrame(test, columns=[\"Null(%) per Client Threshold\", \"Number Rows Drop\", \"Target Ratio(0/1)\"])\ntest","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[[\"Target Ratio(0/1)\"]].hvplot()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Decided to drop any row with more then 0.45 (45%) empty fields in it.","metadata":{}},{"cell_type":"code","source":"keep_rows = nulll_rows_per_client[(nulll_rows_per_client[\"Null(%) per Client\"] <0.45)].index","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.reset_index().set_index([\"customer_ID\", \"S_2\"]).loc[keep_rows, :].reset_index(1).rename(columns = {\"level_1\":\"S_2\"}).copy()\nlast_payment = last_payment.reset_index().set_index([\"customer_ID\", \"S_2\"]).loc[keep_rows, :].reset_index(1).rename(columns = {\"level_1\":\"S_2\"}).copy()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"last_payment.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Final Droppings","metadata":{}},{"cell_type":"code","source":"train = train.drop(drop_list, axis = 1)\nlast_payment = last_payment.drop(drop_list, axis = 1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train.shape)\ndisplay(last_payment.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Only for the last payment data set, get the number of payments per client","metadata":{}},{"cell_type":"code","source":"# last_payment[\"Payments\"] = check_targets_per_client[\"len\"].copy()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature Engineering","metadata":{}},{"cell_type":"code","source":"def date_engineering(df):\n    try:\n        df[\"month_end\"] = df.S_2.dt.is_month_end\n        df[\"month\"] = df.S_2.dt.month\n        #Same as month end thus turned it off\n        # df[\"q_end\"] = df.S_2.dt.is_quarter_end\n        df[\"quarter\"] = df.S_2.dt.quarter\n        df[\"weekofyear\"] = df.S_2.dt.weekofyear\n        del df[\"S_2\"]\n    except:\n        pass","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# date_engineering(train)\ndate_engineering(last_payment)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Check Null Values","metadata":{"tags":[]}},{"cell_type":"code","source":"df = pd.DataFrame({'Count': train.isnull().sum(), 'Percent(%)': 100*train.isnull().sum()/len(train)})\ndf[df['Count'] > 0].sort_values(by='Percent(%)', ascending=False)[\"Percent(%)\"].hvplot(kind = \"bar\", rot = 45, width = 1000)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame({'Count': last_payment.isnull().sum(), 'Percent(%)': 100*last_payment.isnull().sum()/len(last_payment)})\ndf[df['Count'] > 0].sort_values(by='Percent(%)', ascending=False)[\"Percent(%)\"].hvplot(kind = \"bar\", rot = 45, width = 1000)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Most D_64 that are null have a great amount of D_63 category CO. Thus will fill them with whatever CO actually has respective in the D_64 that are not null.","metadata":{}},{"cell_type":"code","source":"# train[train[\"D_64\"]==\"-1\"][\"target\"].value_counts()\ntrain[train.select_dtypes(include=\"object\").isna().any(axis = 1)].select_dtypes(include=\"object\")[\"D_63\"].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[~(train.select_dtypes(include=\"object\").isna().any(axis = 1))\n     & (train[\"D_63\"] == \"CO\")].select_dtypes(include=\"object\").value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[~(train.select_dtypes(include=\"object\").isna().any(axis = 1))\n     & (train[\"D_63\"] == \"CO\")\n     & (train[\"D_64\"] == \"O\")\n     | (train[\"D_64\"] == \"U\")][[\"D_64\",\"target\"]].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"D_64\"].mode()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Its justified to fill the only categorical columns' NaNs with the mode","metadata":{}},{"cell_type":"code","source":"train[\"D_64\"] = train[\"D_64\"].fillna(train[\"D_64\"].mode())\nlast_payment[\"D_64\"] = last_payment[\"D_64\"].fillna(last_payment[\"D_64\"].mode())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Flag numerical NaNs with large negative values","metadata":{}},{"cell_type":"code","source":"@dataclass(init=False)\nclass NullImputa(TransformerMixin):\n    def __init__(self, min_count_na = 5):\n        self.min_count_na = min_count_na\n    def fit(self,X ,y = None):\n        return self\n    def transform(self,X,y = None):\n        for col in X.columns.tolist():\n            if col in X.columns[X.dtypes == object].tolist():\n                X[col] = X[col].fillna(X[col].mode()[0])\n            else:\n                if X[col].isnull().sum() <= self.min_count_na:\n                   if X[col].isnull().sum() > 0:\n                       X.loc[list(X[X[col].isna()].index),f'{col}-mi'] = 1\n                       X[f'{col}-mi'] = X[f'{col}-mi'].fillna(0)\n                       X[col] = X[col].fillna(X[col].median())\n                       # print(col)\n                else:\n                    X[col] = X[col].fillna(-9999)\n        # test.columns[test.isna().sum() > 0]\n        assert X.isna().sum().sum() == 0\n        _ = gc.collect()\n        print(\"Imputation complete.....\")\n        \ndef stratify_data(df, column=\"month\"):\n    '''\n    https://stackoverflow.com/questions/50195187/how-to-do-a-random-stratified-sampling-with-python-not-a-train-test-split\n    '''\n    months_count = df.groupby(column)[column].count()\n    stratified_X_train = list(map(lambda x : df[\n        (\n            df['month'] == months_count.index[x]\n        ) \n    ].iloc[:int(0.95*months_count.iloc[x])], range(len(months_count))))\n    stratified_X_train =  pd.concat(stratified_X_train)\n\n    stratified_X_val = list(map(lambda x : df[\n        (\n            df['month'] == months_count.index[x]\n        ) \n    ].iloc[int(0.95*months_count.iloc[x]):], range(len(months_count))))\n\n    stratified_X_val = pd.concat(stratified_X_val)\n    return stratified_X_train, stratified_X_val","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# imputer = NullImputa(min_count_na=5)\n# _ = imputer.fit_transform(last_payment)\nX,y = last_payment.drop(columns=\"target\"), last_payment[\"target\"]\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2,\n                                                  random_state = 42, stratify = y)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# imputer = NullImputa(min_count_na=5)\n# _ = imputer.fit_transform(last_payment)\n# X_train, X_val = stratify_data(last_payment)\n# X_train, y_train, X_val, y_val = X_train.drop(columns=\"target\") , X_train[\"target\"], X_val.drop(columns=\"target\") , X_val[\"target\"]\n# del train","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape, X_val.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extra_mi_feature_checekr(train, other):\n    '''\n    Imputation leads to addiion of new feature on DataRobot\n    The train colun might have imputation based on median  but the test dosent \n    have and null in that column\n    '''\n    different_columns_found = train.columns.difference(other.columns)\n    for i in different_columns_found:\n        other[i] = 0.0\n    assert train.shape[1] ==  other.shape[1]\n    return 1","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"extra_mi_feature_checekr(X_train,X_val)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train = train.fillna(-9999,axis = 1)\n# last_payment = last_payment.fillna(-9999,axis = 1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train.isna().sum().sum(), last_payment.isna().sum().sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Encoding Categorical Variables","metadata":{}},{"cell_type":"code","source":"def save_model(model_name, model):\n    import pickle\n    '''\n    model_name = name.pkl\n    joblib.load('name.pkl')\n    assign a variable to load model\n    '''\n    with open(str(model_name), 'wb') as f:\n        pickle.dump(model, f)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# encoder = TargetEncoder()\n# X_train[\"D_63_freq\"] = X_train.groupby(\"D_63\")[\"D_63\"].transform(\"count\") / len(train)\n# X_train[\"D_63_mean\"] = encoder1.fit_transform(train[\"D_63\"], train[\"target\"])\n# X_train[\"D_63_freq\"] = X_train.groupby(\"D_63\")[\"D_63\"].transform(\"count\") / len(train)\n# X_train[\"D_63_mean\"] = encoder1.fit_transform(train[\"D_63\"], train[\"target\"])\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in tqdm(X_train.select_dtypes(include = \"object\").columns):\n    encoder = TargetEncoder()\n    X_train[i] = encoder.fit_transform(X_train[i], y_train)\n    save_model(\"%s_encoder.pkl\"%i,encoder)\n    del encoder","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def CategoricalEncoda():\n    from pickle import load\n    if \"X_val\" in globals():\n        encoder_D_63 = load(open('D_63_encoder.pkl', 'rb'))\n        X_val[\"D_63\"] = encoder_D_63.transform(X_val[\"D_63\"])\n        encoder_D_64 = load(open('D_64_encoder.pkl', 'rb'))\n        X_val[\"D_64\"] = encoder_D_64.transform(X_val[\"D_64\"])\n    \n    if \"last_payment\" in globals():\n        encoder_D_63_mini = load(open('D_63_encoder_lastPYT.pkl', 'rb'))\n        encoder_D_64_mini = load(open('D_64_encoder_lastPYT.pkl', 'rb'))\n        last_payment[\"D_63\"] = encoder_D_63_mini.transform(last_payment[\"D_63\"])                              \n        last_payment[\"D_64\"] = encoder_D_64_mini.transform(last_payment[\"D_64\"])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CategoricalEncoda()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for i in tqdm(last_payment.select_dtypes(include = \"object\").columns):\n#     encoder = TargetEncoder()\n#     last_payment[i] = encoder.fit_transform(last_payment[i], last_payment[\"target\"])\n#     save_model(\"%s_encoder_lastPYT.pkl\"%i,encoder)\n#     del encoder","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model Data:\n##### Stratified Train & Hold Out","metadata":{}},{"cell_type":"markdown","source":"- I will set validation data aside after each full fitting round for evaluation.","metadata":{}},{"cell_type":"code","source":"last_payment[[\"month\", \"quarter\"]].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"months_count = train.groupby(\"month\")[\"month\"].count()\nmonths_count","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_train, X_val = stratify_data(train)\n# del train","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_train2, X_val2 = stratify_data(last_payment)\n# del last_payment","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model Testing","metadata":{}},{"cell_type":"code","source":"import xgboost as xgb\nfrom functions import *\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def stratify_data(column=\"month\"):\n# stratified_X_train = list(map(lambda x : train[\n#     (\n#         train['month'] == months_count.index[x]\n#     ) \n# ].sample(int(months_count.iloc[x]*0.05)), range(len(months_count))))\n# def load_data(df):\n    \n\ndef input_output(df):\n    stratified_X_train, stratified_X_val = stratify_data(df)\n    \n    stratified_y_train = stratified_X_train[\"target\"]\n    stratified_X_train = stratified_X_train.drop(columns=\"target\")\n    \n    stratified_y_val = stratified_X_val[\"target\"]\n    stratified_X_val = stratified_X_val.drop(columns=\"target\")\n    return stratified_X_train, stratified_X_val, stratified_y_train, stratified_y_val","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# last_payment = pd.concat(pd.read_csv(\"model_data_mini.csv\",engine=\"c\",chunksize=int(1e6), index_col=0))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# last_payment.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# _ = last_payment.drop(\"Payments\", axis = 1, inplace=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# last_payment = reduce_mem_usage(last_payment)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"last_payment.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_train, X_val = stratify_data(last_payment)\n# del last_payment","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!nvidia-smi","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train.value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def amex_metric_df(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n#     def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n#         df = (pd.concat([y_true, y_pred], axis='columns')\n#               .sort_values('prediction', ascending=False))\n#         df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n#         four_pct_cutoff = int(0.04 * df['weight'].sum())\n#         df['weight_cumsum'] = df['weight'].cumsum()\n#         df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n#         return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n#     def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n#         df = (pd.concat([y_true, y_pred], axis='columns')\n#               .sort_values('prediction', ascending=False))\n#         df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n#         df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n#         total_pos = (df['target'] * df['weight']).sum()\n#         df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n#         df['lorentz'] = df['cum_pos_found'] / total_pos\n#         df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n#         return df['gini'].sum()\n    \n#     def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n#         y_true = pd.DataFrame(y_true, columns = [\"target\"])\n#         y_true_pred = y_true.rename(columns={'target': 'prediction'})\n#         return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n#     g = normalized_weighted_gini(y_true, y_pred)\n#     d = top_four_percent_captured(y_true, y_pred)\n\n#     return \"Amex\", 0.5 * (g + d)\nimport numpy as np\n\ndef amex_metric(y_pred, y_true)-> Tuple[str, float]:\n    # DMatrix prep\n    y_true = y_true.get_label()\n    \n    # count of positives and negatives\n    n_pos = y_true.sum()\n    n_neg = y_true.shape[0] - n_pos\n\n    # sorting by descring prediction values\n    indices = np.argsort(y_pred)[::-1]\n    preds, target = y_pred[indices], y_true[indices]\n\n    # filter the top 4% by cumulative row weights\n    weight = 20.0 - target * 19.0\n    cum_norm_weight = (weight / weight.sum()).cumsum()\n    four_pct_filter = cum_norm_weight <= 0.04\n\n    # default rate captured at 4%\n    d = target[four_pct_filter].sum() / n_pos\n\n    # weighted gini coefficient\n    lorentz = (target / n_pos).cumsum()\n    gini = ((lorentz - cum_norm_weight) * weight).sum()\n\n    # max weighted gini coefficient\n    gini_max = 10 * n_neg * (1 - 19 / (n_pos + 20 * n_neg))\n\n    # normalized weighted gini coefficient\n    g = gini / gini_max\n\n    return \"Amex\", 0.5 * (g + d)\n\ndef amex_metric_mod(y_true, y_pred):\n\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n\n    return 0.5 * (gini[1]/gini[0] + top_four)\n\ndef misclassified(pred_probs, dtrain):\n    labels = dtrain.get_label() # obtain true labels\n    preds = pred_probs > 0.5 # obtain predicted values\n    return 'misclassified', np.sum(labels != preds)","metadata":{"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def stratify_data(df, column=\"month\"):\n    '''\n    https://stackoverflow.com/questions/50195187/how-to-do-a-random-stratified-sampling-with-python-not-a-train-test-split\n    '''\n    months_count = df.groupby(column)[column].count()\n    stratified_X_train = list(map(lambda x : df[\n        (\n            df['month'] == months_count.index[x]\n        ) \n    ].iloc[:int(0.95*months_count.iloc[x])], range(len(months_count))))\n    stratified_X_train =  pd.concat(stratified_X_train)\n\n    stratified_X_val = list(map(lambda x : df[\n        (\n            df['month'] == months_count.index[x]\n        ) \n    ].iloc[int(0.95*months_count.iloc[x]):], range(len(months_count))))\n\n    stratified_X_val = pd.concat(stratified_X_val)\n    return stratified_X_train, stratified_X_val\ndef init_model_params(y_train):\n    '''\n    “binary:logistic” –logistic regression for binary classification, output probability\n\n    “binary:logitraw” –logistic regression for binary classification, output score before logistic transformation\n    '''\n    #define class weight dictionary, negative class has 20x weight\n    w = {0:20, 1:1}\n    global params\n    global plst\n    # plst = \n    params = {\"booster\" :\"gbtree\",\n              \"n_estimators\": 2500,\n              \"early_stopping_rounds\": 10,\n              \"max_depth\" : 5,\n              \"n_jobs\": -1,\n              \"verbosity\" : 3,\n             \"objective\": \"binary:logistic\",\n             \"eta\": 0.05,\n              \"colsample_bytree\" : 0.3, \n             \"tree_method\": \"gpu_hist\",\n              'predictor':'gpu_predictor',\n             \"scale_pos_weight\": int((y_train.value_counts()[0] / y_train.value_counts()[1]) * 20),\n             \"eval_metric\": [\"auc\", \"logloss\", \"error\"],\n             \"subsample\" : 0.8, \"colsample_bylevel\" : 1, \"random_state\" : 42, 'monotone_constraints': {\"P_2\":1}}","metadata":{"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#init params\ninit_model_params(y_train)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xg_train = xgb.DMatrix(X_train,y_train                          \n                       )\nxg_valid = xgb.DMatrix(X_val,y_val                         \n                       )   ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Cross Validation ","metadata":{}},{"cell_type":"code","source":"\nCV = 5\nkfold = StratifiedKFold(n_splits = CV, \n                        shuffle = True,\n                        random_state = 42)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bst_cv = xgb.cv(params, xg_train, maximize=True, seed = 42,verbose_eval=True,\n                num_boost_round = params[\"n_estimators\"], early_stopping_rounds = 10,\n                nfold = CV, folds = kfold, custom_metric = amex_metric)","metadata":{"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bst_cv[[\"train-Amex-std\",\"test-Amex-std\"]].hvplot()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bst_cv[[\"train-Amex-mean\", \"test-Amex-mean\"]].hvplot()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bst_cv.shape[0]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### GridSearch","metadata":{}},{"cell_type":"code","source":"amex_scorer = make_scorer(amex_metric_mod, needs_proba=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = xgb.XGBClassifier(booster = params[\"booster\"], objective = params[\"objective\"],\n                          tree_method = params[\"tree_method\"], n_jobs =-1, predictor = params[\"predictor\"], random_state = 42,\n                         verbose = 2)\n# https://mlfromscratch.com/gridsearch-keras-sklearn/#/\nparam_grid = {\n    'n_estimators': [bst_cv.shape[0]],\n    'colsample_bytree': [0.7, 0.8],\n    'max_depth': [3,7,15,20],\n    'reg_alpha': [1.1, 1.2, 1.3],\n    'reg_lambda': [1.1, 1.2, 1.3],\n    'subsample': [0.7, 0.8, 0.9],\n    \"learning_rate\": [0.001, 0.01, 0.1],\n    \"scale_pos_weight\" : [57,3]\n}\ndef algorithm_pipeline(X_train_data, X_test_data, y_train_data, y_test_data, \n                       model, param_grid, cv=10, scoring_fit=amex_scorer,kfold_ = kfold,\n                       do_probabilities = False):\n    gs = GridSearchCV(\n        estimator=model,\n        param_grid=param_grid, \n        n_jobs=-1, \n        scoring=scoring_fit,\n        verbose=2,\n        cv = kfold_\n    )\n    fitted_model = gs.fit(X_train_data, y_train_data)\n    \n    if do_probabilities:\n      pred = fitted_model.predict_proba(X_test_data)\n    else:\n      pred = fitted_model.predict(X_test_data)\n    \n    return fitted_model, pred\nmodel, pred = algorithm_pipeline(X_train, X_val, y_train, y_val, model, \n                                 param_grid, kfold_ = kfold)\n\n\nprint(model.best_score_)\nprint(model.best_params_)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evals_result = {}\nbaseline_xgb = xgb.train(params, xg_train, evals=[(xg_train, \"train\"), (xg_valid, \"eval\")], num_boost_round= bst_cv.shape[0],\n                verbose_eval=True, feval = amex_metric, evals_result = evals_result)","metadata":{"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_pred = baseline_xgb.predict(X_val.values)\ny_pred = baseline_xgb.predict(xgb.DMatrix(X_val))\ny_pred","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dd = baseline_xgb.get_score(importance_type='weight')\ndf = pd.DataFrame({'feature':dd.keys(),f'importance':dd.values()})\ndf","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(y_pred, bins=100)\nplt.title('OOF Predictions')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.sort_values(by = df.columns[1], ascending=False).set_index(\"feature\").hvplot.barh(width = 2000, height = 800, y_rot = 45)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# amex_metric_df(pd.DataFrame(y_pred, columns=[\"prediction\"]), y_val) ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def split_dataframe(df, chunk_size = int(1e6)): \n    # input - df: a Dataframe, chunkSize: the chunk size\n    # output - a list of DataFrame\n    # purpose - splits the DataFrame into smaller chunks\n    chunks = list()\n    num_chunks = len(df) // chunk_size + 1\n    for i in range(num_chunks):\n        chunks.append(df[i*chunk_size:(i+1)*chunk_size])\n    assert sum([len(i) for i in chunks]) == df.shape[0]\n    _ = gc.collect()\n    return chunks\ndef multiple_predictions(X_train, y_train, metrics):\n    chunks_preds_score = list()\n    dfs = split_dataframe(X_train)\n    ys = split_dataframe(y_train)\n    for i in tqdm(range(len(dfs))):\n        y_pred_train = baseline_xgb.predict(xgb.DMatrix(dfs[i]))\n        chunks_preds_score.append(metrics(ys[i], y_pred_train))\n    del dfs\n    del ys\n    _ = gc.collect()\n    return chunks_preds_score\ndef expected_calibration_error(y, proba, bins = 'fd'):\n  import numpy as np\n  bin_count, bin_edges = np.histogram(proba, bins = bins)\n  n_bins = len(bin_count)\n  bin_edges[0] -= 1e-8 # because left edge is not included\n  bin_id = np.digitize(proba, bin_edges, right = True) - 1\n  bin_ysum = np.bincount(bin_id, weights = y, minlength = n_bins)\n  bin_probasum = np.bincount(bin_id, weights = proba, minlength = n_bins)\n  bin_ymean = np.divide(bin_ysum, bin_count, out = np.zeros(n_bins), where = bin_count > 0)\n  bin_probamean = np.divide(bin_probasum, bin_count, out = np.zeros(n_bins), where = bin_count > 0)\n  ece = np.abs((bin_probamean - bin_ymean) * bin_count).sum() / len(proba)\n  return ece","metadata":{"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# baseline_xgb = xgb.Booster()\n# baseline_xgb.load_model(fname=\"model.xgb\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"multiple_predictions(X_train, y_train, amex_metric_mod)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_val = baseline_xgb.predict(xgb.DMatrix(X_val))\ny_pred_val\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amex_metric_mod(y_val, y_pred_val)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"expected_calibration_error(y_val, y_pred_val)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# baseline_xgb = xgb.XGBClassifier(**params)\n# baseline_xgb.fit(X_train.values,y_train.values, eval_set=[(X_val.values,y_val.values)])","metadata":{"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_pred_val = baseline_xgb.predict_proba(X_val)[:,1]\n# y_pred_val","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amex_metric_mod(y_val, y_pred_val)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"baseline_xgb.save_model(\"model.xgb\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Probability Calibration ","metadata":{}},{"cell_type":"code","source":"y_pred_train = baseline_xgb.predict(xgb.DMatrix(X_train))\ny_pred_train","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.isotonic import IsotonicRegression\niso_reg = IsotonicRegression(y_min = 0, y_max = 1, out_of_bounds = 'clip').fit(y_pred_val, y_val)\nproba_test_forest_isoreg = iso_reg.predict(y_pred_val)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amex_metric_mod(y_val, proba_test_forest_isoreg)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"proba_test_forest_isoreg = iso_reg.predict(y_pred_val)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amex_metric_mod(y_val, proba_test_forest_isoreg)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#save model\ndef save_model(model_name, model):\n    '''\n    model_name = name.pkl\n    joblib.load('name.pkl')\n    assign a variable to load model\n    '''\n    with open(str(model_name), 'wb') as f:\n        pickle.dump(model, f)\n    return 1","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save_model('isotonic.pkl', iso_reg)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}