{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from IPython.core.display import HTML\nHTML(\"\"\"\n<style>\nfigcaption {\n  color: #7db2e0;\n  font-style: italic;\n  \n  font-size: 16px;\n  padding: 0px;\n  text-align: center;\n}\n</style>\n<img width=\"100%\" src=\"https://images.creativemarket.com/0.1.0/ps/2930173/910/607/m1/fpnw/wm0/credit-card-design-by-jan-baca-.jpg?1499199821&s=9dbd168ee67bf12b45e533800f1f9aed&fmt=webp?1499199821&s=772da0dd5276e72af0dc92f36c2f8149?auto=compress&cs=tinysrgb&w=200&h=200&dpr=1\">\n<figcaption>Pop–Art Diamond Bank Card Design</figcaption>\n\"\"\")","metadata":{"execution":{"iopub.status.busy":"2022-10-04T08:55:22.597087Z","iopub.execute_input":"2022-10-04T08:55:22.597656Z","iopub.status.idle":"2022-10-04T08:55:22.630591Z","shell.execute_reply.started":"2022-10-04T08:55:22.597556Z","shell.execute_reply":"2022-10-04T08:55:22.629399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Table of Contents:**","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> **Table of Contents:**\n> * [🚫 Introduction on Payment Default](#1)\n> * [🙏 Credits](#2)\n> * [🛠 Feature Engineering](#3)\n> * [🥢 Feature Selection](#4)\n> * [🤖 Model Training](#5)\n> * [🏅 Feature Importance](#6)\n> * [📈 Cross-Val Roc Curves](#7)\n> * [🤞 Predict Test data & Submission](#8)\n> ---","metadata":{}},{"cell_type":"markdown","source":"<a id=\"1\"></a> \n# <h1 style='display:fill;color:#2d3a41;background-color:#a3d3eb;padding:20px'>   🚫  <b> Introduction on Payment Default </b> </h1>\nEven though credit cards presents many advantages such us avoiding carrying a bulky wallet in your pocket, tracking the spending behaviour, fraud detection... On the other hand the major downside for credit card usage is the increasing tendency to default on their payments. Aggressive marketing strategies can encourage credit card use beyond payment capacity, thus increasing the bearer’s credit risk and resulting in defaults and losses that might have not been properly anticipated\n\n<a id=\"2\"></a> \n# <h1 style='display:fill;color:#2d3a41;background-color:#a3d3eb;padding:20px'>   🙏 <b> Credits </b> </h1>\n In this notebook we build and train a LightGBM model using [@raddar'dataset](https://www.kaggle.com/datasets/raddar/amex-data-integer-dtypes-parquet-format) (discussion [here](https://www.kaggle.com/competitions/amex-default-prediction/discussion/328514) and my notebook explanation [here](https://www.kaggle.com/code/schopenhacker75/data-deanonymization) )\n\nThe feature engineering part  is very inspired from [this notbook](https://www.kaggle.com/code/huseyincot/amex-agg-data-how-it-created) and the design aspect s from [my previous notebook](https://www.kaggle.com/code/schopenhacker75/fancy-complete-eda/notebook) whose it self ispired from [this notebook](https://www.kaggle.com/code/kellibelcher/amex-default-prediction-eda-lgbm-baseline/notebook). The LightGBM wrapper implementation is inspired from [jayjay's notebook](https://www.kaggle.com/code/jayjay75/wids2020-lgb-starter-adversarial-validation)\n","metadata":{}},{"cell_type":"code","source":"## ESSENTIALS\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport gc\nimport sys\nimport pickle\nimport glob\nimport pandas as pd\nfrom sklearn.preprocessing import LabelEncoder\n\npd.set_option('display.max_columns', None)\nimport random\nrandom.seed(75)\nfrom tqdm.notebook import tqdm_notebook\n\nfrom functools import partial, reduce\nimport datetime\nimport time\n\n\n### warnings setting\nimport sys\nimport warnings\nif not sys.warnoptions:\n    warnings.simplefilter(\"ignore\")\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)\n\n#### model\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import roc_auc_score, roc_curve, auc\nimport catboost\nfrom catboost import Pool, CatBoostClassifier \nimport lightgbm as lgb\nimport joblib\nimport pickle\nfrom tqdm.notebook import tqdm_notebook \nimport uuid\n\n##### LOGGING Stettings #####\nimport logging\n# Create logger\nlogger = logging.getLogger()\nlogger.setLevel(logging.INFO)\n# Create STDERR handler\nhandler = logging.StreamHandler(sys.stderr)\n# Create formatter and add it to the handler\nformatter = logging.Formatter('%(asctime)s [%(levelname)s] %(name)s - %(message)s', datefmt='%Y-%m-%d %H:%M:%S',)\nhandler.setFormatter(formatter)\n# Set STDERR handler as the only handler \nlogger.handlers = [handler]\n\n#### plots\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport matplotlib.colors\n\nsns.set(rc={'axes.facecolor':'#f9ecec', 'figure.facecolor':'#f9ecec'})\n\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\nfrom plotly.offline import init_notebook_mode\n\n### Plotly settings\ntheme_palette={\n    'base': '#a3d3eb',\n    'complementary':'#ebbba3',\n    'triadic' : '#eba3d3', \n    'backgound' : '#f6fbfd' \n}\n\ntemp=dict(layout=go.Layout(font=dict(family=\"Ubuntu\", size=14), \n                           height=600, \n                         legend=dict(#traceorder='reversed',\n                            orientation=\"v\",\n                            y=1.15,\n                            x=0.9),\n                    plot_bgcolor = theme_palette['backgound'],\n                      paper_bgcolor = theme_palette['backgound']))\n\n\nSAVED=True\nINTERIM_DATA = \"../input/interim-data-v2\"\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-30T14:56:50.081903Z","iopub.execute_input":"2022-09-30T14:56:50.082431Z","iopub.status.idle":"2022-09-30T14:56:52.577059Z","shell.execute_reply.started":"2022-09-30T14:56:50.082331Z","shell.execute_reply":"2022-09-30T14:56:52.575920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"3\"></a> \n# <h1 style='display:fill;color:#2d3a41;background-color:#a3d3eb;padding:20px'>   🛠 <b>  Feature Engineering </b> </h1>\n\nTo Save the exectution time all the Data Processing part has been caaried out upstream, and stored [here](https://www.kaggle.com/datasets/schopenhacker75/interim-data-v2): we will get directly the prepared [train](https://www.kaggle.com/datasets/schopenhacker75/interim-data-v2?select=train_prepared.ftr) and [test](https://www.kaggle.com/datasets/schopenhacker75/interim-data-v2?select=test_prepared.ftr) datasets ","metadata":{}},{"cell_type":"markdown","source":"<h2 style='color:#2d3a41;background-color:#a3d3eb;padding:10px'>  👉    <b>-- Get Train Data --</b> </h2>","metadata":{}},{"cell_type":"code","source":"if not(SAVED):\n    train = pd.read_parquet('/kaggle/input/amex-data-integer-dtypes-parquet-format/train.parquet')\n    print(f\"Shape = {train.shape}, number of customers = {train['customer_ID'].nunique()}\")\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-30T04:10:45.595986Z","iopub.execute_input":"2022-09-30T04:10:45.597924Z","iopub.status.idle":"2022-09-30T04:10:45.604959Z","shell.execute_reply.started":"2022-09-30T04:10:45.597853Z","shell.execute_reply":"2022-09-30T04:10:45.603341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2 style='color:#2d3a41;background-color:#a3d3eb;padding:10px'>  👉    <b>-- Train Data Feature Engineering  --</b> </h2>","metadata":{}},{"cell_type":"markdown","source":"I chose to refer to **[Martin's](https://www.kaggle.com/code/ragnar123/amex-lgbm-dart-cv-0-7977/notebook) excellent work**. Because it is simple to implement and efficient\n\nThe idea is to aggreagate data by `customer_ID` and apply some statistical functions\n\n<h3 style='color:#2d3a41;background-color:#c8e5f3;padding:10px'>  🫐    <b> a - Numerical features aggregation </b> </h3>\n\nFor each customer ID numerical data we retain:\n* `mean`: the mean of numerical features\n* `std` : standard deviation of each numerical features\n* `min` and  `max`: the minimal and maximal values of each numerical feature\n* `last` : the last values\n\n\n<h3 style='color:#2d3a41;background-color:#c8e5f3;padding:10px'>  🫐    <b> b - Categorical features aggregation </b> </h3>\n\nFor each customer ID categorical data we retain:\n* `count` the count of transactions\n* `last` : the last  values\n* `number` :number of unique values \n\n<h3 style='color:#2d3a41;background-color:#c8e5f3;padding:10px'>  🫐    <b> c - Difference based features</b> </h3>\n\nIt [has been shown](http://www.kaggle.com/code/ragnar123/amex-lgbm-dart-cv-0-7977/notebook) that these features carry a powerful predictive signal:\n* the difference between the last transaction and the second of last\n* the difference between the last transaction and the mean","metadata":{}},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-30T04:10:45.607054Z","iopub.execute_input":"2022-09-30T04:10:45.607654Z","iopub.status.idle":"2022-09-30T04:10:45.838775Z","shell.execute_reply.started":"2022-09-30T04:10:45.607598Z","shell.execute_reply":"2022-09-30T04:10:45.837287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#https://www.kaggle.com/code/ragnar123/amex-lgbm-dart-cv-0-7977\nif not SAVED:\n    features = train.drop(['customer_ID', 'S_2'], axis = 1).columns.to_list()\n\n    cat_features = [\n        \"B_30\",\n        \"B_38\",\n        \"D_114\",\n        \"D_116\",\n        \"D_117\",\n        \"D_120\",\n        \"D_126\",\n        \"D_63\",\n        \"D_64\",\n        \"D_66\",\n        \"D_68\",\n    ]\n    num_features = [col for col in features if col not in cat_features]\n\n    with open('features.pkl', 'wb') as f:\n        pickle.dump(features, f)\n\n    with open('cat_cols.pkl', 'wb') as f:\n        pickle.dump(cat_features, f)\n\n    with open('num_cols.pkl', 'wb') as f:\n        pickle.dump(num_features, f)\nelse:\n    with open(os.path.join(INTERIM_DATA, 'features.pkl'), 'rb') as f:\n        features = pickle.load(f)\n\n    with open(os.path.join(INTERIM_DATA, 'cat_cols.pkl'), 'rb') as f:\n        cat_features = pickle.load(f)\n\n    with open(os.path.join(INTERIM_DATA, 'num_cols.pkl'), 'rb') as f:\n        num_features = pickle.load(f)\n    \n#","metadata":{"execution":{"iopub.status.busy":"2022-09-30T04:10:45.841591Z","iopub.execute_input":"2022-09-30T04:10:45.842064Z","iopub.status.idle":"2022-09-30T04:10:45.874822Z","shell.execute_reply.started":"2022-09-30T04:10:45.842012Z","shell.execute_reply":"2022-09-30T04:10:45.873102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To address the OOM issues, I inplemented **a batch generator** to apply the data processing batch by batch then cancatenate it all. Please note that most of the used aggregation functions are grouped by customer_ID, **so all the statements of a given client must be grouped on a single batch.**","metadata":{}},{"cell_type":"code","source":"\nclass BatchGenerator:\n    def __init__(self, df, batch_feature='customer_ID', keep_features=features, n_batchs=750):\n        self.df = df\n        self.batch_feature = batch_feature\n        self.keep_feature = list(set([batch_feature]+keep_features))\n        self.n_batchs = n_batchs\n        \n    def __iter__(self):\n        unique_vals = self.df[self.batch_feature].unique()\n        batch_size = int(np.ceil(len(unique_vals) / self.n_batchs))\n        groups = self.df.groupby(self.batch_feature).groups\n        n_batchs= min(self.n_batchs, int(np.ceil(len(unique_vals) / batch_size)))\n        for i in range(n_batchs):\n            keys = unique_vals[i*batch_size:(i+1)*batch_size]\n            idx=[i for s in keys for i in groups[s] ]\n            if i == n_batchs-1:\n                keys = unique_vals[(i+1)*batch_size:]\n                idx = idx + [i for s in keys for i in groups[s] ]\n            yield self.df.loc[idx, self.keep_feature]\n        \n        \n# from https://www.kaggle.com/code/ragnar123/amex-lgbm-dart-cv-0-7977\ndef get_difference(data, num_features):\n    df1 = []\n    customer_ids = []\n    for customer_id, df in tqdm_notebook(data.groupby(['customer_ID'])):\n        # Get the differences\n        diff_df1 = df[num_features].diff(1).iloc[[-1]].values.astype(np.float32)\n        # Append to lists\n        df1.append(diff_df1)\n        customer_ids.append(customer_id)\n    # Concatenate\n    df1 = np.concatenate(df1, axis = 0)\n    # Transform to dataframe\n    df1 = pd.DataFrame(df1, columns = [col + '_diff1' for col in df[num_features].columns])\n    # Add customer id\n    df1['customer_ID'] = customer_ids\n    return df1\n\ndef feature_engineering(df, data_type='train'):\n    # num features\n    print(\"num features aggregation\")\n    df_num_agg = df.groupby(\"customer_ID\")[num_features].agg(['mean', 'std', 'min', 'max', 'last'])\n    df_num_agg.columns = ['_'.join(x) for x in df_num_agg.columns]\n    df_num_agg = df_num_agg.reset_index()\n    # cat features\n    print(\"cat features aggregation\")\n    df_cat_agg = df.groupby(\"customer_ID\")[cat_features].agg(['count', 'last', 'nunique'])\n    df_cat_agg.columns = ['_'.join(x) for x in df_cat_agg.columns]\n    df_cat_agg = df_cat_agg.reset_index()\n    \n    # Transform float64 columns to float32\n    print(\"reduce float data size\")\n    cols = list(df_num_agg.dtypes[df_num_agg.dtypes == 'float64'].index)\n    for col in tqdm_notebook(cols):\n        df_num_agg[col] = df_num_agg[col].astype(np.float32)\n    # Transform int64 columns to int32\n    print(\"reduce cat data size\")\n    cols = list(df_cat_agg.dtypes[df_cat_agg.dtypes == 'int64'].index)\n    for col in tqdm_notebook(cols):\n        df_cat_agg[col] = df_cat_agg[col].astype(np.int32)\n    # Get the difference\n    print(\"get diff between last and second of last transaction\")\n    df_diff = get_difference(df, num_features)\n    # merge all\n    print(\"merge all\")\n    df = df_num_agg.merge(df_cat_agg, how = 'inner', on = 'customer_ID').merge(df_diff, how = 'inner', on = 'customer_ID')\n    if data_type=='train':\n        print(\"get labels\")\n        train_labels = pd.read_csv('../input/amex-default-prediction/train_labels.csv')\n        df = df.merge(train_labels, how = 'inner', on = 'customer_ID')\n        del train_labels\n    del df_num_agg, df_cat_agg, df_diff\n    \n    # Round last float features to 2 decimal place\n    num_cols = list(df.dtypes[(df.dtypes == 'float32') | (df.dtypes == 'float64')].index)\n    num_cols = [col for col in num_cols if 'last' in col]\n    print(\"Round last float features to 2 decimal place\")\n    for col in num_cols:\n        df[col + '_round2'] = df[col].round(2)\n    # Get the difference between last and mean\n    num_cols = list(filter(lambda x: 'last' in x, df))\n    num_cols = [col[:-5] for col in num_cols if 'round' not in col]\n    print(\"diff between last and mean transaction\")\n    for col in tqdm_notebook(num_cols):\n        try:\n            df[f'{col}_last_mean_diff'] = df[f'{col}_last'] - df[f'{col}_mean']\n        except:\n            pass\n    # Transform float64 and float32 to float16\n    num_cols = list(df.dtypes[(df.dtypes == 'float32') | (df.dtypes == 'float64')].index)\n    print(\"reduce float data size\")\n    for col in tqdm_notebook(num_cols):\n        df[col] = df[col].astype(np.float16)\n    df.to_feather(f\"{data_type}_{str(uuid.uuid4())}.ftr\")\n    return df['customer_ID'].nunique()\n\n# label encoding dor categorical features\ndef label_encoding(df, train=True):\n    cat_lasts = [f\"{cf}_last\" for cf in cat_features]\n    for cat_col in tqdm_notebook(cat_lasts):\n        print(f\"{cat_col} label encoding\")\n        if train:\n            le = LabelEncoder()\n            df[cat_col] = le.fit_transform(df[cat_col])\n            joblib.dump(le, f'{cat_col}_encoder.pkl')\n        else:\n            le = joblib.load(os.path.join('../input/interim-data-v2', f'{cat_col}_encoder.pkl'))\n            df[cat_col] = le.transform(test[cat_col])\n    return df\n","metadata":{"execution":{"iopub.status.busy":"2022-09-30T04:10:46.428089Z","iopub.execute_input":"2022-09-30T04:10:46.428871Z","iopub.status.idle":"2022-09-30T04:10:46.462554Z","shell.execute_reply.started":"2022-09-30T04:10:46.428790Z","shell.execute_reply":"2022-09-30T04:10:46.461110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-30T04:10:46.990824Z","iopub.execute_input":"2022-09-30T04:10:46.991323Z","iopub.status.idle":"2022-09-30T04:10:47.214496Z","shell.execute_reply.started":"2022-09-30T04:10:46.991285Z","shell.execute_reply":"2022-09-30T04:10:47.212955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not SAVED:\n    N_CID = train['customer_ID'].nunique()\n    samples_df = BatchGenerator(train, feature='customer_ID', n_batchs=50)\n    processed_elements = sum(map(partial(feature_engineering, data_type='train'), tqdm_notebook(samples_df, total=50)))\n    assert processed_elements==N_CID\n    del train\n    gc.collect()\n    train = pd.DataFrame()\n    for path in tqdm_notebook(glob.glob('train_*.ftr')):\n        train = pd.concat([train, pd.read_feather(path)])\n        os.remove(path)\n\n\n    train = label_encoding(train, train=True)\n    train.reset_index(drop=True).to_feather(\"train_prepared.ftr\")\n\nelse: \n    train = pd.read_feather(os.path.join(INTERIM_DATA, \"train_prepared.ftr\"))","metadata":{"execution":{"iopub.status.busy":"2022-09-30T04:10:49.554787Z","iopub.execute_input":"2022-09-30T04:10:49.555266Z","iopub.status.idle":"2022-09-30T04:10:57.991133Z","shell.execute_reply.started":"2022-09-30T04:10:49.555230Z","shell.execute_reply":"2022-09-30T04:10:57.989821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-30T14:57:06.186055Z","iopub.execute_input":"2022-09-30T14:57:06.186477Z","iopub.status.idle":"2022-09-30T14:57:06.351691Z","shell.execute_reply.started":"2022-09-30T14:57:06.186439Z","shell.execute_reply":"2022-09-30T14:57:06.350163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2 style='color:#2d3a41;background-color:#a3d3eb;padding:10px'>  👉    <b>-- Test Data Feature Engineering  --</b> </h2>\n\nApply the same pipeline to the test dataset with some differences:\n* The label merging part is omitted (set `data_type` parameter of the `feature_engineering` function to 'test')\n* Apply the pre-fitted label encoders on the training dataset (set `train` parameter of the `label_encoding` function to **False**)","metadata":{}},{"cell_type":"code","source":"INFER = False\nif not SAVED:\n    if INFER:\n        test = pd.read_parquet(\"../input/amex-data-integer-dtypes-parquet-format/test.parquet\")\n        N_CID = test['customer_ID'].nunique()\n\n        samples_df = BatchGenerator(test, batch_feature='customer_ID', n_batchs=200)\n        processed_elements = sum(map(partial(feature_engineering, data_type='test'), tqdm_notebook(samples_df, total=200)))\n        assert processed_elements==N_CID\n        \n        del test\n        gc.collect()\n\n\n        test = pd.DataFrame()\n        for path in tqdm_notebook(glob.glob('test_*.ftr')):\n            test = pd.concat([test, pd.read_feather(path)])\n            os.remove(path)\n\n        test = label_encoding(test, train=False)\n        test.reset_index(drop=True).to_feather(\"test_prepared.ftr\")\n\n    else:\n        test = pd.read_feather(os.path.join(INTERIM_DATA, \"test_prepared.ftr\"))\n","metadata":{"execution":{"iopub.status.busy":"2022-09-30T04:11:02.018633Z","iopub.execute_input":"2022-09-30T04:11:02.019121Z","iopub.status.idle":"2022-09-30T04:11:02.030601Z","shell.execute_reply.started":"2022-09-30T04:11:02.019085Z","shell.execute_reply":"2022-09-30T04:11:02.028999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"4\"></a> \n# <h1 style='display:fill;color:#2d3a41;background-color:#a3d3eb;padding:20px'>   🥢 <b> Feature Selection </b> </h1>\nFor the feature selection part I used some **filter based techniques** because these methods are faster and less computationally expensive than other feature selection methods such as wrapper methods. Basically:\n* Drop Features with high missing values rate\n* Keep the TOP target correlated features\n\n<h2 style='color:#2d3a41;background-color:#a3d3eb;padding:10px'>  👉    -- Drop Features with high missing rate -- </h2>\n\nAll features having **more than 75% of missing values are ommited**","metadata":{}},{"cell_type":"code","source":"def missing_values_table(df):\n    # Total missing values by column\n    mis_val = df.isnull().sum()\n\n    # Percentage of missing values by column\n    mis_val_percent = 100 * df.isnull().sum() / len(df)\n\n    # build a table with the thw columns\n    mis_val_table = pd.concat([mis_val, mis_val_percent], axis=1)\n\n    # Rename the columns\n    mis_val_table_ren_columns = mis_val_table.rename(\n    columns = {0 : 'Missing Values', 1 : '% of Total Values'})\n\n    # Sort the table by percentage of missing descending\n    mis_val_table_ren_columns = mis_val_table_ren_columns[\n        mis_val_table_ren_columns.iloc[:,1] != 0].sort_values(\n    '% of Total Values', ascending=False).round(1)\n\n    # Print some summary information\n    print (\"Your selected dataframe has \" + str(df.shape[1]) + \" columns.\\n\"      \n        \"There are \" + str(mis_val_table_ren_columns.shape[0]) +\n          \" columns that have missing values.\")\n\n    # Return the dataframe with missing information\n    return mis_val_table_ren_columns\n\n# Missing values for training data\nmissing_values_train = missing_values_table(train)\n#cm = sns.color_palette('Set2', as_cmap=True)\n#missing_values_train[:20]#.style.background_gradient(cmap=cm)\nTHRESHOLD = 75\ndrop_cols = missing_values_train[missing_values_train[ '% of Total Values']>THRESHOLD].index.to_list()\nprint(f\"Drop {len(drop_cols)} features with more than {THRESHOLD}% of missing values\")\ntrain = train.drop(drop_cols, axis=1)\nprint(\"Training data shape after dropping highly missing values columns\", train.shape)","metadata":{"execution":{"iopub.status.busy":"2022-09-30T04:11:03.871778Z","iopub.execute_input":"2022-09-30T04:11:03.872321Z","iopub.status.idle":"2022-09-30T04:11:12.615789Z","shell.execute_reply.started":"2022-09-30T04:11:03.872277Z","shell.execute_reply":"2022-09-30T04:11:12.614521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(drop_cols)","metadata":{"execution":{"iopub.status.busy":"2022-09-30T04:11:12.617943Z","iopub.execute_input":"2022-09-30T04:11:12.618679Z","iopub.status.idle":"2022-09-30T04:11:12.626026Z","shell.execute_reply.started":"2022-09-30T04:11:12.618633Z","shell.execute_reply":"2022-09-30T04:11:12.624571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<h2 style='color:#2d3a41;background-color:#a3d3eb;padding:10px'>  👉    -- Select top correlated features with the target-- </h2>\n\nWith this method we assume that high predictive features are **highly correlated with the target**\n\n","metadata":{}},{"cell_type":"code","source":"corr = train.corrwith(train['target'], axis=0)\ncorr = corr[corr.notna()].sort_values(key=abs, ascending=False)\nTHRESHOLD = 0.15\nCORR_SELECTION=False\nif CORR_SELECTION:\n    selected_feats = corr[corr.abs()>THRESHOLD].index\n    train = train[list(selected_feats)]\n    print(f\"Training data shape after dropping uncorrelated features\"\n          f\"(threshold Pearson correlation = {THRESHOLD})\", \n          train.shape)\n\n    \n","metadata":{"execution":{"iopub.status.busy":"2022-09-30T04:11:14.652275Z","iopub.execute_input":"2022-09-30T04:11:14.652798Z","iopub.status.idle":"2022-09-30T04:11:30.979403Z","shell.execute_reply.started":"2022-09-30T04:11:14.652758Z","shell.execute_reply":"2022-09-30T04:11:30.978034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-30T04:11:30.982018Z","iopub.execute_input":"2022-09-30T04:11:30.982615Z","iopub.status.idle":"2022-09-30T04:11:31.205280Z","shell.execute_reply.started":"2022-09-30T04:11:30.982548Z","shell.execute_reply":"2022-09-30T04:11:31.203470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's display the top correlated features to the target ","metadata":{}},{"cell_type":"code","source":"sorted_corr = corr.sort_values(key=abs, ascending=False)[:11] # top but we have to  drop corr=1\n\n\npos = sorted_corr[(sorted_corr>0) & (sorted_corr<1)]\nneg = sorted_corr[(sorted_corr<0)].sort_values(ascending=False)\n\nfig = go.Figure()\nfig.add_trace(go.Bar(x=pos.index, y= pos.values,\n                     orientation='v',\n                     name='Positive',\n                     marker=dict(color=theme_palette['base'],line=dict(color=theme_palette['complementary'],width=0)),\n                     text = [\"%.2f\" %(round(v ,2) *100) + '%' for v in pos.values],\n                     textposition = 'outside',\n                     textfont_color = '#212a2f'))\n\nfig.add_trace(go.Bar(x=neg.index, y= neg.values,\n                     orientation='v',\n                     name='Negative',\n                     marker=dict(color=theme_palette['complementary'],line=dict(color=theme_palette['base'],width=0)),\n                     text = [\"%.2f\" %(round(v ,2) *100) + '%' for v in neg.values],\n                     textposition = 'outside',\n                     textfont_color = '#212a2f'))\n\n#theme_palette['2']\nfig.update_layout(template = temp,\n                  title={\n                      \"text\": \"<b>Top-10 Correlated Features with the Payment Default Feature</b> <BR />Pearson Values > 0.5<br> <br> \",\n                      \"x\":0.035,\n                      \"font_size\": 18,\n                      \n                  },\n                 plot_bgcolor = theme_palette['backgound'],\n                      paper_bgcolor = theme_palette['backgound'],\n                 legend=dict(\n                            y=1.15,\n                            x=0.88))","metadata":{"execution":{"iopub.status.busy":"2022-09-28T10:17:08.694985Z","iopub.execute_input":"2022-09-28T10:17:08.695315Z","iopub.status.idle":"2022-09-28T10:17:08.850775Z","shell.execute_reply.started":"2022-09-28T10:17:08.695289Z","shell.execute_reply":"2022-09-28T10:17:08.849637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2 style='color:#2d3a41;background-color:#a3d3eb;padding:10px'>  👉    -- Store Interim Data-- </h2>\n\nDue to the redundant crashs we ought to store the used columns to use it later (it has been stored upstream in [this dataset](https://www.kaggle.com/datasets/schopenhacker75/interim-data-v2) ) ","metadata":{}},{"cell_type":"code","source":"if True:#not(SAVED):\n    train=train.set_index('customer_ID')\n    types = train.dtypes\n    target_col = 'target'\n\n    cat_cols = list(types[types.apply(lambda x:not(str(x).startswith('float')))].index)\n    cat_cols = list(filter(lambda x:x!=target_col, cat_cols))\n    features = list(train.drop(target_col, axis=1).columns)\n    gc.collect()\n    print('len cat_col', len(cat_cols))\n    print('len features', len(features))\n    \n    with open('final_features_drop_na_cols.pkl', 'wb') as f:\n        pickle.dump(features, f)\n\n    with open('final_cat_cols_drop_na_cols.pkl', 'wb') as f:\n        pickle.dump(cat_cols, f)\n    \nelse:\n    with open(os.path.join(INTERIM_DATA, 'final_features.pkl'), 'rb') as f:\n        features = pickle.load(f)\n    with open(os.path.join(INTERIM_DATA, 'final_cat_cols.pkl'), 'rb') as f:\n        cat_cols = pickle.load(f)\n","metadata":{"execution":{"iopub.status.busy":"2022-09-30T04:11:31.207158Z","iopub.execute_input":"2022-09-30T04:11:31.207575Z","iopub.status.idle":"2022-09-30T04:11:36.311880Z","shell.execute_reply.started":"2022-09-30T04:11:31.207532Z","shell.execute_reply":"2022-09-30T04:11:36.310312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-30T04:11:36.317370Z","iopub.execute_input":"2022-09-30T04:11:36.318594Z","iopub.status.idle":"2022-09-30T04:11:36.543892Z","shell.execute_reply.started":"2022-09-30T04:11:36.318521Z","shell.execute_reply":"2022-09-30T04:11:36.542423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"5\"></a> \n# <h1 style='display:fill;color:#2d3a41;background-color:#a3d3eb;padding:20px'>   🤖 <b> Model Training </b> </h1>\n\n\nI implemented a gradient boosting model Wrapper **BaseModel** (inspired from [jayjay's notebook](https://www.kaggle.com/code/jayjay75/wids2020-lgb-starter-adversarial-validation)) in which I defined most of in common methods and attributes of gradient boosting based models, such as _cv training, feature importances, prediction_ ...\n**LgbModel** and **CatBoost** would inherit from the BaseModel, Later I can add XGboost, hence we get the overall Gradient Boosting models comparison.\n\n<h2 style='color:#2d3a41;background-color:#a3d3eb;padding:10px'>  👉    -- Competition Metric -- </h2>","metadata":{}},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-30T04:11:36.546215Z","iopub.execute_input":"2022-09-30T04:11:36.546718Z","iopub.status.idle":"2022-09-30T04:11:36.778211Z","shell.execute_reply.started":"2022-09-30T04:11:36.546677Z","shell.execute_reply":"2022-09-30T04:11:36.776833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n    #https://www.kaggle.com/code/inversion/amex-competition-metric-python\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x == 0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n\n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x == 0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n    \n    y_pred=pd.DataFrame(data={'prediction':y_pred})\n    y_true=pd.DataFrame(data={'target':y_true.reset_index(drop=True)})\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)\n","metadata":{"execution":{"iopub.status.busy":"2022-09-30T04:11:38.554918Z","iopub.execute_input":"2022-09-30T04:11:38.555445Z","iopub.status.idle":"2022-09-30T04:11:38.572444Z","shell.execute_reply.started":"2022-09-30T04:11:38.555402Z","shell.execute_reply":"2022-09-30T04:11:38.570690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# a wrapper class  that we can have the same ouput whatever the model we choose\ndef get_partial_pred_df(fold, y_val, pred):\n    df = pd.DataFrame()\n    df['y_true'] = y_val\n    df['y_pred'] = pred\n    df['fold'] = fold\n    return df[['fold','y_true', 'y_pred']]\n\nMODELS_PATH = \"../input/cv-models\"\nclass BaseModel:\n    RAND_SEED = 75\n    # TODO make a longer list of colors\n    COLORS =['#a3d3eb', '#dfa3eb', '#ebbba3', '#afeba3', '#ebe6a3']\n    def __init__(self, name, features, categoricals=[], n_splits=5, verbose=True, ps=None, target_col='target'):\n        self.name = name\n        self.features = features\n        self.n_splits = n_splits\n        self.categoricals = categoricals\n        self.target = target_col\n        self.cv_models = []\n        self.verbose = verbose\n        self.model=None\n        if not ps:\n            self.params = self.get_default_params()\n        else:\n            self.params = ps\n            \n        \n    def train_model(self, train_set, val_set, eval_metric=[]):\n        raise NotImplementedError\n        \n    def get_cv(self, train_df):\n        cv = StratifiedKFold(n_splits=self.n_splits, shuffle=True, random_state=self.RAND_SEED)\n        return cv.split(train_df, train_df[self.target])\n    \n    def get_default_params(self):\n        raise NotImplementedError\n        \n    def convert_dataset(self, x_train, y_train):\n        raise NotImplementedError\n        \n    def convert_x(self, x):\n        return x\n            \n    def fit_cv(self, train_df, save_cv=False):\n        self.oof_pred = np.zeros((len(train_df), ))\n    \n        cv = self.get_cv(train_df)\n        self.partial_oof_scores_=[]\n        self.cv_df=pd.DataFrame()\n        \n        for fold, (train_idx, val_idx) in enumerate(cv):            \n            print(\"*-\"*100)\n            print(f\"*** FOLD == {fold} **\")\n            print(\"*-\"*100)\n            x_train, x_val = train_df[self.features].iloc[train_idx], train_df[self.features].iloc[val_idx]\n            y_train, y_val = train_df[self.target][train_idx], train_df[self.target][val_idx]\n            train_set = self.convert_dataset(x_train, y_train)\n            val_set = self.convert_dataset(x_val, y_val)\n            conv_x_val = self.convert_x(x_val)\n            # OOM\n            if os.path.isfile(os.path.join(MODELS_PATH, f'{self.name}_{fold}.pkl')):\n                print(\"load\", f'{self.name}_{fold}.pkl')\n                model = joblib.load(os.path.join(MODELS_PATH, f'{self.name}_{fold}.pkl'))\n                self.cv_models.append(model)\n            else:    \n                model = self.train_model(train_set, val_set)\n                self.cv_models.append(model)\n\n            self.oof_pred[val_idx] = model.predict(conv_x_val).reshape(self.oof_pred[val_idx].shape)\n            \n            partial_oof_score = amex_metric(y_val, self.oof_pred[val_idx])\n            self.partial_oof_scores_.append(partial_oof_score)\n            print('Partial score of fold {} is: {}'.format(fold,  partial_oof_score))\n            if save_cv:                \n                # save model\n                joblib.dump(model, f'{self.name}_{fold}.pkl')\n                \n            partial_pred_df = get_partial_pred_df(fold, y_val, self.oof_pred[val_idx])\n            self.cv_df = pd.concat([self.cv_df, partial_pred_df])\n\n        self.oof_score_ = amex_metric(train_df[self.target], self.oof_pred) \n        if self.verbose:\n                print('Our oof score is: ', self.oof_score_)\n                \n    def fit(self, train):\n        x_train = train[self.features]\n        y_train = train[self.target]\n        train_set = self.convert_dataset(x_train, y_train)\n        self.model = self.train_model(train_set)\n        print(\"model trained in all training dataset\")\n        \n    def predict_cv(self, test_df):\n        y_pred = np.zeros((len(test_df), ))\n        x_test = self.convert_x(test_df[self.features])\n        for model in self.cv_models:\n            y_pred += model.predict(x_test).reshape(y_pred.shape) / self.n_splits\n        return y_pred\n    \n    def predict(self, test_df):\n        x_test = self.convert_x(test_df[self.features])\n        y_pred = self.model.predict(x_test).reshape((len(test_df), ))\n        return y_pred\n    \n    def get_cv_feature_importance(self):\n        raise NotImplementedError\n    \n    def plot_cv_roc(self): \n        fig=go.Figure()\n        fig.add_trace(go.Scatter(x=np.linspace(0,1,11), y=np.linspace(0,1,11), \n                                 name='Random Chance',mode='lines', showlegend=False,\n                                 line=dict(color=\"Black\", width=1, dash=\"dot\")))\n        for fold, sample_df in self.cv_df.groupby('fold'):\n            fpr, tpr, _ = roc_curve(sample_df.y_true, sample_df.y_pred)\n            roc_auc = auc(fpr,tpr)\n            fig.add_trace(go.Scatter(x=fpr, y=tpr, line=dict(color=self.COLORS[fold], width=3), \n                                     hovertemplate = 'True positive rate = %{y:.3f}<br>False positive rate = %{x:.3f}',\n                                     name='Fold {}: AUC = {:.3f}'.format(fold+1, roc_auc)))\n\n        fig.update_layout(template = temp,\n                      yaxis_automargin=True,\n                      height = 800,\n                      plot_bgcolor = theme_palette['backgound'],\n                      paper_bgcolor = theme_palette['backgound'],\n                      title={\n                          \"text\": \"<b>Cross-Validation ROC Curves</b> \",\n                          \"x\":0.045,\n                          \"font_size\": 20,                    \n                      },\n                      margin={'pad':5},\n                          xaxis_title='False Positive Rate (1 - Specificity)',\n                          yaxis_title='True Positive Rate (Sensitivity)',\n                          legend=dict(orientation='v', y=.07, x=1, xanchor=\"right\",\n                                      bordercolor=\"black\", borderwidth=.5)\n        )\n\n        return fig\n    \n    def plot_cv_feature_importance(self, top=20):\n        feat_imp = self.get_cv_feature_importance()\n        data = feat_imp[:top]\n        threshold = data.iloc[5]['importance']\n        fig = go.Figure()\n        fig.add_trace(go.Bar(y=data.index, x= data['importance'],\n                             orientation='h',\n                             width=[0.6]*len(data),\n                             marker=dict(color=(data['importance'] < threshold).astype('int'),\n                                         colorscale=[[0, theme_palette['complementary']], [1, theme_palette['base']]], ),\n\n                             text = [\"<b>%.2f\"%(round(v ,2))+'</b>' for v in data['importance']],\n                             textposition = 'inside',\n                             textfont_color = theme_palette['backgound']))\n\n        fig.update_layout(template = temp,\n                          yaxis_automargin=True,\n                          height = 800,\n                          plot_bgcolor = theme_palette['backgound'],\n                          paper_bgcolor = theme_palette['backgound'],\n                          yaxis=dict(autorange=\"reversed\"),\n                          title={\n                              \"text\": \"<b>Features Importances</b> \",#\"— Gain - Threshold = 5.2e+04<BR />the last payment 2 is farway in the top of list<br> <br> \",\n                              \"x\":0.045,\n                              \"font_size\": 20,                    \n                          },\n                          margin={'pad':5},\n        )\n        return fig\n\n","metadata":{"execution":{"iopub.status.busy":"2022-09-30T04:12:04.296371Z","iopub.execute_input":"2022-09-30T04:12:04.296983Z","iopub.status.idle":"2022-09-30T04:12:04.337790Z","shell.execute_reply.started":"2022-09-30T04:12:04.296942Z","shell.execute_reply":"2022-09-30T04:12:04.336219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#we choose to try a LightGbM using the Base_Model class\nclass LgbModel(BaseModel):\n    \n    def train_model(self, train_set, val_set=None):\n        verbosity = 100 if self.verbose else 0\n        valid_sets=[train_set]\n        if val_set:\n            valid_sets = [train_set, val_set]\n        return lgb.train(self.params, train_set, \n                         valid_sets=valid_sets, \n                         num_boost_round = 10500,\n                         early_stopping_rounds = 1500,\n                         verbose_eval=verbosity)\n    \n    def convert_dataset(self, x_train, y_train):\n        train_set = lgb.Dataset(x_train, y_train, categorical_feature=self.categoricals)\n        return train_set\n    \n    \n    def get_default_params(self):\n        params = {'n_estimators':50000,\n                    'boosting_type': 'gbdt',\n                    'objective': 'binary',\n                    'metric': 'auc',\n                    'subsample': 0.75,\n                    'subsample_freq': 1,\n                    'learning_rate': 0.1,\n                    'feature_fraction': 0.9,\n                    'max_depth': 15,\n                    'lambda_l1': 1,  \n                    'lambda_l2': 1,\n                    'early_stopping_rounds': 100,\n                    #'is_unbalance' : True ,\n                    'scale_pos_weight' : 3\n                  \n                    }\n        return params\n    \n    def get_cv_feature_importance(self):\n        imp_df = pd.DataFrame(index=self.features)\n        imp_df['importance'] = 0\n        for model in self.cv_models:      \n            imp_df['importance'] = (imp_df['importance'] + pd.Series(model.feature_importance(), index=model.feature_name()))/self.n_splits\n        return imp_df.sort_values('importance', ascending=False)\n                                    \n    \n\n#we choose to try a LightGbM using the Base_Model class\nclass CatBoost(BaseModel):\n    def train_model(self, train_set, val_set=None):\n#        eval_set=[val_set] if val_set else None            \n        verbosity = 100 if self.verbose else 0   \n        return catboost.train(pool=train_set, params=self.params, \n                         eval_set=val_set, verbose_eval=verbosity)   \n    \n    def convert_dataset(self, x_train, y_train):\n        train_set = Pool(data=x_train, label=y_train, \n                         cat_features=self.categoricals)\n        return train_set\n    \n    \n    def get_default_params(self):\n        params = {'iterations' : 5000,\n                  'random_seed':self.RAND_SEED\n#                  'metric': ['auc','binary_logloss'],\n                    }\n        return params\n    \n    def get_cv_feature_importance(self):\n        imp_df = pd.DataFrame(index=self.features)\n        imp_df['importance'] = 0\n        for model in self.cv_models:      \n            imp_df['importance'] = (imp_df['importance'] + pd.Series(model.feature_importances_, index=model.feature_names_))/self.n_splits\n        return imp_df.sort_values('importance', ascending=False)\n                                    \n    ","metadata":{"execution":{"iopub.status.busy":"2022-09-30T04:12:05.168035Z","iopub.execute_input":"2022-09-30T04:12:05.168585Z","iopub.status.idle":"2022-09-30T04:12:05.187146Z","shell.execute_reply.started":"2022-09-30T04:12:05.168537Z","shell.execute_reply":"2022-09-30T04:12:05.186138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['target'] = train[\"target\"].astype('int')\nprint('Transform all String features to category.\\n')\nfor usecol in tqdm_notebook(cat_cols):\n    train[usecol] = train[usecol].replace(np.nan, 0).astype('int').astype('category')\n","metadata":{"execution":{"iopub.status.busy":"2022-09-30T04:12:06.850530Z","iopub.execute_input":"2022-09-30T04:12:06.851401Z","iopub.status.idle":"2022-09-30T04:12:16.000608Z","shell.execute_reply.started":"2022-09-30T04:12:06.851337Z","shell.execute_reply":"2022-09-30T04:12:15.999290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2 style='color:#2d3a41;background-color:#a3d3eb;padding:10px'>  👉    -- LightGBM Training-- </h2>\n","metadata":{}},{"cell_type":"markdown","source":"<img width=\"40%\" src=\"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAdgAAABrCAMAAADjLmPEAAABJlBMVEX///9LS038tRjvSSd2tkQbmtc2NjlFRUdAQELx8fFISEqFhYb39/c7Oz09PUD8sgBaWlyNjY5fX2CmpqbvTy/+3ana2trQ0NC3t7hzc3XwXUPj4+OwsLFqsS2/v787o9qenp8xMTQAktSVlZaKiotvb3DV1dXf39/IyMh5eXpRUVPuMAC9vb1mZmdosClwszklJSjvQx3y+O+jzIe82amz1Z17uUzs9ftXrd7/9ef91pb+79j+4rX5yML0j4DuPRD729f2oZb96Ob3s6rycl7zg3PwVzrxZk/97On1nI/R5cT3rKKUxXPyfGqEvVrk8N3V58pcqwe21qGlzYvQ5vSezOq62u/f7viDv+X8vUP9z3/+6cf9x2eizuv8vDprteH9xFv9zHWLNtRnAAAQ70lEQVR4nO2de3vbthXGpdQGKVKUZbGxpEpRJMuSJdlqbDftkrSbu0vbLcuS9Ja1Tbd23/9LjJRI4AA4uJC62Y/4/mfzBvAnAAfnHICljz5cQdcf/LFU6G7qow9y6/riT7sufSGl8oO9/fDLXRe+kFp5wV5f/HnXRS+kU06wtx/9ZdclL6RVLrDXF3/ddbkLGZQH7MXf/r7rYhcyKTvY69tijnMPlBnsxVe7LnIhG2UEe/vBl7sucSErZQJ7ffGPXZe3kKWygL39+stdF7eQrezBXl/8c9eFLWQva7AX3xRznPskS7CFS+K+yQ7sxb+quy5ooWyyAVu4JO6hLMAWYdf7KCPYwiVxP2UAa3JJ3GypmIWySg/WFHZ9+e2Wilkoq3RgTS6Jm1ePHtk842zUXmoUGs+thtPpsGNXdlzpw9rzzLfp+FmM/06z1Q37w34/bPXG+LN6PqKz4353oilbSGuweGdju9J0uat8HViTS+LloycHVmDrTiLXN506GbjRaYOhXWUwVQfp08hJtivHbTKwBFudDBskqLmuG9fKdWtBQGZ+V3pgGDiY4gvIqFvBb37mcudO7YrEXeTU1WBNLomouR4cWIL1yks5JrCdwfJE0rWrDaIqSR5WDjKB7dWJUyZWYCdtUnPSOlF5jhsQf8KdGbriWeB8d4C3xmOHO48o+AtlCriLGkqwJpfE06i5rh/sNK1TYHNfVLnAVkOnFhfSBmzoBPybh3JqA4hWBzY+m4wQagJY1+pHPuJ/aCqwt9d6l8TN6wXWtYOl7yE4t7kxpjxghyR5sBlsN9Cz4jkYwMZoL6VHCGC9skUNKoS/rwKsySXx3ccHBxsBS/uTbN0oVA6w6QBgBnsyqxlIZQQbPbIvPkQAWw4szKe+8CAUrCkT/M1nTw42BHaenjmwuS+qPGDp78kAtj+QRtaVwZaJOFUQwRrfWgl0dWqwxkzw72lzXT/YyWB5Yi2/WbxBsHNjc9WApQarJ/44BkKLFMFaDBATsWQy2Itv9C6JNwdPDjYHttQfOPHAY2niY9oY2EpZeOGRGRxPc0gQzXuYkawA60zTaexoRvhxWhxEJbBm82kk/lhEsMaw6yewuW4AbGk8dZz2xHiaWpsCW+EnOJ5L6sNW87xTqXTOm63hKAhcTwcW3uuk70C0tRb3JAmsNzNVQDCdJLCmTPC3T54cbBjsytoQWJ6rF9Ql98J5GM+EVWA94ew+YOHVuUMS2LLJ1dKXruDAGsOuC5fEnoKdQa5Bo4medH5GHJsWG+sSkOV9EDJY50xfgXSEZWWEYE2Z4C+l5rpHYEfgZTuBPPdMVTm+sgQL+dUm6IEZI68tfy8B600pWQbWFHa9QZrr/oANgcPObWiN1HNIXQcW+BR464iBpV2s25IuB0oniUGTGscpWGMmeOJB3FOwzIMRveMsFrsOLDBlXW4qS8EGlL0wCgulS87yGqzuCdjbr/VznJvXaHPdG7ANNniZA1RQWrDM5FGAJezBRONiHSbn1y4rPFhjJvjTj/Hmun2w4+5wOm/Mp8NLZTxzNbDo8UvWETujTOXVgu3Sg4qumJQu077VOVY/JKCn82BNYdc3r5VYtwu2OY1m9rHnxnOcaBaZjGXhKNF86cDBwXYmrTDsXo7lYMrU92lMKSogFehxma/IyiEPZAlWYTwR6DxXPiM1nSL27BfaKH1tckl8+rEa69rBtrqpJALNOh8t82rlBcl6TDpWsDQwELAnxy6Jg+KuG5B6KPS2cVye3ZWFqpm3+pL56wYZkzIsu2LFdIewbjbqZ1XPSE2nqLeGYL/S+yGhw38LYAduooE4UZwSObI9iIcmOgy5CrCdOQHkPHfQ524sOW3Sq+Xil92sLmwt2HZaLMGzBMFSn5LXUDyCmU4lDqxe32ub6/rB0rZR48FKftrk5Q/NYOWIjDuDDcQI9gRMS2xqC6UFS+/r9rn/M7BVUD2i6CzSNh1X3xbs2wN9c90a2E6gCJeR0AR2hERkPBeQNYIdKoxXGzGw8m+iRY8N+HGHA9uj5pOit2CmkzXYV98+wpxN2wdbdZVh0EFnrgU7RwOisO8zgmV9hV1GFJQOLPu9CMg4sMx8qqFPuGSmk32LvXn56asnEdztWcU42AZvNTlujcbJvLm2xU4VgW7wLk1gWeQkh02vATunI6zYS/NgmfnUw57QYKZTqXRuPcZGunn7nYbuNsD2YW/qkMaw2+u1hrNl98ymIjLYTjep6CK+zWFjvd9V9CsBxGuprpLjLcWkxEpKsBX2Yx2I020eLDOf5sgDzlPTaeGaygR2ISXdLYCF/jyPDCmRkxGfcSmBdXuLE7zA8YdDfwZtYzBi9SYT4CmI/kqUto8zNinJvpSUgeXmodWQ2fgDqSHyYFm6EGY+HQPTKXolmcEudPP26auDR/y4uwWwbQbEKXN1u+S6UQns0pIOEsdFqTOFPwT4orUuRVp4rb9WIWAVH3eq1UqkziRsExamJXI3IICl5pNgPC+UVnbpMssJdqGI7iefMbqbB3vOQEmv9gSSlcEuagziIi1wKAB9vRYsG2INQVFMMOcpIAvFeTS0QkFDl1eclIa+FLVlnXgcVwG71JuU7ubB+qzButJrbwJUKFje0RGy0drps3/rwDL3a/bJjiFL0Qk81J0kgqUeqkBq3XVoOq0D7EJvnn7/2mq13SpgASeCpC0M2ZvDwIrp2CxMA00RHVhmj2BWKXOBQjFrSAfW8fE0DAksNZ+kCMQJZzqtDay1VgFLR5iy09ZegYF1xNhpkw2zIJCjA8uuqMkcQuoChQILj7Rrd9rHLdSbJIJlgVtxGc8ZZzqVSuN7BJZFXvCQJEuBR8DKnNikB7wkHViWsYsk5Iv59/xrLhm64mhCTmZIZqkElpZBHA140+l+gaX/URilrKtErGLZC8f8gyCotyOwi1q5NSnrRQJbUnicqemUmnX3CCzz+6iypjVgkYlfD50p7w5sfN+58FAZLH1OwA0HaeYkzU5tArA//AF/X+vUCmAZCNX6O7VLEcuzZgyBH8kOLDLGZgFbCxLV4LqBxVspa4IAC1UG6anQaBBNpwgsSGY7PD387ccN010BLHszqgzMthIslk7CjgIjVwd2rLOKM4B1J+NEk17ozwi80vO4x8pgmfkEA0G+Iz4Pgn18eHh4ehrTfbcxuiuApbNYZaCZWleySxHrvNkT7MCCMVy+XQawwkqASggXeTicGxgBS9erA/OpKppOMtiFYrrPNkN3BbD0pyrNXFKpwaJOe9oJWoJlDgrMFBsQJjEaEUu9xINf5MGloCNgqfkEkq66oukExw0ANqV7+uynd//G32FerQCWXarKSlGDRRcM06UalmCZSxGLrlSpSlUxMBxLBxa6zbjQHQaWmU+0UpLppAO7GborgGXR6L7iEg1YLP20nhUsW7Rj2BdjKBQklhZsqcXcJdBbiIFl5lP6CtPBH04DDWAp3nXRXQEsG+FUntpNg2W+asO+GNnBsogcN9JgYFmMK92xSDadLMGmdB//9O4HbYXMutdgQbBWHwXIARYsRAfdAQqWmU9LG646wE6yBruE+6O2QmatAyyMxnBqbxgsM4sNAdkcYEuoMwUFSwelZHaOmE4ZwZ4+XtlQXscYq4yG6pPZ1GWxBQv27VBlgC6VBywYwJn7AwdLb7U0lhDTCbpzzGBP/6OrTOm59miidVjFaMZPLE3O03rAsnWs+uLnAcsW3IG5GQ6WzrsWfhdqOnHeNXuwp8/0zfXhkfZwohXAsrwYPPsS+iM2BJb5nvS73+UBy2JXLpvJ4mDZqwjAObzXxBrs6TtNRaLm+vPRpsGCaAzeDzbVQYA1gS2xHQDwkHBJKGoGsHPMX6IAS2saN+4APaVHH6cFe/qbPi3vxdHRg02DVS01pAJ5+psCy8pQ1mxSkAssiBGYxli26C+aGqVWkvBCL23Anh7qm+vnD44ePNg4WNYPKvbE0WRQrAss3PZMYz/lAAu28QlMVjE0n+gkVthOpmUB9vQ3ZQ0W+iXGunmwYEIQyGEzrjWtBSw6iLbYr4db9sMrB1gQRADBKxVYZj6ltxV/7Gawp4d6p8T7oyXXzYNldiPWZKtw0481gFU4l9hCyrLnqMhmBwtKD0dvFVhmatEnCcOTEezpT4rCJ/o1wboFsC3QJOU4AF0Asyaw+PoYYTGCYqVHdrAgFR6ua1aCBal4+AkGsKeP9R7iL44o182D5XLTxPQgH8ZDVwBrXHfVha+U4F8O8LOChaWHAQYlWGCf46XVgzV4EJ//zrBuASy3Txm/A2x1xMW584PVp0ktxC3bc4j8OYrLumiex9IsfO404FowWDE1WCGFSopLdjVgTc314RHkmhWsaZUEApYGrBb/njNY3cVSCWcNnidg9np0tbvQKcNOf7HmL6QblVTG3TYBS3hRsIJZ1vTh1gse157VYLl3gRgdGrAmD+J/OaxZwZZn7RGqecIRyxzj8k88Uu/3xuNm119uDev4a/AVw8HOI37vpDPuNq6ECxt8a/HcgASzxrxRDuK9bOAhvCsuD8PlOoGwfzYn/EYp/EpKNVjgqSpjM3slWJMH8YWANTPYsocrSMqIpgTWuZHFizcJpgubyDqCANBCi7/XEMRb1EphdWTTg7jo0j8VWYqeQ5cKiF8AEWwHDVjOfJIPq8AaPIhLl8RqYBVK3wUKVvyQAdBgYt41RlEWCJYGN4HkfImhuhx8ZeyWeLBKCC1PAxaaT4jBgoM1Ndf/yVi3A7Y0Higuiz+UsA6wcG0XvVq+dqL5LguTc8WmLhZgPWnlsw4sdG/KKV0hBtbQXN8fYVy3A7bUxD+ysPgAxlrAluQuFctwqvryblNiTQY+sJjNYGszycDWgWV9C+avQcCaPIi/o1i3BbbUKctvyBksWgYDu2wo+cDKGw7hqWsn3G5gUpGCWp+zfk1g3cBmURaUr/I68Y9LwJo8iF/gzXV7YOMUXv6FOqRd4W+ekMoHtlSpC2hVH4g590kNYxuZyYEvOqUMC58dNItKC5ZGRbDP8IlgDR7E5z+rsG4RbKkyrEXTiqjL9OL9gIifUnOES3KCjUzjGXGX305ZPkBdn54f783opJs4OvHXPMg8RNKYQxedBMRXEOfYcuEzryQnBvX18GBNzfWFsrnagi0Tk9L91q/oP7BKj0N/XnZmjWm/yaoscqyye2AJ4x49Km/a3emezWee40UPMO38U5mEx9P5LJqZ10f+UPnV0O6sLiveljfUfGfUT9cXXKHR4St15fp0ZcKs9NjQXD8XXRI5wFbMqopn2tw3lhxy099DeuTdU1X/EjTHqqDuz/QexF+0WC3BblIs41L/QYRCUO8Rl8QdA0sHJGUaYyFJqEvijoFl4ZMVvhK9X1LPce4QWLDgfJXvpu2RqiqXxN0Cy2ZSOXY63Ec9tGmuuwfbZZvKZPu8xp5K55K4S2DBPouKbKVCUFqXxB0Cew6ik8ie+4V4YWHXOwm2C+KjNeO3c/deJpfEXQE7qYOMBuOnc/deZpfEzsBW2ShaaQ49Lhgjbb9eiNev2bBuE2zTDchs1G6P5jMiBs6CHFsJ75OsXBI7AnsWD6hxwAvJH6vl2Pt7j/TcziWxE7DnSAYFa68FV51sXRK7ACt/xgzYTWKSXyEoKRP8DoHtNJDPmKVYg3rGD0Tul+RM8LsDtjoaqD6B5gSzwuGkUSaXxPZbbKnSmsfLKISlZpGVrPpcQqGFLMKuuwUbqdoMp9E0J97GuVYLAkK8KZY8VogpmuOsoq2WtdIZNyeT5vjcOiVqf/X+xcOVtOvyF8L1f1PZxZrIHLyMAAAAAElFTkSuQmCC\">\n","metadata":{}},{"cell_type":"code","source":"#https://www.kaggle.com/code/ragnar123/amex-lgbm-dart-cv-0-7977#kln-129\nparams = {\n        'objective': 'binary',\n        'metric': 'binary_logloss',\n        'boosting': 'dart',\n        'seed': 75,\n        'num_leaves': 100,\n        'learning_rate': 0.01,\n        'feature_fraction': 0.20,\n        'bagging_freq': 10,\n        'bagging_fraction': 0.50,\n        'n_jobs': -1,\n        'lambda_l2': 2,\n        'min_data_in_leaf': 40,\n        }\n\n\nstart_time = time.time()\n\nlgb_model = LgbModel(name='lgb', features=features, n_splits=5, categoricals=cat_cols, ps=params, verbose=None)\nlgb_model.fit_cv(train, save_cv=True)\n\nprint(\"LGB TRAINING TIME\")\nprint(\"--- %s seconds ---\" % (str(datetime.timedelta(seconds=time.time() - start_time))))\n\njoblib.dump(lgb_model, \"lgb_model.sav\")","metadata":{"execution":{"iopub.status.busy":"2022-09-30T04:12:51.819139Z","iopub.execute_input":"2022-09-30T04:12:51.819671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb_model.oof_score_","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2 style='color:#2d3a41;background-color:#a3d3eb;padding:10px'>  👉    -- Catboost Training-- </h2>\n<img width=\"40%\" src=\"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAVsAAACRCAMAAABaFeu5AAAA8FBMVEX/////zAAAAAD/AAD/ygD/7sH/4Y7/4484ODj/zgD/0AD5+fno6Ojb29u2trb/0QBfX1/19fWhoaGoqKjOzs7j4+MXFxdRUVHu7u4jIyPV1dWOjo7Gxsbn5+dDQ0PPz8//+uaZmZmDg4MvLy//9M8yMjJ3d3e6uro+Pj5YWFhtbW1ISEj/4Xz//vX/7a7/+eD/kgD/QQD/MgD/uQD/cAD/UAD/77j/1kX/vwD/9tT/21n/YgD/oAAPDw//sgD/1D3/6qH/XAD/6aT/qAD/3Wj/0ir/iAD/fgL/2lH/4nr/mQD/wqz/qpX/mH//rHD/ABH8eUIkAAAKI0lEQVR4nO2cfX+iuBbHQfEWRAUfEFTa+lCfO7UPttOHddpu63Ru77277//dbEICBEios9bB+9nz/UeJEciPk3NOQkCSAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA/g9Z/GsLDrI++/2moGzDTdanv9cUFHkLvn7J+vz3me20VcArpLCdtrLWzboBe8yW2iqnWTdgj9lSW1k5yboF+8u22so/sm7B/rK1tvJZ1k3YW7bXVoM8TMD22kIeJuKntNU0jVd6l3Uj9hSxtgqSMvJjsajcKkYxWfM+60bsKSJtlfXvJ3cnhXtG2feHi8vrhxcjabuFrFuxn4i09aP/4ivZNl4u84SL24TpriGc8eBry7jQL57pFl/yAZdXcXGVp+wasMfwtWVz1u4j0vo2z3CtxN3CupvV+e8zXG1fI1VuFNl4YLXNnycM91Ww+380PG2V2FBL1pSItPkLI/EXyMOScLWNhaav2m9RbfMJbeV7CGcJuD4hptOrdh7TNuFwN51W0PUdtGFf4Wobq5PUljOA+DgPs3r9kWmO+nVr89Or1Aj1SqP6ky3LHq5PiDlPQ/v2oU/4aFqhWsuFNOvOhqc3YP616pX+XhuzgqttVKYuyhMuI9I+cLSV5UXKYdyjXIT+hqenRv6lNv52O7OA6xOi6eqpIhejTuEbb8omLQ+reNK0RrV6vWfirz1RzWq5zLoMpG3zaH40H66Iup/jGPRyufwLHH9cW0XBBezsyxna1uQlI+13w68XRbhaYYxlmbr+ZsXOCQ0QXQWV6fpIW5OoUK1h/9DZpq0BZbSn9qfsKZWYtj/OTp4eFfYWY8GzUU2+DqR902RlfVC4uY9ruxYcw2phL8AYil4R+luhtkQR81PMLQtt1+TGIrJU5fHGi/vdU/q7VjwnPvf6N8MfXCweYzb/xD/GDDVltOH5pGiLNyabBsFUsLblz9hROhFt/WiE3YD8eP96Kp0Ev2nac375/f1KQfmXH+u6Mb/Ln1Y4zAk1qZbKbuXYOvTlc5x6LjdsO6VSifwhoi3y1EfMfhzLrTSs2I71Unk8bh9WY4WHjYrbpoepOi46Izc8yK5gtA3TAzL3JSvaXTcwTe02f3llFPGdh8cglX2KehT+aoWOKHY5NZuGqBn1xfYKFzQRuZlXENEWuRYz+LPVJ/nZpMPq2DDJHu0KU3g8o3nGtOd459OkB1nZO41ojLZauNTggJQqN9Kr/2vxJf9cjCu4iCcMnNUKOk4MeG1w2fTK9BRiMjWiIqutxeZuPea/Y7+waoaFLb/T6zOmKo6F/XBz8su0DYcMv1NtX0PLNJb5NyNm39JiHdOWs1oBR7Im79BtZGItdTqaTvxWS+YQVV4NVVUdEAfNaHs4R7X8/MzTZ2DP5p7xUiOt2niEMTRnKg6eTVoXd5vBqFfvjSarHO4fnQH+0wAdZPg5oVFEqt3KGh44ULu9yF8RK1X+HdRL2C1n1Wgb93nusVXTJf35GKtMItg46pt9bQ8bPdyRa7QY58tNYq6eARMbHaFvcy/+l/DXibdzHQdSKmGbmr3FXKUdwvrbJ7/wi59eodEvtUw87JWpkuHK0LNEkvvYjR+hLRyGhY4SJ8DE5SbzhFULgb3oyvazYh3ZpXpIN7BQMyxeI8f447p/JbDniYesDHKwYBVHYI6o+9MkrPief/BLAwP/EncJcphDBLhpwzCKtfK9ZlJbn+YoKMdmWw/qdKjhIr+6ChTTsX/AX8b0kyWL/PaRJGHdMPV6lW7IF+M5vNtwg2p49X4kpU1OK2yiraP6TlOsLY5PVNBZxIM7A+Kto/3jGG0eS8Ss46O5TMZl69OTxdk9GkIUqEWuu13yzQhmEbQbFMJ+nC3uTjlWy5kPE/sEDMoyDy2rMRRrazfa7fa43rGbfuevztlkTJLQ5pyIGBqz52e9S4p2nTvqRfLgbMa8ijdPgDr9CY1bJ5KX4WpX+UvDl/Zu7Vfk0Y0dQRzLkJKj0DBF2oZj3gmVyxpGewIy4yHpH4xg2Cd79lpuEpuf9YLwlY22vsQFml6hnOCgaCDe8ktP21B2PslpXGEOJh2zs7MfayuVBsR3llt+5CMgh9siTphNqWy/u1hBgjs/Jj9lqq2sFaQTLK62lv6zfL5+Xl7kX7C26wIzDObxmLz9gBN63twglijXMvt44rG1ibbensrkarHjLpTsDkjQYv6pT0I/2+6oxHhXJFnLVlscsJDlGrdvwfTX8tbAkzmFNKvl5bfCMa8ZJqzOZCNt8Z5QOlFSo3M/yEJVkoKNw0I9GsNKbm2KRwye281aWyzu1+j93ff1QiqkWi132Z2TY7LRkCojeSklT2C0HdE0GKl0xOxp4jl0PXoJnVzUuCUyzPB8SebaymfSf2O3yf4n3aRaLTu0Y+hzo1mbGRxhT7qBtjbVpB8ZD/gGizzFNKzbS3oip0U7ShZzjDGh/vgzpm3+D84NXgbBUzs4Ac3N2IZigdqMf5xFYhlj4xFtsVwt/MWKXCxUZ4iVrrNJWBmNRqax89AHjLaZ2m10DRjlKtVuRcvCGl6QHlN1dXfadNg+7IzCPAFlUk2m2YG2erWMrZU6aHwtOkRzp++XYqe9oh7XwvmaN0TWO4GFj/2yUtQ17wqxtsUlR9tn7h1e32yfREch04mTfq1SqXfm1O3hKa9O23JRktAKxg54ODVtWI0O2VTxHBZiNiWJMO30Dg77dt213DruE3PmKKNKu1zBQY9eODfX6oxx/7DwXXzPwCUHGfXR2Dqu1aRdItQ2sSaB8JLiFUT3yxDleWTw6kVwZv7W7QeBx09Gg/lblsARlIdMqelbZoWtS8eCI7JFpsyb1Mv6s7/qr5pjjJntO1fbxBJGhrTl49V6K2z2cOy1qUM3Wy5WgNqQReUkw43I2o8pM2BwwmnwfujJw2UQ/tSD1GYuT7C+QZ/Skp0u1hFqazxztb0WavvhOlG3N7NV1Z7VghvoVsdU7VHF+60XCFcfTSZmh2zWe/6ipfFxTId2zTwazGf1aHY37kzVod2vRCqO7MngaNZhx3Ljvq2a/d06XbG2F1xtL4XabvLYtF6t6rECXq/klyarcesljiEs2/XyD7G2l1xtecvsiNnCY2ZxPs1u4XmSBJ/mb+Gx3gQ/myd8F2gL68aTiPPbK662gvxWSVsg+k8lZcz7wJFW4BLgOR0ePzmf8I2vLTxfxiNlHqwYf8pB6G3huUguac/4G28xaflr8SH/EpD6/gQjmiuciybB4Dl0LunvpjC+hQHtWeBrZQXen8Dng/d+aMbV+fL64nr5/Yrz3gRaB5435fPhO1W0ooZfAVIU3nCAiQQRG72vhvuamuBHCGQC4P1guwPea7c7ttYW3scoBN4jujvg/be7A97bvDvgfeO7A96TvzsWB1vwlPXZAwAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAJS/AMED0ZQVqMjNAAAAAElFTkSuQmCC\">","metadata":{}},{"cell_type":"code","source":"# very heavy with all feature => we will select more \nprint(THRESHOLD)\ndeeper_feat_sel = True\nif deeper_feat_sel:\n    corr = train.corrwith(train['target'], axis=0)\n    corr = corr[corr.notna()].sort_values(key=abs, ascending=False)\n    THRESHOLD = 0.15\n    CORR_SELECTION=True\n    selected_feats = corr[corr.abs()>THRESHOLD].index\n    train = train[list(selected_feats)]\n\n\n    types = train.dtypes\n    target_col = 'target'\n\n    selected_cat_cols = list(types[types.apply(lambda x:not(str(x).startswith('float')))].index)\n    cb_cat_cols = list(filter(lambda x:x!=target_col, selected_cat_cols))\n    cb_features = list(train.drop(target_col, axis=1).columns)","metadata":{"execution":{"iopub.status.busy":"2022-09-21T20:20:05.061629Z","iopub.execute_input":"2022-09-21T20:20:05.062088Z","iopub.status.idle":"2022-09-21T20:20:17.263483Z","shell.execute_reply.started":"2022-09-21T20:20:05.062054Z","shell.execute_reply":"2022-09-21T20:20:17.262206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"start_time = time.time()\n\ncb_model = CatBoost(name='cb', features=cb_features, categoricals=cb_cat_cols, n_splits=5)\ncb_model.fit_cv(train, save_cv=True)\n\nprint(\"CB TRAINING TIME\")\nprint(\"--- %s seconds ---\" % (str(datetime.timedelta(seconds=time.time() - start_time))))\n\njoblib.dump(cb_model, \"cb_model.sav\")","metadata":{"execution":{"iopub.status.busy":"2022-09-21T20:20:24.085280Z","iopub.execute_input":"2022-09-21T20:20:24.085717Z","iopub.status.idle":"2022-09-21T22:15:05.959678Z","shell.execute_reply.started":"2022-09-21T20:20:24.085681Z","shell.execute_reply":"2022-09-21T22:15:05.958223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"6\"></a> \n# <h1 style='color:#2d3a41;background-color:#a3d3eb;padding:10px'>  🏅     <b>  Feature Importance  </b></h1>\n\nLet's explore the feature importance of each model. One of the advantages of gradient boosting based models is that the feature importances can be directly extracted from the splitting gain of tree algorithm\n\n\n<h2 style='color:#2d3a41;background-color:#c8e5f3;padding:10px'>  🫐   <b>   LightGBM Feature Importance      </b>  </h2>\n","metadata":{}},{"cell_type":"code","source":"fig = lgb_model.plot_cv_feature_importance()\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2 style='color:#2d3a41;background-color:#c8e5f3;padding:10px'>  🫐   <b>   CatBoost Feature Importance      </b>  </h2>","metadata":{}},{"cell_type":"code","source":"fig = cb_model.plot_cv_feature_importance()\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-21T22:15:06.045908Z","iopub.execute_input":"2022-09-21T22:15:06.046319Z","iopub.status.idle":"2022-09-21T22:15:06.075587Z","shell.execute_reply.started":"2022-09-21T22:15:06.046284Z","shell.execute_reply":"2022-09-21T22:15:06.074418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"7\"></a>\n# <h1 style='color:#2d3a41;background-color:#a3d3eb;padding:10px'>  📈     <b> CV Roc Curves </b></h1>\nLets display Cross validation Roc curves for each model\n\n\n<h2 style='color:#2d3a41;background-color:#c8e5f3;padding:10px'>  🫐   <b>   LightGBM CV roc curves  </b>  </h2>\n","metadata":{}},{"cell_type":"code","source":"fig = lgb_model.plot_cv_roc()\n#fig = plot_roc(lgb_model)\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2 style='color:#2d3a41;background-color:#c8e5f3;padding:10px'>  🫐   <b>   CatBoost CV roc curves  </b>  </h2>\n","metadata":{}},{"cell_type":"code","source":"fig = cb_model.plot_cv_roc()\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-21T22:15:06.480858Z","iopub.execute_input":"2022-09-21T22:15:06.481172Z","iopub.status.idle":"2022-09-21T22:15:06.821221Z","shell.execute_reply.started":"2022-09-21T22:15:06.481144Z","shell.execute_reply":"2022-09-21T22:15:06.818607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"8\"></a>\n# <h1 style='color:#2d3a41;background-color:#a3d3eb;padding:10px'>  👉    Infer (or Predict Test data 😅) </h1>\n\nGenerate the test predictions ","metadata":{}},{"cell_type":"code","source":"del train\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_feather(\"../input/interim-data-v2/test_prepared.ftr\").set_index('customer_ID')\nprint('Transform all String features to category.\\n')\nfor usecol in tqdm_notebook(cat_cols):\n    test[usecol] = test[usecol].replace(np.nan, 0).astype('int').astype('category')\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Later\n#lgb_model = joblib.load('../input/interim-data/lgb_model.sav')\n#cb_model = joblib.load('../input/interim-data/cb_model.sav')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from functools import partial, reduce\n\ndef get_batchs(df, batch_size, keep_features):\n    n_batchs = int(len(df)/batch_size)\n    cid = list(df.index)\n    for i in range(n_batchs):\n        idx = cid[i*batch_size:(i+1)*batch_size]\n        if i == n_batchs-1:\n            idx = idx + cid[(i+1)*batch_size:]\n        yield df.loc[idx, keep_features]\n        \ndef predict(sample_df, model):\n    y_pred = model.predict_cv(sample_df)\n    sub = pd.Series(y_pred, index=sample_df.index, name='prediction')\n#    sub.to_frame().to_csv(outfile)\n    return sub","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2 style='color:#2d3a41;background-color:#c8e5f3;padding:10px'>  🫐   <b>   LightGBM submission </b>  </h2>","metadata":{}},{"cell_type":"code","source":"n_batchs=15000\nsamples_df = get_batchs(test,batch_size=n_batchs, keep_features=lgb_model.features)\nsub_lg = pd.concat(map(partial(predict, model=lgb_model), tqdm_notebook(samples_df)))\nprint(len(sub_lg))\nprint(len(test))\n\nsub_lg.to_frame().to_csv('submission_lgb.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2 style='color:#2d3a41;background-color:#c8e5f3;padding:10px'>  🫐   <b>   CatBoost Submission  </b>  </h2>","metadata":{}},{"cell_type":"code","source":"samples_df = get_batchs(test, batch_size=n_batchs, keep_features=cb_model.features)\nsub_cb = pd.concat(map(partial(predict, model=cb_model), tqdm_notebook(samples_df)))\nprint(len(sub_cb))\nprint(len(test))\nsub_cb.to_frame().to_csv('submission_cb.csv')","metadata":{"execution":{"iopub.status.busy":"2022-09-21T22:28:11.828327Z","iopub.execute_input":"2022-09-21T22:28:11.828796Z","iopub.status.idle":"2022-09-21T22:34:26.351652Z","shell.execute_reply.started":"2022-09-21T22:28:11.828750Z","shell.execute_reply":"2022-09-21T22:34:26.350507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2 style='color:#2d3a41;background-color:#c8e5f3;padding:10px'>  🫐   <b>   Weighted submission  </b>  </h2>","metadata":{}},{"cell_type":"code","source":"# weighted submission\nweighted_sub = (sub_lg*lgb_model.oof_score_ + sub_cb*cb_model.oof_score_)/(lgb_model.oof_score_+cb_model.oof_score_)\nweighted_sub.to_frame().to_csv('weighted_sub.csv')","metadata":{"execution":{"iopub.status.busy":"2022-09-21T22:34:26.353195Z","iopub.execute_input":"2022-09-21T22:34:26.353663Z","iopub.status.idle":"2022-09-21T22:34:29.294978Z","shell.execute_reply.started":"2022-09-21T22:34:26.353629Z","shell.execute_reply":"2022-09-21T22:34:29.293678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <h1 style='color:#2d3a41;background-color:#a3d3eb;padding:10px'>  📊    Predictions Distribution </h1>","metadata":{}},{"cell_type":"code","source":"def get_dirtibution(sub,threshold=0.5, model_name='lgb'):\n    target = (sub>threshold).astype(int)\n    target = target.value_counts(normalize=True)\n    target.rename(index={1:'Default',0:'Paid'},inplace=True)\n    fig=go.Figure()\n    fig.add_trace(go.Pie(labels=target.index, values=target*100,# hole=.45, \n                         showlegend=True,#sort=True, \n                         marker=dict(colors=list(theme_palette.values())),\n    #                     marker=dict(colors=color,line=dict(color=pal,width=2.5)),\n                         hovertemplate = \"%{label} Amex Acoounts: <b>%{value:.2f}</b>%<extra></extra>\"))\n    fig.update_layout(template=temp, \n                      title={\n                          \"text\":f'<b>Default VS Paid Transaction</b><BR />{model_name} predictions',\n                          \"x\":0.035,\n                          \"font_size\": 20,\n\n                      },\n\n                      uniformtext_minsize=15,# width=700,\n    #                  height=800,\n                      margin={'t':150, 'l':5})\n    return fig","metadata":{"execution":{"iopub.status.busy":"2022-09-21T22:36:53.527155Z","iopub.execute_input":"2022-09-21T22:36:53.528409Z","iopub.status.idle":"2022-09-21T22:36:53.537810Z","shell.execute_reply.started":"2022-09-21T22:36:53.528358Z","shell.execute_reply":"2022-09-21T22:36:53.536529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = get_dirtibution(sub_lg, model_name='LightGBM')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-21T22:36:54.228325Z","iopub.execute_input":"2022-09-21T22:36:54.228899Z","iopub.status.idle":"2022-09-21T22:36:54.265419Z","shell.execute_reply.started":"2022-09-21T22:36:54.228854Z","shell.execute_reply":"2022-09-21T22:36:54.263689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = get_dirtibution(sub_cb, model_name='CatBoost')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-21T22:37:09.797089Z","iopub.execute_input":"2022-09-21T22:37:09.797634Z","iopub.status.idle":"2022-09-21T22:37:09.829330Z","shell.execute_reply.started":"2022-09-21T22:37:09.797593Z","shell.execute_reply":"2022-09-21T22:37:09.828112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}