{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Import Libraries","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np \nimport pandas as pd\nimport seaborn as sns\n!pip install kds\nimport kds\nimport time\nimport gc\nimport matplotlib.pyplot as plt\nimport matplotlib.colors\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\nfrom plotly.offline import init_notebook_mode\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import StratifiedKFold \nfrom sklearn.metrics import roc_auc_score, roc_curve, auc\nfrom lightgbm import LGBMClassifier, early_stopping, log_evaluation\nimport warnings, gc\nfrom sklearn.metrics import accuracy_score, confusion_matrix, classification_report, fbeta_score, matthews_corrcoef\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.feature_selection import SelectKBest, chi2\nfrom sklearn.metrics import f1_score\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.metrics import confusion_matrix,accuracy_score,classification_report\nfrom sklearn.metrics import roc_auc_score,roc_curve\nfrom sklearn.metrics import precision_score,recall_score\nfrom sklearn.metrics import precision_recall_curve\nfrom sklearn.metrics import brier_score_loss,log_loss\n#\nimport lightgbm as lgb\nfrom lightgbm import LGBMClassifier\nfrom xgboost import XGBClassifier\n# Hyperopt\nimport hyperopt\nfrom functools import partial\nfrom hyperopt import fmin, hp, tpe, Trials, space_eval, STATUS_OK, STATUS_RUNNING\nfrom sklearn.base import TransformerMixin, BaseEstimator\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.metrics import fbeta_score, make_scorer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.feature_selection import SelectKBest\nfrom sklearn.feature_selection import chi2\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.feature_selection import RFE\nfrom sklearn.feature_selection import SelectFromModel\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import GridSearchCV, cross_val_score, StratifiedKFold, RepeatedStratifiedKFold, learning_curve\nfrom sklearn.model_selection._split import _BaseKFold, indexable, _num_samples\nfrom sklearn.utils.validation import _deprecate_positional_args\nfrom sklearn.calibration import calibration_curve,CalibratedClassifierCV\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier, ExtraTreesClassifier,BaggingClassifier, VotingClassifier, RandomTreesEmbedding\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.linear_model import RidgeClassifier, SGDClassifier, LogisticRegression\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.discriminant_analysis import LinearDiscriminantAnalysis\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.svm import SVC\nwarnings.filterwarnings(\"ignore\")\ninit_notebook_mode(connected=True)\n\ntemp=dict(layout=go.Layout(font=dict(family=\"Franklin Gothic\", size=12), \n                           height=500, width=1000))\n#display full dataframe\npd.set_option(\"display.max_rows\", None)\npd.set_option(\"display.max_columns\", None)\nwarnings.filterwarnings('ignore')\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-31T22:34:39.641400Z","iopub.execute_input":"2022-07-31T22:34:39.641880Z","iopub.status.idle":"2022-07-31T22:34:58.392491Z","shell.execute_reply.started":"2022-07-31T22:34:39.641839Z","shell.execute_reply":"2022-07-31T22:34:58.391028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### EDA Functions","metadata":{}},{"cell_type":"code","source":"'''Create function for providing summary statistics in a table'''\ndef resumetable(df):\n    \"\"\"\n    Objective: For a given dataframe this function provides information\n    regarding Missing and Unique values per column.\n\n    Input: param df: Dataframe to check the information.\n\n    Output: return summary: a dataframe with columns providing summary per column of the input dataframe.\n    \n    \"\"\"\n    df = df.copy()\n    print(f\"Dataset Shape: {df.shape}\")\n    summary = pd.DataFrame(df.dtypes,columns=['dtypes'])\n    summary = summary.reset_index()\n    summary['Name'] = summary['index']\n    summary = summary[['Name','dtypes']]\n    summary['Missing'] = df.isnull().sum().values\n    summary['Missing Percentage'] = df.isnull().sum().values/len(df)\n    summary['Uniques'] = df.nunique().values\n    return summary","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:34:58.395067Z","iopub.execute_input":"2022-07-31T22:34:58.395873Z","iopub.status.idle":"2022-07-31T22:34:58.405216Z","shell.execute_reply.started":"2022-07-31T22:34:58.395816Z","shell.execute_reply":"2022-07-31T22:34:58.404330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> ### Class to inspect missing values","metadata":{}},{"cell_type":"code","source":"class MissingValuesInsights:\n    @classmethod\n    def insights(cls,df,perc_limit):\n        \"\"\"\n        DP 3 pricing service orchestration from request processing,\n        to actual price selection, logging, benchmarking\n        and finally response creation.\n\n        :param df:\n        :param df:\n        :return missing: sorted dataframe in descending order by missing percentage of column names\n        \"\"\"\n        df = df.copy()\n        #Create missing value df\n        missing = cls.count_sort(df)\n        #Plot missing values\n        cls.plot(df,perc_limit)\n        return missing\n    '''Create function for plotting missing features'''\n    @classmethod\n    def plot(cls,df,perc_limit):\n        df = df.copy()\n        total = df.isnull().sum().sort_values(ascending=False)\n        percent = (df.isnull().sum()/df.isnull().count()).sort_values(ascending=False)\n        missing_data = pd.concat([total, percent], axis=1,join='outer', keys=['missing_count', 'percent'])\n        missing_data = missing_data.loc[missing_data['percent'] > perc_limit]\n        missing_data = missing_data.sort_values(by='missing_count')\n        missing_data = missing_data[['missing_count']].drop_duplicates()\n        #\n        ind = np.arange(missing_data.shape[0])\n        width = 0.2\n        fig, ax = plt.subplots(figsize=(15,5))\n        rects = ax.barh(ind, missing_data.missing_count.values, color='b')\n        ax.set_yticks(ind)\n        ax.set_yticklabels(missing_data.index.values, rotation='horizontal')\n        ax.set_xlabel(\"Missing Observations Count\")\n        ax.set_title(\"Missing Observations Count - Features\")\n        plt.show()\n    '''Create function for providing missing data insights'''\n    @classmethod\n    def count_sort(cls,df):\n        df = df.copy()\n        total = df.isnull().sum().sort_values(ascending=False)\n        percent = (df.isnull().sum()/df.isnull().count()).sort_values(ascending=False)\n        missing_data = pd.concat([total, percent], axis=1,join='outer', keys=['Total Missing Count', '% of Total Observations'])\n        missing_data = missing_data.loc[missing_data['% of Total Observations']>=0.05]\n        missing_data.index.name =' Feature'\n        missing_data = missing_data.reset_index()\n        return missing_data","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:34:58.407060Z","iopub.execute_input":"2022-07-31T22:34:58.408012Z","iopub.status.idle":"2022-07-31T22:34:58.429303Z","shell.execute_reply.started":"2022-07-31T22:34:58.407972Z","shell.execute_reply":"2022-07-31T22:34:58.427600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> ### Plot Target Variable","metadata":{}},{"cell_type":"code","source":"'''Plot target'''\ndef PlotTarget(df, target):\n    \"\"\"\n\n    Objective: For a given dataframe and defined target column plot target distribution (binary) showing both percentage and count.\n\n    Input: param df: Dataframe to perform aggregation.\n           param target: Target column name.\n\n    Output: return Plot of target distribution.\n\n    \"\"\"\n    total = len(df)\n    plt.figure(figsize=(12,7))\n\n    g = sns.countplot(x=target, data=df, color='blue')\n    g.set_title(f\"Target Distribution \\nTotal Customers: {total}\", \n                fontsize=20)\n    g.set_xlabel(\"Default?\", fontsize=18)\n    g.set_ylabel('Count', fontsize=18)\n    for p in g.patches:\n        height = p.get_height()\n        g.text(p.get_x()+p.get_width()/2.,\n                height + 4,\n                '{:1.2f}%'.format(height/total*100),\n                ha=\"center\", fontsize=15) \n    g.set_ylim(0, total *1.00)\n\n    plt.show()    \n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:34:58.433150Z","iopub.execute_input":"2022-07-31T22:34:58.433955Z","iopub.status.idle":"2022-07-31T22:34:58.446417Z","shell.execute_reply.started":"2022-07-31T22:34:58.433898Z","shell.execute_reply":"2022-07-31T22:34:58.445347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> ### Plot Numeric Features both for 0 and 1 and check differences","metadata":{}},{"cell_type":"code","source":"'''Plot Numeric Features vs target in order to assess differences between 1 and 0'''\ndef TargetvsNumeric(data, numeric, target, numeric_name):\n\n\n    df1 = data[(data[numeric] >= 0) & \n                                  (data[target] == 1)]\n    df2 = data[(data[numeric] >= 0) & \n                                  (data[target] == 0)]\n\n    #figure size\n    title = 'Distribution of ' + numeric_name\n    suptitle = numeric_name + ' Distribution'\n    xlabel = numeric_name + ' Range'\n    plt_title = \"Distribution of \" + numeric_name + \" by Target\"\n    plt.figure(figsize=(20,5))\n\n    plt.subplot(121)\n    plt.suptitle(suptitle, fontsize=20)\n    sns.distplot(data[(data[numeric] >= 0)][numeric], bins=24)\n    plt.title(title,fontsize=18)\n    plt.xlabel(xlabel,fontsize=15)\n    plt.ylabel(\"Probability\",fontsize=15)\n\n    plt.subplot(122)\n\n    sns.distplot(df1[numeric], bins=24, color='r', label='Yes')\n    sns.distplot(df2[numeric], bins=24, color='blue', label='No')\n    plt.title(plt_title,fontsize=18)\n    plt.xlabel(numeric_name,fontsize=15)\n    plt.ylabel(\"Probability\",fontsize=15)\n    plt.legend()\n\n\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:34:58.448403Z","iopub.execute_input":"2022-07-31T22:34:58.449872Z","iopub.status.idle":"2022-07-31T22:34:58.465795Z","shell.execute_reply.started":"2022-07-31T22:34:58.449816Z","shell.execute_reply":"2022-07-31T22:34:58.464375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Check working space","metadata":{}},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:34:58.467565Z","iopub.execute_input":"2022-07-31T22:34:58.468489Z","iopub.status.idle":"2022-07-31T22:34:58.493365Z","shell.execute_reply.started":"2022-07-31T22:34:58.468435Z","shell.execute_reply":"2022-07-31T22:34:58.492141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Metrics to optimize\n\n- ","metadata":{}},{"cell_type":"code","source":"# Copied from https://www.kaggle.com/code/thedevastator/lag-features-are-all-you-need\ndef amex_metric(y_true, y_pred):\n    labels = np.transpose(np.array([y_true, y_pred]))\n    labels = labels[labels[:, 1].argsort()[::-1]]\n    weights = np.where(labels[:,0]==0, 20, 1)\n    cut_vals = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n    gini = [0,0]\n    for i in [1,0]:\n        labels = np.transpose(np.array([y_true, y_pred]))\n        labels = labels[labels[:, i].argsort()[::-1]]\n        weight = np.where(labels[:,0]==0, 20, 1)\n        weight_random = np.cumsum(weight / np.sum(weight))\n        total_pos = np.sum(labels[:, 0] *  weight)\n        cum_pos_found = np.cumsum(labels[:, 0] * weight)\n        lorentz = cum_pos_found / total_pos\n        gini[i] = np.sum((lorentz - weight_random) * weight)\n    return 0.5 * (gini[1]/gini[0] + top_four)\n\ndef amex_metric_np(preds, target):\n    indices = np.argsort(preds)[::-1]\n    preds, target = preds[indices], target[indices]\n    weight = 20.0 - target * 19.0\n    cum_norm_weight = (weight / weight.sum()).cumsum()\n    four_pct_mask = cum_norm_weight <= 0.04\n    d = np.sum(target[four_pct_mask]) / np.sum(target)\n    weighted_target = target * weight\n    lorentz = (weighted_target / weighted_target.sum()).cumsum()\n    gini = ((lorentz - cum_norm_weight) * weight).sum()\n    n_pos = np.sum(target)\n    n_neg = target.shape[0] - n_pos\n    gini_max = 10 * n_neg * (n_pos + 20 * n_neg - 19) / (n_pos + 20 * n_neg)\n    g = gini / gini_max\n    return 0.5 * (g + d)\n\ndef plot_roc(y_val,y_prob):\n    colors=px.colors.qualitative.Prism\n    fig=go.Figure()\n    fig.add_trace(go.Scatter(x=np.linspace(0,1,11), y=np.linspace(0,1,11), \n                             name='Random Chance',mode='lines', showlegend=False,\n                             line=dict(color=\"Black\", width=1, dash=\"dot\")))\n    for i in range(len(y_val)):\n        y=y_val[i]\n        prob=y_prob[i]\n        fpr, tpr, _ = roc_curve(y, prob)\n        roc_auc = auc(fpr,tpr)\n        fig.add_trace(go.Scatter(x=fpr, y=tpr, line=dict(color=colors[::-1][i+1], width=3), \n                                 hovertemplate = 'True positive rate = %{y:.3f}<br>False positive rate = %{x:.3f}',\n                                 name='Fold {}:  Gini = {:.3f}, AUC = {:.3f}'.format(i+1, gini[i],roc_auc)))\n    fig.update_layout(template=temp, title=\"Cross-Validation ROC Curves\", \n                      hovermode=\"x unified\", width=700,height=600,\n                      xaxis_title='False Positive Rate (1 - Specificity)',\n                      yaxis_title='True Positive Rate (Sensitivity)',\n                      legend=dict(orientation='v', y=.07, x=1, xanchor=\"right\",\n                                  bordercolor=\"black\", borderwidth=.5))\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:34:58.495282Z","iopub.execute_input":"2022-07-31T22:34:58.496064Z","iopub.status.idle":"2022-07-31T22:34:58.528166Z","shell.execute_reply.started":"2022-07-31T22:34:58.496015Z","shell.execute_reply":"2022-07-31T22:34:58.527154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Import train valid test","metadata":{}},{"cell_type":"code","source":"def import_train_data():\n    ### train/test\n    start = time.time()\n    train = pd.read_feather('/kaggle/input/amexfeather/train_data.ftr')\n    #keep last statement for each customer to train the model\n    train = train.groupby('customer_ID').tail(1)\n    end = time.time()\n    print(\"Train Data time to import:\")\n    print(end - start)\n    print(\"The training data begins on {} and ends on {}.\".format(train['S_2'].min().strftime('%m-%d-%Y'),train['S_2'].max().strftime('%m-%d-%Y')))\n    print(\"There are {:,.0f} customers in the training set and {} features.\".format(train.shape[0],train.shape[1]))\n    \n    return train","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:34:58.529411Z","iopub.execute_input":"2022-07-31T22:34:58.530242Z","iopub.status.idle":"2022-07-31T22:34:58.544305Z","shell.execute_reply.started":"2022-07-31T22:34:58.530203Z","shell.execute_reply":"2022-07-31T22:34:58.543095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def import_test_data():\n    ### train/test\n    start = time.time()\n    test = pd.read_feather('/kaggle/input/amexfeather/test_data.ftr')\n    test = test.groupby('customer_ID').tail(1).set_index('customer_ID')\n    end = time.time()\n    print(\"Test Data time to import:\")\n    print(end - start)\n    print(\"\\nThe test data begins on {} and ends on {}.\".format(test['S_2'].min().strftime('%m-%d-%Y'),test['S_2'].max().strftime('%m-%d-%Y')))\n    print(\"There are {:,.0f} customers in the test set and {} features.\".format(test.shape[0],test.shape[1]))\n    \n    return test","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:34:58.546060Z","iopub.execute_input":"2022-07-31T22:34:58.546844Z","iopub.status.idle":"2022-07-31T22:34:58.557388Z","shell.execute_reply.started":"2022-07-31T22:34:58.546794Z","shell.execute_reply":"2022-07-31T22:34:58.556078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''Function for lgbm evaluation metric similar to roc auc'''\ndef gini(y, pred):\n    g = np.asarray(np.c_[y, pred, np.arange(len(y)) ], dtype=np.float)\n    g = g[np.lexsort((g[:,2], -1*g[:,1]))]\n    gs = g[:,0].cumsum().sum() / g[:,0].sum()\n    gs -= (len(y) + 1) / 2.\n    return gs / len(y)\n    \ndef gini_lgb(preds,dtrain):\n    y = list(dtrain.get_label())\n    score = gini(y,preds) / gini(y,y)\n    return 'gini', score, True        \n        \n'''Function for lgbm evaluation metric'''       \ndef lgb_fbeta_score(y_hat, data):\n    y_true = data.get_label()\n    y_hat = np.round(y_hat) # scikits f1 doesn't like probabilities\n    return 'fbeta', fbeta_score(y_true, y_hat, average='binary', beta=0.5), True        \n\n'''Function for lgbm evaluation metric'''       \ndef lgb_precision_score(y_hat, data):\n    y_true = data.get_label()\n    y_hat = np.round(y_hat)\n    return 'precision', precision_score(y_true, y_hat), True   \n\n# Copied from https://www.kaggle.com/code/thedevastator/lag-features-are-all-you-need\n\ndef lgb_amex_metric(y_pred, y_true):\n    y_true = y_true.get_label()\n    return 'amex_metric', amex_metric(y_true, y_pred), True\n\ndef mcc(tp, tn, fp, fn):\n    sup = tp * tn - fp * fn\n    inf = (tp + fp) * (tp + fn) * (tn + fp) * (tn + fn)\n    if inf==0:\n        return 0\n    else:\n        return sup / np.sqrt(inf)\n    \ndef eval_mcc(y_true, y_prob, show=False):\n    idx = np.argsort(y_prob)\n    y_true_sort = y_true[idx]\n    n = y_true.shape[0]\n    nump = 1.0 * np.sum(y_true) # number of positive\n    numn = n - nump # number of negative\n    tp = nump\n    tn = 0.0\n    fp = numn\n    fn = 0.0\n    best_mcc = 0.0\n    best_id = -1\n    prev_proba = -1\n    best_proba = -1\n    mccs = np.zeros(n)\n    for i in range(n):\n        # all items with idx < i are predicted negative while others are predicted positive\n        # only evaluate mcc when probability changes\n        proba = y_prob[idx[i]]\n        if proba != prev_proba:\n            prev_proba = proba\n            new_mcc = mcc(tp, tn, fp, fn)\n            if new_mcc >= best_mcc:\n                best_mcc = new_mcc\n                best_id = i\n                best_proba = proba\n        mccs[i] = new_mcc\n        if y_true_sort[i] == 1:\n            tp -= 1.0\n            fn += 1.0\n        else:\n            fp -= 1.0\n            tn += 1.0\n    if show:\n        y_pred = (y_prob >= best_proba).astype(int)\n        score = matthews_corrcoef(y_true, y_pred)\n        print(score, best_mcc)\n        plt.plot(mccs)\n        return best_proba, best_mcc, y_pred\n    else:\n        return best_mcc\n    \n'''Function for lgbm evaluation metric'''     \ndef mcc_eval(preds, data):\n    y_true = data.get_label()\n    best_mcc = eval_mcc(y_true, preds)\n    return 'MCC', best_mcc, True\n\nmcc_scorer = make_scorer(matthews_corrcoef)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:34:58.563441Z","iopub.execute_input":"2022-07-31T22:34:58.564314Z","iopub.status.idle":"2022-07-31T22:34:58.593453Z","shell.execute_reply.started":"2022-07-31T22:34:58.564259Z","shell.execute_reply":"2022-07-31T22:34:58.591882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature Engineering","metadata":{}},{"cell_type":"markdown","source":"### Label Encoding","metadata":{}},{"cell_type":"code","source":"class FeatureEngineering:\n    @classmethod\n    def pipeline(cls, df, init_cat_cols):\n        \"\"\"\n        Objective: Function which encapsulates all previously defined steps for Feature Engineering.\n        This function defines the order which each function should be performed.\n        Also, except the extra features both the imputation of Nas is included.\n\n        Input: param df: Dataframe to perform Feature Engineering.\n\n        Output: return Dataframe after Feature Engineering.\n\n        \"\"\"\n        start = time.time()\n        df = df.copy()\n        # Fix categorical columns\n        df = cls.fix_data_types(df, init_cat_cols)\n        # Rest Features\n        df = cls.rest_features(df, init_cat_cols)\n        # drop cols\n        df = cls.drop_cols_na(df)\n        gc.collect()\n        # fill na\n        df = cls.fill_na(df)\n        gc.collect()\n        # Drop na\n        df = cls.drop_na_category(df, init_cat_cols)\n        gc.collect()\n        print(\"Total time for Feature Engineering is: \")\n        end = time.time()\n        print((end - start) / 60)\n        return df\n    \n    def rest_features(df, cat_cols):\n        \"\"\"\n\n        Objective: Create extra features to add into the model.\n\n        Input: param df: Dataframe to perform Feature Engineering.\n\n        Output: return Dataframe after Feature Engineering.\n\n        \"\"\"\n        start = time.time()\n        df = df.copy()\n        print('Starting Feature Engineering...')\n        df['P_2-B_9'] = df['P_2'] * df['B_9']\n        df['P_2-D_43'] = df['P_2'] * df['D_43']\n        df['B_4-S_3'] = df['B_4'] * df['S_3']\n        df['D_46-B_3'] = df['D_46'] * df['B_3']\n        df['D_61-D_41'] = df['D_61'] * df['D_41']\n        df['B_5-D_59'] = df['B_5'] * df['D_59']\n        df['D_48-D_41'] = df['D_48'] * df['D_41']\n        df['D_133-D_140'] = df['D_133'] * df['D_140']\n        df['R_1-D_56'] = df['R_1'] * df['D_56']\n        df['R_7-R_11'] = df['R_7'] * df['R_11']\n        df['D_121-D_119'] = df['D_121'] * df['D_119']\n        df['D_60-D_78'] = df['D_60'] * df['D_78']\n        df['D_91-D_51'] = df['D_91'] * df['D_51']\n        df['S_12-S_11'] = df['S_12'] * df['S_11']\n        df['R_7-S_11'] = df['R_7'] * df['S_27']\n        df['R_6-R_20'] = df['R_6'] * df['R_20']\n        \n        \n        df['D_41-D_51'] = df['D_41'] * df['D_51']\n        df['D_133-D_112'] = df['D_133'] * df['D_112']\n        df['B_8-D_106'] = df['B_8'] * df['D_106']\n        df['B_3-B_23'] = df['B_3'] * df['B_23']\n        df['S_8-B_17'] = df['S_8'] * df['B_17']\n        df['S_12-D_106'] = df['S_12'] * df['D_106']\n        df['S_8-B_17'] = df['S_8'] * df['B_17']\n        df['S_11-R_5'] = df['S_11'] * df['R_5']\n        df['B_1-S_3'] = df['B_1'] * df['S_3']\n        df['R_26-S_7'] = df['R_26'] * df['S_7']\n\n        \n        \n        df['P_2/B_9'] = df['P_2'] / df['B_9']\n        df['P_2/S_7'] = df['P_2'] / df['S_7']\n        df['P_3/R_3'] = df['P_3'] / df['R_3']\n        df['P_3/B_3'] = df['P_3'] / df['B_3']\n        df['S_3/B_3'] = df['S_3'] / df['B_3']\n        df['S_12/D_51'] = df['S_12'] / df['D_51']\n        df['B_4/B_2'] = df['B_4'] / df['B_2']\n        df['D_41/D_46'] = df['D_41'] / df['D_46']\n        df['D_39/D_59'] = df['D_39'] / df['D_59']\n        df['D_124/D_122'] = df['D_124'] / df['D_122']\n        \n        df['R_1+R_7+R_11'] = df['R_1'] + df['R_7'] + df['R_11']\n        df['B_4+B_3+B_9'] = df['B_4'] + df['B_3'] + df['B_9']\n        df['D_46+D_41+D_56'] = df['D_46'] + df['D_41'] + df['D_56']\n        df['S_22+S_23+S_24'] = df['S_22'] + df['S_23'] + df['S_24']\n        df['S_25+S_26'] = df['S_25'] + df['S_26']\n        df['S_12+D_51'] = df['S_12'] + df['D_51']\n        df['R_1+R_2+R_3'] = df['R_1'] + df['R_2'] + df['R_3']\n        df['B_1+B_2+B_3'] = df['B_1'] + df['B_2'] + df['B_3']\n        df['R_7+R_8+R_10'] = df['R_7'] + df['R_8'] + df['R_10']\n        df['B_40+B_41'] = df['B_40'] + df['B_41']\n        df['B_18+B_19'] = df['B_18'] + df['B_19']\n        df['D_59+D_60+R_10'] = df['D_59'] + df['D_60'] + df['D_61']\n        \n        # Cat cols\n        df['D_63-D_64'] = df['D_63'] + df['D_64']\n        df['D_63-D_68'] = df['D_63'] + df['D_68']\n        df['D_63-B_30'] = df['D_63'] + df['B_30']\n        df['D_63-B_38'] = df['D_63'] + df['B_38']\n        df['D_63-D_114'] = df['D_63'] + df['D_114']\n        df['D_63-D_116'] = df['D_63'] + df['D_116']\n        df['D_63-D_117'] = df['D_63'] + df['D_117']\n        df['D_63-D_120'] = df['D_63'] + df['D_120']\n        df['D_63-D_126'] = df['D_63'] + df['D_126']\n        df['D_64-D_68'] = df['D_64'] + df['D_68']\n        df['D_64-B_30'] = df['D_64'] + df['B_30']\n        \n        df['D_64-D_68'] = df['D_64'] + df['D_68']\n        df['D_64-B_30'] = df['D_64'] + df['B_30']\n        df['D_64-B_38'] = df['D_64'] + df['B_38']\n        df['D_64-D_114'] = df['D_64'] + df['D_114']\n        df['D_64-D_116'] = df['D_64'] + df['D_116']\n        df['D_64-D_117'] = df['D_64'] + df['D_117']\n        df['D_64-D_120'] = df['D_64'] + df['D_120']\n        df['D_64-D_126'] = df['D_64'] + df['D_126']\n        df['D_64-D_68'] = df['D_64'] + df['D_68']\n        df['D_64-B_30'] = df['D_64'] + df['B_30']\n        \n        df['D_68-B_30'] = df['D_68'] + df['B_30']\n        df['D_68-B_38'] = df['D_68'] + df['B_38']\n        df['D_68-D_114'] = df['D_68'] + df['D_114']\n        df['D_68-D_116'] = df['D_68'] + df['D_116']\n        df['D_68-D_117'] = df['D_68'] + df['D_117']\n        df['D_68-D_120'] = df['D_68'] + df['D_120']\n        df['D_68-D_126'] = df['D_68'] + df['D_126']\n\n\n        gc.collect()\n        end = time.time()\n        print((end - start) / 60)\n        print(\"Rest Features time:\")\n        return df\n    \n    '''Drop columns which have above a certain percentage na values'''\n\n    @classmethod\n    def drop_cols_na(cls, df, pct = 99):\n        df = df.copy()\n        # Below code gives percentage of null in every column\n        null_percentage = df.isnull().sum()/df.shape[0]*100\n\n        # Below code gives list of columns having more than x% null\n        col_to_drop = null_percentage[null_percentage>pct].keys()\n        print(\"Cols to be dropped: \")\n        print(len(col_to_drop))\n        output_df = df.drop(col_to_drop, axis=1)\n        \n        return output_df\n    \n\n    '''Impute NA values with distinct values'''\n\n    @classmethod\n    def fill_na(cls, df):\n        \"\"\"\n\n        Objective: Impute features having NA values.\n\n        Input: param df: Main Dataframe.\n\n        Output: return df Imputed Dataframe after NA treatment.\n\n        \"\"\"\n        start = time.time()\n        df = df.copy()\n        numeric_columns = df.select_dtypes([np.number]).columns\n        # for every column which is coming from na in return need to fill for na values\n        for col in numeric_columns:\n            # fill all na columns\n            df[col] = np.where((df[col].isna()),\n                               (df[col].min() - 1),\n                               df[col])\n\n        print(\"FillNa: \")\n        end = time.time()\n        print((end - start) / 60)\n        return df\n\n    @classmethod\n    def drop_na_category(cls, df, cat_cols):\n        \"\"\"\n\n        Objective: Drop rows for certain columns having NA values.\n        Case is that for certain columns on prediction level\n        there would never be cases with NA.\n\n        Input: param df: Main Dataframe.\n\n        Output: return df Dataframe after dropping rows with NA.\n\n        \"\"\"\n        start = time.time()\n        df = df.copy()\n        if len(cat_cols) < 1:\n            print(\"No Categorical\")\n        else:\n            # Drop rows with NA\n            df = df.dropna(subset=cat_cols)\n            # Resetting the indices using df.reset_index()\n            df = df.reset_index(drop=True)\n            print(\"Drop NA: \")\n            end = time.time()\n            print((end - start) / 60)\n        return df\n    \n\n    '''Fix all columns datatypes to correct ones'''\n\n    @classmethod\n    def fix_data_types(cls, df, cat_cols):\n        \"\"\"\n\n        Objective: Define datatypes for certain columns.\n        Fix datatypes to correct ones.\n\n        Input: param df: Dataframe to fix datatypes.\n\n        Output: return Dataframe after fixing datatypes.\n\n        \"\"\"\n        start = time.time()\n        df = df.copy()\n        if len(cat_cols) < 1:\n            print(\"No Categorical\")\n        else:\n            # convert categorical columns to str\n            for col in cat_cols:\n                df[col] = df[col].astype(str)\n            print(\"Total time for Fixing Data Types is: \")\n            end = time.time()\n            print((end - start) / 60)\n        return df\n    \n    @classmethod\n    def keep_columns(cls, df, keep_cols):\n        df = df.copy()\n        df = df[keep_cols]\n        return df\n    \n    @classmethod\n    def keep_last_statement(cls, df):\n        df = df.copy()\n        df = df.groupby('customer_ID').tail(1).set_index('customer_ID')\n        return df\n    \n    \n    @classmethod\n    def AggCol(cls, df, group_by, col, agg):\n           \"input is df, groupbycol(s), desired aggregation. Output should be a column\"\n           col = df.groupby(group_by)[col].transform(agg)\n           return col","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:34:58.596023Z","iopub.execute_input":"2022-07-31T22:34:58.596601Z","iopub.status.idle":"2022-07-31T22:34:58.656285Z","shell.execute_reply.started":"2022-07-31T22:34:58.596551Z","shell.execute_reply":"2022-07-31T22:34:58.655245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class LabelEncoding:\n    \n    @classmethod\n    def pipeline(cls,\n                 df,\n                 date_column,\n                 target_name,\n                 date_train_valid,\n                 date_train,\n                 date_valid,\n                 date_test,\n                 cat_cols,\n#                  drop_cols\n                ):\n        # Split main dataframe to train,test\n        train_valid,test = cls.split_train_test(df,\n                                          date_column,\n                                          date_train_valid,\n                                          date_test)\n        # Label Encoding \n        label_encoder = LabelEncodingColumns(cat_cols)\n        label_encoder_dictionary = label_encoder.final_dictionary\n        train_valid_enc = label_encoder.fit_transform(train_valid)\n        test_enc = label_encoder.fit_transform(test)\n        gc.collect()\n        # Impute inf values\n        train_valid_enc = cls.impute_inf_nas(train_valid_enc)\n        test_enc = cls.impute_inf_nas(test_enc)\n        gc.collect()\n        # Split Train Valid\n        train_enc,valid_enc = cls.split_train_valid(train_valid_enc,\n                                                    date_column,\n                                                    date_train,\n                                                    date_valid)\n        gc.collect()\n        # Drop certain columns before ML procedure\n#         train_enc = cls.drop_columns(train_enc,drop_cols)\n#         valid_enc = cls.drop_columns(valid_enc,drop_cols)\n#         test_enc = cls.drop_columns(test_enc,drop_cols)\n        print (\"After FE\")\n        print (\"Train: \",train_enc.shape[0],\"observations, and \",train_enc.shape[1],\"features\")\n        print (\"Valid: \",valid_enc.shape[0],\"observations, and \",valid_enc.shape[1],\"features\")\n        print (\"Test: \",test_enc.shape[0],\"observations, and \",test_enc.shape[1],\"features\")\n        return train_enc, valid_enc, test_enc, label_encoder_dictionary\n\n    @classmethod\n    def drop_columns(cls,\n                     df,\n                     dropcols):\n        \"\"\"\n\n        Objective: Drop columns not needed for final model training.\n\n        Input: param df: Dataframe to perform change.\n                param dropcols: List of columns to be dropped.\n\n        Output: return Dataframe after dropping columns.\n\n\n        \"\"\"\n        start = time.time()\n        #keep certain columns for ML process \n        # drop list of columns \n        df = df.drop(dropcols, axis=1)\n        # drop duplicates\n        df = df.drop_duplicates()\n        print(\"Drop Columns: \")\n        end = time.time()\n        print((end - start)/60) \n        return df\n\n    @classmethod\n    def split_train_test(cls,\n                         df,\n                         date_column,\n                         date_train_valid,\n                         date_test):\n        \"\"\"\n\n        Objective: Split Dataframe into certain time periods.\n        First split is between Train/Valid and Test datasets.\n\n        Input: param df: Dataframe to perform split.\n                param date_train_valid: Date to define train/valid split.\n                param date_test: Date to define test split.\n\n        Output: return df_train,df_test: Return train/valid and test dataset.\n\n\n        \"\"\"\n        df = df.copy()\n        # set train dataset based on search date\n        df_train = df[df[date_column] <= date_train_valid]\n        # set test dataset based on search date\n        df_test = df[df[date_column] >= date_test]\n\n        return df_train,df_test\n\n    @classmethod\n    def split_train_valid(cls,\n                          df,\n                          date_column,\n                          date_train,\n                          date_valid):\n        \"\"\"\n\n         Objective: Split Dataframe into certain time periods.\n         First split is between Train/Valid and Test datasets.\n\n         Input: param df: Dataframe to perform split.\n                param date_train_valid: Date to define train/valid split.\n                param date_test: Date to define test split.\n\n         Output: return df_train,df_test: Return train/valid and test dataset.\n\n\n        \"\"\"\n        df = df.copy()\n        # set train dataset based on search date\n        df_train = df[df[date_column] <= date_train]\n        # set valid dataset based on search date\n        df_valid = df[df[date_column] >= date_valid]\n\n        return df_train,df_valid\n\n    @classmethod\n    def impute_inf_nas(cls,\n                       df):\n        \"\"\"\n\n        Objective: Impute NA values for all columns into zero.\n\n        Input: param df: Dataframe to perform split.\n\n        Output: return df: Return Dataframe after NA impute.\n\n\n        \"\"\"\n        start = time.time()\n        df = df.copy()\n        #inf/-inf\n        df = df.replace((np.inf, -np.inf, np.nan), 0).reset_index(drop=True)\n        print(\"Impute Inf and NAs: \")\n        end = time.time()\n        print((end - start)/60) \n        return df\n    \nclass LabelEncodingColumns(BaseEstimator, TransformerMixin):\n\n    def __init__(self, cols=None):\n        self.cols = cols\n        self.label_encoders = {col: LabelEncoder() for col in cols}\n        self.feauture_dictionary = {}\n        self.final_dictionary = {}\n\n    def fit(self, df, y=None):\n        for col in self.cols:\n            self.label_encoders[col].fit(df[col].append(pd.Series(['unknown'])))\n            self.feauture_dictionary[col] = self.label_encoders[col].classes_\n\n            le_name_mapping = dict(zip(self.label_encoders[col].classes_,\n                                       self.label_encoders[col].transform(self.label_encoders[col].classes_)))\n            self.final_dictionary.update({col: le_name_mapping})\n\n        return self\n\n    def transform(self, df_, y=None):\n\n        df = df_.copy()\n\n        label_enc_dict = {}\n\n        for col in self.cols:\n            temp_dic = dict(zip(self.label_encoders[col].classes_, self.label_encoders[col].classes_))\n            df[col] = df[col].map(temp_dic)\n            df[col].fillna('unknown', inplace=True)\n            label_enc_dict[col] = self.label_encoders[col].transform(df[col])\n            del temp_dic\n\n        # The index of the resulting DataFrame should be assigned and\n        # equal to the one of the original DataFrame. Otherwise, upon\n        # concatenation NaNs will be introduced.\n        labelenc_cols = pd.DataFrame(label_enc_dict, index=df.index)\n\n        for col in self.cols:\n            df[col] = labelenc_cols[col]\n\n        return df","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:34:58.658081Z","iopub.execute_input":"2022-07-31T22:34:58.658891Z","iopub.status.idle":"2022-07-31T22:34:58.694443Z","shell.execute_reply.started":"2022-07-31T22:34:58.658833Z","shell.execute_reply":"2022-07-31T22:34:58.693197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Split Train Test","metadata":{}},{"cell_type":"code","source":"class SplitTrainTest:\n    \n    @classmethod\n    def pipeline(cls,\n                 df,\n                 date_column,\n                 target_name,\n                 date_train_valid,\n                 date_train,\n                 date_valid,\n                 date_test,\n#                  drop_cols\n                ):\n        # Split main dataframe to train,test\n        train_valid_df, test_df = cls.split_train_test(df,\n                                          date_column,\n                                          date_train_valid,\n                                          date_test)\n        gc.collect()\n\n        # Impute inf values\n        train_valid_df = cls.impute_inf_nas(train_valid_df)\n        test_df = cls.impute_inf_nas(test_df)\n        # Split Train Valid\n        train_df,valid_df = cls.split_train_valid(train_valid_df, \n                                                  date_column,\n                                                  date_train, \n                                                  date_valid)\n        gc.collect()\n        # Drop certain columns before ML procedure\n#         train_df = cls.drop_columns(train_df, drop_cols)\n#         valid_df = cls.drop_columns(valid_df, drop_cols)\n#         test_df = cls.drop_columns(test_df, drop_cols)\n        gc.collect()\n        print (\"After FE\")\n        print (\"Train: \",train_df.shape[0],\"observations, and \",train_df.shape[1],\"features\")\n        print (\"Valid: \",valid_df.shape[0],\"observations, and \",valid_df.shape[1],\"features\")\n        print (\"Test: \",test_df.shape[0],\"observations, and \",test_df.shape[1],\"features\")\n        return train_df, valid_df, test_df\n        \n    @classmethod\n    def drop_columns(cls,\n                     df,\n                     dropcols):\n        \"\"\"\n\n        Objective: Drop columns not needed for final model training.\n\n        Input: param df: Dataframe to perform change.\n                param dropcols: List of columns to be dropped.\n\n        Output: return Dataframe after dropping columns.\n\n\n        \"\"\"\n        start = time.time()\n        #keep certain columns for ML process \n        # drop list of columns \n        df = df.drop(dropcols, axis=1)\n        # drop duplicates\n        df = df.drop_duplicates()\n        print(\"Drop Columns: \")\n        end = time.time()\n        print((end - start)/60) \n        return df\n\n    @classmethod\n    def split_train_test(cls,\n                         df,\n                         date_column,\n                         date_train_valid,\n                         date_test):\n        \"\"\"\n\n        Objective: Split Dataframe into certain time periods.\n        First split is between Train/Valid and Test datasets.\n\n        Input: param df: Dataframe to perform split.\n                param date_train_valid: Date to define train/valid split.\n                param date_test: Date to define test split.\n\n        Output: return df_train,df_test: Return train/valid and test dataset.\n\n\n        \"\"\"\n        df = df.copy()\n        # set train dataset based on search date\n        df_train = df[df[date_column] <= date_train_valid]\n        # set test dataset based on search date\n        df_test = df[df[date_column] >= date_test]\n\n        return df_train, df_test\n\n    @classmethod\n    def split_train_valid(cls,\n                          df,\n                          date_column,\n                          date_train,\n                          date_valid):\n        \"\"\"\n\n         Objective: Split Dataframe into certain time periods.\n         First split is between Train/Valid and Test datasets.\n\n         Input: param df: Dataframe to perform split.\n                param date_train_valid: Date to define train/valid split.\n                param date_test: Date to define test split.\n\n         Output: return df_train,df_test: Return train/valid and test dataset.\n\n\n        \"\"\"\n        df = df.copy()\n        # set train dataset based on search date\n        df_train = df[df[date_column] <= date_train]\n        # set valid dataset based on search date\n        df_valid = df[df[date_column] >= date_valid]\n\n        return df_train,df_valid\n\n    @classmethod\n    def impute_inf_nas(cls,\n                       df):\n        \"\"\"\n\n        Objective: Impute NA values for all columns into zero.\n\n        Input: param df: Dataframe to perform split.\n\n        Output: return df: Return Dataframe after NA impute.\n\n\n        \"\"\"\n        start = time.time()\n        df = df.copy()\n        #inf/-inf\n        df = df.replace((np.inf, -np.inf, np.nan), 0).reset_index(drop=True)\n        print(\"Impute Inf and NAs: \")\n        end = time.time()\n        print((end - start)/60) \n        return df\n        ","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:34:58.696537Z","iopub.execute_input":"2022-07-31T22:34:58.697321Z","iopub.status.idle":"2022-07-31T22:34:58.720378Z","shell.execute_reply.started":"2022-07-31T22:34:58.697271Z","shell.execute_reply":"2022-07-31T22:34:58.719247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Custom Cross Validation","metadata":{}},{"cell_type":"code","source":"class CustomCrossValidation:\n\n    @classmethod\n    def split(cls,\n              X: pd.DataFrame,\n              y: np.ndarray = None,\n              groups: np.ndarray = None):\n        \"\"\"Returns to a grouped time series split generator.\"\"\"\n        assert len(X) == len(groups),  (\n            \"Length of the predictors is not\"\n            \"matching with the groups.\")\n        # The min max index must be sorted in the range\n        for group_idx in range(groups.min(), groups.max()):\n\n            training_group = group_idx\n            # Gets the next group right after\n            # the training as test\n            test_group = group_idx + 1\n            training_indices = np.where(\n                groups == training_group)[0]\n            test_indices = np.where(groups == test_group)[0]\n            if len(test_indices) > 0:\n                # Yielding to training and testing indices\n                # for cross-validation generator\n                yield training_indices, test_indices","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:34:58.722555Z","iopub.execute_input":"2022-07-31T22:34:58.723314Z","iopub.status.idle":"2022-07-31T22:34:58.737827Z","shell.execute_reply.started":"2022-07-31T22:34:58.723262Z","shell.execute_reply":"2022-07-31T22:34:58.736625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Prepare Model","metadata":{}},{"cell_type":"code","source":"class PrepareModel:\n    \n    '''Concentrate ML Pipeline into one function and check results'''  \n    @classmethod\n    def pipeline(cls,\n                 train,\n                 valid,\n                 test,\n                 date_column,\n                 target_name,\n                 top_features,\n                 keep_features,\n                 scoring,\n                 id_level):\n        #Copy dfs\n        train = train.copy()\n        valid = valid.copy()\n        test = test.copy()\n        # Necessary for Time Based CrossValidation\n        groups_experiment = cls.group_experiment(train, date_column)\n        #Drop search date\n        train = cls.drop_search_date(train, date_column)\n        valid = cls.drop_search_date(valid, date_column)\n        test = cls.drop_search_date(test, date_column)\n        #Variable Selection\n        train_best = cls.variable_selection(train,\n                                            top_features,\n                                            id_level,\n                                            target_name,\n                                            keep_features)\n        gc.collect()\n        #keep those features only\n        feat_list= list(train_best.columns)\n        print(\"Selected Features are: \")\n        # print chosen vars \n        print(feat_list)\n        # Split train test\n        x_train, x_test, y_train, y_test = cls.model_creation(train,\n                                                              valid,\n                                                              target_name,\n                                                              feat_list,\n                                                              id_level)\n        gc.collect()\n        # Out of time sample x,y\n        x_oot, y_oot = cls.CreateOos(test, target_name, id_level, feat_list)\n        gc.collect()\n        # Fit candidate models and check performance\n#         clfs = cls.candidate_models(x_train,y_train,x_test,y_test,groups_experiment,scoring)\n        return x_train, x_test, y_train, y_test,x_oot, y_oot, feat_list, groups_experiment\n    \n    \n    '''Convert main dataframe to train test split ready to train/validate model'''\n    @classmethod\n    def CreateOos(cls,\n                  df,\n                  target_name,\n                  id_level,\n                  selected_features):\n        \"\"\"\n        Return 2 dataframes for x and y\n        param df:\n            dataframe to split into x and y oos\n        return:\n            return 2 frames\n        \"\"\"\n        #set x and y\n        y = df[target_name].values\n        df = df.set_index(id_level)\n        x = df.drop([target_name], axis=1)\n        x = x[selected_features]\n\n        return x,y\n    \n    '''Create an ordered numpy object for time based cross validation step''' \n    @classmethod\n    def group_experiment(cls, df, date_column):\n        df = df.copy()\n        #sort values first\n        df = df.sort_values([date_column])\n\n        search_date_df = pd.DataFrame(df[date_column].drop_duplicates())\n        search_date_df = search_date_df.sort_values([date_column]).reset_index()\n        #drop column\n        search_date_df = search_date_df.drop(['index'], axis=1)\n        search_date_df.rename(columns={ search_date_df.columns[0]: date_column }, inplace = True)\n        #sort them with ascending order\n        search_date_df['order'] = 0\n\n        for i in range(len(search_date_df)):\n             search_date_df.order[i] = i + 1\n\n        # join with data\n        #bring in the order of observations\n        df = pd.merge(df,search_date_df, on=[date_column]) \n\n        #\n        order_numpy = df['order'].to_numpy()\n        return order_numpy\n\n    '''Fit consecutive candidate ML models and decide upon which is best'''  \n    @classmethod\n    def candidate_models(cls,\n                        x_train,\n                        y_train,\n                        x_test,\n                        y_test,\n                        groups_experiment,\n                        scoring):\n\n\n            clfs = []\n\n\n            clfs.append((\"LogReg\",\n                 Pipeline([(\"Scaler\", StandardScaler()),\n                           (\"LogReg\", LogisticRegression(class_weight='balanced'))])))\n\n            clfs.append((\"XGBClassifier\",\n                         Pipeline([(\"Scaler\", StandardScaler()),\n                                   (\"XGB\", XGBClassifier(scale_pos_weight=10))])))\n\n            clfs.append((\"DecisionTreeClassifier\",\n                         Pipeline([(\"Scaler\", StandardScaler()),\n                                   (\"DecisionTrees\", DecisionTreeClassifier(class_weight='balanced'))])))\n\n            clfs.append((\"RandomForestClassifier\",\n                         Pipeline([(\"Scaler\", StandardScaler()),\n                                   (\"RandomForest\", RandomForestClassifier(class_weight='balanced'))])))\n\n            clfs.append((\"LGBMClassifier\",\n                         Pipeline([(\"Scaler\", StandardScaler()),\n                                   (\"GradientBoosting\",\n                                    LGBMClassifier(class_weight='balanced'))])))\n\n            clfs.append((\"RidgeClassifier\",\n                         Pipeline([(\"Scaler\", StandardScaler()),\n                                   (\"RidgeClassifier\", RidgeClassifier(class_weight='balanced'))])))\n\n            clfs.append((\"BaggingRidgeClassifier\",\n                         Pipeline([(\"Scaler\", StandardScaler()),\n                                   (\"BaggingClassifier\", BaggingClassifier())])))\n\n            clfs.append((\"ExtraTreesClassifier\",\n                         Pipeline([(\"Scaler\", StandardScaler()),\n                                   (\"ExtraTrees\", ExtraTreesClassifier(class_weight='balanced'))])))\n\n\n            results, names = [], []\n\n            for name, model in clfs:\n                # Define time based validation splitter\n                custom_splitter = CustomCrossValidation.split(\n                                    X=x_train,\n                                    y=y_train,\n                                    groups=groups_experiment)\n\n                # Perform Cross Validation\n                cv_results = cross_val_score(model,\n                                             x_train,\n                                             y_train,\n                                             cv=custom_splitter,\n                                             scoring=scoring,\n                                             n_jobs=-1)\n                names.append(name)\n                results.append(cv_results)\n                msg = \"%s: %f (+/- %f)\" % (name, cv_results.mean(), cv_results.std())\n                print(msg)\n\n            # boxplot algorithm comparison\n            fig = plt.figure(figsize=(21, 8))\n            fig.suptitle('Classifier Algorithm Comparison', fontsize=22)\n            ax = fig.add_subplot(111)\n\n\n            foo = pd.DataFrame({'Names':names,'Results':results})\n            result = foo.explode('Results').reset_index(drop=True)\n            result = result.assign(Names=result['Names'].astype('category'), \n                                   Values=result['Results'].astype(np.float32))\n\n            sns.boxplot(x='Names', y='Results', data=result)\n            ax.set_xticklabels(names)\n            ax.set_xlabel(\"Algorithmn\", fontsize=20)\n            ax.set_ylabel(\"F1 of Models\", fontsize=18)\n            ax.set_xticklabels(ax.get_xticklabels(), rotation=45)\n\n            return clfs\n    '''Convert main dataframe to train test split ready to train/validate model'''\n    @classmethod\n    def model_creation(cls,\n                      df_train,\n                      df_valid,\n                      target_name,\n                      selected_features,\n                      id_level):\n        \"\"\"\n        Return 4 dataframes for train and test\n        param df:\n            dataframe to split into train test\n        return:\n            return 4 frames\n        \"\"\"\n        df_train = df_train.copy()\n        df_valid = df_valid.copy()\n        #set x and y\n        y_train = df_train[target_name].values\n        x_train = df_train.set_index(id_level)\n        x_train = x_train.drop([target_name], axis=1)\n        x_train= x_train[selected_features]\n        # valid\n        y_test = df_valid[target_name].values\n        x_test = df_valid.set_index(id_level)\n        x_test = x_test.drop([target_name], axis=1)\n        x_test= x_test[selected_features]\n\n        return x_train, x_test, y_train, y_test\n    \n    @classmethod\n    def drop_search_date(cls, df, date_column):\n        df = df.copy()\n        df = df.drop([date_column], axis=1)\n        return df\n    \n    @classmethod\n    def xy_creation(cls,\n                    df,\n                    target_name,\n                    id_level):\n\n           df = df.copy()\n           #set x and y\n           y_train = df[target_name].values\n           x_train = df.drop([target_name,id_level], axis=1)\n\n           return x_train,y_train \n        \n    @classmethod\n    def cor_selector(cls,X, y, features_to_select, feature_name):\n        cor_list = []\n        # calculate the correlation with y for each feature\n        for i in X.columns.tolist():\n            cor = np.corrcoef(X[i], y)[0, 1]\n            cor_list.append(cor)\n        # replace NaN with 0\n        cor_list = [0 if np.isnan(i) else i for i in cor_list]\n        # feature name\n        cor_feature = X.iloc[:,np.argsort(np.abs(cor_list))[-features_to_select:]].columns.tolist()\n        # feature selection? 0 for not select, 1 for select\n        cor_support = [True if i in cor_feature else False for i in feature_name]\n        return cor_support, cor_feature\n    \n    @classmethod\n    def remove_collinear_features(cls,x, threshold):\n        '''\n        Objective:\n            Remove collinear features in a dataframe with a correlation coefficient\n            greater than the threshold. Removing collinear features can help a model \n            to generalize and improves the interpretability of the model.\n\n        Inputs: \n            x: features dataframe\n            threshold: features with correlations greater than this value are removed\n\n        Output: \n            dataframe that contains only the non-highly-collinear features\n        '''\n\n        # Calculate the correlation matrix\n        corr_matrix = x.corr()\n        iters = range(len(corr_matrix.columns) - 1)\n        drop_cols = []\n\n        # Iterate through the correlation matrix and compare correlations\n        for i in iters:\n            for j in range(i+1):\n                item = corr_matrix.iloc[j:(j+1), (i+1):(i+2)]\n                col = item.columns\n                row = item.index\n                val = abs(item.values)\n\n                # If correlation exceeds the threshold\n                if val >= threshold:\n                    # Print the correlated features and the correlation value\n                    #print(col.values[0], \"|\", row.values[0], \"|\", round(val[0][0], 2))\n                    drop_cols.append(col.values[0])\n\n        # Drop one of each pair of correlated columns\n        drops = set(drop_cols)\n        x = x.drop(columns=drops)\n        print('Removed Columns {}'.format(drops))\n        return x\n\n\n    #dataframe from upper step and number of best variables to output\n    @classmethod\n    def variable_selection(cls,\n                          df,\n                          best,\n                          id_level,\n                          target_name,\n                          features_to_select):\n        \n            df = df.copy()\n            df = df.head(25000)\n            start = time.time()\n\n            #Creating the X and y variables\n\n            X,y = cls.xy_creation(df, target_name, id_level)\n\n     # =============================================================================\n     #        #Remove missing\n     # =============================================================================\n            print('Remove Missing..')\n            for col in X.columns:\n                   if X[col].isnull().sum()/X.shape[0] > 0.9:\n                          print(col)\n                          del X[col]\n     # =============================================================================\n     #        #Remove low variance features\n     # =============================================================================       \n            for col in X.columns:\n                   if X[col].nunique()==1:\n                          print(col)\n                          del X[col]\n     # =============================================================================\n     #        #Pearson Correlation\n     # =============================================================================\n            print('Pearson Correlation..')\n\n            feature_name = X.columns.tolist()     \n\n            cor_support, cor_feature = cls.cor_selector(X, y, features_to_select, feature_name)\n            gc.collect()\n     # =============================================================================\n     #        #Chi-2\n     # =============================================================================\n            print('Chi-2..')\n\n            X_norm = MinMaxScaler().fit_transform(X)\n            chi_selector = SelectKBest(chi2, k=features_to_select)\n            chi_selector.fit(X_norm, y)\n\n            chi_support = chi_selector.get_support()\n            chi_feature = X.loc[:,chi_support].columns.tolist()\n            print(str(len(chi_feature)), 'selected features')\n\n            gc.collect()\n     # =============================================================================\n     #        #Wrapper RFE selector\n     # =============================================================================\n            print('RFE selector..')\n            rfe_selector = RFE(estimator=LogisticRegression(class_weight = 'balanced'), n_features_to_select=features_to_select, step=100, verbose=5)\n            rfe_selector.fit(X_norm, y)\n\n            rfe_support = rfe_selector.get_support()\n            rfe_feature = X.loc[:,rfe_support].columns.tolist()\n            print(str(len(rfe_feature)), 'selected features')\n            gc.collect()\n     # =============================================================================\n     #        #Embedded\n     # =============================================================================\n\n            print('LogisticRegression l2 ..')\n\n            embeded_lr_selector = SelectFromModel(LogisticRegression(penalty=\"l2\",class_weight = 'balanced'))\n            embeded_lr_selector.fit(X_norm, y)\n\n            embeded_lr_support = embeded_lr_selector.get_support()\n            embeded_lr_feature = X.loc[:,embeded_lr_support].columns.tolist()\n            print(str(len(embeded_lr_feature)), 'selected features')\n            gc.collect()\n     # =============================================================================\n     #        #Random Forest\n     # =============================================================================\n            print('Random Forest ..')\n\n            embeded_rf_selector = SelectFromModel(RandomForestClassifier(n_estimators=features_to_select,class_weight = 'balanced'))\n            embeded_rf_selector.fit(X, y)\n\n            embeded_rf_support = embeded_rf_selector.get_support()\n            embeded_rf_feature = X.loc[:,embeded_rf_support].columns.tolist()\n            print(str(len(embeded_rf_feature)), 'selected features')\n            gc.collect()\n     # =============================================================================\n     #        #Light GBM\n     # =============================================================================\n            print('Light GBM ..')\n\n\n            lgbc=LGBMClassifier(n_estimators=features_to_select, class_weight = 'balanced',learning_rate=0.13)\n\n            embeded_lgb_selector = SelectFromModel(lgbc)\n            embeded_lgb_selector.fit(X, y)\n\n            embeded_lgb_support = embeded_lgb_selector.get_support()\n            embeded_lgb_feature = X.loc[:,embeded_lgb_support].columns.tolist()\n            print(str(len(embeded_lgb_feature)), 'selected features')\n            gc.collect()\n            del chi_feature, rfe_feature, embeded_lr_feature, embeded_rf_feature, embeded_lgb_feature\n            gc.collect()\n\n\n\n            pd.set_option('display.max_rows', None)\n            # put all selection together\n            feature_selection_df = pd.DataFrame({'Feature':feature_name, \n                                                 'Pearson':cor_support, \n                                                 'Chi-2':chi_support, \n                                                 'RFE':rfe_support, \n                                                 'Logistics':embeded_lr_support,\n                                                 'Random Forest':embeded_rf_support,\n                                                 'LightGBM':embeded_lgb_support\n                                                })\n            # count the selected times for each feature\n            feature_selection_df['Total'] = np.sum(feature_selection_df, axis=1)\n            # display the top x\n            feature_selection_df = feature_selection_df.sort_values(['Total','Feature'] , ascending=False)\n            feature_selection_df.index = range(1, len(feature_selection_df)+1)\n\n            best_features = feature_selection_df.head(best)\n            best_features = list(best_features['Feature'])\n            print(best_features)\n            Xbest = X[best_features]\n\n            # remove uncorrelated features \n            Xbest = cls.remove_collinear_features(Xbest,0.9)\n\n\n            end = time.time()\n            print((end - start)/60)\n\n            return Xbest","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:34:58.740406Z","iopub.execute_input":"2022-07-31T22:34:58.741202Z","iopub.status.idle":"2022-07-31T22:34:58.832367Z","shell.execute_reply.started":"2022-07-31T22:34:58.741151Z","shell.execute_reply":"2022-07-31T22:34:58.831283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Create Class for Lightgbm Model","metadata":{}},{"cell_type":"code","source":"class LightgbmDefault:\n    \n    @classmethod\n    def pipeline(cls,\n                 x_train,\n                 y_train,\n                 x_test,\n                 y_test,\n                 hyper_space,\n                 scoring,\n                 groups_experiment):\n        \n            # run optimization\n            optimization = fmin(fn = partial(cls.to_minimize,\n                                             scoring = scoring,\n                                             features = x_train,\n                                             target = y_train,\n                                             groups_experiment = groups_experiment),\n                      space = hyper_space, \n                      algo = tpe.suggest,\n                      trials = Trials(),\n                      max_evals = int(5))\n            \n            # chosen space\n            best_params = space_eval(hyper_space, optimization)\n            gc.collect()\n            # fit model\n            clf_best = LGBMClassifier(**best_params)\n            clf_best.fit(x_train, y_train)\n            # \n            lgbm_chosen = cls.run_lgb(x_train, y_train, x_test, y_test, best_params)\n            \n            return lgbm_chosen, best_params\n        \n    \n    @classmethod\n    def run_lgb(cls,x_train, y_train, x_test, y_test, best_params):\n        # Init params for lgbm\n        params_lgbm = {'boosting_type': 'gbdt',\n                  'max_depth' : 8,\n                  'objective': 'binary',\n                  'nthread': 5,\n                  'num_leaves': 120,\n                  'learning_rate': 0.15,\n                  'max_bin': 512,\n                  'subsample_for_bin': 100,\n                  'subsample': 0.65,\n                  'subsample_freq': 1,\n                  'colsample_bytree': 0.65,\n                  'reg_alpha': 2,\n                  'reg_lambda': 2,\n                  'min_split_gain': 0.1,\n                  'min_child_weight': 1,\n                  'min_child_samples': 2,\n                  'scale_pos_weight': 10,\n                  'num_class' : 1,\n                  'metric' : {'auc', 'mcc_eval'},\n                  'eval_metric': {'auc', 'mcc_eval'}\n                  }\n\n\n        # Using parameters already set above, replace in the best from the grid search\n        params_lgbm['learning_rate'] = best_params['learning_rate']\n        params_lgbm['scale_pos_weight'] = best_params['scale_pos_weight']\n        params_lgbm['max_depth'] = best_params['max_depth']\n        params_lgbm['num_leaves'] = best_params['num_leaves']\n        # params_lgbm['n_estimators'] = best_params['n_estimators']\n        params_lgbm['min_child_samples'] = best_params['min_child_samples']\n        params_lgbm['reg_lambda'] = best_params['reg_lambda']\n        params_lgbm['reg_alpha'] = best_params['reg_alpha']\n        params_lgbm['colsample_bytree'] = best_params['colsample_bytree']\n        params_lgbm['subsample'] = best_params['subsample']\n\n        # define dataset\n        train_data_df = lgb.Dataset(x_train, label=y_train)\n        test_data_df = lgb.Dataset(x_test, label=y_test, reference=train_data_df)\n        evals_results_df = {}\n        print(\"Training the model...\")\n        #Train model on selected parameters and number of iterations\n        lgbm_best = lgb.train(params_lgbm,\n                         train_data_df,\n                         valid_sets=[train_data_df, test_data_df], \n                         valid_names=['train','test'], \n                         evals_result=evals_results_df, \n                         num_boost_round= best_params['n_estimators'],\n                         callbacks=[lgb.early_stopping(stopping_rounds=50)],\n                         feval=lgb_amex_metric,\n                         verbose_eval= True\n                         )\n        \n        return lgbm_best\n    \n    \n    @classmethod\n    def to_minimize(cls, hyperparameters, scoring, features, target, groups_experiment):\n        # create an instance of the model \n        clf = LGBMClassifier(**hyperparameters)\n        cv1 = StratifiedKFold(n_splits=10, random_state=13,shuffle=True)\n        cv = CustomCrossValidation.split(\n                        X=features,\n                        y=target,\n                        groups=groups_experiment)\n\n        # train with cross-validation\n        result = cross_val_score(estimator = clf, \n                                    X = features, \n                                    y = target, \n                                    groups=pd.DataFrame(groups_experiment),\n                                    scoring = scoring,\n                                    cv = cv1, \n                                    n_jobs = -1,\n                                    error_score='raise')\n\n        return -result.mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:34:58.833915Z","iopub.execute_input":"2022-07-31T22:34:58.834519Z","iopub.status.idle":"2022-07-31T22:34:58.858326Z","shell.execute_reply.started":"2022-07-31T22:34:58.834483Z","shell.execute_reply":"2022-07-31T22:34:58.857310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Plot Metrics","metadata":{}},{"cell_type":"code","source":"class PlotMetrics:\n    \n    @classmethod\n    def plot_confusion_matrix(cls, y_true, y_pred, normalize=False, title = None, cmap=plt.cm.Blues):\n\n\n        \"\"\"\n        This function prints and plots the confusion matrix.\n        Normalization can be applied by setting `normalize=True`.\n        \"\"\"\n        if not title:\n            if normalize:\n                title = 'Normalized confusion matrix'\n            else:\n                title = 'Confusion matrix, without normalization'\n\n        # Compute confusion matrix\n        cm = confusion_matrix(y_true, y_pred)\n\n        if normalize:\n            cm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n            print(\"Normalized confusion matrix\")\n        else:\n            print('Confusion matrix, without normalization')\n\n        print(cm)\n\n        fig, ax = plt.subplots()\n        im = ax.imshow(cm, interpolation='nearest', cmap=cmap)\n        ax.figure.colorbar(im, ax=ax)\n        # We want to show all ticks...\n        ax.set(xticks=[],\n               yticks=[],\n               # ... and label them with the respective list entries\n               #xticklabels=classes, yticklabels=classes,\n               title=title,\n               ylabel='True label',\n               xlabel='Predicted label')\n\n        # Rotate the tick labels and set their alignment.\n        plt.setp(ax.get_xticklabels(), rotation=45, ha=\"right\",\n                 rotation_mode=\"anchor\")\n\n\n        # Loop over data dimensions and create text annotations.\n        fmt = '.2f' if normalize else 'd'\n        thresh = cm.max() / 2.\n        for i in range(cm.shape[0]):\n            for j in range(cm.shape[1]):\n                ax.text(j, i, format(cm[i, j], fmt),\n                        ha=\"center\", va=\"center\",\n                        color=\"white\" if cm[i, j] > thresh else \"black\")\n        fig.tight_layout()\n        return ax\n\n    @classmethod\n    def plot_precision_recall_vs_thresholds(cls, precisions, recalls, thresholds):\n        plt.plot(thresholds, precisions[:-1], \"b--\", label=\"Precision\")\n        plt.plot(thresholds, recalls[:-1], \"g--\", label=\"Recall\")\n        plt.xlabel(\"Threshold\")\n        plt.legend(bbox_to_anchor=(1.05, 1), loc='upper left', borderaxespad=0.)\n        plt.grid(b=True, which=\"both\", axis=\"both\", color='gray', linestyle='-', linewidth=1)\n        plt.title('Precision-Recall Thresholds')\n        \n    @classmethod\n    def plot_predictions(cls, x, y, clf):\n\n            #Predict on test set\n            #Predicting proba\n            predictions_clf_prob = clf.predict(x)\n            predictions_clf_01 = np.where(predictions_clf_prob > 0.5, 1, 0) #Turn probability to 0-1 binary output\n            y_df = pd.DataFrame(y, columns=['target'])\n            y_pred_df = pd.DataFrame(predictions_clf_01, columns=['prediction'])\n            print(\"accuracy is:\", accuracy_score(y,predictions_clf_01))\n            print(\"\\n\")\n            print(\"confusion matrix is:\", confusion_matrix(y, predictions_clf_01))\n            print(\"\\n\")\n            print(\"fbeta is:\", fbeta_score(y, predictions_clf_01, beta=2))\n            print(\"\\n\")\n            print(\"amex metric is:\", amex_metric(y, predictions_clf_01))\n            print(\"\\n\")\n            print(classification_report(y, predictions_clf_01))\n\n            # Generate ROC curve values: fpr, tpr, thresholds\n            fpr, tpr, thresholds = roc_curve(y, predictions_clf_prob)\n\n            # Plot ROC curve\n            plt.plot([0, 1], [0, 1], 'k--')\n            plt.plot(fpr, tpr)\n            plt.xlabel('False Positive Rate')\n            plt.ylabel('True Positive Rate')\n            plt.title('ROC Curve')\n            plt.show()\n\n            # Plot non-normalized confusion matrix\n            cls.plot_confusion_matrix(y, predictions_clf_01,\n                                  title='Confusion matrix, without normalization')\n\n            # Plot normalized confusion matrix\n            cls.plot_confusion_matrix(y, predictions_clf_01, normalize=True,\n                                  title='Normalized confusion matrix')\n            plt.show()\n            # Plot Precision Recall Curve\n            precisions, recalls, thresholds = precision_recall_curve(y, predictions_clf_prob)\n            cls.plot_precision_recall_vs_thresholds(precisions, recalls, thresholds)\n            plt.show()\n            # Report \n            kds.metrics.report(y, predictions_clf_prob)\n    \n    @classmethod\n    def plot_importance(cls, model, X , num = 10):\n             feature_imp = pd.DataFrame({'Value':model.feature_importance(),'Feature':X.columns})\n             plt.figure(figsize=(20, 10))\n             sns.set(font_scale = 1)\n             sns.barplot(x=\"Value\", y=\"Feature\", data=feature_imp.sort_values(by=\"Value\", \n                                                                 ascending=False)[0:num])\n             plt.title('LightGBM Features (avg over folds)')\n             plt.tight_layout()\n             plt.savefig('lgbm_importance.png')\n             plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:34:58.860209Z","iopub.execute_input":"2022-07-31T22:34:58.860951Z","iopub.status.idle":"2022-07-31T22:34:58.895728Z","shell.execute_reply.started":"2022-07-31T22:34:58.860895Z","shell.execute_reply":"2022-07-31T22:34:58.894313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(props):\n    start_mem_usg = props.memory_usage().sum() / 1024**2 \n    print(\"Memory usage of properties dataframe is :\",start_mem_usg,\" MB\")\n    NAlist = [] # Keeps track of columns that have missing values filled in. \n    for col in props.columns:\n        if props[col].dtype == float:  # Exclude strings\n            \n            # Print current column type\n            print(\"******************************\")\n            print(\"Column: \",col)\n            print(\"dtype before: \",props[col].dtype)\n            \n            # make variables for Int, max and min\n            IsInt = False\n            mx = props[col].max()\n            mn = props[col].min()\n            \n            # Integer does not support NA, therefore, NA needs to be filled\n            if not np.isfinite(props[col]).all(): \n                NAlist.append(col)\n                props[col].fillna(mn-1,inplace=True)  \n                   \n            # test if column can be converted to an integer\n            asint = props[col].fillna(0).astype(np.int64)\n            result = (props[col] - asint)\n            result = result.sum()\n            if result > -0.01 and result < 0.01:\n                IsInt = True\n\n            \n            # Make Integer/unsigned Integer datatypes\n            if IsInt:\n                if mn >= 0:\n                    if mx < 255:\n                        props[col] = props[col].astype(np.uint8)\n                    elif mx < 65535:\n                        props[col] = props[col].astype(np.uint16)\n                    elif mx < 4294967295:\n                        props[col] = props[col].astype(np.uint32)\n                    else:\n                        props[col] = props[col].astype(np.uint64)\n                else:\n                    if mn > np.iinfo(np.int8).min and mx < np.iinfo(np.int8).max:\n                        props[col] = props[col].astype(np.int8)\n                    elif mn > np.iinfo(np.int16).min and mx < np.iinfo(np.int16).max:\n                        props[col] = props[col].astype(np.int16)\n                    elif mn > np.iinfo(np.int32).min and mx < np.iinfo(np.int32).max:\n                        props[col] = props[col].astype(np.int32)\n                    elif mn > np.iinfo(np.int64).min and mx < np.iinfo(np.int64).max:\n                        props[col] = props[col].astype(np.int64)    \n            \n            # Make float datatypes 32 bit\n            else:\n                props[col] = props[col].astype(np.float32)\n            \n            # Print new column type\n            print(\"dtype after: \",props[col].dtype)\n            print(\"******************************\")\n    \n    # Print final result\n    print(\"___MEMORY USAGE AFTER COMPLETION:___\")\n    mem_usg = props.memory_usage().sum() / 1024**2 \n    print(\"Memory usage is: \",mem_usg,\" MB\")\n    print(\"This is \",100*mem_usg/start_mem_usg,\"% of the initial size\")\n    return props, NAlist","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:34:58.897855Z","iopub.execute_input":"2022-07-31T22:34:58.898757Z","iopub.status.idle":"2022-07-31T22:34:58.928544Z","shell.execute_reply.started":"2022-07-31T22:34:58.898702Z","shell.execute_reply":"2022-07-31T22:34:58.927567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = import_train_data()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:34:58.932359Z","iopub.execute_input":"2022-07-31T22:34:58.933043Z","iopub.status.idle":"2022-07-31T22:35:23.772710Z","shell.execute_reply.started":"2022-07-31T22:34:58.932993Z","shell.execute_reply":"2022-07-31T22:35:23.769763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> #### Inpect how the dataframe looks like\n* We see that many of **Nans** in certain columns. Need to further check which are","metadata":{}},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:35:23.775706Z","iopub.execute_input":"2022-07-31T22:35:23.776654Z","iopub.status.idle":"2022-07-31T22:35:23.990759Z","shell.execute_reply.started":"2022-07-31T22:35:23.776613Z","shell.execute_reply":"2022-07-31T22:35:23.989149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> ### Check the resume of the train df","metadata":{}},{"cell_type":"code","source":"# resumetable(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:35:23.992312Z","iopub.execute_input":"2022-07-31T22:35:23.992694Z","iopub.status.idle":"2022-07-31T22:35:23.999737Z","shell.execute_reply.started":"2022-07-31T22:35:23.992657Z","shell.execute_reply":"2022-07-31T22:35:23.998626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> ### Check missing values in features\n* 25 Features have >70% missing values.\n* Need to decide on feature imputation strategy for rest features with less missing values.","metadata":{}},{"cell_type":"code","source":"# MissingValuesInsights.insights(train, 0.2)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:35:24.000970Z","iopub.execute_input":"2022-07-31T22:35:24.001346Z","iopub.status.idle":"2022-07-31T22:35:24.014921Z","shell.execute_reply.started":"2022-07-31T22:35:24.001311Z","shell.execute_reply":"2022-07-31T22:35:24.013295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> ### Target variable is 25% in distribution","metadata":{}},{"cell_type":"code","source":"PlotTarget(train, 'target')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:35:24.016640Z","iopub.execute_input":"2022-07-31T22:35:24.017179Z","iopub.status.idle":"2022-07-31T22:35:32.407122Z","shell.execute_reply.started":"2022-07-31T22:35:24.017130Z","shell.execute_reply":"2022-07-31T22:35:32.405569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# take the first 100K observations so that time for plot the features will be manageable\ntrain_head = train.head(50000)\nnum_cols = train_head.columns[2:27]\nfor col in num_cols:\n    TargetvsNumeric(train_head, col, 'target', col)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:35:32.408983Z","iopub.execute_input":"2022-07-31T22:35:32.409368Z","iopub.status.idle":"2022-07-31T22:36:04.114505Z","shell.execute_reply.started":"2022-07-31T22:35:32.409325Z","shell.execute_reply":"2022-07-31T22:36:04.113254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Categorical Columns\nCAT_COLS_AGG_TRAIN = train.select_dtypes(include=['category']).columns.tolist()\nOBJ_COLS_AGG_TRAIN = train.select_dtypes(include=['object']).columns.tolist()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:36:04.115996Z","iopub.execute_input":"2022-07-31T22:36:04.117031Z","iopub.status.idle":"2022-07-31T22:36:04.149018Z","shell.execute_reply.started":"2022-07-31T22:36:04.116988Z","shell.execute_reply":"2022-07-31T22:36:04.147303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CAT_COLS_AGG_TRAIN","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:36:04.150822Z","iopub.execute_input":"2022-07-31T22:36:04.151561Z","iopub.status.idle":"2022-07-31T22:36:04.163423Z","shell.execute_reply.started":"2022-07-31T22:36:04.151519Z","shell.execute_reply":"2022-07-31T22:36:04.162073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Config","metadata":{}},{"cell_type":"code","source":"# Target name for target variable\nTARGET_NAME = \"target\"\n# Date column\nDATE_COLUMN = \"S_2\"\n# MaxDate of training-validation period\nDATE_TRAIN_VALID = \"2018-03-30\"\n# MaxDate of training-validation period\nDATE_TRAIN = \"2018-03-23\"\n# MinDate of validation period\nDATE_VALID = \"2018-03-24\"\n# MinDate of testing period\nDATE_TEST = \"2018-03-31\"\n# Top features\nTOP_FEATURES = 80\n# Keep final number of features\nKEEP_FEATURES = 60\n# Scoring Optimization\nSCORING = mcc_scorer\n# Id Level\nID_LEVEL = 'customer_ID'\n# Define searched space\nLGBM_HYPER_SPACE = {'objective': 'binary',\n                    'metric': 'roc_auc',\n                    'boosting': 'gbdt',\n                    'n_estimators': hp.choice('n_estimators', [450, 500, 600, 700, 900, 1100]),\n                    'scale_pos_weight': hp.choice('scale_pos_weight', list(range(6, 14, 1))),\n                    'max_depth': hp.choice('max_depth', list(range(8, 20, 2))),\n                    'num_leaves': hp.choice('num_leaves', list(range(50, 3000, 100))),\n                    'subsample': hp.choice('subsample', [.6, .7, .8, .9, 1]),\n                    'colsample_bytree': hp.uniform('colsample_bytree', 0.6, 1),\n                    'learning_rate': hp.uniform('learning_rate', 0.03, 0.18),\n                    'reg_alpha': hp.choice('reg_alpha', [.1, .2, .3, .4, .5, .6]),\n                    'reg_lambda': hp.choice('reg_lambda', [.1, .2, .3, .4, .5, .6]),\n                    'min_child_samples': hp.choice('min_child_samples', list(range(200, 4000, 100))),\n                    'feature_fraction': hp.choice('feature_fraction', [.6, .7, .8, .9]),\n                    'bagging_fraction': hp.choice('bagging_fraction', [.6, .7, .8, .9])\n                    }\n\n# Categorical Columns\nINIT_CAT_COLS = ['D_63', 'D_64', 'D_68', 'B_30', 'B_38',\n'D_114', 'D_116', 'D_117','D_120', 'D_126']\n                                                   \nCAT_COLS = ['D_63', 'D_64', 'D_68', 'B_30', 'B_38',\n'D_114', 'D_116', 'D_117','D_120', 'D_126',\n'D_63-D_126','D_63-D_120','D_63-D_117','D_63-D_116',\n'D_63-D_114','D_63-B_38','D_63-B_30','D_63-D_68',\n'D_63-D_64', 'D_64-D_68','D_64-B_30',\n'D_64-B_38','D_64-D_114','D_64-D_116',\n'D_64-D_117','D_64-D_120','D_64-D_126',\n'D_64-D_68','D_64-B_30',\n'D_68-B_30','D_68-B_38','D_68-D_114','D_68-D_116',\n'D_68-D_117','D_68-D_120','D_68-D_126']   \n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:36:04.165627Z","iopub.execute_input":"2022-07-31T22:36:04.166033Z","iopub.status.idle":"2022-07-31T22:36:04.186102Z","shell.execute_reply.started":"2022-07-31T22:36:04.165999Z","shell.execute_reply":"2022-07-31T22:36:04.185054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:36:04.193439Z","iopub.execute_input":"2022-07-31T22:36:04.193916Z","iopub.status.idle":"2022-07-31T22:36:04.365219Z","shell.execute_reply.started":"2022-07-31T22:36:04.193868Z","shell.execute_reply":"2022-07-31T22:36:04.363866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train, NAlist = reduce_mem_usage(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:36:04.367056Z","iopub.execute_input":"2022-07-31T22:36:04.367472Z","iopub.status.idle":"2022-07-31T22:36:04.395111Z","shell.execute_reply.started":"2022-07-31T22:36:04.367435Z","shell.execute_reply":"2022-07-31T22:36:04.393793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# feature engineering\ntraindf = FeatureEngineering.pipeline(train, INIT_CAT_COLS)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:36:04.396961Z","iopub.execute_input":"2022-07-31T22:36:04.397631Z","iopub.status.idle":"2022-07-31T22:36:19.024357Z","shell.execute_reply.started":"2022-07-31T22:36:04.397590Z","shell.execute_reply":"2022-07-31T22:36:19.023147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:36:19.025891Z","iopub.execute_input":"2022-07-31T22:36:19.026272Z","iopub.status.idle":"2022-07-31T22:36:19.185050Z","shell.execute_reply.started":"2022-07-31T22:36:19.026224Z","shell.execute_reply":"2022-07-31T22:36:19.183854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train, valid, test, target_mean_dictionary = LabelEncoding.pipeline(traindf,\n                                  DATE_COLUMN,           \n                                  TARGET_NAME,\n                                  DATE_TRAIN_VALID,\n                                  DATE_TRAIN,\n                                  DATE_VALID,\n                                  DATE_TEST,\n                                  CAT_COLS)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:36:19.186354Z","iopub.execute_input":"2022-07-31T22:36:19.187176Z","iopub.status.idle":"2022-07-31T22:36:40.307043Z","shell.execute_reply.started":"2022-07-31T22:36:19.187138Z","shell.execute_reply":"2022-07-31T22:36:40.305742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del traindf\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:36:40.309065Z","iopub.execute_input":"2022-07-31T22:36:40.309881Z","iopub.status.idle":"2022-07-31T22:36:40.768580Z","shell.execute_reply.started":"2022-07-31T22:36:40.309843Z","shell.execute_reply":"2022-07-31T22:36:40.767234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split main dataframe to train,test\nx_train, x_test, y_train, y_test, x_oot, y_oot, feat_list, groups_experiment = PrepareModel.pipeline(train,\n                                  valid,\n                                  test,\n                                  DATE_COLUMN,                                                                   \n                                  TARGET_NAME,\n                                  TOP_FEATURES,\n                                  KEEP_FEATURES,\n                                  SCORING,\n                                  ID_LEVEL)\n\ndel train, valid, test\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:36:40.770303Z","iopub.execute_input":"2022-07-31T22:36:40.770761Z","iopub.status.idle":"2022-07-31T22:37:23.591230Z","shell.execute_reply.started":"2022-07-31T22:36:40.770723Z","shell.execute_reply":"2022-07-31T22:37:23.589863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(feat_list)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:37:23.592976Z","iopub.execute_input":"2022-07-31T22:37:23.593360Z","iopub.status.idle":"2022-07-31T22:37:23.601721Z","shell.execute_reply.started":"2022-07-31T22:37:23.593325Z","shell.execute_reply":"2022-07-31T22:37:23.600184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Choose Best Lightgbm model\nlgbm_chosen,best_params = LightgbmDefault.pipeline(x_train,\n             y_train,\n             x_test,\n             y_test,\n             LGBM_HYPER_SPACE,\n             SCORING,\n             groups_experiment)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:37:23.603530Z","iopub.execute_input":"2022-07-31T22:37:23.604381Z","iopub.status.idle":"2022-07-31T23:06:46.369208Z","shell.execute_reply.started":"2022-07-31T22:37:23.604329Z","shell.execute_reply":"2022-07-31T23:06:46.366819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"    # Plot Metrics\nPlotMetrics.plot_predictions(x_train, y_train, lgbm_chosen)\nPlotMetrics.plot_predictions(x_test, y_test, lgbm_chosen)\nPlotMetrics.plot_predictions(x_oot,y_oot ,lgbm_chosen)\nPlotMetrics.plot_importance(lgbm_chosen, x_test , num = 15)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:06:46.372232Z","iopub.execute_input":"2022-07-31T23:06:46.372760Z","iopub.status.idle":"2022-07-31T23:06:59.846191Z","shell.execute_reply.started":"2022-07-31T23:06:46.372710Z","shell.execute_reply":"2022-07-31T23:06:59.844870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_params","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:06:59.847836Z","iopub.execute_input":"2022-07-31T23:06:59.848608Z","iopub.status.idle":"2022-07-31T23:06:59.858615Z","shell.execute_reply.started":"2022-07-31T23:06:59.848549Z","shell.execute_reply":"2022-07-31T23:06:59.856820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_params['n_estimators']","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:06:59.861642Z","iopub.execute_input":"2022-07-31T23:06:59.862307Z","iopub.status.idle":"2022-07-31T23:06:59.877003Z","shell.execute_reply.started":"2022-07-31T23:06:59.862252Z","shell.execute_reply":"2022-07-31T23:06:59.875373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# del x_train, x_test, y_train, y_test, x_oot, y_oot\n# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:06:59.878943Z","iopub.execute_input":"2022-07-31T23:06:59.879484Z","iopub.status.idle":"2022-07-31T23:06:59.890023Z","shell.execute_reply.started":"2022-07-31T23:06:59.879430Z","shell.execute_reply":"2022-07-31T23:06:59.888609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Score Test Data","metadata":{}},{"cell_type":"markdown","source":"- del vars","metadata":{}},{"cell_type":"code","source":"test = import_test_data()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:06:59.891868Z","iopub.execute_input":"2022-07-31T23:06:59.893980Z","iopub.status.idle":"2022-07-31T23:07:49.669451Z","shell.execute_reply.started":"2022-07-31T23:06:59.893922Z","shell.execute_reply":"2022-07-31T23:07:49.667858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# feature engineering\ntest = FeatureEngineering.pipeline(test, INIT_CAT_COLS)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:07:49.670978Z","iopub.execute_input":"2022-07-31T23:07:49.671369Z","iopub.status.idle":"2022-07-31T23:08:15.723861Z","shell.execute_reply.started":"2022-07-31T23:07:49.671334Z","shell.execute_reply":"2022-07-31T23:08:15.722823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### choose certain cols\ntest = test[feat_list].drop_duplicates()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:08:19.924415Z","iopub.execute_input":"2022-07-31T23:08:19.924901Z","iopub.status.idle":"2022-07-31T23:08:25.470011Z","shell.execute_reply.started":"2022-07-31T23:08:19.924862Z","shell.execute_reply":"2022-07-31T23:08:25.468572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Categorical Columns\nCAT_COLS_TST = test.select_dtypes(include=['object']).columns.tolist()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:08:25.472732Z","iopub.execute_input":"2022-07-31T23:08:25.473324Z","iopub.status.idle":"2022-07-31T23:08:25.499914Z","shell.execute_reply.started":"2022-07-31T23:08:25.473266Z","shell.execute_reply":"2022-07-31T23:08:25.498453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CAT_COLS_TST","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:08:25.501578Z","iopub.execute_input":"2022-07-31T23:08:25.502124Z","iopub.status.idle":"2022-07-31T23:08:25.511063Z","shell.execute_reply.started":"2022-07-31T23:08:25.502071Z","shell.execute_reply":"2022-07-31T23:08:25.509602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# resumetable(test)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:09:46.749522Z","iopub.execute_input":"2022-07-24T10:09:46.750020Z","iopub.status.idle":"2022-07-24T10:09:48.017344Z","shell.execute_reply.started":"2022-07-24T10:09:46.749984Z","shell.execute_reply":"2022-07-24T10:09:48.016146Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if len(CAT_COLS_TST) > 0:\n# set the whole column to this value.\n    for feature in CAT_COLS_TST:\n        test[feature] = target_mean_dictionary[feature].get(\n            test[feature].loc[0],\n            np.nan,\n        )\nelse:\n    print(\"Continue to Scoring!\")\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:08:29.440835Z","iopub.execute_input":"2022-07-31T23:08:29.441283Z","iopub.status.idle":"2022-07-31T23:08:29.477611Z","shell.execute_reply.started":"2022-07-31T23:08:29.441247Z","shell.execute_reply":"2022-07-31T23:08:29.476239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gbm_test_preds = lgbm_chosen.predict(test)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:08:31.451648Z","iopub.execute_input":"2022-07-31T23:08:31.452130Z","iopub.status.idle":"2022-07-31T23:08:38.263457Z","shell.execute_reply.started":"2022-07-31T23:08:31.452088Z","shell.execute_reply":"2022-07-31T23:08:38.262278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(\"/kaggle/input/amex-default-prediction/sample_submission.csv\")\nsub['prediction']=gbm_test_preds\nsub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:08:38.265729Z","iopub.execute_input":"2022-07-31T23:08:38.267194Z","iopub.status.idle":"2022-07-31T23:08:46.107675Z","shell.execute_reply.started":"2022-07-31T23:08:38.267134Z","shell.execute_reply":"2022-07-31T23:08:46.106142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:08:46.109422Z","iopub.execute_input":"2022-07-31T23:08:46.109851Z","iopub.status.idle":"2022-07-31T23:08:46.128464Z","shell.execute_reply.started":"2022-07-31T23:08:46.109800Z","shell.execute_reply":"2022-07-31T23:08:46.127496Z"},"trusted":true},"execution_count":null,"outputs":[]}]}