{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**The solution uses pytorch tabnet algos, so pls have the relevant pytorch modules iunstalled before testing the solution.**","metadata":{}},{"cell_type":"markdown","source":"**Load libs**","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport warnings\nimport time\nfrom datetime import datetime\n\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.preprocessing import OrdinalEncoder, LabelEncoder, OneHotEncoder\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.metrics import accuracy_score, f1_score, confusion_matrix, roc_auc_score\nfrom imblearn.over_sampling import SMOTE\n\nimport torch\nfrom pytorch_tabnet.pretraining import TabNetPretrainer\nfrom pytorch_tabnet.tab_model import TabNetClassifier\nimport xgboost as xgb\nimport matplotlib.pyplot as plt\nimport pickle\nfrom typing import Tuple\n\nwarnings.filterwarnings(\"ignore\")\n%matplotlib inline\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**timer fn for timing the executions...**","metadata":{}},{"cell_type":"code","source":"def timer(myFn):\n    def fTimer(*args, **kwargs):\n        start_time = time.time()\n        result = myFn(*args, **kwargs)\n        end_time = time.time()\n        computation_time = round(end_time - start_time, 2)\n        print(\"{} is executed\".format(myFn.__name__), end=\" \")\n        print('(took: {:.2f} seconds)'.format(computation_time))\n        return result\n    return fTimer","metadata":{"execution":{"iopub.status.busy":"2022-08-29T10:07:23.952560Z","iopub.execute_input":"2022-08-29T10:07:23.953007Z","iopub.status.idle":"2022-08-29T10:07:23.961503Z","shell.execute_reply.started":"2022-08-29T10:07:23.952971Z","shell.execute_reply":"2022-08-29T10:07:23.959728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Load data, preprocessing and transformations**\n\n**load headers and labels**","metadata":{}},{"cell_type":"code","source":"root=\"/home/sadagopan/amex-default-prediction/\"\nheaders=pd.read_csv(root+\"headers.csv\")\nheaders = headers.columns\nscols = [col for col in headers if col[0] == 'S']\ndcols = [col for col in headers if col[0] == 'D']\nrcols = [col for col in headers if col[0] == 'R']\nbcols = [col for col in headers if col[0] == 'B']\npcols = [col for col in headers if col[0] == 'P']\nlabels =pd.read_csv(root+\"train_labels.csv\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**group data by customer_ID to identify the min/max dates\nmost (~ 90%) data had 13 dates info while remaining (~ 10%) had less than 13 dates info anywhere between 1 to 12 dates.\nBeyond 3 dates, there were few input that had missing in between date info.**","metadata":{}},{"cell_type":"code","source":"headers=pd.read_csv(root+\"headers.csv\")\nheaders = headers.columns\n\ntraindf = pd.read_csv(root+\"/tdata\")\ntdf = traindf.groupby('customer_ID')['S_2'].agg(['min', 'max', 'count']).reset_index()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**transform data for data less than 13 dates**","metadata":{}},{"cell_type":"code","source":"traindfs = []\n\nprint(\"Loading...\", end=\" \")\nfor j in range(12):\n    print(\"x\" + str(j), end=\" \")\n    traindfs.append(pd.read_csv(\"x\" + str(j) + \".csv\"))\n    traindfs[j]['S_2'] = pd.to_datetime(traindfs[j]['S_2'])\n    traindfs[j]['day_of_month'] = traindfs[j]['S_2'].dt.strftime(\"%d\")\n    traindfs[j]['day_of_week'] = traindfs[j]['S_2'].dt.strftime(\"%u\")\n    traindfs[j]['day_of_year'] = traindfs[j]['S_2'].dt.strftime(\"%j\")\n    traindfs[j]['week_of_year'] = traindfs[j]['S_2'].dt.isocalendar().week\n    traindfs[j]['yearmonth'] = traindfs[j]['S_2'].dt.strftime(\"%Y%m\")\n\ni=0\nprint(\"\")\nprint(\"Transforming...\", end=\" \")\nfor j in range(12):\n    print(\"x\" + str(j), end=\" \")\n\n    maxdate = datetime.strptime('Apr 30 2019 1:33PM', '%b %d %Y %I:%M%p')\n    mindate = maxdate + pd.offsets.DateOffset(months= -1-j)\n    dates=pd.DataFrame(pd.date_range(mindate, maxdate, freq='MS').tolist(), columns=['S_2_dt'])\n    dates['yearmonth'] = dates['S_2_dt'].dt.strftime(\"%Y%m\")\n    dates.reset_index(inplace=True)\n    dates.rename(columns={'index': 'seq'}, inplace=True)\n    dates['seq']=(dates['seq']+1)\n\n    traindfs[i]['S_2'] = traindfs[i]['S_2'] + pd.offsets.DateOffset(years=1)\n    traindfs[i]['S_2'] = traindfs[i]['S_2'] + pd.offsets.DateOffset(months=1)\n    traindfs[i]['yearmonthorig'] = traindfs[i]['S_2'].dt.strftime(\"%Y%m\")\n    traindfs[i]['seq'] = traindfs[i].groupby(['customer_ID']).cumcount().add(1)\n    traindfs[i][['sncnt', 'dncnt', 'rncnt', 'bncnt', 'pncnt']] = traindfs[i].apply(lambda x: (sum([(2^scols.index(col)) for col in scols if pd.isnull(x[col])]),\n                sum([(2^dcols.index(col)) for col in dcols if pd.isnull(x[col])]),\n                sum([(2^rcols.index(col)) for col in rcols if pd.isnull(x[col])]),\n                sum([(2^bcols.index(col)) for col in bcols if pd.isnull(x[col])]),\n                sum([(2^pcols.index(col)) for col in pcols if pd.isnull(x[col])])), axis=1, result_type='expand')\n\n    dateshift = pd.DataFrame(traindfs[i].groupby(['customer_ID'])['S_2'].agg(['min', 'max', 'count'])).rename(columns={\"max\": \"S_2\"}).reset_index()\n    dateshift['S_2'] = pd.to_datetime(dateshift['S_2'])\n    dateshift['mondiff'] = dateshift['S_2'].dt.month - 4\n    dateshift['mondiff'] = dateshift['mondiff'].apply(lambda x: 12+x if x <= 0 else x)\n\n    traindfs[i] = pd.merge(traindfs[i], dateshift[['customer_ID', 'mondiff']], on=['customer_ID'], how='inner')\n    traindfs[i]['yearmonth'] = traindfs[i]['S_2'].dt.strftime(\"%Y%m\")\n\n    traindfs[i]=pd.merge(traindfs[i], pd.merge(pd.DataFrame(traindfs[i]['customer_ID'].unique(), columns=['customer_ID']), dates, how='cross'), on=['customer_ID', 'yearmonth'], how='outer')\n    traindfs[i]['origmissrcnt'] = traindfs[i].apply(lambda x: 1 if (pd.isnull(x['S_2'])) else 0, axis=1)\n    traindfs[i]['totmissrcnt'] = traindfs[i].apply(lambda x: 1 if (pd.isnull(x['S_2'])) else 0, axis=1)\n    traindfs[i]['S_2'] = traindfs[i].apply(lambda x: x['S_2_dt'] if (pd.isnull(x['S_2']) or x['S_2'].month != x['S_2_dt'].month) else x['S_2'], axis=1)\n    traindfs[i].drop(columns=['S_2_dt', 'seq_x', 'seq_y'], inplace=True)\n    traindfs[i] = traindfs[i].rename(columns={\"yearmonth_x\": \"yearmonth\"})\n\n    traindfs[i]['yearmonthorig'].fillna(0, inplace=True)\n    traindfs[i][['sncnt', 'dncnt', 'rncnt', 'bncnt', 'pncnt']].fillna(-1, inplace=True)\n    traindfs[i].drop(columns=['mondiff'], inplace=True)\n\n    traindfs[i].drop(columns=['S_2'], inplace=True)\n    traindfs[i]=traindfs[i].set_index(['customer_ID', 'yearmonth']).unstack(1).sort_index(axis=1, level=1)\n    traindfs[i] = traindfs[i].reset_index()\n    traindfs[i] = traindfs[i].merge(labels, on=\"customer_ID\", how=\"inner\")\n    traindfs[i].drop(columns=[traindfs[i].columns.to_list()[1]], inplace=True)\n\n    coldict={}\n    for col in traindfs[i].columns:\n        coldict[col] = ''.join(col)\n    traindfs[i].rename(coldict, axis=1, inplace=True)\n\n    traindfs[i].to_csv(\"x\" + str(j) +\"13dts.csv\", index=False)\n    i+=1\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**transform data for data with 13 dates**","metadata":{}},{"cell_type":"code","source":"traindfs = []\np=0\nt=20\nprint(\"Loading data parse 1...\", end=\" \")\nfor j in range(p):\n    print(\"x12-\" + str(j), end=\" \")\n    traindfs.append(pd.read_csv(\"x12-\" + str(j) + \"dtcnt.csv\"))\n\nprint(\"Loading data parse 2...\", end=\" \")\nfor j in range(p, t):\n    print(\"x12-\" + str(j), end=\" \")\n    traindfs.append(pd.read_csv(\"x12-\" + str(j) + \".csv\"))\n    traindfs[j]['S_2'] = pd.to_datetime(traindfs[j]['S_2'])\n    traindfs[j]['day_of_month'] = traindfs[j]['S_2'].dt.strftime(\"%d\")\n    traindfs[j]['day_of_week'] = traindfs[j]['S_2'].dt.strftime(\"%u\")\n    traindfs[j]['day_of_year'] = traindfs[j]['S_2'].dt.strftime(\"%j\")\n    traindfs[j]['week_of_year'] = traindfs[j]['S_2'].dt.isocalendar().week\n    traindfs[j]['yearmonth'] = traindfs[j]['S_2'].dt.strftime(\"%Y%m\")\n\nprint(\"\")\nprint(\"Transforming data...\", end=\" \")\ni=p\nj=12\nfor k in range(p, t):\n    print(\"x\" + str(j) + \"-\" + str(k), end=\" \")\n    traindfs.append(pd.read_csv(\"x\" + str(j) + \"-\" + str(k) + \".csv\"))\n\n    maxdate = datetime.strptime('Apr 30 2019 1:33PM', '%b %d %Y %I:%M%p')\n    mindate = maxdate + pd.offsets.DateOffset(months= -1-j)\n    dates=pd.DataFrame(pd.date_range(mindate, maxdate, freq='MS').tolist(), columns=['S_2_dt'])\n    dates['yearmonth'] = dates['S_2_dt'].dt.strftime(\"%Y%m\")\n    dates.reset_index(inplace=True)\n    dates.rename(columns={'index': 'seq'}, inplace=True)\n    dates['seq']=(dates['seq']+1)\n\n    traindfs[i]['S_2'] = pd.to_datetime(traindfs[i]['S_2'])\n    traindfs[i]['S_2'] = traindfs[i]['S_2'] + pd.offsets.DateOffset(years=1)\n    traindfs[i]['S_2'] = traindfs[i]['S_2'] + pd.offsets.DateOffset(months=1)\n    traindfs[i]['yearmonthorig'] = traindfs[i]['S_2'].dt.strftime(\"%Y%m\")\n    traindfs[i]['seq'] = traindfs[i].groupby(['customer_ID']).cumcount().add(1)\n    traindfs[i][['sncnt', 'dncnt', 'rncnt', 'bncnt', 'pncnt']] = traindfs[i].apply(lambda x: (sum([(2^scols.index(col)) for col in scols if pd.isnull(x[col])]),\n                sum([(2^dcols.index(col)) for col in dcols if pd.isnull(x[col])]),\n                sum([(2^rcols.index(col)) for col in rcols if pd.isnull(x[col])]),\n                sum([(2^bcols.index(col)) for col in bcols if pd.isnull(x[col])]),\n                sum([(2^pcols.index(col)) for col in pcols if pd.isnull(x[col])])), axis=1, result_type='expand')\n\n    dateshift = pd.DataFrame(traindfs[i].groupby(['customer_ID'])['S_2'].agg(['max', 'count'])).rename(columns={\"max\": \"S_2\"}).reset_index()\n    dateshift['S_2'] = pd.to_datetime(dateshift['S_2'])\n    dateshift['mondiff'] = dateshift['S_2'].dt.month - 4\n    dateshift['mondiff'] = dateshift['mondiff'].apply(lambda x: 12+x if x <= 0 else x)\n\n    traindfs[i] = pd.merge(traindfs[i], dateshift[['customer_ID', 'mondiff']], on=['customer_ID'], how='inner')\n    traindfs[i]['yearmonth'] = traindfs[i]['S_2'].dt.strftime(\"%Y%m\")\n\n    traindfs[i]=pd.merge(traindfs[i], pd.merge(pd.DataFrame(traindfs[i]['customer_ID'].unique(), columns=['customer_ID']), dates, how='cross'), on=['customer_ID', 'yearmonth'], how='outer')\n    traindfs[i]['origmissrcnt'] = traindfs[i].apply(lambda x: 1 if (pd.isnull(x['S_2'])) else 0, axis=1)\n    traindfs[i]['totmissrcnt'] = traindfs[i].apply(lambda x: 1 if (pd.isnull(x['S_2'])) else 0, axis=1)\n    traindfs[i]['S_2'] = traindfs[i].apply(lambda x: x['S_2_dt'] if (pd.isnull(x['S_2']) or x['S_2'].month != x['S_2_dt'].month) else x['S_2'], axis=1)\n    traindfs[i].drop(columns=['S_2_dt', 'seq_x', 'seq_y'], inplace=True)\n\n    traindfs[i]['yearmonthorig'].fillna(0, inplace=True)\n    traindfs[i]['day_of_month'] = traindfs[i]['S_2'].dt.strftime(\"%d\")\n    traindfs[i]['day_of_week'] = traindfs[i]['S_2'].dt.strftime(\"%u\")\n    traindfs[i]['day_of_year'] = traindfs[i]['S_2'].dt.strftime(\"%j\")\n    traindfs[i]['week_of_year'] = traindfs[i]['S_2'].dt.isocalendar().week\n    traindfs[i].drop(columns=['mondiff'], inplace=True)\n    traindfs[i].drop(columns=['S_2'], inplace=True)\n\n    traindfs[i]=traindfs[i].set_index(['customer_ID', 'yearmonth']).unstack(1).sort_index(axis=1, level=1)\n    traindfs[i] = traindfs[i].reset_index()\n\n    traindfs[i] = traindfs[i].merge(labels, on=\"customer_ID\", how=\"inner\")\n    traindfs[i].drop(columns=[traindfs[i].columns.to_list()[1]], inplace=True)\n\n    coldict={}\n    for col in traindfs[i].columns:\n        coldict[col] = ''.join(col)\n    traindfs[i].rename(coldict, axis=1, inplace=True)\n\n    traindfs[i].to_csv(\"x\" + str(j) + \"-\" + str(k) +\"dtcnt.csv\", index=False)\n    i+=1\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**preprocess data**","metadata":{}},{"cell_type":"code","source":"@timer\ndef prepareData(X: \"pd.dataFrame\", impute_numericals, impute_categories, encode_categories, imputation_type='simple', numcores=6) -> \"pd.dataFrame\":\n    try:\n        X.drop(columns=['customer_ID'], inplace=True)\n        print(\"dropped customer_ID.\", end=\" \")\n    except:\n        None\n\n    if impute_categories:\n        cols = [col for col in X.columns if col[:-6] in ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']]\n        imp = SimpleImputer(missing_values=np.nan, strategy='constant')\n        imp.fit(X[cols])\n        X[cols]=imp.transform(X[cols])\n        X[cols] = X[cols].astype(\"str\")\n        print(\"imputed categoricals.\", end=\" \")\n\n    if encode_categories:\n        cols = [col for col in X.columns if col[:-6] in ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']]\n        X[cols] = X[cols].astype(\"str\")\n        X[cols] = X[cols].astype(\"category\")\n        enc = OrdinalEncoder()\n        enc.fit(X[cols])\n        X[cols]=enc.transform(X[cols])\n        print(\"encoded categories.\", end=\" \")\n        X[cols] = X[cols].astype(\"category\")\n\n    if impute_numericals:\n        cols=list(X.select_dtypes(include=['int64', 'float64']).columns[X.select_dtypes(include=['int64', 'float64']).isna().any()])\n        imp = SimpleImputer(strategy='median')\n        X[cols] = imp.fit_transform(X[cols])\n        print(\"imputed numericals.\", end=\" \")\n\n    return X\n\n@timer\ndef splitData(X: \"pd.dataFrame\"):\n    y=X['target']\n    X.drop(columns=['target'], inplace=True)\n\n    X_train, X_test, y_train, y_test = train_test_split(X, y,test_size=0.1)\n    print(\"train/test done.\")\n\n    return X_train, y_train, X_test, y_test\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**custom metric function for evaluation**","metadata":{}},{"cell_type":"code","source":"def amex_metric(y_pred: pd.DataFrame, y_true: pd.DataFrame)  -> Tuple[str, float]:\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    yt =pd.DataFrame(y_true.get_label(), columns=['target'])\n    yp =pd.DataFrame(y_pred, columns=['prediction'])\n    g = normalized_weighted_gini(yt, yp)\n    d = top_four_percent_captured(yt, yp)\n    print('g:', g, 'd:', d, end=\" \")\n\n    return 'amexmetric', (0.5 * (g+d))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**xgboost model**","metadata":{}},{"cell_type":"code","source":"@timer\ndef trainXgbModel(X_train, y_train, X_test, y_test, FEATS, ROUNDS) -> \"XGBoost model obj\":    \n    params = {\n                'eta': 0.02,\n                'max_depth': 10,\n                'min_child_weight': 7,\n                'subsample': 0.6,\n                'objective': 'binary:logistic',\n                'eval_metric': ['rmse', 'rmsle', 'auc'],\n                'gpu_id' : 0,\n                'tree_method' : 'gpu_hist',\n                'disable_default_eval_metric': 1\n        }\n\n    dtrain, dtest = xgb.DMatrix(X_train, y_train, feature_names=FEATS, enable_categorical=True), xgb.DMatrix(X_test, y_test, feature_names=FEATS, enable_categorical=True)\n\n    EVAL_LIST = [(dtrain, \"train\"),(dtest, \"test\")]\n\n    xgb_model = xgb.train(params,dtrain,ROUNDS,EVAL_LIST, feval=amex_metric)\n    \n    return xgb_model","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**pytorch tabnet model**","metadata":{}},{"cell_type":"code","source":"@timer\ndef trainTabNetModel(X_train, y_train, pretrain):    \n    tabNet_model = TabNetClassifier(\n                                   n_d=16,\n                                   n_a=16,\n                                   n_steps=4,\n                                   gamma=1.9,\n                                   n_independent=4,\n                                   n_shared=5,\n                                   seed=42,\n                                   optimizer_fn = torch.optim.Adam,\n                                   scheduler_params = {\"milestones\": [150,250,300,350,400,450],'gamma':0.2},\n                                   scheduler_fn=torch.optim.lr_scheduler.MultiStepLR\n                                  )\n\n    tabNet_model.fit(\n        X_train = X_train.to_numpy(),\n        y_train = y_train.to_numpy(),\n        eval_set=[(X_train.to_numpy(), y_train.to_numpy()),\n                  (X_test.to_numpy(), y_test.to_numpy())],\n        max_epochs = 100,\n        batch_size = 256,\n        patience = 10,\n        from_unsupervised = pretrain\n        )\n    \n    return tabNet_model","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**pretrainer**","metadata":{}},{"cell_type":"code","source":"@timer\ndef tabNetPretrain(X_train):\n    tabnet_params = dict(n_d=8, n_a=8, n_steps=3, gamma=1.3,\n                             n_independent=2, n_shared=2,\n                             seed=42, lambda_sparse=1e-3,\n                             optimizer_fn=torch.optim.Adam,\n                             optimizer_params=dict(lr=2e-2,\n                                                   weight_decay=1e-5\n                                                  ),\n                             mask_type=\"entmax\",\n                             scheduler_params=dict(max_lr=0.05,\n                                                   steps_per_epoch=int(X_train.shape[0] / 256),\n                                                   epochs=200,\n                                                   is_batch_level=True\n                                                  ),\n                             scheduler_fn=torch.optim.lr_scheduler.OneCycleLR,\n                             verbose=10\n                        )\n\n    pretrainer = TabNetPretrainer(**tabnet_params)\n\n    pretrainer.fit(\n        X_train=X_train.to_numpy(),\n        eval_set=[X_train.to_numpy()],\n        max_epochs = 100,\n        patience = 10, \n        batch_size = 256, \n        virtual_batch_size = 128,\n        num_workers = 1, \n        drop_last = True)\n    \n    return pretrainer","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**make predictions...**","metadata":{}},{"cell_type":"code","source":"@timer\ndef makePredictions(X_test, xgb_model, tabNet_model):\n    y_xgb_pred = None\n    y_d1_cnn_pred = None\n    y_tabNet_pred = None\n\n    if xgb_model is not None:\n        y_xgb_pred = xgb_model.predict(xgb.DMatrix(X_test, enable_categorical=True))\n    if tabNet_model is not None:\n        y_tabNet_pred = tabNet_model.predict_proba(X_test.to_numpy())[:,1]\n    \n    return [y_xgb_pred, y_tabNet_pred]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**evaluate predictions**","metadata":{}},{"cell_type":"code","source":"@timer\ndef evaluate(y_test, y_xgb_pred, y_tabNet_pred) -> None:\n    preds = {}\n    if y_xgb_pred is not None:\n        preds['XGBoost'] = y_xgb_pred\n    if y_tabNet_pred is not None:\n        preds['TabNet'] = y_tabNet_pred\n\n    for key in preds:\n        print(\"The ROC AUC score of \"+ str(key) +\" model is \" +\n              str(round(roc_auc_score(y_test, preds[key]), 4))\n             )\n\n    for key in preds:\n        print(\"The F1 score of \"+ str(key) +\" model at threshold = 0.27 is \" +\n              str(round(f1_score(y_test, np.where(preds[key] > 0.27, 1, 0)), 4))\n             )","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**plot predictions**","metadata":{}},{"cell_type":"code","source":"# Plot prediction distribution\ndef plotPredictionDistribution(y_xgb_pred, y_tabNet_pred) -> None:\n    preds = {}\n    if y_xgb_pred is not None:\n        preds['XGBoost'] = y_xgb_pred\n    if y_tabNet_pred is not None:\n        preds['TabNet'] = y_tabNet_pred\n\n    for key in preds:\n        plt.hist(preds[key], bins = 100)\n        plt.title(f\"Predicted probability distribution of {key}\")\n        plt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**read base data**","metadata":{}},{"cell_type":"code","source":"@timer\ndef readData(tag, frange, fstart=0):\n    traindf=pd.DataFrame()\n    for i in range(fstart, frange):\n        print(\"x\" + tag + str(i), end=\" \")\n        traindf = pd.concat([traindf, pd.read_csv(\"x\" + tag + str(i) + \"dtcnt.csv\")], ignore_index=True)\n\n    [traindf[col].astype('category') for col in traindf.columns if (col[:len(col)-6] in ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']) or ('yearmonth' in col) or ('day_of_week' in col)]\n    #    traindf[col] = traindf[col].astype('category')\n    #print(\"updated categories...\")\n\n    return traindf","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**balance the data using SMOTE**","metadata":{}},{"cell_type":"code","source":"@timer\ndef balancedata(X_train: \"pd.dataFrame\", y_train: \"pd.dataFrame\") -> (\"pd.dataFrame\", \"pd.dataFrame\"):\n    sm = SMOTE(random_state = 2)\n    X_train_res, y_train_res = sm.fit_sample(X_train, y_train.ravel())\n    print(\"The ratio of target class in training set is \" + str(round(y_train_res.sum()/len(y_train_res) * 100, 2)) + \"%\")\n    print(\"Len of X_train is\", len(X_train), \" & X_train_res is\", len(X_train_res))\n    return X_train_res, y_train_res\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**load data and balance it**","metadata":{}},{"cell_type":"code","source":"#Load and split the data\n\nROUNDS = 51\nloads=[False, True, True, True] # [csv, xgb, d1cnn, tabnet]\nnof, fstart=12, 0\nfprefix=\"\" #str(nof)\nftag = \"\".join([str(i) for i in range(nof)])\nftag=\"all\"\n\nif loads[0]:\n    print(\"Loading the data\", \"X\" + fprefix + ftag + \"bal.csv\", end=\" \")\n    X = pd.read_csv(\"X\" + fprefix + ftag + \"bal.csv\")\nelse:\n    print(\"Reading the data\", end=\" \")\n    df = readData(fprefix, nof, fstart)\n    print(\"Preprocessing the data\", end=\" \")\n    X=prepareData(df.loc[:, ~df.isnull().all()], True, True, True)\n    X.to_csv(\"X\" + fprefix + ftag + \".csv\", index=False)\n    y=X['target']\n    X_train_res, y_train_res = balancedata(X, y)\n    if 'target' not in X.columns:\n        X=pd.concat([X_train_res, pd.DataFrame(y_train_res, columns=[\"target\"])], axis='columns')\n    else:\n        X= X_train_res\n    X.to_csv(\"X\" + fprefix + ftag + \"bal.csv\", index=False)\n\nprint(\"\")\ny=X['target']\n\nFEATS = [col for col in X.columns if (col[:len(col)-6] in ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']) or ('yearmonth' in col) or ('day_of_week' in col) or ('day_of_month') in col or ('day_of_year' in col)]\n\nX_train, y_train, X_test, y_test = splitData(X)\n\nprint(\"The ratio of target class in training set is \" + str(round(y_train.sum()/len(y_train) * 100, 2)) + \"%\")\nprint(\"The ratio of target class in test set is \" + str(round(y_test.sum()/len(y_test) * 100, 2)) + \"%\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**train the models**","metadata":{}},{"cell_type":"code","source":"torch.cuda.empty_cache() # PyTorch thing\n\nloads=[True, True, True, False] # [csv, xgb, d1cnn, tabnet]\nftag = \"\".join([str(i) for i in range(nof)])\ntag='xgb'\nROUNDS=1200\n\nif loads[1]:\n    print(\"Loading XGBoost model\")\n    xgb_model = pickle.load(open(tag+\"X\" + ftag + \"model\", \"rb\"))\nelse:\n    print(\"Training XGBoost model\")\n    xgb_model = trainXgbModel(X_train, y_train, X_test, y_test, X_train.columns, ROUNDS)\n    pickle.dump(xgb_model, open(tag+\"X\" + str(nof) + \"model\", \"wb\"))\n\n\ntag='tabnet'\nif loads[3]:\n    print(\"Loading TabNet model\")\n    tabNet_model = TabNetClassifier()\n    tabNet_model.load_model(tag+\"X\" + ftag + \"model.zip\")\nelse:\n    print(\"Training TabNet model\")\n    tabNet_model = trainTabNetModel(X_train_res, y_train_res, tabNetPretrain(X_train))\n    tabNet_model.save_model(tag+\"X\" + ftag + \"model\")\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**predictions**","metadata":{}},{"cell_type":"code","source":"###Train Predicitons\nprint(\"Making predictions\")\ny_xgb_pred, y_tabNet_pred = makePredictions(X_train, xgb_model, tabNet_model)\n\nprint(\"Evaluation of the model\")\nevaluate(y_train, y_xgb_pred, y_tabNet_pred)\n\nprint(\"Prediction distribution\")\nplotPredictionDistribution(y_xgb_pred, y_tabNet_pred)\n\n\n###Test Predictions\nprint(\"Making predictions\")\ny_xgb_pred, y_tabNet_pred = makePredictions(X_test, xgb_model, tabNet_model)\n\nprint(\"Evaluation of the model\")\nevaluate(y_test, y_xgb_pred, y_tabNet_pred)\n\nprint(\"Prediction distribution\")\nplotPredictionDistribution(y_xgb_pred, y_tabNet_pred)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**predictions for competition test**","metadata":{}},{"cell_type":"markdown","source":"**for data with less than 13 dates**","metadata":{}},{"cell_type":"code","source":"testdf = pd.DataFrame()\nfor i in range(12):\n    print(\"t\" + str(i), end=\" \")\n    testdf = pd.concat([testdf, pd.read_csv(\"t\" + str(i) + \"13dts.csv\")], ignore_index=True)\ncust = testdf[['customer_ID']]\ntestdf.drop(columns=['origmissrcnt', 'totmissrcnt'], inplace=True)\nXtest=prepareData(testdf, False, False, True)\nprint(\"Making predictions...\", end=\" \")\ny_xgb_pred, y_d1_cnn_pred, y_tabNet_pred = makePredictions(Xtest[X.columns], xgb_model, None, None)\npd.concat([cust, pd.DataFrame(y_xgb_pred, columns=[\"Label\"])], axis='columns').to_csv(\"test/t\" + \"01234567891011\" + \".preds\", index=False)\nprint(\"saved predictions.\")\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**for data with 13 dates**","metadata":{}},{"cell_type":"code","source":"for i in range(41):\n    print(\"t12-\" + str(i) + \"dtcnt.csv\", end=\" \")\n    testdf = pd.read_csv(\"t12-\" + str(i) + \"dtcnt.csv\")\n    cust = testdf[['customer_ID']]\n    testdf.drop(columns=['origmissrcnt', 'totmissrcnt'], inplace=True)\n    X=prepareData(testdf, False, False, True)\n    print(\"Making predictions...\", end=\" \")\n    y_xgb_pred, y_d1_cnn_pred, y_tabNet_pred = makePredictions(X, xgb_model, None, None)\n    pd.concat([cust, pd.DataFrame(y_xgb_pred, columns=[\"Label\"])], axis='columns').to_csv(\"test/t12-\" + str(i) + \".preds\", index=False)\n    print(\"saved predictions.\")\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**evaluating the predictions**","metadata":{}},{"cell_type":"code","source":"for i in range(41):\n    print(\"t12-\" + str(i) + \"dtcnt.csv\", end=\" \")\n    testdf = pd.read_csv(\"t12-\" + str(i) + \"dtcnt.csv\")\n    cust = testdf[['customer_ID']]\n    testdf.drop(columns=['origmissrcnt', 'totmissrcnt'], inplace=True)\n    X=prepareData(testdf, False, False, True)\n    print(\"Making predictions...\", end=\" \")\n    y_xgb_pred, y_d1_cnn_pred, y_tabNet_pred = makePredictions(X, xgb_model, None, None)\n    pd.concat([cust, pd.DataFrame(y_xgb_pred, columns=[\"Label\"])], axis='columns').to_csv(\"test/t12-\" + str(i) + \".preds\", index=False)\n    print(\"saved predictions.\")\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**rolling up the test predictions into single entity**","metadata":{}},{"cell_type":"code","source":"samples = pd.read_csv(root + \"sample_submission.csv\")\n\npreddf= pd.DataFrame()\nrangemin, rangemax = 0, 1\n\npreds = pd.read_csv(root+\"/test/t\"+ \"01234567891011\" + \".preds\")\nprint(preds['Label'].min(), preds['Label'].max())\npreds['Label'] = ( (preds[['Label']] - rangemin) / (rangemax - rangemin) ) * (1 - 0) + 0\nprint(preds['Label'].min(), preds['Label'].max())\npreddf = pd.concat([preddf, preds], ignore_index=True, axis=0)\n\nfor i in range(41):\n    preds = pd.read_csv(root+\"/test/t12-\"+ str(i) + \".preds\") #.rename(columns={\"prediction\":\"Label\"}).drop(columns=\"Unnamed: 0\")\n    if (preds['Label'].agg(['min', 'max'])[0] < 0) or (preds['Label'].agg(['min', 'max'])[1]>1):\n        print(i, preds['Label'].agg(['min', 'max'])[0], preds['Label'].agg(['min', 'max'])[1])\n        rangemin = min(preds['Label'].min(), rangemin)\n        rangemax = max(preds['Label'].max(), rangemax)\n        preds['Label'] = ( (preds[['Label']] - rangemin) / (rangemax - rangemin) ) * (1 - 0) + 0\n    preddf = pd.concat([preddf, preds], ignore_index=True, axis=0)\n\nprint(\"preds min max:\", (preddf['Label'].min(), preddf['Label'].max()))\n\n\nvalids = pd.merge(samples, preddf, on='customer_ID', how='inner')\nvalids.rename(columns={'prediction': 'target'}, inplace=True)\nvalids.rename(columns={'Label': 'prediction'}, inplace=True)\n\nvalids['prediction'].fillna(0, inplace=True)\n\nprint(\"matching 1's\", valids[(valids['target'] == round(valids['prediction'])) & (valids['target'] == 1)].count()[0], valids[(valids['target'] == 1)].count()[0])\nprint(\"matching 0's\", valids[(valids['target'] == round(valids['prediction'])) & (valids['target'] == 0)].count()[0], valids[(valids['target'] == 0)].count()[0])\nprint(\"matching\", valids[valids['target'] == round(valids['prediction'])].count()[0], \"total\", valids.count()[0], \"%\", valids[valids['target'] == round(valids['prediction'])].count()[0]/valids.count()[0])\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Saving the output for submission**","metadata":{}},{"cell_type":"code","source":"valids[['customer_ID', 'prediction']].to_csv('submission.csv', index=False)\n","metadata":{},"execution_count":null,"outputs":[]}]}