{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Reference\nhttps://www.kaggle.com/code/medali1992/amex-tabnetclassifier-feature-eng-0-791/notebook CV 0.78920 LB 0.791\n\nhttps://github.com/DataCanvasIO/DeepTables\n\nhttps://deeptables.readthedocs.io/en/latest/model_config.html#parameters","metadata":{}},{"cell_type":"markdown","source":"Please play with this. It's a very good stuff, but underestimated for now, I guess. Haven't spent much time on this yet. So try another more suitable dataset with more or fewer number features, choose another preprocess options, explore deep models, configurate layers, sum them togerther to construct new models. Also, if you have more computational resources try autotune trough DeepTables AutoML. It provides feature selection and model explanation as well. After making number of good scored models, check its diversity and make an ensemble.","metadata":{}},{"cell_type":"code","source":"!pip -q install deeptables","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport random\nimport pickle\nimport time\nimport os\nimport gc\n\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.model_selection import StratifiedKFold\n\nimport tensorflow as tf, deeptables as dt\nfrom tensorflow.keras.utils import plot_model\nfrom deeptables.models import DeepTable, ModelConfig\nfrom deeptables.models import deepnets\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nprint('TensorFlow version',tf.__version__)\nprint('DeepTables version',dt.__version__)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    debug = False\n    n_folds = 5\n    seed = 42\n    batch_size = 128","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\nseed_everything(seed=CFG.seed)\n\n\ndef amex_metric_mod(y_true, y_pred):\n\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n        \n    return 0.5 * (gini[1]/gini[0] + top_four)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfeatures_avg = ['B_11', 'B_13', 'B_14', 'B_15', 'B_16', 'B_17', 'B_18', 'B_19', 'B_2', \n                'B_20', 'B_28', 'B_29', 'B_3', 'B_33', 'B_36', 'B_37', 'B_4', 'B_42', \n                'B_5', 'B_8', 'B_9', 'D_102', 'D_103', 'D_105', 'D_111', 'D_112', 'D_113', \n                'D_115', 'D_118', 'D_119', 'D_121', 'D_124', 'D_128', 'D_129', 'D_131', \n                'D_132', 'D_133', 'D_139', 'D_140', 'D_141', 'D_143', 'D_144', 'D_145', \n                'D_39', 'D_41', 'D_42', 'D_43', 'D_44', 'D_45', 'D_46', 'D_47', 'D_48', \n                'D_49', 'D_50', 'D_51', 'D_52', 'D_56', 'D_58', 'D_62', 'D_70', 'D_71', \n                'D_72', 'D_74', 'D_75', 'D_79', 'D_81', 'D_83', 'D_84', 'D_88', 'D_91', \n                'P_2', 'P_3', 'R_1', 'R_10', 'R_11', 'R_13', 'R_18', 'R_19', 'R_2', 'R_26', \n                'R_27', 'R_28', 'R_3', 'S_11', 'S_12', 'S_22', 'S_23', 'S_24', 'S_26', \n                'S_27', 'S_5', 'S_7', 'S_8', ]\nfeatures_min = ['B_13', 'B_14', 'B_15', 'B_16', 'B_17', 'B_19', 'B_2', 'B_20', 'B_22', \n                'B_24', 'B_27', 'B_28', 'B_29', 'B_3', 'B_33', 'B_36', 'B_4', 'B_42', \n                'B_5', 'B_9', 'D_102', 'D_103', 'D_107', 'D_109', 'D_110', 'D_111', \n                'D_112', 'D_113', 'D_115', 'D_118', 'D_119', 'D_121', 'D_122', 'D_128', \n                'D_129', 'D_132', 'D_133', 'D_139', 'D_140', 'D_141', 'D_143', 'D_144', \n                'D_145', 'D_39', 'D_41', 'D_42', 'D_45', 'D_46', 'D_48', 'D_50', 'D_51', \n                'D_53', 'D_54', 'D_55', 'D_56', 'D_58', 'D_59', 'D_60', 'D_62', 'D_70', \n                'D_71', 'D_74', 'D_75', 'D_78', 'D_79', 'D_81', 'D_83', 'D_84', 'D_86', \n                'D_88', 'D_96', 'P_2', 'P_3', 'P_4', 'R_1', 'R_11', 'R_13', 'R_17', 'R_19', \n                'R_2', 'R_27', 'R_28', 'R_4', 'R_5', 'R_8', 'S_11', 'S_12', 'S_23', 'S_25', \n                'S_3', 'S_5', 'S_7', 'S_9', ]\nfeatures_max = ['B_1', 'B_11', 'B_13', 'B_15', 'B_16', 'B_17', 'B_18', 'B_19', 'B_2', \n                'B_22', 'B_24', 'B_27', 'B_28', 'B_29', 'B_3', 'B_31', 'B_33', 'B_36', \n                'B_4', 'B_42', 'B_5', 'B_7', 'B_9', 'D_102', 'D_103', 'D_105', 'D_109', \n                'D_110', 'D_112', 'D_113', 'D_115', 'D_121', 'D_124', 'D_128', 'D_129', \n                'D_131', 'D_139', 'D_141', 'D_144', 'D_145', 'D_39', 'D_41', 'D_42', \n                'D_43', 'D_44', 'D_45', 'D_46', 'D_47', 'D_48', 'D_50', 'D_51', 'D_52', \n                'D_53', 'D_56', 'D_58', 'D_59', 'D_60', 'D_62', 'D_70', 'D_72', 'D_74', \n                'D_75', 'D_79', 'D_81', 'D_83', 'D_84', 'D_88', 'D_89', 'P_2', 'P_3', \n                'R_1', 'R_10', 'R_11', 'R_26', 'R_28', 'R_3', 'R_4', 'R_5', 'R_7', 'R_8', \n                'S_11', 'S_12', 'S_23', 'S_25', 'S_26', 'S_27', 'S_3', 'S_5', 'S_7', 'S_8', ]\nfeatures_last = ['B_1', 'B_11', 'B_12', 'B_13', 'B_14', 'B_16', 'B_18', 'B_19', 'B_2', \n                 'B_20', 'B_21', 'B_24', 'B_27', 'B_28', 'B_29', 'B_3', 'B_30', 'B_31', \n                 'B_33', 'B_36', 'B_37', 'B_38', 'B_39', 'B_4', 'B_40', 'B_42', 'B_5', \n                 'B_8', 'B_9', 'D_102', 'D_105', 'D_106', 'D_107', 'D_108', 'D_110', \n                 'D_111', 'D_112', 'D_113', 'D_114', 'D_115', 'D_116', 'D_117', 'D_118', \n                 'D_119', 'D_120', 'D_121', 'D_124', 'D_126', 'D_128', 'D_129', 'D_131', \n                 'D_132', 'D_133', 'D_137', 'D_138', 'D_139', 'D_140', 'D_141', 'D_142', \n                 'D_143', 'D_144', 'D_145', 'D_39', 'D_41', 'D_42', 'D_43', 'D_44', 'D_45', \n                 'D_46', 'D_47', 'D_48', 'D_49', 'D_50', 'D_51', 'D_52', 'D_53', 'D_55', \n                 'D_56', 'D_59', 'D_60', 'D_62', 'D_63', 'D_64', 'D_66', 'D_68', 'D_70', \n                 'D_71', 'D_72', 'D_73', 'D_74', 'D_75', 'D_77', 'D_78', 'D_81', 'D_82', \n                 'D_83', 'D_84', 'D_88', 'D_89', 'D_91', 'D_94', 'D_96', 'P_2', 'P_3', \n                 'P_4', 'R_1', 'R_10', 'R_11', 'R_12', 'R_13', 'R_16', 'R_17', 'R_18', \n                 'R_19', 'R_25', 'R_28', 'R_3', 'R_4', 'R_5', 'R_8', 'S_11', 'S_12', \n                 'S_23', 'S_25', 'S_26', 'S_27', 'S_3', 'S_5', 'S_7', 'S_8', 'S_9', ]\nfeatures_categorical = ['B_30_last', 'B_38_last', 'D_114_last', 'D_116_last',\n                        'D_117_last', 'D_120_last', 'D_126_last',\n                        'D_63_last', 'D_64_last', 'D_66_last', 'D_68_last']\n\nINFERENCE = True\nfor i in [0, 1] if INFERENCE else [0]:\n    # i == 0 -> process the train data\n    # i == 1 -> process the test data\n    df = pd.read_feather(['../input/amexfeather/train_data.ftr',\n                          '../input/amexfeather/test_data.ftr'][i]) \n    cid = pd.Categorical(df.pop('customer_ID'), ordered=True)\n    last = (cid != np.roll(cid, -1)) # Mask for last statement of every customer\n    if i == 0: # train\n        target = df.loc[last, 'target']\n    print('Read', i)\n    gc.collect()\n    df_avg = (df\n              .groupby(cid)\n              .mean()[features_avg]\n              .rename(columns={f: f\"{f}_avg\" for f in features_avg})\n             )\n    print('Computed avg', i)\n    gc.collect()\n    df_max = (df\n              .groupby(cid)\n              .max()[features_max]\n              .rename(columns={f: f\"{f}_max\" for f in features_max})\n             )\n    print('Computed max', i)\n    gc.collect()\n    df_min = (df\n              .groupby(cid)\n              .min()[features_min]\n              .rename(columns={f: f\"{f}_min\" for f in features_min})\n             )\n    print('Computed min', i)\n    gc.collect()\n    df_last = (df.loc[last, features_last]\n               .rename(columns={f: f\"{f}_last\" for f in features_last})\n               .set_index(np.asarray(cid[last]))\n              )\n    df = None # We no longer need the original data\n    print('Computed last', i)\n    \n    df_categorical = df_last[features_categorical].astype(object)\n    features_not_cat = [f for f in df_last.columns if f not in features_categorical]\n    if i == 0: # train\n        ohe = OneHotEncoder(drop='first', sparse=False, dtype=np.float32, handle_unknown='ignore')\n        ohe.fit(df_categorical)\n        with open(\"ohe.pickle\", 'wb') as f: pickle.dump(ohe, f)\n    df_categorical = pd.DataFrame(ohe.transform(df_categorical).astype(np.float16),\n                                  index=df_categorical.index).rename(columns=str)\n    print('Computed categorical', i)\n    \n    categorical_columns = df_categorical.columns.values.tolist()\n    print('Categorical columns num', len(categorical_columns))\n    \n    df = pd.concat([df_last[features_not_cat], df_categorical, df_avg, df_min, df_max], axis=1)\n    \n    # Impute missing values\n    df.fillna(value=0, inplace=True)\n    \n    del df_avg, df_max, df_min, df_last, df_categorical, cid, last, features_not_cat\n    \n    if i == 0: # train\n        # Free the memory\n        df.reset_index(drop=True, inplace=True) # Frees 0.2 GByte\n        df.to_feather('train_processed.ftr')\n        df = None\n        gc.collect()\n\ntrain = pd.read_feather('train_processed.ftr')\ntest = df\ntarget = target.reset_index(drop=True)\ndel df, ohe\n\nprint('Train Shapes:', train.shape, target.shape)\nif INFERENCE: print('Test Shapes:', test.shape)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#https://www.kaggle.com/code/cdeotte/tensorflow-transformer-0-790/notebook\nimport math\nimport matplotlib.pyplot as plt\n\nLR_START = 5e-6\nLR_MAX = 1e-3\nLR_MIN = 1e-7\nLR_RAMPUP_EPOCHS = 3\nLR_SUSTAIN_EPOCHS = 0\nEPOCHS = 12\n\ndef lrfn(epoch):\n    if epoch < LR_RAMPUP_EPOCHS:\n        lr = (LR_MAX - LR_START) / LR_RAMPUP_EPOCHS * epoch + LR_START\n    elif epoch < LR_RAMPUP_EPOCHS + LR_SUSTAIN_EPOCHS:\n        lr = LR_MAX\n    else:\n        decay_total_epochs = EPOCHS - LR_RAMPUP_EPOCHS - LR_SUSTAIN_EPOCHS - 1\n        decay_epoch_index = epoch - LR_RAMPUP_EPOCHS - LR_SUSTAIN_EPOCHS\n        phase = math.pi * decay_epoch_index / decay_total_epochs\n        cosine_decay = 0.5 * (1 + math.cos(phase))\n        lr = (LR_MAX - LR_MIN) * cosine_decay + LR_MIN\n        \n    return lr\n\nrng = [i for i in range(EPOCHS)]\nlr_y = [lrfn(x) for x in rng]\nplt.figure(figsize=(10, 4))\nplt.plot(rng, lr_y, '-o')\nplt.xlabel('Epoch'); plt.ylabel('LR')\nprint(\"Learning rate schedule: {:.3g} to {:.3g} to {:.3g}\". \\\n      format(lr_y[0], max(lr_y), lr_y[-1]))\nLR = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose = True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CFG.epochs = 12\n\nfrom tensorflow_addons.optimizers import AdamW\noptimizer = AdamW(learning_rate=1e-3, weight_decay=4.3e-5)\n\nconf = ModelConfig(auto_imputation=False,\n                   auto_categorize=False,\n                   categorical_columns=categorical_columns,\n                   embeddings_output_dim=4,\n                   embedding_dropout=0.3,\n                   nets=deepnets.FGCNN,\n                   #nets=['fgcnn_dnn_nets', 'fgcnn_ipnn_nets'],\n                   #nets=deepnets.FGCNN+deepnets.xDeepFM,\n                   #nets=deepnets.FGCNN+['linear'],\n                   #nets=['linear', 'dnn_nets'],\n                   #dnn_params={\n                   #    'hidden_units': ((512, 0.3, True), (256, 0.3, True)),\n                   #    'dnn_activation': 'relu',\n                   #},\n                   stacking_op='concat',  # 'add'\n                   output_use_bias=False,\n                   metrics=['binary_crossentropy', 'AUC'],  # First metric is used for early stopping\n                   earlystopping_patience=3,\n                   )","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_predictions = np.zeros((train.shape[0]))\ntest_predictions = np.zeros(test.shape[0])\nkfold = StratifiedKFold(n_splits = CFG.n_folds, shuffle=True, random_state = CFG.seed)\nfor fold, (train_idx, valid_idx) in enumerate(kfold.split(train, target)):\n    print(f\"**************************** \")\n    print(f\"********** Fold {fold+1} ********** \",'\\n')\n    # DEBUG MODE\n    if CFG.debug == True:\n        if fold > 0:\n            print('\\nDebug mode activated: Will train only one fold...\\n')\n            break\n            \n    start = time.time() \n     \n    X_train, y_train = train.loc[train_idx], target.loc[train_idx]\n    X_valid, y_valid = train.loc[valid_idx], target.loc[valid_idx]\n    \n    for i in range(len(optimizer.weights)):\n        optimizer.weights[i]._handle_name = optimizer.weights[i].name + '_' + str(fold) + str(i)\n    conf = conf._replace(optimizer=optimizer)\n    \n    model = DeepTable(config=conf)\n    model.fit(X_train, y_train, batch_size=CFG.batch_size, epochs=CFG.epochs, verbose=2,\n                                                           callbacks=[LR])\n    \n    val_score_fold = model.evaluate(X_valid, y_valid, batch_size=CFG.batch_size*4, verbose=0)\n    print('##############################')\n    print(f'########## Model Metric Fold {fold+1}:',list(val_score_fold.items())[1:],'\\n')\n    preds = model.predict_proba(X_valid, batch_size=CFG.batch_size*4)[:, 1]\n    val_score_fold = amex_metric_mod(y_valid.values.flatten(), preds)\n    print('#############################')\n    print(f'########## Amex Metric Fold {fold+1} =',val_score_fold,'\\n')\n    oof_predictions[valid_idx] = oof_predictions[valid_idx] + preds\n    del X_train, y_train\n    del X_valid, y_valid\n    gc.collect()\n    \n    test_predictions += model.predict_proba(test, batch_size=CFG.batch_size*4)[:, 1]/5\n    \n    end = time.time()\n    time_delta = np.round((end - start)/60, 2)\n    print(f'\\nFold {fold+1}/{CFG.n_folds} | {time_delta:.2f} min','\\n')\n    \nprint(f'OOF Amex metric across folds: {amex_metric_mod(target.values.flatten(), oof_predictions.flatten())}')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame({'customer_ID': test.index, 'prediction': test_predictions})\nsub.to_csv('submission_deeptables.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_model(model.get_model().model)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}