{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Info\n\nThis Notebook was initially built off of [this Notebook](https://www.kaggle.com/code/ambrosm/amex-keras-quickstart-1-training) by AmbrosM\n\nI mainly changed the data, model structure, optimizer, and loss function. \n\nI use the adabelief optimizer\n\nThe datasets I use as train datasets are essentially the datasets from [this lgbm dart notebook](https://www.kaggle.com/code/ragnar123/amex-lgbm-dart-cv-0-7977) with standard scaling and some additional features including second-to-last features and their respective differences from the last feature\n\nThis model gets 0.790-0.791 cv each seed and 0.7928 averaged cv across 4 seeds.","metadata":{}},{"cell_type":"code","source":"%%capture\n!pip install adabelief-tf --no-cache-dir \nfrom adabelief_tf import AdaBeliefOptimizer","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:04:06.763371Z","iopub.execute_input":"2022-08-08T20:04:06.763873Z","iopub.status.idle":"2022-08-08T20:04:50.334192Z","shell.execute_reply.started":"2022-08-08T20:04:06.763775Z","shell.execute_reply":"2022-08-08T20:04:50.332969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport dill as pickle   \nfrom matplotlib import pyplot as plt\nimport random\nimport datetime\nimport math\nimport gc\nimport os\nimport warnings\nimport seaborn as sns\nimport itertools\nimport multiprocessing\nimport joblib\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\nfrom matplotlib.ticker import MaxNLocator\nfrom colorama import Fore, Back, Style\nfrom tqdm import tqdm\nimport h5py\n\nfrom sklearn.model_selection import StratifiedKFold,KFold\nfrom sklearn.preprocessing import StandardScaler, QuantileTransformer, OneHotEncoder, PowerTransformer\nfrom sklearn.metrics import roc_curve, roc_auc_score, average_precision_score\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.utils import class_weight \nfrom sklearn.utils.class_weight import compute_class_weight\n\nimport tensorflow_addons as tfa\nimport tensorflow as tf\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '1'\ntf.config.threading.set_inter_op_parallelism_threads(4)\nfrom tensorflow import keras\nfrom tensorflow.keras.models import Model, load_model,model_from_json\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, LearningRateScheduler, EarlyStopping\nfrom tensorflow.keras.layers import Dense, Input, InputLayer, Add, Concatenate, Dropout, BatchNormalization, Conv1D, Reshape, Flatten, AveragePooling1D, MaxPool1D\nfrom tensorflow.keras.utils import plot_model\nfrom tensorflow.keras.losses import binary_crossentropy\nimport tensorflow.keras.backend as K","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-08T20:04:50.336489Z","iopub.execute_input":"2022-08-08T20:04:50.337330Z","iopub.status.idle":"2022-08-08T20:04:51.347608Z","shell.execute_reply.started":"2022-08-08T20:04:50.337274Z","shell.execute_reply":"2022-08-08T20:04:51.346604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex_metric(y_true, y_pred):\n\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n\n    return 0.5 * (gini[1]/gini[0] + top_four)\n    \nbest_score = 0\nclass MyCustomMetricCallback(tf.keras.callbacks.Callback):\n\n    def __init__(self, save_path, train=None, validation=None, best_score = 0):\n        super(MyCustomMetricCallback, self).__init__()\n        self.train = train\n        self.validation = validation\n        self.best_score = best_score\n        self.save_path =save_path\n        self.best_epoch = 0\n\n    def on_epoch_end(self, epoch, logs={}):\n        if self.train:\n#             logs['my_metric_train'] = float('inf')\n#             X_train, y_train = self.train[0], self.train[1]\n#             y_pred = self.model.predict(X_train)\n#             score = amex_metric_tensor(y_train, y_pred)\n#             logs['my_metric_train'] = np.round(score, 5)\n            pass\n\n        if self.validation:\n            X_valid, y_valid = self.validation[0], self.validation[1]\n            y_pred = self.model.predict(X_valid).reshape( (len(X_valid), )) \n            val_score = amex_metric(y_valid, y_pred)\n            logs['my_metric_val'] = val_score\n            if val_score>self.best_score:\n                self.best_score = val_score\n                self.best_epoch = epoch\n                self.model.save(f\"{self.save_path}.h5\")\n                print('best_val_score: ', self.best_score)\n            elif self.best_epoch==0:\n                if epoch-self.best_epoch > 40:\n                    self.model.stop_training = True\n            elif epoch-self.best_epoch > 12:\n                self.model.stop_training = True\n                \n            del X_valid, y_valid, y_pred, val_score\n            gc.collect()\n#Keras\n# def DiceBCELoss(targets, inputs, smooth=1e-6):  \n#     inputs = K.flatten(inputs)\n#     targets = K.flatten(targets)\n#     BCE =  binary_crossentropy(targets, inputs)\n#     intersection = K.sum(K.dot(targets, inputs))    \n#     dice_loss = 1 - (2*intersection + smooth) / (K.sum(targets) + K.sum(inputs) + smooth)\n#     Dice_BCE = BCE + dice_loss\n    \n#     return Dice_BCE\n# ALPHA = 0.8\nALPHA= 5\nGAMMA = 2\ndef FocalLoss(targets, inputs, alpha=ALPHA, gamma=GAMMA):    \n    BCE = K.binary_crossentropy(targets, inputs)\n    BCE_EXP = K.exp(-BCE)\n    focal_loss = K.mean(alpha * K.pow((1-BCE_EXP), gamma) * BCE)\n    \n    return focal_loss","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-08T20:04:51.349200Z","iopub.execute_input":"2022-08-08T20:04:51.349818Z","iopub.status.idle":"2022-08-08T20:04:51.370388Z","shell.execute_reply.started":"2022-08-08T20:04:51.349778Z","shell.execute_reply":"2022-08-08T20:04:51.369341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# The model\n\nEssentially the data I pass into the model is in the format of 15 features per category(P_2, B_3 etc.). So for example the first few features would be something like P_2_mean, P_2_max, P_2_last_mean_diff up to 15 features related to P_2 then there would be 15 features related to the next feature and so on. \n\nThe Conv1D processes all of the features related to that one feature individually, which leads to better model performance","metadata":{}},{"cell_type":"code","source":"def my_model(n_inputs):\n    \"\"\"Sequential neural network with a skip connection.\n    \n    Returns a compiled instance of tensorflow.keras.models.Model.\n    \"\"\"\n    activation = 'swish'\n    inputs = Input(shape=(n_inputs, ))\n    x = Reshape((n_inputs, 1))(inputs)\n    # 15 agg features per main feature, size = 15, step = 15.\n    x = Conv1D(24,15,strides=15, activation=activation)(x)\n    x = BatchNormalization()(x)\n    x = Conv1D(12,1, activation=activation)(x)\n    x = BatchNormalization()(x)\n    x = Conv1D(4,1, activation=activation)(x)\n    x = BatchNormalization()(x)\n    x = Flatten()(x)\n    x = Dropout(0.33)(x)\n    x = Dense(32, activation = activation)(x)\n    x = BatchNormalization()(x)\n    x = Dropout(0.1)(x)\n    x = Dense(16, activation = activation)(x)\n    outputs = Dense(1, activation='sigmoid')(x)\n    gc.collect()\n    return Model(inputs, outputs)\nmodel = my_model(190*15)\nmodel.summary()\ndel model","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:04:51.372452Z","iopub.execute_input":"2022-08-08T20:04:51.373062Z","iopub.status.idle":"2022-08-08T20:04:51.831505Z","shell.execute_reply.started":"2022-08-08T20:04:51.373009Z","shell.execute_reply":"2022-08-08T20:04:51.830778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Cross-validation\n\nWe use a standard cross-validation loop. In the loop, we train a model. We use a StratifiedKFold because the data is imbalanced.\n\neach fold takes approx ~1 hour to run","metadata":{}},{"cell_type":"code","source":"VERBOSE = 0\nCYCLES = 1\nEPOCHS = 400\nBATCH_SIZE = 2048\nFOLDS = 10\nSEED = 0\nCURRENT_FOLD = 9\n\ndef fit_model(seed, fold):\n    train = joblib.load('../input/amex-keras-dataset/train_agg_extra2_scaled.pkl').astype('float16')\n    #turn 2d feature array of shape (190, 15) into 1d array\n    train = np.array([i.flatten('C') for i in train])\n    gc.collect()\n    \n    target = pd.read_csv(f'../input/amex-default-prediction/train_labels.csv').target.astype('float32')\n    idx_tr, idx_va = list(StratifiedKFold(n_splits=FOLDS, shuffle= True, random_state= SEED).split(target,target))[fold]\n    X_va = train[idx_va]\n    X_tr = train[idx_tr]\n    y_tr, y_va = target[idx_tr], target[idx_va]\n    del train, target, idx_tr\n    gc.collect()\n    lr = ReduceLROnPlateau(monitor=\"val_loss\", \n                           factor=0.3, \n                           patience=5, \n                           mode = 'min', \n                           verbose=VERBOSE)\n    es = EarlyStopping(monitor=\"val_loss\",\n                       patience=7, \n                       min_delta=0.00001,\n                       verbose=VERBOSE,\n                       mode=\"min\", \n                       restore_best_weights=True)\n    best_score= 0\n    for seed1 in range(10):\n        print('seed: ',seed1)\n        np.random.seed(seed1)\n        random.seed(seed1)\n        tf.random.set_seed(seed1)\n        custom = MyCustomMetricCallback(save_path = f'model_fold{fold}_seed{seed}', validation=(X_va, y_va), best_score=best_score)\n                    \n        callbacks = [lr, \n                     es, \n                     tf.keras.callbacks.TerminateOnNaN(), \n                     custom,\n                     ]\n        model = my_model(X_tr.shape[1])\n        model.compile(optimizer=AdaBeliefOptimizer(learning_rate=0.02,\n                                                   weight_decay = 1e-5,\n                                                   epsilon = 1e-7,\n                                                   print_change_log = False,\n            ),\n            loss=FocalLoss,\n            )\n        gc.collect()\n        model.fit(X_tr, y_tr, \n                validation_data=(X_va, y_va),\n                epochs=EPOCHS,\n                verbose=VERBOSE,\n                batch_size=BATCH_SIZE,\n                shuffle=True,\n                callbacks=callbacks)\n        best_score = custom.best_score\n        gc.collect()\n        K.clear_session()\n        \n\ndef fit_train_models(current_fold = 0):\n    print(f'KFOLD WITH SEED {SEED}:')\n    fit_model(SEED, current_fold)\n    gc.collect()\n# fit_train_models(current_fold = CURRENT_FOLD)\n# 0.2170, shape=(), dtype=float32) 0.7919 5","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-08T20:04:51.832821Z","iopub.execute_input":"2022-08-08T20:04:51.833666Z","iopub.status.idle":"2022-08-08T20:04:51.848738Z","shell.execute_reply.started":"2022-08-08T20:04:51.833631Z","shell.execute_reply":"2022-08-08T20:04:51.847710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference\n\nBecause my dataset is too big, I need to run the inference after training the model on a separate run. \n\nWe also need to split the test data into multiple parts because of memory.\n\nThe following code is the result of running the above training 4 times with seeds 0,1,3,4","metadata":{}},{"cell_type":"code","source":"# # get oof predictions\n# train = joblib.load('../input/amex-keras-dataset/train_agg_extra2_scaled.pkl').astype('float16')\n# train = np.array([i.flatten('C') for i in train])\n# gc.collect()\n# target = pd.read_csv(f'../input/amex-default-prediction/train_labels.csv').target.astype('float32')\n\n# oof_predictions = np.zeros(len(train))\n\n# folds = [i for i in range(10)]\n# seeds = [0]\n\n# gc.collect()\n# for seed in tqdm(seeds):\n#     oof_pred = np.zeros(len(train))\n#     for fold, (idx_tr, idx_va) in enumerate(StratifiedKFold(n_splits=10, shuffle= True, random_state= seed).split(target,target)):\n#         model = tf.keras.models.load_model(f'../input/amex-keras-models/model_fold{fold}_seed{seed}.h5',custom_objects={\"AdaBeliefOptimizer\":AdaBeliefOptimizer(learning_rate=0.02,weight_decay = 1e-5,epsilon = 1e-7,print_change_log = False,),                                                                                                            'FocalLoss': FocalLoss})\n#         oof = model.predict(train[idx_va]).reshape((len(idx_va),) ) \n#         print(amex_metric(target[idx_va], oof))\n#         oof_pred[idx_va] += oof\n#         gc.collect()\n#         del model, oof\n#         gc.collect()\n#         K.clear_session()  \n#     print(amex_metric(target, oof_pred))\n#     oof_predictions += oof_pred/ len(seeds)\n# print(amex_metric(target, oof_predictions))\n# sub = pd.read_csv('../input/amex-default-prediction/train_labels.csv').drop('target',axis=1)\n# sub['prediction'] = oof_predictions\n# sub.to_csv('keras-cnn_oof.csv')\n# del train,target,oof_predictions,sub","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:04:51.850122Z","iopub.execute_input":"2022-08-08T20:04:51.850737Z","iopub.status.idle":"2022-08-08T20:06:36.997999Z","shell.execute_reply.started":"2022-08-08T20:04:51.850694Z","shell.execute_reply":"2022-08-08T20:06:36.996803Z"},"scrolled":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # get submission predictions\n# test_pred=None\n# for pt in tqdm([1,2]):\n#     test = joblib.load(f'../input/amex-keras-dataset/test_agg_extra2_pt{pt}_scaled.pkl')\n#     test = np.array([i.flatten('C') for i in test])\n#     test_predictions = np.zeros(len(test))\n#     for seed,fold in tqdm(itertools.product(seeds,folds)):\n#         model = tf.keras.models.load_model(f'../input/amex-keras-models/model_fold{fold}_seed{seed}.h5',custom_objects={\"AdaBeliefOptimizer\":AdaBeliefOptimizer, 'FocalLoss': FocalLoss})\n#         def split(a, n):\n#             k, m = divmod(len(a), n)\n#             return [a[i*k+min(i, m):(i+1)*k+min(i+1, m)] for i in range(n)]\n#         split_ids = split(range(len(test)),2)\n#         for (j,ids) in enumerate(split_ids):\n#             test_predictions[ids] += model.predict(test[ids]).reshape((len(ids),) ) / len(folds) / len(seeds)\n#             gc.collect()\n#         del model, split_ids\n#         gc.collect()\n#         K.clear_session()\n#     if test_pred is None:\n#         test_pred = test_predictions\n#     else:\n#         test_pred = np.concatenate([test_pred,test_predictions])\n# sub = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\n# sub['prediction'] = test_pred\n# sub.to_csv('keras-cnn_sub.csv')","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-08-08T20:07:13.306985Z","iopub.execute_input":"2022-08-08T20:07:13.307555Z","iopub.status.idle":"2022-08-08T20:30:27.321416Z","shell.execute_reply.started":"2022-08-08T20:07:13.307507Z","shell.execute_reply":"2022-08-08T20:30:27.320255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('../input/amex-keras-models/keras-cnn_sub.csv').drop('Unnamed: 0',axis=1)\nsub.to_csv('keras-cnn_sub.csv',index = False)\nsub","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:53:46.978544Z","iopub.execute_input":"2022-08-08T20:53:46.979055Z","iopub.status.idle":"2022-08-08T20:53:51.577478Z","shell.execute_reply.started":"2022-08-08T20:53:46.979017Z","shell.execute_reply":"2022-08-08T20:53:51.576340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Other things to work on\n\n- looking at [this notebook](https://www.kaggle.com/code/nawidsayed/lightgbm-and-cnn-3rd-place-solution/notebook) which I used for reference from a similar competition, I could probably try using tree predictors to generate fake features that probably would help information extraction.\n\n- better model architecture and parameters\n\n- I had to cut some one-hot-encoded categorical features since they exceeded the 15 feature limit, perhaps add some embedding instead of one-hot-encoding them","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}