{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This Notebook is built off of [this Notebook](https://www.kaggle.com/code/ambrosm/amex-keras-quickstart-1-training) by AmbrosM","metadata":{}},{"cell_type":"code","source":"#tpu\nimport tensorflow as tf\nresolver = tf.distribute.cluster_resolver.TPUClusterResolver(tpu='')\ntf.config.experimental_connect_to_cluster(resolver)\n# This is the TPU initialization code that has to be at the beginning.\ntf.tpu.experimental.initialize_tpu_system(resolver)\nprint(\"All devices: \", tf.config.list_logical_devices('TPU'))\nstrategy = tf.distribute.TPUStrategy(resolver)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T00:04:54.758916Z","iopub.execute_input":"2022-07-03T00:04:54.759775Z","iopub.status.idle":"2022-07-03T00:05:07.750272Z","shell.execute_reply.started":"2022-07-03T00:04:54.759672Z","shell.execute_reply":"2022-07-03T00:05:07.749407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport pickle\nfrom matplotlib import pyplot as plt\nimport random\nimport datetime\nimport math\nimport gc\nimport warnings\nimport seaborn as sns\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\nfrom matplotlib.ticker import MaxNLocator\nfrom colorama import Fore, Back, Style\nimport h5py\n\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.preprocessing import StandardScaler, QuantileTransformer, OneHotEncoder, PowerTransformer\nfrom sklearn.metrics import roc_curve, roc_auc_score, average_precision_score\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.utils import class_weight \nfrom sklearn.utils.class_weight import compute_class_weight\n\n# tf.config.threading.set_inter_op_parallelism_threads(4)\nimport tensorflow_addons as tfa\nfrom tensorflow.keras.models import Model, load_model\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, LearningRateScheduler, EarlyStopping\nfrom tensorflow.keras.layers import Dense, Input, InputLayer, Add, Concatenate, Dropout, BatchNormalization\nfrom tensorflow.keras.utils import plot_model\nimport tensorflow.keras.backend as K","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-03T00:05:07.75174Z","iopub.execute_input":"2022-07-03T00:05:07.751968Z","iopub.status.idle":"2022-07-03T00:05:09.140281Z","shell.execute_reply.started":"2022-07-03T00:05:07.751942Z","shell.execute_reply":"2022-07-03T00:05:09.139277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot training history\ndef plot_history(history, *, n_epochs=None, plot_lr=False, title=None, bottom=None, top=None):\n    \"\"\"Plot (the last n_epochs epochs of) the training history\n    \n    Plots loss and optionally val_loss and lr.\"\"\"\n    plt.figure(figsize=(15, 6))\n    from_epoch = 0 if n_epochs is None else max(len(history['loss']) - n_epochs, 0)\n    \n    # Plot training and validation losses\n    plt.plot(np.arange(from_epoch, len(history['loss'])), history['loss'][from_epoch:], label='Training loss')\n    try:\n        plt.plot(np.arange(from_epoch, len(history['loss'])), history['val_loss'][from_epoch:], label='Validation loss')\n        best_epoch = np.argmin(np.array(history['val_loss']))\n        best_val_loss = history['val_loss'][best_epoch]\n        if best_epoch >= from_epoch:\n            plt.scatter([best_epoch], [best_val_loss], c='r', label=f'Best val_loss = {best_val_loss:.5f}')\n        if best_epoch > 0:\n            almost_epoch = np.argmin(np.array(history['val_loss'])[:best_epoch])\n            almost_val_loss = history['val_loss'][almost_epoch]\n            if almost_epoch >= from_epoch:\n                plt.scatter([almost_epoch], [almost_val_loss], c='orange', label='Second best val_loss')\n    except KeyError:\n        pass\n    if bottom is not None: plt.ylim(bottom=bottom)\n    if top is not None: plt.ylim(top=top)\n    plt.gca().xaxis.set_major_locator(MaxNLocator(integer=True))\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')\n    plt.legend(loc='lower left')\n    if title is not None: plt.title(title)\n        \n    # Plot learning rate\n    if plot_lr and 'lr' in history:\n        ax2 = plt.gca().twinx()\n        ax2.plot(np.arange(from_epoch, len(history['lr'])), np.array(history['lr'][from_epoch:]), color='g', label='Learning rate')\n        ax2.set_ylabel('Learning rate')\n        ax2.legend(loc='upper right')\n        \n    plt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-03T00:05:09.141857Z","iopub.execute_input":"2022-07-03T00:05:09.142195Z","iopub.status.idle":"2022-07-03T00:05:09.159048Z","shell.execute_reply.started":"2022-07-03T00:05:09.142126Z","shell.execute_reply":"2022-07-03T00:05:09.158147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex_metric(y_true, y_pred, return_components=False) -> float:\n    \"\"\"Amex metric for ndarrays\"\"\"\n    def top_four_percent_captured(df) -> float:\n        \"\"\"Corresponds to the recall for a threshold of 4 %\"\"\"\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(df) -> float:\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(df) -> float:\n        \"\"\"Corresponds to 2 * AUC - 1\"\"\"\n        df2 = pd.DataFrame({'target': df.target, 'prediction': df.target})\n        df2.sort_values('prediction', ascending=False, inplace=True)\n        return weighted_gini(df) / weighted_gini(df2)\n\n    df = pd.DataFrame({'target': y_true.ravel(), 'prediction': y_pred.ravel()})\n    df.sort_values('prediction', ascending=False, inplace=True)\n    g = normalized_weighted_gini(df)\n    d = top_four_percent_captured(df)\n\n    if return_components: return g, d, 0.5 * (g + d)\n    return 0.5 * (g + d)\ndef create_weighted_binary_crossentropy(zero_weight, one_weight):\n\n    def weighted_cross_entropy_fn(y_true, y_pred):\n        tf_y_true = tf.cast(y_true, dtype=y_pred.dtype)\n        tf_y_pred = tf.cast(y_pred, dtype=y_pred.dtype)\n\n        weights_v = tf.where(tf.equal(tf_y_true, 1), 2*one_weight, 2*zero_weight)\n        ce = K.binary_crossentropy(tf_y_true, tf_y_pred)\n        loss = K.mean(tf.multiply(ce, weights_v))\n        return loss\n\n    return weighted_cross_entropy_fn","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-03T00:05:09.161711Z","iopub.execute_input":"2022-07-03T00:05:09.162059Z","iopub.status.idle":"2022-07-03T00:05:09.178782Z","shell.execute_reply.started":"2022-07-03T00:05:09.162017Z","shell.execute_reply":"2022-07-03T00:05:09.177788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Reading and preprocessing the training data\n\nThis aggregation dataset is one-hot encoded and NaN's are filled with zeros\n","metadata":{}},{"cell_type":"code","source":"train = pd.read_feather('../input/amex-aggreation-dataset/train_processed.ftr')\ntrain.sort_values(by = 'customer_ID', inplace = True)\ntrain.reset_index(drop = True, inplace = True)\ncid = train['customer_ID']\ntarget = train['target']\ntrain = train.drop(['target','customer_ID'], axis = 1)\ntrain = pd.DataFrame(StandardScaler().fit_transform(train),columns = train.columns).astype('float16')","metadata":{"execution":{"iopub.status.busy":"2022-07-03T00:05:09.180094Z","iopub.execute_input":"2022-07-03T00:05:09.180353Z","iopub.status.idle":"2022-07-03T00:05:49.913529Z","shell.execute_reply.started":"2022-07-03T00:05:09.180326Z","shell.execute_reply":"2022-07-03T00:05:49.911845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T00:05:49.91598Z","iopub.execute_input":"2022-07-03T00:05:49.916974Z","iopub.status.idle":"2022-07-03T00:05:49.956854Z","shell.execute_reply.started":"2022-07-03T00:05:49.916929Z","shell.execute_reply":"2022-07-03T00:05:49.95627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T00:05:49.957825Z","iopub.execute_input":"2022-07-03T00:05:49.95838Z","iopub.status.idle":"2022-07-03T00:05:49.964694Z","shell.execute_reply.started":"2022-07-03T00:05:49.95835Z","shell.execute_reply":"2022-07-03T00:05:49.963565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# The model\n\nOur model has four hidden layers, enriched by a skip connection and a Dropout layer.","metadata":{}},{"cell_type":"code","source":"def my_model(n_inputs=len(train.columns)):\n    \"\"\"Sequential neural network with a skip connection.\n    \n    Returns a compiled instance of tensorflow.keras.models.Model.\n    \"\"\"\n    activation = 'swish'\n    l1 = 1e-7\n    l2 = 4e-4\n    inputs = Input(shape=(n_inputs, ))\n    x0 = BatchNormalization()(inputs)\n    x0 = Dense(256, \n               kernel_regularizer=tf.keras.regularizers.L1L2(l1=l1,l2=l2),\n#                activity_regularizer=tf.keras.regularizers.L1L2(l1=l1,l2=l2),\n              activation=activation,\n             )(x0)\n    x0 = Dropout(0.1)(x0)\n    x = Dense(64, \n              kernel_regularizer=tf.keras.regularizers.L1L2(l1=l1,l2=l2),\n#               activity_regularizer=tf.keras.regularizers.L1L2(l1=l1,l2=l2),\n              activation=activation,\n             )(x0)\n    x = Dense(64, \n              kernel_regularizer=tf.keras.regularizers.L1L2(l1=l1,l2=l2),\n#               activity_regularizer=tf.keras.regularizers.L1L2(l1=l1,l2=l2),\n              activation=activation,\n             )(x)\n    x = Concatenate()([x, x0])\n    x = Dropout(0.1)(x)\n    x = Dense(16, \n              kernel_regularizer=tf.keras.regularizers.L1L2(l1=l1,l2=l2),\n#               activity_regularizer=tf.keras.regularizers.L1L2(l1=l1,l2=l2),\n              activation=activation,\n             )(x)\n    x = Dense(1,\n              activation='sigmoid',\n             )(x)\n    return Model(inputs, x)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T00:05:49.966336Z","iopub.execute_input":"2022-07-03T00:05:49.966813Z","iopub.status.idle":"2022-07-03T00:05:49.97721Z","shell.execute_reply.started":"2022-07-03T00:05:49.966784Z","shell.execute_reply":"2022-07-03T00:05:49.976526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Cross-validation\n\nWe use a standard cross-validation loop. In the loop, we scale the data and train a model. We use a StratifiedKFold because the data is imbalanced.\n\nI use multiple different seeds (SEEDS) for the cv folds because I want to use this model's oof predictions for ensembling later. Averaging across multiple folds makes the oof predictions more closely resemble the test predictions that the model outputs\n\nAlso, since sometimes the model has exploding gradients, we need to try multiple seeds (SEEDSTRY) and hope that at least one will not explode.","metadata":{}},{"cell_type":"code","source":"FOLDS = 10\nSEEDS = 2\nSEEDSTRY = 4\nVERBOSE = 0\nLR_START = 0.012\nCYCLES = 1\nEPOCHS = 400\nDIAGRAMS = False\nBATCH_SIZE = 2048\n\n#initialize custom loss function (loss prioritizes 1 over 0)\none_weight = 0.6\nloss = create_weighted_binary_crossentropy(1-one_weight, one_weight)\n\nnp.random.seed(1)\nrandom.seed(1)\n\ndef fit_model(X_tr, y_tr, X_va=None, y_va=None, fold=0):\n    global oof_pred\n    start_time = datetime.datetime.now()\n    \n    lr = ReduceLROnPlateau(monitor=\"val_loss\", factor=0.50, \n                           patience=5, verbose=VERBOSE)\n    es = EarlyStopping(monitor=\"val_loss\",\n                       patience=20, \n                       verbose=1,\n                       mode=\"min\", \n                       restore_best_weights=True)\n    callbacks = [lr, es, tf.keras.callbacks.TerminateOnNaN()]\n            \n    #reset model\n    with strategy.scope():\n        model = my_model(train.shape[1])\n        model.compile(optimizer=tf.keras.optimizers.Nadam(learning_rate=LR_START,\n                                                      clipvalue= 0.5,\n                                                      clipnorm = 1.0 # prevent gradient explosion\n                                                     ),\n                  loss=loss,\n                 )\n            \n    # Train the model\n    history = model.fit(X_tr, y_tr, \n                        validation_data=(X_va, y_va),\n                        epochs=EPOCHS,\n                        verbose=VERBOSE,\n                        batch_size=BATCH_SIZE,\n                        shuffle=True,\n                        callbacks=callbacks).history\n        \n    \n    X_tr, y_tr, callbacks, es, lr = None, None, None, None, None\n    \n    lastloss = f\"Training loss: {history['loss'][-1]:.4f} | Val loss: {history['val_loss'][-1]:.4f}\"\n\n    # Inference for validation\n    oof_pred = model.predict(X_va, batch_size=len(X_va), verbose=0).ravel()\n\n    # Evaluation: Execution time, loss and metrics\n    score = amex_metric(y_va.values, oof_pred)\n    print(f\"{Fore.GREEN}{Style.BRIGHT}Fold {fold} | {str(datetime.datetime.now() - start_time)[-12:-7]}\"\n          f\" | {len(history['loss']):3} ep\"\n          f\" | {lastloss} | Score: {score:.5f}{Style.RESET_ALL}\")\n\n    if DIAGRAMS:\n        # Plot training history\n        plot_history(history, \n                     title=f\"Learning curve\",\n                     plot_lr=True)\n        \n    return score, model\n\n\nprint(f\"{len(train.columns)} features\")\noof_predictions = np.zeros(len(train))\nfor seed in range(SEEDS):\n    print(f'KFOLD WITH SEED {seed}:')\n    kf = StratifiedKFold(n_splits=FOLDS, shuffle= True, random_state= seed)\n    for fold, (idx_tr, idx_va) in enumerate(kf.split(train, target)):\n        print('-' * 15 + f'fold {fold}' + '-'*15)\n        gc.collect()\n        b_score = 0\n        b_oof = None\n        b_model = None\n        init_seed = seed\n        for i in range(SEEDSTRY):\n            print('seed:', init_seed)\n            tf.random.set_seed(init_seed)\n            score, model = fit_model(train.iloc[idx_tr], target.iloc[idx_tr], train.iloc[idx_va], target.iloc[idx_va], fold=fold)\n            if score > b_score:\n                b_model = model\n                b_score = score\n                b_oof = oof_pred\n            init_seed+=1\n            \n        oof_predictions[idx_va] += b_oof / SEEDS\n        #save model\n        b_model.save_weights(f\"model_fold{fold}_seed{seed}.h5\")\n        b_model = b_model.to_json()\n        with open(f\"model_fold{fold}_seed{seed}.json\", \"w\") as json_file:\n            json_file.write(b_model)\n        del b_score, b_oof, b_model, score, model\n        gc.collect()\nprint(f\"{Fore.GREEN}{Style.BRIGHT}OOF Score: {amex_metric(target, oof_predictions)}{Style.RESET_ALL}\")","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-03T01:09:21.848213Z","iopub.execute_input":"2022-07-03T01:09:21.848772Z","iopub.status.idle":"2022-07-03T01:11:02.045521Z","shell.execute_reply.started":"2022-07-03T01:09:21.848736Z","shell.execute_reply":"2022-07-03T01:11:02.044486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# write to files and clear memory \ndel kf, train, target\ngc.collect()\noof_pred= pd.read_csv('../input/amex-default-prediction/train_labels.csv')\noof_pred.drop(['target'],axis = 1,inplace = True)\noof_pred['prediction'] = oof_predictions\noof_pred.to_csv(f'keras_oof_folds{FOLDS}_seeds{SEEDS}_w{one_weight}.csv')\ndel oof_pred","metadata":{"execution":{"iopub.status.busy":"2022-07-03T01:13:30.250093Z","iopub.execute_input":"2022-07-03T01:13:30.250447Z","iopub.status.idle":"2022-07-03T01:13:31.087997Z","shell.execute_reply.started":"2022-07-03T01:13:30.250412Z","shell.execute_reply":"2022-07-03T01:13:31.087087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission\n\nWe need to split the test data into multiple parts because of memory.","metadata":{}},{"cell_type":"code","source":"test = pd.read_feather('../input/amex-aggreation-dataset/test_processed.ftr')\ntest.drop(['customer_ID'], axis = 1, inplace = True)\ntest.reset_index(drop = True, inplace = True)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T01:13:45.171945Z","iopub.execute_input":"2022-07-03T01:13:45.172297Z","iopub.status.idle":"2022-07-03T01:14:06.576718Z","shell.execute_reply.started":"2022-07-03T01:13:45.172254Z","shell.execute_reply":"2022-07-03T01:14:06.574834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#scale test and predict\nSPLITS = 3\ndef split(a, n):\n    k, m = divmod(len(a), n)\n    return (a[i*k+min(i, m):(i+1)*k+min(i+1, m)] for i in range(n))\n\ny_pred_avg = np.zeros(len(test))\nfor seed in range(SEEDS):\n    for fold in range(FOLDS):\n        from tensorflow.keras.models import model_from_json\n        json_file = open(f\"model_fold{fold}_seed{seed}.json\", 'r')\n        model = json_file.read()\n        json_file.close()\n        model = model_from_json(model, custom_objects={\"weighted_cross_entropy_fn\": loss})\n        model.load_weights(f\"model_fold{fold}_seed{seed}.h5\")\n\n        pred = np.array([])\n        split_ids = split(test.index, SPLITS)\n        for (j,ids) in enumerate(split_ids):\n            df = StandardScaler().fit_transform(X= test.iloc[ids])\n            pred = np.append(pred, model.predict(df).reshape(1, -1)[0])\n        del model\n        gc.collect()\n        y_pred_avg += pred / FOLDS / SEEDS\n    \ndel test\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T01:18:33.305549Z","iopub.execute_input":"2022-07-03T01:18:33.306559Z","iopub.status.idle":"2022-07-03T01:21:21.246098Z","shell.execute_reply.started":"2022-07-03T01:18:33.306515Z","shell.execute_reply":"2022-07-03T01:21:21.244922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\nsub['prediction'] = y_pred_avg\nsub.to_csv(f'keras_sub_folds{FOLDS}_seeds{SEEDS}_w{one_weight}.csv', index=False)\nsub","metadata":{"execution":{"iopub.status.busy":"2022-07-03T01:21:54.614256Z","iopub.execute_input":"2022-07-03T01:21:54.615249Z","iopub.status.idle":"2022-07-03T01:22:01.474338Z","shell.execute_reply.started":"2022-07-03T01:21:54.615194Z","shell.execute_reply":"2022-07-03T01:22:01.473439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As a plausibility test, we plot a histogram of the predictions and the OOF predictions. ","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16, 5))\nplt.hist(sub.prediction, bins=np.linspace(0, 1, 21), density=True)\nplt.title(\"Plausibility check\", fontsize=20)\nplt.xlabel('Prediction')\nplt.ylabel('Density')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T01:22:14.254075Z","iopub.execute_input":"2022-07-03T01:22:14.255102Z","iopub.status.idle":"2022-07-03T01:22:14.65253Z","shell.execute_reply.started":"2022-07-03T01:22:14.255047Z","shell.execute_reply":"2022-07-03T01:22:14.651601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16, 5))\nplt.hist(oof_predictions, bins=np.linspace(0, 1, 21), density=True)\nplt.title(\"Plausibility check\", fontsize=20)\nplt.xlabel('Prediction')\nplt.ylabel('Density')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T01:22:37.217551Z","iopub.execute_input":"2022-07-03T01:22:37.217883Z","iopub.status.idle":"2022-07-03T01:22:37.499456Z","shell.execute_reply.started":"2022-07-03T01:22:37.217848Z","shell.execute_reply":"2022-07-03T01:22:37.498609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}