{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Intro\n\nThis notebook uses 3D data of shape `[n_customers, 13, n_features]` that has been processed in a similar way to this great notebook: https://www.kaggle.com/code/cdeotte/tensorflow-gru-starter-0-790\n\nThe training records in the DataFrame with less than 13 rows were pre-padded and reshaped to 3D. Then the arrays were sequently sliced to shape `[14000,13,192]` and saved separately.\n\nSome variation of the model from this paper was used for training.: https://downloads.hindawi.com/journals/misy/2022/9103437.pdf\n","metadata":{}},{"cell_type":"markdown","source":"# Imports","metadata":{"execution":{"iopub.status.busy":"2022-08-14T10:13:38.144899Z","iopub.execute_input":"2022-08-14T10:13:38.145395Z","iopub.status.idle":"2022-08-14T10:13:38.170949Z","shell.execute_reply.started":"2022-08-14T10:13:38.145298Z","shell.execute_reply":"2022-08-14T10:13:38.169795Z"}}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport gc\nimport sys\nimport math\nimport pickle\nimport tensorflow as tf\nimport tensorflow_addons as tfa\nfrom tensorflow.keras import backend as K\nfrom tensorflow import keras\nfrom tensorflow.keras import layers as L\nfrom tensorflow.keras import Model\nimport matplotlib.pyplot as plt\n\nfrom sklearn.model_selection import KFold\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score, accuracy_score\nimport tqdm\n\npd.set_option('display.max_columns', None)\n\ncustomers_train = pd.read_csv('../input/amex-numpy-arrays/customers_train_prepad.csv')\ncat_cols = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68', 'quarter', 'day_of_week', 'year', 'counts'] \n\nn_features = 192 #train.shape[-1]","metadata":{"execution":{"iopub.status.busy":"2022-08-14T14:46:04.026191Z","iopub.execute_input":"2022-08-14T14:46:04.027257Z","iopub.status.idle":"2022-08-14T14:46:11.089056Z","shell.execute_reply.started":"2022-08-14T14:46:04.026783Z","shell.execute_reply":"2022-08-14T14:46:11.087960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T14:46:11.091603Z","iopub.execute_input":"2022-08-14T14:46:11.092344Z","iopub.status.idle":"2022-08-14T14:46:11.112634Z","shell.execute_reply.started":"2022-08-14T14:46:11.092305Z","shell.execute_reply":"2022-08-14T14:46:11.108090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# COMPETITION METRIC FROM Konstantin Yakovlev\n# https://www.kaggle.com/kyakovlev\n# https://www.kaggle.com/competitions/amex-default-prediction/discussion/327534\ndef amex_metric_mod(y_true, y_pred):\n\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n\n    return 0.5 * (gini[1]/gini[0] + top_four)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T14:46:11.114366Z","iopub.execute_input":"2022-08-14T14:46:11.115088Z","iopub.status.idle":"2022-08-14T14:46:11.141307Z","shell.execute_reply.started":"2022-08-14T14:46:11.115043Z","shell.execute_reply":"2022-08-14T14:46:11.140059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train DataFrame\nAll customers are padded to 13 records, numeric `NaN` values are filled with `-1`, categorical - with `0`.","metadata":{}},{"cell_type":"code","source":"def display_df(path):\n    df = pd.read_parquet(path)\n    df = df.sort_values(['customer_ID', 'counts'])\n    display(df)\n    \npath = '../input/amex-datasets-processed/train_processed.parquet'\ndisplay_df(path)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T14:46:11.147472Z","iopub.execute_input":"2022-08-14T14:46:11.147761Z","iopub.status.idle":"2022-08-14T14:46:57.367841Z","shell.execute_reply.started":"2022-08-14T14:46:11.147721Z","shell.execute_reply":"2022-08-14T14:46:57.366843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training the model","metadata":{}},{"cell_type":"code","source":"def add_max_avg(tensor):\n    _avr = L.GlobalAveragePooling1D()(tensor)\n    _max = L.GlobalMaxPooling1D()(tensor)\n    return L.Add()([_avr, _max])\n\n\ndef get_model(columns = n_features, rnn_units=128, dense_units=32):\n       \n    cat_embeds = []\n    \n    inputs = L.Input(shape=(13, columns))  \n    \n    #Masking padded rows\n    inp = L.Activation('relu')(inputs)\n    inp = L.Masking(0.)(inp)\n    \n    num_inputs = inp[:,:,:-len(cat_cols)]\n    \n    for i in range(2, len(cat_cols)+1):\n        emb = L.Lambda(lambda x: x[:,:,-i])(inp)\n        emb = L.Embedding(10, 4, trainable=True, mask_zero=False, name=f\"embedding_{i}\")(emb)\n        cat_embeds.append(emb)       \n        \n    concated_cats = L.Concatenate()(cat_embeds)\n    #Add positional embedding\n    n_features = concated_cats.shape[-1]\n    emb = L.Embedding(14, n_features, trainable=True, mask_zero=False, name=f\"embedding_pos\")(inp[:,:,-1])\n    concated_cats = L.Add()([concated_cats, emb])\n    \n    concated_cat_embeds = L.Concatenate()([num_inputs, concated_cats])\n    \n    rnn = L.Bidirectional(L.LSTM(units=rnn_units, return_sequences=True))(concated_cat_embeds)\n    rnn_w = L.Lambda(lambda x: x/10)(rnn)\n    rnn_w = L.Activation('tanh')(rnn_w)\n    rnn_w = L.Lambda(lambda x: x*10)(rnn_w)\n    rnn_w = L.Activation('softmax')(rnn_w)\n    rnn = L.Lambda(lambda x: tf.reduce_sum(x[0]*x[1], axis=1))([rnn, rnn_w])\n\n    cnn = L.Conv1D(concated_cat_embeds.shape[-1], 1, padding='same')(concated_cat_embeds)\n    cnn = L.BatchNormalization()(cnn)\n    cnn = L.Add()([cnn,concated_cat_embeds])\n    cnn = L.Activation('relu')(cnn)\n    cnn_avr = L.GlobalAveragePooling1D(keepdims=True)(cnn)\n    cnn_max = L.GlobalMaxPooling1D(keepdims=True)(cnn)\n    cnn = L.Add()([cnn_avr, cnn_max])\n    cnn = L.Conv1D(rnn_units*2, 1, padding='same')(cnn)\n    cnn = L.Lambda(lambda x: tf.keras.backend.squeeze(x,axis=1))(cnn)\n    cnn = L.Activation('sigmoid')(cnn)\n    \n    last_state = L.Concatenate()([rnn, cnn])\n    last_state = L.Dropout(0.2)(last_state)\n    \n    last_state = L.Concatenate()([last_state, num_inputs[:,-1,:]])\n    last_state = L.Dropout(0.35)(last_state)\n    \n    for n in [20,10,5,2,1]:\n        last_state = L.Dense(dense_units*n, activation=\"gelu\")(last_state)\n        last_state = L.Dropout(0.1)(last_state)\n    \n    out = L.Dense(1, activation=\"sigmoid\")(last_state)\n    \n    \n    model = Model(inputs=inputs, outputs=out)\n    model.compile(optimizer = keras.optimizers.Adam(lr=1e-3), loss=\"binary_crossentropy\", metrics=[tf.keras.metrics.AUC(from_logits=True)])\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-08-14T14:46:57.368912Z","iopub.execute_input":"2022-08-14T14:46:57.369261Z","iopub.status.idle":"2022-08-14T14:46:57.388098Z","shell.execute_reply.started":"2022-08-14T14:46:57.369225Z","shell.execute_reply":"2022-08-14T14:46:57.387058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_model().summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T14:46:57.390239Z","iopub.execute_input":"2022-08-14T14:46:57.390875Z","iopub.status.idle":"2022-08-14T14:47:01.125466Z","shell.execute_reply.started":"2022-08-14T14:46:57.390841Z","shell.execute_reply":"2022-08-14T14:47:01.124504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nLR_START = 1e-6\nLR_MAX = 1e-3\nLR_MIN = 1e-6\nLR_RAMPUP_EPOCHS = 0\nLR_SUSTAIN_EPOCHS = 0\nEPOCHS = 30\n\ndef lrfn(epoch):\n    if epoch < LR_RAMPUP_EPOCHS:\n        lr = (LR_MAX - LR_START) / LR_RAMPUP_EPOCHS * epoch + LR_START\n    elif epoch < LR_RAMPUP_EPOCHS + LR_SUSTAIN_EPOCHS:\n        lr = LR_MAX\n    else:\n        decay_total_epochs = EPOCHS - LR_RAMPUP_EPOCHS - LR_SUSTAIN_EPOCHS - 1\n        decay_epoch_index = epoch - LR_RAMPUP_EPOCHS - LR_SUSTAIN_EPOCHS\n        phase = math.pi * decay_epoch_index / decay_total_epochs\n        cosine_decay = 0.5 * (1 + math.cos(phase))\n        lr = (LR_MAX - LR_MIN) * cosine_decay + LR_MIN\n    return lr\n\nrng = [i for i in range(EPOCHS)]\nlr_y = [lrfn(x) for x in rng]\nplt.figure(figsize=(10, 4))\nplt.plot(rng, lr_y, '-o')\nplt.xlabel('Epoch'); plt.ylabel('LR')\nprint(\"Learning rate schedule: {:.3g} to {:.3g} to {:.3g}\". \\\n      format(lr_y[0], max(lr_y), lr_y[-1]))\nLR = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose = True)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-14T14:47:01.127137Z","iopub.execute_input":"2022-08-14T14:47:01.127834Z","iopub.status.idle":"2022-08-14T14:47:01.356147Z","shell.execute_reply.started":"2022-08-14T14:47:01.127795Z","shell.execute_reply":"2022-08-14T14:47:01.354355Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH_TO_DATA = '../input/amex-split-arrays-prepad/' \n\ndef load_arrays(indices, mode='train'):\n    data = []\n    targets = []\n    base_len = 14000\n    customer_ids = []\n    \n    for k in indices:\n        array = np.load(f'{PATH_TO_DATA}{mode}_{k}.npy')\n        data.append(array)\n        customer_ids += (list(range(k*base_len, k*base_len + len(array))))\n    data = np.concatenate(data,axis=0)\n    \n    if mode == 'train':\n        for k in indices:\n            targets.append( np.load(f'{PATH_TO_DATA}targets_{k}.npy'))\n        targets = np.concatenate(targets,axis=0)\n        return data, targets, customer_ids\n    else:\n        return data, customer_ids","metadata":{"execution":{"iopub.status.busy":"2022-08-14T14:47:01.357684Z","iopub.execute_input":"2022-08-14T14:47:01.358263Z","iopub.status.idle":"2022-08-14T14:47:01.365563Z","shell.execute_reply.started":"2022-08-14T14:47:01.358224Z","shell.execute_reply":"2022-08-14T14:47:01.364577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_folds = 5\ncustomers_train['prediction'] = 1 \n\ntrain_arrays = [x for x in os.listdir(PATH_TO_DATA) if 'train' in x]\nidx = list(range(len(train_arrays)))\n\nestop = tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=5, verbose=0, mode='min',restore_best_weights=True)\nkf_split = KFold(n_splits=n_folds, shuffle=True, random_state=2)\n\nfor fold,(tr_idx, val_idx) in enumerate(kf_split.split(idx)):\n    \n\n    X_train, y_train, train_cases = load_arrays(tr_idx, mode='train')\n    X_test, y_test, test_cases  = load_arrays(val_idx, mode='train')\n    \n    path_to_checkpoints = f\"best_fold_{fold+1}.hdf5\"\n    checkpointer = tf.keras.callbacks.ModelCheckpoint(filepath=path_to_checkpoints, monitor=\"val_loss\", mode='min', verbose=1, save_best_only=True)\n\n    model = get_model(rnn_units=64)\n\n    print(f'Training Model Fold {fold+1}...')\n    print(f'Val indices: {val_idx}')\n    \n    history = model.fit(\n        X_train, y_train,\n        epochs = 25,\n        batch_size=512,\n        callbacks = [estop, LR, checkpointer],\n        validation_data = (X_test, y_test),\n        shuffle=True\n        #class_weight = {0:1, 1:2}\n    )   \n    \n    del X_train, y_train,\n    \n    model.save(f\"end_fold_{fold+1}.hdf5\")\n    model.load_weights(f\"best_fold_{fold+1}.hdf5\")\n    \n    test_preds = model.predict(X_test).squeeze()\n    customers_train.iloc[test_cases, -1] = test_preds\n    \n    print(f'\\n Fold {fold+1}:\\n'\n    f' ROC_AUC: {roc_auc_score(y_test, test_preds)}',\n    '\\n',\n    f'ACC: {accuracy_score(y_test, test_preds.round())}',\n         '\\n',\n    f'AMEX metric: {amex_metric_mod(y_test.squeeze(), test_preds)}\\n'\n    )\n\n    del model, X_test, y_test, test_preds\n    K.clear_session()\n    gc.collect()\n\ncustomers_train.to_csv('customers_train_oofs.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T14:47:01.367166Z","iopub.execute_input":"2022-08-14T14:47:01.367844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# OOF evaluation","metadata":{}},{"cell_type":"code","source":"all_customers = amex_metric(customers_train[['target']],customers_train[['prediction']])\n\ncustomer_13 = customers_train[customers_train.counts==13]\ncustomer_13 = amex_metric(customer_13[['target']],customer_13[['prediction']])\n\ncustomer_not_13 = customers_train[customers_train.counts!=13]\ncustomer_not_13 = amex_metric(customer_not_13[['target']],customer_not_13[['prediction']])\n\nprint(f' All customers score: {all_customers}\\n',\n      f'Customers 13 rec. score: {customer_13}\\n',\n      f'Customers less 13 rec. score: {customer_not_13}\\n',\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference","metadata":{}},{"cell_type":"code","source":"customers_test = pd.read_csv('../input/amex-numpy-arrays/customers_test_prepad.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = []\nfor model_path in sorted([x for x in os.listdir() if 'best' in x]):\n    model =  tf.keras.models.load_model(model_path)\n    models.append(model)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH_TO_DATA = '../input/amex-split-arrays-prepad/'\ntest_arrays = [x for x in os.listdir(PATH_TO_DATA) if 'test' in x]\n\ndef infer_test(models, ind):\n    preds = []\n    test_data,_ = load_arrays([ind], mode='test')\n    for model in models:\n        pred = model.predict(test_data)\n        preds.append(pred.squeeze())\n        \n    del test_data\n    return preds","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = []\ndf = pd.DataFrame()\n\nidx = range(len(test_arrays))\nfor i in idx:\n    p = infer_test(models, i)\n    \n    df = pd.concat([df, pd.DataFrame(np.array(p).transpose())])\n    \n    p = np.sum(p, axis=0)/len(models)\n    \n    predictions.append(p)\n    gc.collect()\n    \npredictions = np.hstack(predictions)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers_test['prediction'] = predictions\ncustomers_test.to_csv('predictions_test.csv',index=False)\ncustomers_test","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\nsample_submission = sample_submission[['customer_ID']].merge(customers_test)\n\nsample_submission[['customer_ID', 'prediction']].to_csv('submission.csv',index=False)\nsample_submission","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}