{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#This notebook is built using codes and procedures from Chris Deotte's primer on Transformers https://www.kaggle.com/code/cdeotte/tensorflow-transformer-0-790\n#The custom attention layers and architecture provides for 0.79 LB score (the same as the original work), but slightly better local scores\nimport numpy as np, pandas as pd # CPU LIBRARIES\nimport matplotlib.pyplot as plt, gc, os\nos.environ[\"TF_GPU_ALLOCATOR\"]=\"cuda_malloc_async\"  # TF will not use all memory\nimport tensorflow as tf\nimport tensorflow.keras.backend as K\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow import math\n\nprint('Using TensorFlow version',tf.__version__)\n\nPATH_TO_DATA = '../input/amex-data-for-transformers-and-rnns/data/'\n\n# IF YOU WISH TO INFER A MODEL YOU TRAINED OFFLINE\n# THEN SET TO FALSE AND PROVIDE KAGGLE DATASET URL\nTRAIN_MODEL = True\nPATH_TO_MODEL = './model/'\n\nINFER_TEST = True","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:50:14.620833Z","iopub.execute_input":"2022-08-13T21:50:14.621414Z","iopub.status.idle":"2022-08-13T21:50:20.354927Z","shell.execute_reply.started":"2022-08-13T21:50:14.621369Z","shell.execute_reply":"2022-08-13T21:50:20.353733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# COMPETITION METRIC FROM Konstantin Yakovlev\n# https://www.kaggle.com/kyakovlev\n# https://www.kaggle.com/competitions/amex-default-prediction/discussion/327534\ndef amex_metric_mod(y_true, y_pred):\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n\n    return 0.5 * (gini[1]/gini[0] + top_four)  ","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:50:20.359929Z","iopub.execute_input":"2022-08-13T21:50:20.360799Z","iopub.status.idle":"2022-08-13T21:50:20.378187Z","shell.execute_reply.started":"2022-08-13T21:50:20.360759Z","shell.execute_reply":"2022-08-13T21:50:20.377031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Position embedding is added\nclass PositionEmbeddingLayer(layers.Layer):\n    def __init__(self, sequence_length, vocab_size, output_dim, **kwargs):\n        super(PositionEmbeddingLayer, self).__init__(**kwargs)\n        self.word_embedding_layer = layers.Embedding(\n            input_dim=vocab_size, output_dim=output_dim\n        )\n        self.position_embedding_layer = layers.Embedding(\n            input_dim=sequence_length, output_dim=output_dim\n        )\n \n    def call(self, inputs):        \n        position_indices = tf.range(tf.shape(inputs)[-1])\n        embedded_words = self.word_embedding_layer(inputs)\n        embedded_indices = self.position_embedding_layer(position_indices)\n        return embedded_words + embedded_indices","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:50:20.382971Z","iopub.execute_input":"2022-08-13T21:50:20.385372Z","iopub.status.idle":"2022-08-13T21:50:20.397530Z","shell.execute_reply.started":"2022-08-13T21:50:20.385332Z","shell.execute_reply":"2022-08-13T21:50:20.396413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math as mth\n\n#This layer is a custom self attention mechanism realisation. It implies shared queries such that one query corresponds\n#to a number of keys defined by num_vals\nclass SelfAtt_Block(keras.layers.Layer):\n    def __init__(self,key_dim=None, value_dim=None, t_steps=None, num_vals=None ):\n        super(SelfAtt_Block, self).__init__()\n        self.keys = []\n        self.vals = []\n        self.q = layers.Dense(key_dim, use_bias=True)\n        for i in range(num_vals):\n            self.keys.append(layers.Dense(key_dim, use_bias=True))\n            self.vals.append(keras.Sequential([layers.Dense(value_dim, use_bias=True), layers.Dropout(0.1)]))        \n        self.t_steps = t_steps\n        self.value_dim = value_dim\n        self.key_dim = key_dim\n        self.num_vals = num_vals\n\n    def call(self, inputs, queries):\n        keys=[x(inputs) for x in self.keys]\n        vals=[x(inputs) for x in self.vals]\n        query = self.q(queries)\n        score_mtx = [layers.Dot(axes=(-1,-1))([query,x])/mth.sqrt(self.key_dim) for x in keys] \n        score_mtx = [keras.activations.softmax(x, axis=-1) for x in score_mtx]\n        score_mtx = [layers.Reshape((self.t_steps ,self.t_steps ,1))(x) for x in score_mtx]\n        value=[layers.Reshape((1,self.t_steps ,self.value_dim))(x) for x in vals]\n        weighted_comps = []\n        for i in range(self.num_vals):\n            weighted_comps.append(tf.math.multiply(score_mtx[i],value[i]))\n        weighted_comps = layers.concatenate(weighted_comps, axis=-1) if self.num_vals>1 else weighted_comps[0]\n        return tf.math.reduce_sum(weighted_comps, axis=-2, keepdims=False)\n\n#This layer applies the attention mechanism to a sequence in kind of original way. Each timestep produces a key and a query, but values are taken directly from \n#timestep data score for each query is calculated only once in contrast with the original approach - against the key of that query.\n# The idea is to provide a bottleneck and reduce ability on nn to memorize the results.\n#After scoring and summing up the results we obtain a single vector output of reduced  dimensionality [batch_size, seq_feature_dimension*num_vals],\n#here num_vals is equivalent to the number of attention heads in the original approach\nclass Attention_to_time(keras.layers.Layer):\n    def __init__(self,key_dim=None, num_vals=None ):\n        super(Attention_to_time, self).__init__()\n        self.keys = []\n        self.q = []\n        for i in range(num_vals):\n            self.q.append(layers.Dense(key_dim, use_bias=True))\n            self.keys.append(layers.Dense(key_dim, use_bias=True)) \n        self.key_dim = key_dim\n        self.num_vals = num_vals\n\n    def call(self, inputs, queries):\n        keys = [x(inputs) for x in self.keys]\n        queries = [x(queries) for x in self.q]\n        score_mtx = [tf.math.multiply(x[0],x[1])/mth.sqrt(self.key_dim) for x in zip(queries, keys)]\n        score_mtx = [tf.math.reduce_sum(x, axis=-1, keepdims=True) for x in score_mtx]\n        score_mtx = [keras.activations.softmax(x, axis=-2) for x in score_mtx]\n        new_vals = [tf.math.multiply(x, inputs) for x in  score_mtx]\n        new_vals = [tf.math.reduce_sum(x, axis=-2, keepdims=False) for x in  new_vals]\n        return layers.concatenate(new_vals, axis=-1)\n\n\n    \nclass MH_SelfAtt(keras.layers.Layer):\n    def __init__(self,key_dim=None, value_dim=None, t_steps=None, num_heads=None, num_vals=None ):\n        super(MH_SelfAtt, self).__init__()\n        self.TB=[]\n        for i in range(num_heads):\n            self.TB.append(SelfAtt_Block(key_dim=key_dim, value_dim=value_dim, t_steps=t_steps ,num_vals=num_vals ))       \n    def call(self, inputs):\n        rez=[x(inputs,inputs) for x in self.TB]\n        return layers.concatenate(rez, axis=-1)\n            \n    \n    \nkey_dim=32\nvalue_dim=40\nt_steps=16\n\n\n\ndef build_model():\n    \n    inp = layers.Input(shape=(13,188))\n    embeddings = []\n    inp1 = layers.GlobalMaxPooling1D(keepdims=True)(inp)\n    inp2 = layers.GlobalAveragePooling1D(keepdims=True)(inp)\n    inp3 = - layers.GlobalMaxPooling1D(keepdims=True)(-inp)\n    inp1 = layers.Concatenate(axis=-2)([inp1, inp2, inp3, inp])\n    for k in range(11):\n        emb = PositionEmbeddingLayer(16,10,3)\n        embeddings.append( emb(inp1[:,:,k]) )\n        \n        \n     \n    x = layers.Concatenate()([inp1[:,:,11:]]+embeddings)\n    t1 = MH_SelfAtt(key_dim=key_dim, value_dim=value_dim, t_steps=t_steps, num_heads=1,num_vals=14)(x)\n    t1 = layers.Dense(210)(t1)\n    t1 = layers.LayerNormalization(epsilon=1e-6)(t1+x)\n    t2 = layers.Dense(210, activation='gelu')(t1)\n    t2 = layers.Dropout(0.1)(t2)\n    t2 = layers.Dense(210)(t2)\n    t2 = layers.LayerNormalization(epsilon=1e-6)(t1+t2)\n    t2 = Attention_to_time(key_dim=32, num_vals=2 )(t2,t2)\n\n    x = layers.Dense(32)(t2)\n\n\n    catN2 = layers.Dense(1,activation='sigmoid')(x)\n    model = keras.Model(inputs=inp, outputs=catN2)\n    opt = tf.keras.optimizers.Adam(learning_rate=0.001)\n    loss = tf.keras.losses.BinaryCrossentropy()\n    model.compile(loss=loss, optimizer = opt)\n        \n    return model\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:50:20.403738Z","iopub.execute_input":"2022-08-13T21:50:20.406260Z","iopub.status.idle":"2022-08-13T21:50:20.445608Z","shell.execute_reply.started":"2022-08-13T21:50:20.406223Z","shell.execute_reply":"2022-08-13T21:50:20.444303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nLR_START = 1e-6\nLR_MAX = 1e-3\nLR_MIN = 1e-6\nLR_RAMPUP_EPOCHS = 0\nLR_SUSTAIN_EPOCHS = 0\nEPOCHS = 8\n\ndef lrfn(epoch):\n    if epoch < LR_RAMPUP_EPOCHS:\n        lr = (LR_MAX - LR_START) / LR_RAMPUP_EPOCHS * epoch + LR_START\n    elif epoch < LR_RAMPUP_EPOCHS + LR_SUSTAIN_EPOCHS:\n        lr = LR_MAX\n    else:\n        decay_total_epochs = EPOCHS - LR_RAMPUP_EPOCHS - LR_SUSTAIN_EPOCHS - 1\n        decay_epoch_index = epoch - LR_RAMPUP_EPOCHS - LR_SUSTAIN_EPOCHS\n        phase = mth.pi * decay_epoch_index / decay_total_epochs\n        cosine_decay = 0.5 * (1 + mth.cos(phase))\n        lr = (LR_MAX - LR_MIN) * cosine_decay + LR_MIN\n    return lr\n\ncallback = keras.callbacks.LearningRateScheduler(lrfn)\n\n\n        \nif TRAIN_MODEL:\n    VERBOSE = 1 # use 1 for interactive \n    folds=[[1,2],[3,4],[5,6],[7,8],[9,10]]\n    for fold,valid in enumerate(folds):\n        # INDICES OF TRAIN AND VALID FOLDS\n        valid_idx = valid\n        train_idx = [x for x in [1,2,3,4,5,6,7,8,9,10] if x not in valid_idx]\n        print('#'*25)\n        print(f'### Fold {fold} with valid files', valid_idx)\n\n        # READ TRAIN DATA FROM DISK\n        X_train = []; y_train = []\n        for k in train_idx:\n            X_train.append( np.load(f'{PATH_TO_DATA}data_{k}.npy').astype(dtype=np.float16))\n            y_train.append( pd.read_parquet(f'{PATH_TO_DATA}targets_{k}.pqt') )\n        X_train = np.concatenate(X_train,axis=0)\n        y_train = pd.concat(y_train).target.values\n        print('### Training data shapes', X_train.shape, y_train.shape)\n\n        # READ VALID DATA FROM DISK\n        X_valid = []; y_valid = []\n        for k in valid_idx:\n            X_valid.append( np.load(f'{PATH_TO_DATA}data_{k}.npy'))\n            y_valid.append( pd.read_parquet(f'{PATH_TO_DATA}targets_{k}.pqt') )\n        X_valid = np.concatenate(X_valid,axis=0)\n        y_valid = pd.concat(y_valid).target.values\n        print('### Validation data shapes', X_valid.shape, y_valid.shape)\n        print('#'*25)\n\n        # BUILD AND TRAIN MODEL\n        K.clear_session()\n        m0 = build_model()\n        class_weight={0:1, 1:1}      \n        m0.summary() \n        m0.fit(X_train, y_train, epochs = 8, validation_data=(X_valid,y_valid), callbacks=[callback], batch_size=400, shuffle=True,class_weight=class_weight ) \n        if not os.path.exists(PATH_TO_MODEL): os.makedirs(PATH_TO_MODEL)\n        m0.save_weights(f'{PATH_TO_MODEL}fold_{fold}.h5')\n        \n\n\n        # INFER VALID DATA\n        print('Inferring validation data...')\n        p = m0.predict(X_valid, batch_size=512, verbose=VERBOSE).flatten()\n        print()\n        print(f'Fold {fold+1} CV=', amex_metric_mod(y_valid, p) )\n        print()\n\n        \n        # CLEAN MEMORY\n        del m0, X_train, y_train, X_valid, y_valid, p\n        gc.collect()\n\n       \n\n    # PRINT OVERALL RESULTS\n    print('#'*25)\n    #print(f'Overall CV =', amex_metric_mod(true, oof) )\n    #del true, oof\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:50:20.447027Z","iopub.execute_input":"2022-08-13T21:50:20.447429Z","iopub.status.idle":"2022-08-13T22:26:54.273138Z","shell.execute_reply.started":"2022-08-13T21:50:20.447397Z","shell.execute_reply":"2022-08-13T22:26:54.270763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n    \n    \nif INFER_TEST:\n\n    K.clear_session()\n    # LOAD SAMPLE SUBMISSION\n    start = 0; end = 0\n    sub = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\n\n    \n    # REARANGE SUB ROWS TO MATCH 20 TEST FILES\n    sub['hash'] = sub['customer_ID'].str[-16:].apply(lambda x: int(x, 16)).astype('int64')\n    test_hash_index = np.load(f'{PATH_TO_DATA}test_hashes_data.npy')\n    sub = sub.set_index('hash').loc[test_hash_index].reset_index(drop=True)\n\n    \n    for k in range(20):\n        K.clear_session()\n        model = build_model()\n        print(f'Inferring Test_File_{k+1}')\n        X_test = np.load(f'{PATH_TO_DATA}test_data_{k+1}.npy')\n        end = start + X_test.shape[0]\n\n        # INFER 5 FOLD MODELS\n        \n        j=0\n        model.load_weights(f'{PATH_TO_MODEL}fold_{j}.h5')\n        p = model.predict(X_test, batch_size=512, verbose=0).flatten()\n        for j in range(len(folds)-1):\n            model.load_weights(f'{PATH_TO_MODEL}fold_{j+1}.h5')\n            p+= model.predict(X_test, batch_size=512, verbose=0).flatten()\n        p = p/len(folds)\n\n        sub.loc[start:end-1,'prediction'] = p\n        start = end\n        gc.collect()\n        del model\n        ","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:26:54.278958Z","iopub.execute_input":"2022-08-13T22:26:54.279318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if INFER_TEST:\n    sub.to_csv(f'submission.csv',index=False)\n    print('Submission file shape is', sub.shape )\n    display( sub.head() )","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}