{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <span style=\"color:#008bf8ff; font-weight: bold;\"> Summary </span>","metadata":{}},{"cell_type":"markdown","source":"Hello, I would be happy if this notebook could help in someway. <span style=\"color:#008bf8ff; font-weight: bold;\">Please, fill free to use and comment it. </span>\n\nA preliminary exploration work, including graphs, has been done in an <a href='https://www.kaggle.com/code/laetitialanfranchi/exploration-with-graphs'><span style=\"color:#008bf8ff; font-weight: bold;\">exploration notebook</span></a>.\n\nThis notebook includes all in one Preprocessing, Transformer Model, Inference and Submission, and <span style=\"color:#008bf8ff; font-weight: bold;\">might be updated in the following weeks.</span>\n\n<span style=\"color:#008bf8ff; font-weight: bold;\"> A similar version of this notebook has been trained and obtained a val loss of 0.0716 on the train dataset </span>\n\n**** \n\nDetails about what is included in this notebook:\n- **Preprocessing, mainly composed of:** \n    - Reshaping data between DMS and 2A3 experiments\n    - Filtering on SN = True lines\n    - Cliping reactivities between 0 and 1\n    - Creating words from sequences based on a sliding window\n    - Padding and encoding sequence \n    - Splitting data between train and val and creating an input pipeline dataset to feed the model\n    - <span style=\"color:#008bf8ff; font-weight: bold;\">The total duration of the preprocessing steps is less than 5 minutes</span>\n- **A transformer model, mainly composed of an embedding layer and encoder blocks:** \n    - The embedding is a trainable layer, with a normalization and a positionnal encoding\n    - The encoder blocks all based on attention score and ends up with some appropriate dense layers\n    - DMS and 2A3 experiments are predicted jointly\n    - Input sequence length is variable\n    - <span style=\"color:#008bf8ff; font-weight: bold;\">Training duration is about 50-80s per epoch</span>\n- **Inference and submission:** \n    - The test data are preprocessed and inference is done\n    - Inference is done by sequence lenght, in or to ease the prediction rearrangement required to meet the competition requirements\n    - <span style=\"color:#008bf8ff; font-weight: bold;\">Inference duration is less than 15 minutes</span>\n    \n**** ","metadata":{}},{"cell_type":"markdown","source":"# <span style=\"color:#008bf8ff; font-weight: bold;\"> Imports </span>","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras import mixed_precision\n\nfrom sklearn.model_selection import train_test_split\n\nfrom itertools import product\nimport math\nfrom datetime import datetime\nimport pytz\nimport matplotlib.pyplot as plt\nimport pickle\n\nfrom script_cfg import *","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Tensorflow version \" + tf.__version__)\n\ntpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect()\nprint('Running on TPU ', tpu.master())\n\ntpu_strategy = tf.distribute.TPUStrategy(tpu)\nprint ('Number of devices: {}'.format(tpu_strategy.num_replicas_in_sync))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"color:#008bf8ff; font-weight: bold;\"> Preprocessing </span>","metadata":{}},{"cell_type":"code","source":"# Preprocessing variables\nSN_FILTER = True\nWINDOW = 4\nMAX_SEQ = 457","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@simple_time_and_memory_tracker\ndef create_train_array():\n    ''' Load data, impute nan value, and deduplicate sequences\n        It returns sequences, exp_type and reactivities '''\n    \n    # Load usefull train data: sequence SN_filter, experiment type and reactivities\n    data = pd.read_csv(CFG.TRAIN, dtype=CFG.DTYPES, usecols=['sequence', 'SN_filter', 'experiment_type'] + CFG.COLS_REACT)\n    \n    # Filter (or not) on Signal to Noise = 1\n    if SN_FILTER:\n        data = data[ (data.SN_filter == True) ]\n\n    # Shuffle data\n    data = data.sample(frac=1, random_state=42)\n\n    # Pass to numpy array\n    sequence = np.array(data.sequence)\n    exp_type = np.array(data.experiment_type)\n    react = np.array(data[CFG.COLS_REACT])\n\n    # Clip reactivities between 0 and 1\n    react = np.clip(react, 0, 1)\n\n    # Drop duplicated sequences\n    sequence_exp_type = exp_type + sequence\n    _, unique_indices = np.unique(sequence_exp_type, return_index=True)\n    sequence = sequence[unique_indices]\n    exp_type = exp_type[unique_indices]\n    react = react[unique_indices]\n    \n    return sequence, exp_type, react\n\n@simple_time_and_memory_tracker\ndef split_dms_2a3(sequence, exp_type, react):\n    ''' Split sequence and react according to exp-type '''\n    \n    # Keep only dms experiment\n    indices_dms = np.where(exp_type == 'DMS_MaP')\n    react_dms = react[indices_dms]\n    sequence_dms = sequence[indices_dms]\n    sequence_len_dms = np.vectorize(len)(sequence_dms)\n\n    # Keep only 2a3 experiment\n    indices_2a3 = np.where(exp_type == '2A3_MaP')\n    react_2a3 = react[indices_2a3]\n    sequence_2a3 = sequence[indices_2a3]\n    sequence_len_2a3 = np.vectorize(len)(sequence_2a3)\n    \n    # Tensorize\n    sequence_dms, react_dms, sequence_2a3, react_2a3 = tf.constant(sequence_dms), tf.constant(react_dms), tf.constant(sequence_2a3), tf.constant(react_2a3)\n\n    return sequence_dms, react_dms, sequence_2a3, react_2a3\n\n@simple_time_and_memory_tracker\ndef merge_dms_2a3(sequence_dms, sequence_2a3, react_dms, react_2a3):\n    ''' Re-merge dms and 2a3 experiments after preprocessing '''\n\n    sequence_dms_pd = pd.DataFrame(sequence_dms, columns=[f\"seq_{i:004d}\" for i in range(sequence_dms.shape[1])])\n    sequence_2a3_pd = pd.DataFrame(sequence_2a3, columns=[f\"seq_{i:004d}\" for i in range(sequence_2a3.shape[1])])\n    react_dms_pd = pd.DataFrame(react_dms, columns=[i for i in range(react_dms.shape[1])])\n    react_2a3_pd = pd.DataFrame(react_2a3, columns=[i for i in range(react_2a3.shape[1])])\n   \n    df_dms = pd.concat([sequence_dms_pd, react_dms_pd], axis=1)\n    df_2a3 = pd.concat([sequence_2a3_pd, react_2a3_pd], axis=1)\n    df_dms = df_dms.sample(frac=1, random_sate=42)\n    df_2a3 = df_2a3.sample(frac=1, random_sate=42)\n\n    df = df_dms.merge(df_2a3, how='outer', on=[f\"seq_{i:004d}\" for i in range(sequence_dms.shape[1])])\n    df = df.sample(frac=1, random_state=42)\n\n    sequence = np.array(df.iloc[:, :sequence_dms.shape[1]], dtype=np.int32)\n    react = np.array(df.iloc[:, sequence_dms.shape[1]:], dtype=np.float32)\n\n    react = np.reshape(react, (react.shape[0], react_dms.shape[1], 2), order='F')\n    \n    sequence = tf.constant(sequence)\n    react = tf.constant(react)\n    \n    return sequence, react","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@simple_time_and_memory_tracker\ndef window_tokenize_pad(sequence):\n    ''' Pad and tokenise sequence according to the window size '''\n\n    sequence_ragged = tf.strings.bytes_split(sequence)\n    sequence_padded = tf.squeeze(sequence_ragged.to_tensor('E'))\n    sequence_pre_padded = tf.concat([tf.fill((tf.shape(sequence_padded)[0], WINDOW), 'E'), sequence_padded], axis=1)\n    sequence_ngrams = tf.strings.ngrams(sequence_pre_padded, WINDOW, separator='')\n    \n    return sequence_ngrams[:, 1:]\n\ndef window_encoding_table():\n    ''' Build the encoding table according to the window size '''\n\n    def build_window_words():\n        ''' Build all possible words, according to the window size '''\n        \n        window_words = [''.join(p) for w in range(1, WINDOW+1) for p in product([\"A\", \"G\", \"U\", \"C\"], repeat=w) ]\n        window_words_padded = []\n        window_words_padded.append('E' * (WINDOW))\n        \n        for word in window_words:\n            if len(word) < WINDOW:\n                window_words_padded.append('E' * (WINDOW - len(word)) + word)\n                window_words_padded.append(word + 'E' * (WINDOW - len(word)))\n            else:\n                window_words_padded.append(word)\n        \n        return window_words_padded\n    \n    window_words = build_window_words()    \n    table = tf.lookup.StaticHashTable(initializer=tf.lookup.KeyValueTensorInitializer(keys=tf.constant(window_words), \n                                                                                      values=tf.constant(range(len(window_words)), dtype=tf.int32),), \n                                      default_value=tf.constant(-1), name=\"char_encoding_window\")\n    \n    return table, window_words\n    \n@simple_time_and_memory_tracker\ndef window_encode(sequence_padded):\n    ''' Encode sequences, according to the encoding table '''\n\n    table = window_encoding_table()[0]\n    sequence_encoded = table.lookup(sequence_padded)\n    return sequence_encoded","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocessing():\n    ''' Complete preprocessing: load train data, rearrange dms and 2a3 experiments, apply a sliding window, tokenise, pad and encode sequences '''\n\n    # Load train data\n    sequence, exp_type, react = create_train_array()\n\n    # Split dms and 2a3 experiments\n    sequence_dms, react_dms, sequence_2a3, react_2a3 = split_dms_2a3(sequence, exp_type, react)\n\n    # Sequence preprocessing: apply a sliding window, tokenize, pad and encode \n    sequence_dms, sequence_2a3 = window_tokenize_pad(sequence_dms), window_tokenize_pad(sequence_2a3)\n    sequence_dms, sequence_2a3 = window_encode(sequence_dms), window_encode(sequence_2a3)\n\n    # Re-merge dms and 2a3 experiments \n    sequence, react = merge_dms_2a3(sequence_dms, sequence_2a3, react_dms, react_2a3)\n    \n    return sequence, react","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Execute preprocessing\nsequence, react = preprocessing()\n\n# Save locally \nnp.save('sequence.npy', sequence)\nnp.save('react.npy', react)\n\nsequence.shape, react.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load from local\nsequence = tf.constant(np.load('sequence.npy', allow_pickle=True))\nreact = tf.constant(np.load('react.npy', allow_pickle=True))\n\nsequence.shape, react.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sequence[0]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"react[0, :, 0]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Embedding variables directly resulting from the preprocessing steps\n\nvocab_size = len(window_encoding_table()[1])\nprint('Vocab size of sequence after the sliding window:', vocab_size)\n\nencoded_padded_value = window_encoding_table()[0].lookup(tf.constant('E' * WINDOW)).numpy()\nprint('Encoded padded value for sequence:', encoded_padded_value)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"color:#008bf8ff; font-weight: bold;\"> Input Pipeline </span>","metadata":{}},{"cell_type":"code","source":"# Batchsize with regards to TPU connection\ntry: \n    batchsize_train = 256 * tpu_strategy.num_replicas_in_sync\n    batchsize_val = batchsize_train\n\nexcept:\n    batchsize_train = 8\n    batchsize_val = 4\n    \nbatchsize_train, batchsize_val","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_val_split(sequence, react):\n    ''' Simple split of train data between a train and a val '''\n    \n    sequence_train, sequence_test, react_train, react_test = train_test_split(sequence, react, test_size=0.2)\n    return sequence_train, sequence_test, react_train, react_test\n\ndef build_dataset(sequence, react, batchsize):\n    ''' Simple pipeline of the input dataset '''\n    \n    dataset = tf.data.Dataset.from_tensors((tf.constant(sequence), tf.constant(react)))\n    dataset = dataset.cache()\n    dataset = dataset.shuffle(batchsize * 8)\n    dataset = dataset.unbatch().batch(batchsize, drop_remainder=True)\n    dataset = dataset.prefetch(tf.data.AUTOTUNE)\n    return dataset","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split train between train and val\nsequence_train, sequence_val, react_train, react_val = train_val_split(np.array(sequence), np.array(react))\nsequence_train, sequence_val, react_train, react_val = tf.constant(sequence_train), tf.constant(sequence_val), tf.constant(react_train), tf.constant(react_val)\nprint('Train/val shapes:', sequence_train.shape, sequence_val.shape, react_train.shape, react_val.shape)\n\n# Build train and val datasets\ndataset_train = build_dataset(sequence_train, react_train, batchsize_train)\ndataset_val = build_dataset(sequence_val, react_val, batchsize_val)\n\n# Check data shape and count the number of train and val batches \ndataset_train_iter = next(iter(dataset_train))\ndataset_val_iter = next(iter(dataset_val))\n\ntrain_batch_nb = round(sequence_train.shape[0]/dataset_train_iter[0].shape[0])\nval_batch_nb = round(sequence_val.shape[0]/dataset_val_iter[0].shape[0])\n\nprint('Train - dataset shapes', dataset_train_iter[0].shape, dataset_train_iter[1].shape, 'Train batch nb:', train_batch_nb)\nprint('Val - dataset shapes', dataset_val_iter[0].shape, dataset_val_iter[1].shape, 'Val batch nb:', val_batch_nb)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"color:#008bf8ff; font-weight: bold;\"> Model architecture </span>","metadata":{}},{"cell_type":"code","source":"# Embedding parameters\nd_model = 256\n\n# Encoder parameters\nnum_layers = 12\nnum_heads = 8\ndff = 256 * 2\n\n# Hyper-parameters\nemb_rate = 0.4\nenc_rate = 0.4","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class positional_encoding_layer(tf.keras.layers.Layer):\n    ''' Positionnal encoding, as defined is the Transformer paper '''\n\n    def __init__(self, vocab_size, maxlen, d_model):\n        super().__init__()\n        self.d_model = d_model\n        self.pos_emb = self.positional_encoding(maxlen, d_model)\n        self.supports_masking = True\n\n    def call(self, x):\n        maxlen = tf.shape(x)[-2]\n        x = tf.math.multiply(x, tf.math.sqrt(tf.cast(self.d_model, tf.float32)))\n        return x + self.pos_emb[0, :maxlen, :]\n\n    def positional_encoding(self, maxlen, d_model):\n\n        def get_angles(pos, i, d_model):\n            angle_rates = 1 / np.power(10000, (2 * (i//2)) / np.float32(d_model))\n            return pos * angle_rates\n\n        angle_rads = get_angles(np.arange(maxlen)[:, np.newaxis], np.arange(d_model)[np.newaxis, :], d_model)\n        angle_rads[:, 0::2] = np.sin(angle_rads[:, 0::2])\n        angle_rads[:, 1::2] = np.cos(angle_rads[:, 1::2])\n        pos_encoding = angle_rads[np.newaxis, ...]\n        pos_encoding = tf.cast(pos_encoding, dtype=tf.float32)\n        \n        return pos_encoding","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_encoder():\n    ''' Encoder is composed of: \n            - Trainable embedding layer + a dropout layer\n            - Positionnal Encoding layer, with normalisation \n            - Encoder Blocks with multi-heads attention, normalization and drop out layers\n            - Dense layers to fit the reactivities output\n        The input sequence lenght is variable (shape set to None). The output lenght is the length of the input sequence, so is variable '''\n    \n    # Input Layer\n    inp = tf.keras.layers.Input(shape=(None, ))\n    \n    # Embedding and Positionnal encoding\n    x = tf.keras.layers.Embedding(input_dim=vocab_size + 1, output_dim=d_model, mask_zero=True)(inp)\n    x = tf.keras.layers.Dropout(emb_rate)(inputs=x)\n    x = positional_encoding_layer(vocab_size=vocab_size, maxlen=MAX_SEQ, d_model=d_model)(x)\n    \n    # Loop on Encoder blocks    \n    for i in range(num_layers):\n        \n        # Multi head attention\n        attn_output, _ = tf.keras.layers.MultiHeadAttention(num_heads=num_heads, key_dim=d_model)(query=x, key=x, value=x, return_attention_scores=True)\n        attn_output = tf.keras.layers.Dropout(enc_rate)(attn_output)\n        mha_output = tf.keras.layers.LayerNormalization(epsilon=1e-6)(x + attn_output)\n        \n        # Feed Forward\n        ffn_output = tf.keras.layers.Dense(dff, activation='relu')(mha_output)\n        ffn_output = tf.keras.layers.Dense(d_model, activation='relu')(ffn_output)\n        ffn_output = tf.keras.layers.Dropout(enc_rate)(ffn_output)\n        \n        # Normalisation\n        x = tf.keras.layers.LayerNormalization(epsilon=1e-6)(mha_output + ffn_output)\n        \n    # Ouput Dense layers\n    x = tf.keras.layers.Dense(d_model / 2, activation='relu')(x)\n    x = tf.keras.layers.Dense(d_model / 4, activation='relu')(x)\n    x = tf.keras.layers.Dense(d_model / 8, activation='relu')(x)\n    out = tf.keras.layers.Dense(2, activation='relu')(x)\n       \n    model = tf.keras.Model(inputs=inp, outputs=out)\n    return model","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"color:#008bf8ff; font-weight: bold;\"> Compilation </span>","metadata":{}},{"cell_type":"code","source":"# Compile Hyper-parameters\ninitial_learning_rate = 1e-4\ndecay_steps = 8000\ndecay_rate = 0.01\nnb_epoch_max = 100\n\n# Model_name for saving\nmodel_name = f\"encoder_sn{SN_FILTER}_w{WINDOW}_d{d_model}_l{num_layers}_h{num_heads}_ff{dff}_r{emb_rate}-{enc_rate}_lr{initial_learning_rate}-{decay_steps}-{decay_rate}_b{batchsize_train}\"\nmodel_name","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_lr_scheduler(initial_learning_rate, decay_steps, decay_rate, nb_epoch_max):\n    ''' Simple visualisation of the learning rate scheduler '''\n    \n    nb_steps_per_epoch = round(sequence_train.shape[0] / batchsize_train)\n\n    def get_lr_cosine_decay(lr_init, decay_step, alpha):\n        lr_schedule = tf.keras.optimizers.schedules.CosineDecay(lr_init, decay_step, alpha)\n        return [lr_schedule(i).numpy() for i in range(nb_steps_per_epoch*nb_epoch_max) if i%nb_steps_per_epoch == 0]\n\n    fig, ax = plt.subplots(nrows=1, ncols=1, figsize=(6, 2))\n    fig.suptitle('Cosine decay', color='#3f3f3f', y=1.2, fontweight='bold', fontsize='large')\n    fig.subplots_adjust(left=0.05, bottom=0.05, right=0.95, top=1, wspace=0.2, hspace=0.7)\n    lr_schedule = get_lr_cosine_decay(initial_learning_rate, decay_steps, decay_rate)\n    ax.plot(lr_schedule, color='#ffc2b4ff');\n    ax.set_xlabel('epoch', loc='center', color='#3f3f3f', fontsize='small')\n    ax.set_title(f\"Decay step {decay_steps}, Alpha {decay_rate:.0e}, Final lr {lr_schedule[-1]:.0e} \", color='#3f3f3f', fontsize='small')\n    ax.set_ylim(ymin=0, ymax=initial_learning_rate)\n    ax.grid(axis=\"both\", lw=0.5, ls=':')\n    ax.spines[['right', 'top']].set_visible(False)\n    ax.spines['left'].set(color='#3f3f3f', linewidth=0.5, position=('axes', 0))\n    ax.spines['bottom'].set(color='#3f3f3f', linewidth=0.5, position=('axes', 0))\n    ax.tick_params(axis='both', color='#3f3f3f', colors='#3f3f3f', labelsize='x-small')\n    \nplot_lr_scheduler(initial_learning_rate, decay_steps, decay_rate, nb_epoch_max)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def masked_mae_loss(react_true, react_pred):\n    ''' Computation of the loss: mae with a mask on value imputed '''\n    \n    mask = tf.math.is_finite(react_true)\n    react_true_masked = tf.boolean_mask(react_true, mask)\n    react_pred_masked = tf.boolean_mask(react_pred, mask)\n    loss = loss_object(react_true_masked, react_pred_masked)\n    \n    return loss","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Build and compile the model\nwith tpu_strategy.scope():\n    \n    loss_object = tf.keras.losses.MeanSquaredError()\n    lr_schedule = tf.keras.optimizers.schedules.ExponentialDecay(initial_learning_rate=initial_learning_rate, decay_steps=decay_steps, decay_rate=decay_rate)\n    optimizer = tf.keras.optimizers.Adam(learning_rate=lr_schedule, beta_1=0.9, beta_2=0.98, epsilon=1e-9)\n\n    encoder = create_encoder()\n    \n    encoder.compile(optimizer=optimizer, loss=masked_mae_loss)\n    encoder.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"color:#008bf8ff; font-weight: bold;\"> Training </span>","metadata":{}},{"cell_type":"code","source":"print(datetime.now(pytz.timezone('Europe/Paris')).strftime('%H:%M'))\n\n# Train the model\nwith tpu_strategy.scope():\n    \n    es = tf.keras.callbacks.EarlyStopping(patience=4, restore_best_weights=True, monitor='val_loss', min_delta=0.0005, mode='auto', verbose=0)\n    ckpt = tf.keras.callbacks.ModelCheckpoint(f\"ckpt_{model_name}\", monitor=\"val_loss\", mode='auto', save_freq='epoch', save_best_only=True, save_weights_only=False, verbose=0)\n\n    history = encoder.fit(dataset_train, validation_data=dataset_val, shuffle=True, epochs=1, callbacks=[es, ckpt], verbose=1)\n    \nprint(datetime.now(pytz.timezone('Europe/Paris')).strftime('%H:%M'))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"color:#008bf8ff; font-weight: bold;\"> Registry </span>","metadata":{}},{"cell_type":"code","source":"# Save the model\nencoder.save(f\"saved_model/{model_name}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"color:#008bf8ff; font-weight: bold;\"> Inference </span>","metadata":{}},{"cell_type":"code","source":"# Inference parameters\nbatchsize_test = 4096\ntest_sequence_lenght_unique = [177, 207, 307, 457]\nchunksize = 800000","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ''' Load and preprocess test data '''\n\n# Data is saved by sequence lenght to ease the inference process\nfor seq_len in test_sequence_lenght_unique:\n    print(f\"\\033[94m Processing {seq_len} \\033[0m\")\n\n    # A chunk logic is applied to fit in RAM\n    test_data = pd.read_csv('/kaggle/input/stanford-ribonanza-rna-folding/test_sequences.csv', chunksize=chunksize)\n    test_sequence = tf.zeros((0, seq_len), dtype=tf.int32)\n\n    for chunk_nb, chunk in enumerate(test_data):\n        test_sequence_ = tf.constant(chunk.sequence)\n        test_sequence_lenght = tf.strings.length(test_sequence_)\n\n        indices_ = tf.where(test_sequence_lenght == seq_len)\n        if indices_.shape[0] > 0:\n            test_sequence_ = tf.squeeze(tf.gather(test_sequence_, indices_))\n            test_sequence_ = window_tokenize_pad(test_sequence_)\n            test_sequence_ = window_encode(test_sequence_)\n            test_sequence = tf.concat([test_sequence, test_sequence_], axis=0)\n\n    ''' Save ''' \n    print(f\"Saving seq_len {seq_len}, with sequence shape {test_sequence.shape}\\n\")\n    np.save(f\"test_sequence_{seq_len}.npy\", test_sequence)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with tpu_strategy.scope():\n    encoder = tf.keras.models.load_model(f\"saved_model/{model_name}\", custom_objects={'masked_mae_loss': 'masked_mae_loss'})    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_dataset_test(sequence, batchsize):\n    dataset = tf.data.Dataset.from_tensors(sequence)\n    dataset = dataset.unbatch().batch(batchsize, drop_remainder=False)\n    return dataset","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(datetime.now(pytz.timezone('Europe/Paris')).strftime('%H:%M'))\ndf_submission_final = pd.DataFrame(columns = ['id', 'reactivity_DMS_MaP', 'reactivity_2A3_MaP'])\nstart_id = 0\n\n# Infering is done by sequence lenght\nfor seq_len in test_sequence_lenght_unique:\n    print(f\"\\033[94m \\n Infering {seq_len} \\033[0m\")\n    start_time = time.time()\n\n    # Load test data preprocessed\n    test_sequence = tf.constant(np.load(f\"test_sequence_{seq_len}.npy\", allow_pickle=True), dtype=tf.float64)\n    dataset_test = build_dataset_test(test_sequence, batchsize_test)\n    \n    # Predict\n    with tpu_strategy.scope():\n        prediction = encoder.predict(dataset_test, verbose=1)\n        \n    # Concatenate id, dms and 2a3 prediction       \n    df_submission_len = pd.DataFrame({'id': np.arange(start_id, start_id + test_sequence.shape[0] * seq_len),\n                                      'reactivity_DMS_MaP': np.reshape(prediction[:, :, 0], (test_sequence.shape[0] * seq_len)), \n                                      'reactivity_2A3_MaP': np.reshape(prediction[:, :, 1], (test_sequence.shape[0] * seq_len))})\n       \n    # Concatenate and save submission\n    df_submission_final = pd.concat([df_submission_final, df_submission_len], axis=0)\n    start_id += test_sequence.shape[0] * seq_len\n    print(f\"Submission lines {df_submission_final.shape[0]}, duration {time.time()-start_time:0.1f}sec\")\n\ndf_submission_final.to_parquet(path=f\"submission.parquet\")    \nprint(f\"\\n \\033[94m Checks: {df_submission_final.shape[0] == 59440671 + 207000000 + 614000 + 2742000}, {df_submission_final.shape[0] == df_submission_final.id.nunique()} \\033[0m \\n\")\nprint(datetime.now(pytz.timezone('Europe/Paris')).strftime('%H:%M'))\ndf_submission_final.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}