{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport time\nimport numpy as np\nimport tensorflow as tf\nfrom sklearn.model_selection import KFold\nimport random\nimport joblib","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = '/kaggle/input/stanford-ribonanza-rna-folding/train_data.csv'\ntest_sequences ='/kaggle/input/stanford-ribonanza-rna-folding/test_sequences.csv'\ninputs_length = 457\n\ntrain_file = \"train_file.csv\"\n#model = 'model_DMS.ml'\nmodel = 'model.ml'","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:58:15.397228Z","iopub.execute_input":"2023-11-13T08:58:15.397639Z","iopub.status.idle":"2023-11-13T08:58:15.401681Z","shell.execute_reply.started":"2023-11-13T08:58:15.397614Z","shell.execute_reply":"2023-11-13T08:58:15.400939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cp /kaggle/input/stanford-rrf-tensorflow-training-tpu/models.pkl2 models.pkl2","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:58:15.402534Z","iopub.execute_input":"2023-11-13T08:58:15.402776Z","iopub.status.idle":"2023-11-13T08:58:17.306181Z","shell.execute_reply.started":"2023-11-13T08:58:15.402754Z","shell.execute_reply":"2023-11-13T08:58:17.304816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## read and create data","metadata":{}},{"cell_type":"markdown","source":"df_c = pd.read_csv(train_data, skiprows=0, chunksize=400000)\nfor n in range(4):\n    print(\"###############:\",n)\n    df = df_c.get_chunk()\n    print(df.shape)\n    df = df[df[\"SN_filter\"] >0 ]\n    df_DMS_MaP = df[df[\"experiment_type\"] == \"DMS_MaP\"]\n    df_2A3_MaP = df[df[\"experiment_type\"] == \"2A3_MaP\"]    \n    delete_list = []\n    for k in df.columns:\n        if 'reactivity' in k and 'error'  in k:\n            delete_list.append(k)\n    df_DMS_MaP = df_DMS_MaP.drop(delete_list, axis=1)\n    df_2A3_MaP = df_2A3_MaP.drop(delete_list, axis=1)\n    print(df_DMS_MaP.shape)\n    print(df_2A3_MaP.shape)\n    if n == 0 :\n        df_DMS_MaP.to_csv(\"DMS_MaP_\"+train_file, header=df_DMS_MaP.keys() ,index=False)\n        df_2A3_MaP.to_csv(\"2A3_MaP_\"+train_file, header=df_2A3_MaP.keys() ,index=False)\n    else:\n        df_DMS_MaP.to_csv(\"DMS_MaP_\"+train_file, mode='a', header=False,index=False)        \n        df_2A3_MaP.to_csv(\"2A3_MaP_\"+train_file, mode='a', header=False,index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-06T19:45:04.009748Z","iopub.execute_input":"2023-11-06T19:45:04.010033Z","iopub.status.idle":"2023-11-06T19:47:32.224475Z","shell.execute_reply.started":"2023-11-06T19:45:04.010007Z","shell.execute_reply":"2023-11-06T19:47:32.223684Z"}}},{"cell_type":"markdown","source":"# model","metadata":{}},{"cell_type":"code","source":"import logging\nimport time\n\nimport numpy as np\n\nimport tensorflow_datasets as tfds\nimport tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:58:17.309028Z","iopub.execute_input":"2023-11-13T08:58:17.309327Z","iopub.status.idle":"2023-11-13T08:58:19.388486Z","shell.execute_reply.started":"2023-11-13T08:58:17.309300Z","shell.execute_reply":"2023-11-13T08:58:19.387373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## encode","metadata":{}},{"cell_type":"code","source":"def positional_encoding(length, depth):\n  depth = depth/2\n\n  positions = np.arange(length)[:, np.newaxis]     # (seq, 1)\n  depths = np.arange(depth)[np.newaxis, :]/depth   # (1, depth)\n\n  angle_rates = 1 / (10000**depths)         # (1, depth)\n  angle_rads = positions * angle_rates      # (pos, depth)\n\n  pos_encoding = np.concatenate(\n      [np.sin(angle_rads), np.cos(angle_rads)],\n      axis=-1) \n\n  return tf.cast(pos_encoding, dtype=tf.float32)\n                 \nclass PositionalEmbedding(tf.keras.layers.Layer):\n  def __init__(self, vocab_size, d_model):\n    super().__init__()\n    self.d_model = d_model\n    self.embedding = tf.keras.layers.Embedding(vocab_size, d_model, mask_zero=True) \n    self.pos_encoding = positional_encoding(length=2048, depth=d_model)\n\n  def compute_mask(self, *args, **kwargs):\n    return self.embedding.compute_mask(*args, **kwargs)\n\n  def call(self, x):\n    length = tf.shape(x)[1]\n    x = self.embedding(x)\n    # This factor sets the relative scale of the embedding and positonal_encoding.\n    x *= tf.math.sqrt(tf.cast(self.d_model, tf.float32))\n    x = x + self.pos_encoding[tf.newaxis, :length, :]\n    return x","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:58:19.389782Z","iopub.execute_input":"2023-11-13T08:58:19.390066Z","iopub.status.idle":"2023-11-13T08:58:19.399825Z","shell.execute_reply.started":"2023-11-13T08:58:19.390040Z","shell.execute_reply":"2023-11-13T08:58:19.398885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BaseAttention(tf.keras.layers.Layer):\n  def __init__(self, **kwargs):\n    super().__init__()\n    self.mha = tf.keras.layers.MultiHeadAttention(**kwargs)\n    self.layernorm = tf.keras.layers.LayerNormalization()\n    self.add = tf.keras.layers.Add()","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:58:19.400931Z","iopub.execute_input":"2023-11-13T08:58:19.401196Z","iopub.status.idle":"2023-11-13T08:58:19.420834Z","shell.execute_reply.started":"2023-11-13T08:58:19.401172Z","shell.execute_reply":"2023-11-13T08:58:19.419964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class GlobalSelfAttention(BaseAttention):\n  def call(self, x):\n    attn_output = self.mha(\n        query=x,\n        value=x,\n        key=x)\n    x = self.add([x, attn_output])\n    x = self.layernorm(x)\n    return x","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:58:19.421870Z","iopub.execute_input":"2023-11-13T08:58:19.422144Z","iopub.status.idle":"2023-11-13T08:58:19.454953Z","shell.execute_reply.started":"2023-11-13T08:58:19.422120Z","shell.execute_reply":"2023-11-13T08:58:19.454109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class FeedForward(tf.keras.layers.Layer):\n  def __init__(self, d_model, dff, dropout_rate=0.1):\n    super().__init__()\n    self.seq = tf.keras.Sequential([\n      tf.keras.layers.Dense(dff, activation='relu'),\n      tf.keras.layers.Dense(d_model),\n      tf.keras.layers.Dropout(dropout_rate)\n    ])\n    self.add = tf.keras.layers.Add()\n    self.layer_norm = tf.keras.layers.LayerNormalization()\n\n  def call(self, x):\n    x = self.add([x, self.seq(x)])\n    x = self.layer_norm(x) \n    return x","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:58:19.456027Z","iopub.execute_input":"2023-11-13T08:58:19.456277Z","iopub.status.idle":"2023-11-13T08:58:19.478216Z","shell.execute_reply.started":"2023-11-13T08:58:19.456254Z","shell.execute_reply":"2023-11-13T08:58:19.477322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class EncoderLayer(tf.keras.layers.Layer):\n  def __init__(self,*, d_model, num_heads, dff, dropout_rate=0.1):\n    super().__init__()\n\n    self.self_attention = GlobalSelfAttention(\n        num_heads=num_heads,\n        key_dim=d_model,\n        dropout=dropout_rate)\n\n    self.ffn = FeedForward(d_model, dff)\n\n  def call(self, x):\n    x = self.self_attention(x)\n    x = self.ffn(x)\n    return x","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:58:19.479433Z","iopub.execute_input":"2023-11-13T08:58:19.479838Z","iopub.status.idle":"2023-11-13T08:58:19.490614Z","shell.execute_reply.started":"2023-11-13T08:58:19.479811Z","shell.execute_reply":"2023-11-13T08:58:19.489785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class RnnModel(tf.keras.Model):\n  def __init__(self, *, num_layers, d_model, num_heads,\n               dff, vocab_size, dropout_rate=0.1):\n    super().__init__()\n\n    self.d_model = d_model\n    self.num_layers = num_layers\n\n    self.pos_embedding = PositionalEmbedding(\n        vocab_size=vocab_size, d_model=d_model)\n\n    self.enc_layers = [\n        EncoderLayer(d_model=d_model,\n                     num_heads=num_heads,\n                     dff=dff,\n                     dropout_rate=dropout_rate)\n        for _ in range(num_layers)]\n    \n    self.dropout = tf.keras.layers.Dropout(dropout_rate)\n    self.layer1_100 = tf.keras.layers.Dense(400, activation='relu')\n    self.layer1_10 = tf.keras.layers.Dense(40,activation='relu')\n    self.out1_layer = tf.keras.layers.Dense(2)\n    #self.layer2_100 = tf.keras.layers.Dense(400, activation='relu')\n    #self.layer2_10 = tf.keras.layers.Dense(40,activation='relu')\n    #self.out2_layer = tf.keras.layers.Dense(1)\n    self.gausi1 = tf.keras.layers.GaussianNoise(0.01)\n    self.gausi2 = tf.keras.layers.GaussianNoise(0.01)\n    #self.gausi3 = tf.keras.layers.GaussianNoise(0.01)\n  def call(self, x):\n    #print(x.shape)\n    # `x` is token-IDs shape: (batch, seq_len)\n    x = self.pos_embedding(x)  # Shape `(batch_size, seq_len, d_model)`.\n    # Add dropout.\n    x = self.dropout(x)\n    for i in range(self.num_layers):\n        x = self.enc_layers[i](x)\n    x = self.gausi1(x)\n    x1 = self.layer1_100(x)\n    x1 = self.gausi2(x1)\n    x1 = self.layer1_10(x1)\n    o1 = self.out1_layer(x1) \n    #x2 = self.layer2_100(x)\n    #x2 = self.gausi3(x2)\n    #x2 = self.layer2_10(x2)\n    #o2 = self.out2_layer(x2) \n    #o = tf.concat([o1,o2], axis = 1)\n    #return tf.squeeze(o,axis = -1)\n    return o1 ","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:58:19.491666Z","iopub.execute_input":"2023-11-13T08:58:19.491945Z","iopub.status.idle":"2023-11-13T08:58:19.504945Z","shell.execute_reply.started":"2023-11-13T08:58:19.491922Z","shell.execute_reply":"2023-11-13T08:58:19.504078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_train_col(str_col):\n    return np.array([*(str_col+(inputs_length-len(str_col))*\"0\")])","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:58:19.506037Z","iopub.execute_input":"2023-11-13T08:58:19.506302Z","iopub.status.idle":"2023-11-13T08:58:19.520077Z","shell.execute_reply.started":"2023-11-13T08:58:19.506280Z","shell.execute_reply":"2023-11-13T08:58:19.519217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def scheduler(epoch, lr):\n    if epoch < 1:\n        return lr\n    return lr * tf.math.exp(-0.1)\ndef loss_fn(labels, targets):\n    labels_mask = tf.math.is_nan(labels)\n    labels = tf.where(labels_mask, tf.zeros_like(labels), labels)\n    mask_count = tf.math.reduce_sum(tf.where(labels_mask, tf.zeros_like(labels), tf.ones_like(labels)))\n    loss = tf.math.abs(labels - targets)\n    loss = tf.where(labels_mask, tf.zeros_like(loss), loss)\n    loss = tf.math.reduce_sum(loss)/mask_count\n    return loss","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:58:19.521216Z","iopub.execute_input":"2023-11-13T08:58:19.521477Z","iopub.status.idle":"2023-11-13T08:58:19.531705Z","shell.execute_reply.started":"2023-11-13T08:58:19.521447Z","shell.execute_reply":"2023-11-13T08:58:19.530858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect(tpu=\"local\") # \"local\" for 1VM TPU\n    print('Running on TPU ')#, tpu.cluster_spec().as_dict()['worker'])\nexcept ValueError:\n    tpu = None\nif tpu:\n    strategy = tf.distribute.TPUStrategy(tpu)\n    print(\"on TPU\")\n    print(\"REPLICAS: \", strategy.num_replicas_in_sync)\nelse:\n    strategy = tf.distribute.get_strategy()\n'''\nstrategy = tf.distribute.MirroredStrategy()\n'''\nmodel_list = []\nwith strategy.scope():\n    '''\n    kf = KFold(n_splits=N_Folds, shuffle=True, random_state=400)\n    for fold, (train_idx, valid_idx) in enumerate(kf.split(train_dt, target)):    \n        print(\"####################\",fold,\"/\",N_Folds)\n        learning_rate = 0.0005#1e-4\n        epsilon = 1e-9\n        #loss = tf.keras.losses.MeanSquaredError()\n        loss = custom_loss\n        metric = custom_metric\n        optimizer = tf.keras.optimizers.Adam(learning_rate=learning_rate, epsilon=epsilon)\n        #optimizer = tf.keras.optimizers.Adam(learning_rate=0.0005)\n        model_T = RnnModel(num_layers=6,d_model=198,num_heads=8,dff=500,vocab_size=5)\n        model_T.compile(optimizer=optimizer, loss=loss, metrics=[metric])\n\n        X_train, X_valid = train_dt[train_idx,:], train_dt[valid_idx,:]\n        y_train, y_valid = target[train_idx], target[valid_idx]\n        print(\"train shape:\",X_train.shape)\n        print(\"valid shape:\",X_valid.shape)\n        callback1 = tf.keras.callbacks.EarlyStopping(monitor='loss', patience=3)\n        callback2 = tf.keras.callbacks.LearningRateScheduler(scheduler)\n        model_T.fit(x=X_train, y=y_train,validation_data=(X_valid, y_valid),\n                      steps_per_epoch = 500,epochs=60,\n                      callbacks=[callback1,callback2])\n        #model_T.save(f'model_fold_{fold+fold}_{model}')\n        model_list.append(model_T)\n    '''\n    model_list = joblib.load(\"/kaggle/input/stanford-rrf-tensorflow-training-tpu/models.pkl2\") ","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:58:19.534905Z","iopub.execute_input":"2023-11-13T08:58:19.535257Z","iopub.status.idle":"2023-11-13T09:02:18.867053Z","shell.execute_reply.started":"2023-11-13T08:58:19.535231Z","shell.execute_reply":"2023-11-13T09:02:18.865808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# prediection","metadata":{}},{"cell_type":"markdown","source":"test_df = temp_df = pd.read_csv(test_sequences)#get_big_data(test_sequences)\ntest_df.head(1)","metadata":{"execution":{"iopub.status.busy":"2023-11-13T07:21:53.414694Z","iopub.execute_input":"2023-11-13T07:21:53.414985Z","iopub.status.idle":"2023-11-13T07:21:59.955417Z","shell.execute_reply.started":"2023-11-13T07:21:53.414960Z","shell.execute_reply":"2023-11-13T07:21:59.954374Z"}}},{"cell_type":"markdown","source":"test_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-13T07:21:59.958108Z","iopub.execute_input":"2023-11-13T07:21:59.958417Z","iopub.status.idle":"2023-11-13T07:21:59.963859Z","shell.execute_reply.started":"2023-11-13T07:21:59.958394Z","shell.execute_reply":"2023-11-13T07:21:59.962839Z"}}},{"cell_type":"markdown","source":"for i,j in zip(['A','G','U','C'], ['1','2','3','4']):\n    test_df['sequence'] = test_df['sequence'].str.replace(i,j)\ntest_df.head(1)","metadata":{"execution":{"iopub.status.busy":"2023-11-13T07:21:59.964925Z","iopub.execute_input":"2023-11-13T07:21:59.965197Z","iopub.status.idle":"2023-11-13T07:22:05.474417Z","shell.execute_reply.started":"2023-11-13T07:21:59.965175Z","shell.execute_reply":"2023-11-13T07:22:05.473379Z"}}},{"cell_type":"markdown","source":"test_df[\"sequence\"] = test_df[\"sequence\"].apply(create_train_col)\ntest_df.head(1)","metadata":{"execution":{"iopub.status.busy":"2023-11-13T07:22:05.475552Z","iopub.execute_input":"2023-11-13T07:22:05.475823Z","iopub.status.idle":"2023-11-13T07:24:03.858132Z","shell.execute_reply.started":"2023-11-13T07:22:05.475800Z","shell.execute_reply":"2023-11-13T07:24:03.857260Z"}}},{"cell_type":"markdown","source":"span = 59999\nn_loop = int(test_df.shape[0]/span)+1\n#g_i = n_loop-1","metadata":{"execution":{"iopub.status.busy":"2023-11-13T07:24:03.859310Z","iopub.execute_input":"2023-11-13T07:24:03.859594Z","iopub.status.idle":"2023-11-13T07:24:03.864240Z","shell.execute_reply.started":"2023-11-13T07:24:03.859569Z","shell.execute_reply":"2023-11-13T07:24:03.863450Z"}}},{"cell_type":"markdown","source":"g_id = 0\nfor g_i in range(n_loop):\n    print(\"####################\",g_i,\"/\",n_loop)\n    #prd_2A3 = model_2A3.predict(dt[span*g_i:span*(g_i+1),:])   \n    #prd_DMS = model_DMS.predict(dt[span*g_i:span*(g_i+1),:])\n    #prd_2A3 = model_2A3.predict(np.array(test_df.loc[span*g_i:span*(g_i+1)-1,\"sequence\"].to_list()).astype(int))   \n    #prd_DMS = model_DMS.predict(np.array(test_df.loc[span*g_i:span*(g_i+1)-1,\"sequence\"].to_list()).astype(int))    \n    pre_dt = np.array(test_df.loc[span*g_i:span*(g_i+1)-1,\"sequence\"].to_list()).astype(int)\n    #prd_DMS = 0\n    #prd_2A3 = 0\n    prd_all = 0\n    for i in range(len(model_list)):\n        prd_all += np.clip(model_list[i].predict(pre_dt),0,1)\n    #    prd_DMS += model_DMS_list[i].predict(pre_dt) \n    #    prd_2A3 += model_2A3_list[i].predict(pre_dt)  \n    #prd_DMS /= N_Folds\n    #prd_2A3 /= N_Folds  \n    #prd_DMS = np.clip(prd_DMS, 0, 1)\n    #prd_2A3 = np.clip(prd_2A3, 0, 1)\n    prd_all /= len(model_list) \n    #prd_all = model_all.predict(pre_dt)\n    #prd_all = np.clip(prd_all, 0, 1)     \n    dfs = []\n    for i in range(np.array(test_df.loc[span*g_i:span*(g_i+1)-1,\"sequence\"].to_list()).shape[0]):\n        id_min = test_df.loc[span*g_i+i,\"id_min\"]\n        id_max = test_df.loc[span*g_i+i,\"id_max\"]\n        len_ids = id_max+1-id_min\n        ids = list(range(g_id,g_id +len_ids))\n        g_id += len(ids)\n        df = pd.DataFrame({'id': ids,'reactivity_DMS_MaP': prd_all[i][:,1][:len(ids)],'reactivity_2A3_MaP': prd_all[i][:,0][:len(ids)]})\n        dfs.append(df)\n    pred_df = pd.concat(dfs)\n    pred_df = pred_df.set_index('id')   \n    if g_i == 0:\n        pred_df.to_csv('predict_submission_new.csv', header=pred_df.keys())\n    else:\n        pred_df.to_csv('predict_submission_new.csv', mode='a', header=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-13T07:34:26.815475Z","iopub.execute_input":"2023-11-13T07:34:26.816455Z"}}},{"cell_type":"code","source":"test_sequences_df = pd.read_csv('/kaggle/input/stanford-ribonanza-rna-folding/test_sequences.csv')\ntest_sequences_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-13T09:02:18.868307Z","iopub.execute_input":"2023-11-13T09:02:18.868596Z","iopub.status.idle":"2023-11-13T09:02:25.622589Z","shell.execute_reply.started":"2023-11-13T09:02:18.868570Z","shell.execute_reply":"2023-11-13T09:02:25.621635Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_sequences = test_sequences_df.sequence.to_numpy()\nencoding_dict = {'A':1, 'C': 2, 'G': 3, 'U': 4}\nencoding_dict","metadata":{"execution":{"iopub.status.busy":"2023-11-13T09:02:25.623889Z","iopub.execute_input":"2023-11-13T09:02:25.624206Z","iopub.status.idle":"2023-11-13T09:02:25.630624Z","shell.execute_reply.started":"2023-11-13T09:02:25.624178Z","shell.execute_reply":"2023-11-13T09:02:25.629782Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_len = 457 \ntest_sequences_encoded = []\nfor seq in test_sequences:\n    test_sequences_encoded.append(\n        np.concatenate([np.asarray([encoding_dict[x] for x in seq]), np.zeros((max_len - len(seq)))]).astype(np.float32))","metadata":{"execution":{"iopub.status.busy":"2023-11-13T09:02:25.631631Z","iopub.execute_input":"2023-11-13T09:02:25.631905Z","iopub.status.idle":"2023-11-13T09:03:08.610572Z","shell.execute_reply.started":"2023-11-13T09:02:25.631881Z","shell.execute_reply":"2023-11-13T09:03:08.609191Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ds = tf.data.Dataset.from_tensor_slices(test_sequences_encoded)\nbatch_size = 256\n#test_ds = test_ds.take(10000)\ntest_ds = test_ds.padded_batch(batch_size, padding_values=(0.0), padded_shapes=([max_len]), drop_remainder=False)\ntest_ds = test_ds.prefetch(tf.data.AUTOTUNE)\nbatch = next(iter(test_ds))\nbatch.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-13T09:04:41.220904Z","iopub.execute_input":"2023-11-13T09:04:41.221268Z","iopub.status.idle":"2023-11-13T09:06:04.879592Z","shell.execute_reply.started":"2023-11-13T09:04:41.221242Z","shell.execute_reply":"2023-11-13T09:06:04.878625Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = model_list[0]","metadata":{"execution":{"iopub.status.busy":"2023-11-13T09:06:10.020834Z","iopub.execute_input":"2023-11-13T09:06:10.021211Z","iopub.status.idle":"2023-11-13T09:06:10.025807Z","shell.execute_reply.started":"2023-11-13T09:06:10.021180Z","shell.execute_reply":"2023-11-13T09:06:10.024964Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = model.predict(test_ds)\npreds_processed = []\nfor i, pred in enumerate(preds):\n    preds_processed.append(pred[:len(test_sequences[i])])\nconcat_preds = np.concatenate(preds_processed)","metadata":{"execution":{"iopub.status.busy":"2023-11-13T09:06:14.115773Z","iopub.execute_input":"2023-11-13T09:06:14.116134Z","iopub.status.idle":"2023-11-13T09:11:23.898984Z","shell.execute_reply.started":"2023-11-13T09:06:14.116104Z","shell.execute_reply":"2023-11-13T09:11:23.897866Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({'id':np.arange(0, len(concat_preds), 1), 'reactivity_DMS_MaP':concat_preds[:,1], 'reactivity_2A3_MaP':concat_preds[:,0]})\nsubmission.to_csv('submission.csv', index=False)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-13T09:11:27.920460Z","iopub.execute_input":"2023-11-13T09:11:27.920816Z","iopub.status.idle":"2023-11-13T09:25:23.364941Z","shell.execute_reply.started":"2023-11-13T09:11:27.920788Z","shell.execute_reply":"2023-11-13T09:25:23.364073Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}