{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":51294,"databundleVersionId":6923401,"sourceType":"competition"},{"sourceId":6933839,"sourceType":"datasetVersion","datasetId":3981418}],"dockerImageVersionId":30580,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import tensorflow as tf\n\n\nresolver = tf.distribute.cluster_resolver.TPUClusterResolver()\ntf.config.experimental_connect_to_cluster(resolver)\ntf.tpu.experimental.initialize_tpu_system(resolver)\nprint(\"All devices: \", tf.config.list_logical_devices('TPU'))\nstrategy = tf.distribute.experimental.TPUStrategy(resolver)\n\n#strategy = tf.distribute.MirroredStrategy()\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-04T18:40:47.829262Z","iopub.execute_input":"2023-12-04T18:40:47.829575Z","iopub.status.idle":"2023-12-04T18:40:56.117264Z","shell.execute_reply.started":"2023-12-04T18:40:47.829547Z","shell.execute_reply":"2023-12-04T18:40:56.116354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pandas import read_csv, concat\ndataset = read_csv('/kaggle/input/stanford-ribonanza-rna-folding/train_data_QUICK_START.csv', chunksize=100000)\ndataset = concat(dataset)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:40:56.118777Z","iopub.execute_input":"2023-12-04T18:40:56.119076Z","iopub.status.idle":"2023-12-04T18:41:13.179745Z","shell.execute_reply.started":"2023-12-04T18:40:56.119044Z","shell.execute_reply":"2023-12-04T18:41:13.178493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.sequence.size","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:41:13.180764Z","iopub.execute_input":"2023-12-04T18:41:13.181062Z","iopub.status.idle":"2023-12-04T18:41:13.188934Z","shell.execute_reply.started":"2023-12-04T18:41:13.181032Z","shell.execute_reply":"2023-12-04T18:41:13.188257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dms, a3 = dataset[dataset.experiment_type=='DMS_MaP'].reset_index(drop=True), dataset[dataset.experiment_type=='2A3_MaP'].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:41:13.190660Z","iopub.execute_input":"2023-12-04T18:41:13.190943Z","iopub.status.idle":"2023-12-04T18:41:17.853103Z","shell.execute_reply.started":"2023-12-04T18:41:13.190916Z","shell.execute_reply":"2023-12-04T18:41:17.852156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rmdb = read_csv('/kaggle/input/rmdb-rna-mapping-database-2023-data/rmdb_data.v1.3.0.csv')","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:41:17.854007Z","iopub.execute_input":"2023-12-04T18:41:17.854245Z","iopub.status.idle":"2023-12-04T18:41:34.173513Z","shell.execute_reply.started":"2023-12-04T18:41:17.854221Z","shell.execute_reply":"2023-12-04T18:41:34.172514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a3_exp = ['1M7','NMIA','BzCN']\na3 = concat([a3, rmdb.query('experiment_type in @a3_exp and SN_filter==1')], axis=0).reset_index(drop=True)\n               \ndms_exp = ['BzCN_cotx', 'DMS_cotx', 'DMS_M2_seq', 'DMS']\ndms = concat([dms, rmdb.query('(experiment_type in @dms_exp) and SN_filter==1')], axis=0).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:41:34.174499Z","iopub.execute_input":"2023-12-04T18:41:34.174757Z","iopub.status.idle":"2023-12-04T18:41:34.178440Z","shell.execute_reply.started":"2023-12-04T18:41:34.174725Z","shell.execute_reply":"2023-12-04T18:41:34.177795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dms.experiment_type.unique()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:41:34.179193Z","iopub.execute_input":"2023-12-04T18:41:34.179412Z","iopub.status.idle":"2023-12-04T18:41:34.201942Z","shell.execute_reply.started":"2023-12-04T18:41:34.179389Z","shell.execute_reply":"2023-12-04T18:41:34.201255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_a3 = a3.sequence\ny_a3 = a3.filter(regex='reactivity_[0-9]')#.values.where((a3.filter(regex='reactivity_error*')<1).values, a3.filter(regex='reactivity_[0-9]').values, np.nan )\ny_a3 = y_a3.query('@y_a3.fillna(0).sum(1)!=0')\nx_a3 = x_a3.iloc[y_a3.index].reset_index(drop=True)\ny_a3 = y_a3.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:41:34.202760Z","iopub.execute_input":"2023-12-04T18:41:34.203029Z","iopub.status.idle":"2023-12-04T18:41:34.793115Z","shell.execute_reply.started":"2023-12-04T18:41:34.203001Z","shell.execute_reply":"2023-12-04T18:41:34.792144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_dms = dms.sequence\ny_dms = dms.filter(regex='reactivity_[0-9]')#.values(.where(dms.filter(regex='reactivity_error*')<1)\ny_dms = y_dms.query('@y_dms.fillna(0).sum(1)!=0')\nx_dms = x_dms.iloc[y_dms.index].reset_index(drop=True)\ny_dms = y_dms.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:41:34.794158Z","iopub.execute_input":"2023-12-04T18:41:34.794407Z","iopub.status.idle":"2023-12-04T18:41:35.382619Z","shell.execute_reply.started":"2023-12-04T18:41:34.794382Z","shell.execute_reply":"2023-12-04T18:41:35.381639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!pip install keras_nlp","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:41:35.385482Z","iopub.execute_input":"2023-12-04T18:41:35.385856Z","iopub.status.idle":"2023-12-04T18:41:35.389081Z","shell.execute_reply.started":"2023-12-04T18:41:35.385825Z","shell.execute_reply":"2023-12-04T18:41:35.388390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras import Model\n#from keras_nlp.layers import TransformerDecoder, TransformerEncoder, PositionEmbedding\nfrom tensorflow.keras import Sequential\nfrom itertools import repeat\nfrom tensorflow.keras.layers import Dropout\nfrom transformers import TFAutoModel\n\nclass RNA(Model):\n    def __init__(self, hidden_dim):\n        super().__init__()\n        self.encoder = TFAutoModel.from_pretrained('AmelieSchreiber/esm2_t6_8M_UR50D_rna_binding_site_predictor')#Sequential([*repeat(TransformerEncoder(hidden_dim*4, 6, activation='gelu', dropout=0.1, normalize_first=True), 12)])\n        self.dp = Dropout(0.2)\n        self.fc = tf.keras.layers.Dense(1)\n    def call(self, x):\n        x = self.encoder(x).last_hidden_state\n        return tf.squeeze(self.fc(x), -1)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:41:35.389950Z","iopub.execute_input":"2023-12-04T18:41:35.390206Z","iopub.status.idle":"2023-12-04T18:41:58.981928Z","shell.execute_reply.started":"2023-12-04T18:41:35.390181Z","shell.execute_reply":"2023-12-04T18:41:58.980848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.layers import TextVectorization\nimport tensorflow as tf\n\ntokenizer = TextVectorization(\n    output_mode='int',\n    ngrams=1,\n    output_sequence_length=433,\n    split='character',\n    vocabulary=['a', 'c', 'g', 'u']\n)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:41:58.983072Z","iopub.execute_input":"2023-12-04T18:41:58.983655Z","iopub.status.idle":"2023-12-04T18:41:59.011040Z","shell.execute_reply.started":"2023-12-04T18:41:58.983622Z","shell.execute_reply":"2023-12-04T18:41:59.010111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\n\ndef loss_fn(labels, targets):\n    labels_mask = tf.math.is_nan(labels)\n    labels = tf.where(labels_mask, tf.zeros_like(labels), labels)\n    mask_count = tf.math.reduce_sum(tf.where(labels_mask, tf.zeros_like(labels), tf.ones_like(labels)))\n    loss = tf.math.abs(labels - targets)\n    loss = tf.where(labels_mask, tf.zeros_like(loss), loss)\n    loss = tf.math.reduce_sum(loss)/(mask_count if mask_count != 0.0 else 1.0)\n    return loss","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:41:59.012116Z","iopub.execute_input":"2023-12-04T18:41:59.012366Z","iopub.status.idle":"2023-12-04T18:41:59.018063Z","shell.execute_reply.started":"2023-12-04T18:41:59.012342Z","shell.execute_reply":"2023-12-04T18:41:59.017262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport tensorflow as tf\nfrom tensorflow.keras import Model, Sequential\nfrom tensorflow.keras.layers import Input, Embedding, Dense, ReLU, Flatten, Softmax, SimpleRNN, LSTM, MultiHeadAttention\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Layer, LayerNormalization, Dense, SimpleRNNCell, RNN, LSTM, Bidirectional, LSTMCell\nfrom sklearn.model_selection import KFold\nimport numpy as np\n\nwith strategy.scope():\n    model_a3 = RNA(192)\n    model_a3.compile(\n                optimizer=tf.keras.optimizers.Adam(learning_rate=5e-4),\n                loss=loss_fn,\n            )\n\n    train_dataset = tf.data.Dataset.from_tensor_slices((x_a3.values, y_a3.values)).batch(128).map(lambda x, y: (tokenizer(x), tf.clip_by_value(y, 0, 1)))\n    train_dataset = train_dataset.shuffle(train_dataset.cardinality())\n    train_split = train_dataset.take(int(len(train_dataset)*0.8))\n    val_split = train_dataset.skip(int(len(train_dataset)*0.8)).take(int(len(train_dataset)*0.2))\n    model_a3.fit(train_split, validation_data=val_split, epochs=50, batch_size=128)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:41:59.019055Z","iopub.execute_input":"2023-12-04T18:41:59.019312Z","iopub.status.idle":"2023-12-04T18:48:35.202040Z","shell.execute_reply.started":"2023-12-04T18:41:59.019287Z","shell.execute_reply":"2023-12-04T18:48:35.200357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport tensorflow as tf\nfrom tensorflow.keras import Model, Sequential\nfrom tensorflow.keras.layers import Input, Embedding, Dense, ReLU, Flatten, Softmax, SimpleRNN, LSTM, MultiHeadAttention\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Layer, LayerNormalization, Dense, SimpleRNNCell, RNN, LSTM, Bidirectional, LSTMCell\nfrom sklearn.model_selection import KFold\nimport numpy as np\n\nwith strategy.scope():\n    model_dms = RNA(192)\n    model_dms.compile(\n                optimizer=tf.keras.optimizers.Adam(learning_rate=5e-4),\n                loss=loss_fn,\n            )\n    train_dataset = tf.data.Dataset.from_tensor_slices((x_dms.values, y_dms.values)).batch(128).map(lambda x, y: (tokenizer(x), tf.clip_by_value(y, 0, 1)))\n    train_dataset = train_dataset.shuffle(train_dataset.cardinality())\n\n    train_split = train_dataset.take(int(len(train_dataset)*0.8))\n    val_split = train_dataset.skip(int(len(train_dataset)*0.8)).take(int(len(train_dataset)*0.2))\n    model_dms.fit(train_split, validation_data=val_split, epochs=50, batch_size=128)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:48:35.202909Z","iopub.status.idle":"2023-12-04T18:48:35.203223Z","shell.execute_reply.started":"2023-12-04T18:48:35.203074Z","shell.execute_reply":"2023-12-04T18:48:35.203089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.layers import TextVectorization\nimport tensorflow as tf\n\ntokenizer = TextVectorization(\n    output_mode='int',\n    ngrams=1,\n    output_sequence_length=457,\n    split='character',\n    vocabulary=['a', 'c', 'g', 'u']\n)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:48:35.204623Z","iopub.status.idle":"2023-12-04T18:48:35.204953Z","shell.execute_reply.started":"2023-12-04T18:48:35.204784Z","shell.execute_reply":"2023-12-04T18:48:35.204800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = read_csv('/kaggle/input/stanford-ribonanza-rna-folding/test_sequences.csv')","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:48:35.205868Z","iopub.status.idle":"2023-12-04T18:48:35.206177Z","shell.execute_reply.started":"2023-12-04T18:48:35.206034Z","shell.execute_reply":"2023-12-04T18:48:35.206048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    predictions_a3 = model_a3.predict(tf.data.Dataset.from_tensor_slices((test.sequence)).batch(128).map(lambda x: tokenizer(x)))\n","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:48:35.207314Z","iopub.status.idle":"2023-12-04T18:48:35.207595Z","shell.execute_reply.started":"2023-12-04T18:48:35.207454Z","shell.execute_reply":"2023-12-04T18:48:35.207468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    predictions_dms = model_dms.predict(tf.data.Dataset.from_tensor_slices((test.sequence)).batch(128).map(lambda x: tokenizer(x)))\n","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:48:35.208604Z","iopub.status.idle":"2023-12-04T18:48:35.208948Z","shell.execute_reply.started":"2023-12-04T18:48:35.208771Z","shell.execute_reply":"2023-12-04T18:48:35.208788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lens = test.sequence.str.len()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:48:35.210172Z","iopub.status.idle":"2023-12-04T18:48:35.210465Z","shell.execute_reply.started":"2023-12-04T18:48:35.210316Z","shell.execute_reply":"2023-12-04T18:48:35.210333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy\npredictions_a3_by_length = []\nfor i in range(lens.size):\n    x = numpy.reshape(predictions_a3[i, :lens[i]], (-1, 1))\n    x = numpy.clip(x, 0, 1)\n    #x[:26] = 0\n    #x[:-21] = 0\n    predictions_a3_by_length.append(x)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:48:35.211454Z","iopub.status.idle":"2023-12-04T18:48:35.211726Z","shell.execute_reply.started":"2023-12-04T18:48:35.211590Z","shell.execute_reply":"2023-12-04T18:48:35.211604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_a3 = numpy.concatenate(predictions_a3_by_length, 0)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:48:35.212672Z","iopub.status.idle":"2023-12-04T18:48:35.212976Z","shell.execute_reply.started":"2023-12-04T18:48:35.212810Z","shell.execute_reply":"2023-12-04T18:48:35.212824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy\npredictions_dms_by_length = []\nfor i in range(lens.size):\n    x = numpy.reshape(predictions_dms[i, :lens[i]], (-1, 1))\n    x = numpy.clip(x, 0, 1)\n    #x[:26] = 0\n    #x[:-21] = 0\n    predictions_dms_by_length.append(x)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:48:35.213691Z","iopub.status.idle":"2023-12-04T18:48:35.213997Z","shell.execute_reply.started":"2023-12-04T18:48:35.213831Z","shell.execute_reply":"2023-12-04T18:48:35.213845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_dms = numpy.concatenate(predictions_dms_by_length, 0)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:48:35.215010Z","iopub.status.idle":"2023-12-04T18:48:35.215295Z","shell.execute_reply.started":"2023-12-04T18:48:35.215150Z","shell.execute_reply":"2023-12-04T18:48:35.215164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_dms.shape","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:48:35.216389Z","iopub.status.idle":"2023-12-04T18:48:35.216667Z","shell.execute_reply.started":"2023-12-04T18:48:35.216530Z","shell.execute_reply":"2023-12-04T18:48:35.216544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pandas import DataFrame\nsubmission = DataFrame({'id':np.arange(0, 269796671, 1), 'reactivity_DMS_MaP': predictions_dms[:, 0], 'reactivity_2A3_MaP': predictions_a3[:, 0]})","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:48:35.217824Z","iopub.status.idle":"2023-12-04T18:48:35.218168Z","shell.execute_reply.started":"2023-12-04T18:48:35.218004Z","shell.execute_reply":"2023-12-04T18:48:35.218021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.reactivity_2A3_MaP.describe()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:48:35.219528Z","iopub.status.idle":"2023-12-04T18:48:35.219892Z","shell.execute_reply.started":"2023-12-04T18:48:35.219710Z","shell.execute_reply":"2023-12-04T18:48:35.219728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.reactivity_DMS_MaP.describe()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:48:35.221002Z","iopub.status.idle":"2023-12-04T18:48:35.221295Z","shell.execute_reply.started":"2023-12-04T18:48:35.221148Z","shell.execute_reply":"2023-12-04T18:48:35.221163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pyarrow","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:48:35.222303Z","iopub.status.idle":"2023-12-04T18:48:35.222601Z","shell.execute_reply.started":"2023-12-04T18:48:35.222447Z","shell.execute_reply":"2023-12-04T18:48:35.222462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:48:35.223582Z","iopub.status.idle":"2023-12-04T18:48:35.223875Z","shell.execute_reply.started":"2023-12-04T18:48:35.223730Z","shell.execute_reply":"2023-12-04T18:48:35.223745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_parquet('submission.parquet', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:48:35.224905Z","iopub.status.idle":"2023-12-04T18:48:35.225199Z","shell.execute_reply.started":"2023-12-04T18:48:35.225056Z","shell.execute_reply":"2023-12-04T18:48:35.225070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}