{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":51294,"databundleVersionId":6923401,"sourceType":"competition"},{"sourceId":6933839,"sourceType":"datasetVersion","datasetId":3981418}],"dockerImageVersionId":30589,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"https://www.kaggle.com/code/mirenaborisova/srrnaf-8/notebook","metadata":{}},{"cell_type":"markdown","source":"https://www.kaggle.com/code/pranshubahadur/esm2-rmdb-rna-dataset","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.simplefilter('ignore')\n\nimport pandas as pd\npd.set_option('display.max_columns', 30)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-07T13:36:16.199507Z","iopub.execute_input":"2023-12-07T13:36:16.200385Z","iopub.status.idle":"2023-12-07T13:36:17.116167Z","shell.execute_reply.started":"2023-12-07T13:36:16.200305Z","shell.execute_reply":"2023-12-07T13:36:17.115216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\n\nresolver = tf.distribute.cluster_resolver.TPUClusterResolver()\ntf.config.experimental_connect_to_cluster(resolver)\ntf.tpu.experimental.initialize_tpu_system(resolver)\nstrategy = tf.distribute.experimental.TPUStrategy(resolver)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T13:36:17.117564Z","iopub.execute_input":"2023-12-07T13:36:17.117893Z","iopub.status.idle":"2023-12-07T13:36:39.646790Z","shell.execute_reply.started":"2023-12-07T13:36:17.117864Z","shell.execute_reply":"2023-12-07T13:36:39.646000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rmdb = pd.read_csv('/kaggle/input/rmdb-rna-mapping-database-2023-data/rmdb_data.v1.3.0.csv')","metadata":{"execution":{"iopub.status.busy":"2023-12-07T13:36:39.647805Z","iopub.execute_input":"2023-12-07T13:36:39.648060Z","iopub.status.idle":"2023-12-07T13:36:51.045560Z","shell.execute_reply.started":"2023-12-07T13:36:39.648033Z","shell.execute_reply":"2023-12-07T13:36:51.044369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rmdb_SN_filter = rmdb[rmdb.SN_filter == 1]\nerror_feats = [feat for feat in rmdb_SN_filter.columns if 'error' in feat]\nrmdb_SN_filter_no_error = rmdb_SN_filter.drop(columns = error_feats)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T13:36:51.047359Z","iopub.execute_input":"2023-12-07T13:36:51.047646Z","iopub.status.idle":"2023-12-07T13:36:51.423948Z","shell.execute_reply.started":"2023-12-07T13:36:51.047619Z","shell.execute_reply":"2023-12-07T13:36:51.422955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# experiment_type = rmdb_SN_filter_no_error.experiment_type.unique()\n\n# rmdb_etype_2A3 = [experiment_type[0]]\n# rmdb_etype_DMS = (experiment_type[1:]).tolist()","metadata":{"execution":{"iopub.status.busy":"2023-12-07T13:36:51.425088Z","iopub.execute_input":"2023-12-07T13:36:51.425400Z","iopub.status.idle":"2023-12-07T13:36:51.429006Z","shell.execute_reply.started":"2023-12-07T13:36:51.425370Z","shell.execute_reply":"2023-12-07T13:36:51.428358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rmdb_etype_2A3 = ['1M7','NMIA','BzCN']\nrmdb_etype_DMS = ['BzCN_cotx', 'DMS_cotx', 'DMS_M2_seq', 'DMS']","metadata":{"execution":{"iopub.status.busy":"2023-12-07T13:36:51.429912Z","iopub.execute_input":"2023-12-07T13:36:51.430151Z","iopub.status.idle":"2023-12-07T13:36:51.441524Z","shell.execute_reply.started":"2023-12-07T13:36:51.430127Z","shell.execute_reply":"2023-12-07T13:36:51.440764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_quick = pd.read_csv('/kaggle/input/stanford-ribonanza-rna-folding/train_data_QUICK_START.csv')\ntrain_quick_2A3 = train_quick[train_quick.experiment_type == '2A3_MaP'].reset_index(drop=True)\ntrain_quick_DMS = train_quick[train_quick.experiment_type == 'DMS_MaP'].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T13:36:51.442405Z","iopub.execute_input":"2023-12-07T13:36:51.442663Z","iopub.status.idle":"2023-12-07T13:37:07.303525Z","shell.execute_reply.started":"2023-12-07T13:36:51.442637Z","shell.execute_reply":"2023-12-07T13:37:07.302482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rmdb_2A3 = rmdb_SN_filter_no_error[rmdb_SN_filter_no_error.experiment_type.isin(rmdb_etype_2A3)].reset_index(drop=True)\nrmdb_2A3 = pd.concat([train_quick_2A3, rmdb_2A3]).reset_index(drop=True)\n\nrmdb_DMS = rmdb_SN_filter_no_error[rmdb_SN_filter_no_error.experiment_type.isin(rmdb_etype_DMS)].reset_index(drop=True)\nrmdb_DMS = pd.concat([train_quick_DMS, rmdb_DMS]).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T13:37:07.304671Z","iopub.execute_input":"2023-12-07T13:37:07.304954Z","iopub.status.idle":"2023-12-07T13:37:10.439466Z","shell.execute_reply.started":"2023-12-07T13:37:07.304926Z","shell.execute_reply":"2023-12-07T13:37:10.438360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_2A3 = rmdb_2A3.sequence\nrmdb_2A3 = rmdb_2A3.filter(regex='reactivity_[0-9]')\nrmdb_2A3 = rmdb_2A3.fillna(0)\nX_2A3 = X_2A3.iloc[rmdb_2A3.index].reset_index(drop=True)\nY_2A3 = rmdb_2A3.reset_index(drop=True)\n\nX_DMS = rmdb_DMS.sequence\nrmdb_DMS = rmdb_DMS.filter(regex='reactivity_[0-9]')\nrmdb_DMS = rmdb_DMS.fillna(0)\nX_DMS = X_DMS.iloc[rmdb_DMS.index].reset_index(drop=True)\nY_DMS = rmdb_DMS.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T13:37:10.440511Z","iopub.execute_input":"2023-12-07T13:37:10.440774Z","iopub.status.idle":"2023-12-07T13:37:12.305698Z","shell.execute_reply.started":"2023-12-07T13:37:10.440748Z","shell.execute_reply":"2023-12-07T13:37:12.304531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\n\ndel train_quick\ndel train_quick_2A3\ndel train_quick_DMS\n\ndel rmdb\ndel rmdb_SN_filter\ndel error_feats\ndel rmdb_SN_filter_no_error\ndel rmdb_etype_2A3\ndel rmdb_etype_DMS\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-12-07T13:37:12.308497Z","iopub.execute_input":"2023-12-07T13:37:12.308781Z","iopub.status.idle":"2023-12-07T13:37:12.578967Z","shell.execute_reply.started":"2023-12-07T13:37:12.308753Z","shell.execute_reply":"2023-12-07T13:37:12.578046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras import Model\nfrom transformers import TFAutoModel\n\nclass RNA(Model):\n    \n    def __init__(self):\n        \n        super().__init__()\n        self.encoder = TFAutoModel.from_pretrained('AmelieSchreiber/esm2_t6_8M_UR50D_rna_binding_site_predictor')\n        self.dropout = tf.keras.layers.Dropout(0.2)\n        self.dense = tf.keras.layers.Dense(1)\n        \n    def call(self, x):\n        \n        x = self.encoder(x).last_hidden_state\n        \n        return tf.squeeze(self.dense(x), -1)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T13:37:12.579975Z","iopub.execute_input":"2023-12-07T13:37:12.580246Z","iopub.status.idle":"2023-12-07T13:37:35.702427Z","shell.execute_reply.started":"2023-12-07T13:37:12.580217Z","shell.execute_reply":"2023-12-07T13:37:35.701440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def loss(X, Y):\n    \n    X_mask = tf.math.is_nan(X)\n    X = tf.where(X_mask, tf.zeros_like(X), X)\n    sum_mask = tf.math.reduce_sum(tf.where(X_mask, tf.zeros_like(X), tf.ones_like(X)))\n    loss = tf.math.abs(X - Y)\n    loss = tf.where(X_mask, tf.zeros_like(loss), loss)\n    loss = tf.math.reduce_sum(loss) / (sum_mask if sum_mask != 0.0 else 1.0)\n    \n    return loss","metadata":{"execution":{"iopub.status.busy":"2023-12-07T13:37:35.703620Z","iopub.execute_input":"2023-12-07T13:37:35.704158Z","iopub.status.idle":"2023-12-07T13:37:35.709410Z","shell.execute_reply.started":"2023-12-07T13:37:35.704127Z","shell.execute_reply":"2023-12-07T13:37:35.708772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_vectorization = tf.keras.layers.TextVectorization(output_mode='int',\n                                                       ngrams=1,\n                                                       output_sequence_length=Y_DMS.shape[1],\n                                                       split='character',\n                                                       vocabulary=['a', 'c', 'g', 'u'])","metadata":{"execution":{"iopub.status.busy":"2023-12-07T13:37:35.710256Z","iopub.execute_input":"2023-12-07T13:37:35.710507Z","iopub.status.idle":"2023-12-07T13:37:35.747830Z","shell.execute_reply.started":"2023-12-07T13:37:35.710483Z","shell.execute_reply":"2023-12-07T13:37:35.747149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    \n    model_DMS = RNA()\n    model_DMS.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=5e-4), loss=loss)\n    \n    train_ds = tf.data.Dataset.from_tensor_slices((X_DMS.values, Y_DMS.values)) \\\n        .batch(128) \\\n        .map(lambda x, y: (text_vectorization(x), tf.clip_by_value(y, 0, 1)))\n    train_ds = train_ds.shuffle(train_ds.cardinality())\n    train_data = train_ds.take(int(len(train_ds) * 0.8))\n    validation_data = train_ds.skip(int(len(train_ds) * 0.8)).take(int(len(train_ds) * 0.2))\n\n    model_DMS.fit(train_data, validation_data=validation_data, epochs=25, batch_size=128)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T13:37:35.748723Z","iopub.execute_input":"2023-12-07T13:37:35.748980Z","iopub.status.idle":"2023-12-07T14:50:09.444831Z","shell.execute_reply.started":"2023-12-07T13:37:35.748955Z","shell.execute_reply":"2023-12-07T14:50:09.443639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_vectorization_preds = tf.keras.layers.TextVectorization(output_mode='int',\n                                                             ngrams=1,\n                                                             output_sequence_length=457,\n                                                             split='character',\n                                                             vocabulary=['a', 'c', 'g', 'u'])","metadata":{"execution":{"iopub.status.busy":"2023-12-07T14:50:09.446915Z","iopub.execute_input":"2023-12-07T14:50:09.447592Z","iopub.status.idle":"2023-12-07T14:50:09.461334Z","shell.execute_reply.started":"2023-12-07T14:50:09.447553Z","shell.execute_reply":"2023-12-07T14:50:09.460360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/stanford-ribonanza-rna-folding/test_sequences.csv')","metadata":{"execution":{"iopub.status.busy":"2023-12-07T14:50:09.462441Z","iopub.execute_input":"2023-12-07T14:50:09.462910Z","iopub.status.idle":"2023-12-07T14:50:16.347727Z","shell.execute_reply.started":"2023-12-07T14:50:09.462879Z","shell.execute_reply":"2023-12-07T14:50:16.346622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    \n    preds_DMS = model_DMS.predict(tf.data.Dataset.from_tensor_slices((test.sequence)) \\\n        .batch(128) \\\n        .map(lambda x: text_vectorization_preds(x)))","metadata":{"execution":{"iopub.status.busy":"2023-12-07T14:50:16.348969Z","iopub.execute_input":"2023-12-07T14:50:16.349275Z","iopub.status.idle":"2023-12-07T14:57:41.968462Z","shell.execute_reply.started":"2023-12-07T14:50:16.349228Z","shell.execute_reply":"2023-12-07T14:57:41.967224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_DMS = pd.DataFrame(preds_DMS)\n# df_DMS.to_csv('preds_DMS.csv')","metadata":{"execution":{"iopub.status.busy":"2023-12-07T14:57:41.973297Z","iopub.execute_input":"2023-12-07T14:57:41.973586Z","iopub.status.idle":"2023-12-07T14:57:41.977463Z","shell.execute_reply.started":"2023-12-07T14:57:41.973559Z","shell.execute_reply":"2023-12-07T14:57:41.976498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    \n    model_2A3 = RNA()\n    model_2A3.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=5e-4), loss=loss)\n    \n    train_ds = tf.data.Dataset.from_tensor_slices((X_2A3.values, Y_2A3.values)) \\\n        .batch(128) \\\n        .map(lambda x, y: (text_vectorization(x), tf.clip_by_value(y, 0, 1)))\n    train_ds = train_ds.shuffle(train_ds.cardinality())\n    train_data = train_ds.take(int(len(train_ds) * 0.8))\n    validation_data = train_ds.skip(int(len(train_ds) * 0.8)).take(int(len(train_ds) * 0.2))\n\n    model_2A3.fit(train_data, validation_data=validation_data, epochs=25, batch_size=128)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T14:57:41.978685Z","iopub.execute_input":"2023-12-07T14:57:41.978953Z","iopub.status.idle":"2023-12-07T16:22:31.336360Z","shell.execute_reply.started":"2023-12-07T14:57:41.978927Z","shell.execute_reply":"2023-12-07T16:22:31.335246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    \n    preds_2A3 = model_2A3.predict(tf.data.Dataset.from_tensor_slices((test.sequence)) \\\n        .batch(128) \\\n        .map(lambda x: text_vectorization_preds(x)))","metadata":{"execution":{"iopub.status.busy":"2023-12-07T16:22:31.338377Z","iopub.execute_input":"2023-12-07T16:22:31.338774Z","iopub.status.idle":"2023-12-07T16:29:55.089609Z","shell.execute_reply.started":"2023-12-07T16:22:31.338740Z","shell.execute_reply":"2023-12-07T16:29:55.088531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_2A3 = pd.DataFrame(preds_2A3)\n# df_2A3.to_csv('preds_2A3.csv')","metadata":{"execution":{"iopub.status.busy":"2023-12-07T16:29:55.092769Z","iopub.execute_input":"2023-12-07T16:29:55.093071Z","iopub.status.idle":"2023-12-07T16:29:55.097416Z","shell.execute_reply.started":"2023-12-07T16:29:55.093040Z","shell.execute_reply":"2023-12-07T16:29:55.096477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_seq_lengths = test.sequence.str.len()","metadata":{"execution":{"iopub.status.busy":"2023-12-07T16:29:55.098403Z","iopub.execute_input":"2023-12-07T16:29:55.098689Z","iopub.status.idle":"2023-12-07T16:29:55.559129Z","shell.execute_reply.started":"2023-12-07T16:29:55.098662Z","shell.execute_reply":"2023-12-07T16:29:55.558189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\npreds_2A3_bylength = []\n\nfor i in range(test_seq_lengths.size):\n    \n    x = np.reshape(preds_2A3[i, :test_seq_lengths[i]], (-1, 1))\n    x = np.clip(x, 0, 1)\n    \n    preds_2A3_bylength.append(x)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T16:29:55.560077Z","iopub.execute_input":"2023-12-07T16:29:55.560327Z","iopub.status.idle":"2023-12-07T16:30:07.898324Z","shell.execute_reply.started":"2023-12-07T16:29:55.560288Z","shell.execute_reply":"2023-12-07T16:30:07.897219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_2A3 = np.concatenate(preds_2A3_bylength, 0)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T16:30:07.899469Z","iopub.execute_input":"2023-12-07T16:30:07.899765Z","iopub.status.idle":"2023-12-07T16:30:08.866700Z","shell.execute_reply.started":"2023-12-07T16:30:07.899735Z","shell.execute_reply":"2023-12-07T16:30:08.865578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_DMS_bylength = []\n\nfor i in range(test_seq_lengths.size):\n    \n    x = np.reshape(preds_DMS[i, :test_seq_lengths[i]], (-1, 1))\n    x = np.clip(x, 0, 1)\n    \n    preds_DMS_bylength.append(x)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T16:30:08.867836Z","iopub.execute_input":"2023-12-07T16:30:08.868132Z","iopub.status.idle":"2023-12-07T16:30:21.633571Z","shell.execute_reply.started":"2023-12-07T16:30:08.868101Z","shell.execute_reply":"2023-12-07T16:30:21.632574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_DMS = np.concatenate(preds_DMS_bylength, 0)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T16:30:21.634632Z","iopub.execute_input":"2023-12-07T16:30:21.634912Z","iopub.status.idle":"2023-12-07T16:30:22.531355Z","shell.execute_reply.started":"2023-12-07T16:30:21.634882Z","shell.execute_reply":"2023-12-07T16:30:22.530262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_DMS.shape[0]","metadata":{"execution":{"iopub.status.busy":"2023-12-07T16:30:22.532636Z","iopub.execute_input":"2023-12-07T16:30:22.533021Z","iopub.status.idle":"2023-12-07T16:30:22.538712Z","shell.execute_reply.started":"2023-12-07T16:30:22.532970Z","shell.execute_reply":"2023-12-07T16:30:22.537798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({'id': np.arange(0, preds_DMS.shape[0], 1),\n                           'reactivity_DMS_MaP': preds_DMS[:, 0],\n                           'reactivity_2A3_MaP': preds_2A3[:, 0]})","metadata":{"execution":{"iopub.status.busy":"2023-12-07T16:30:22.543907Z","iopub.execute_input":"2023-12-07T16:30:22.544151Z","iopub.status.idle":"2023-12-07T16:30:24.545327Z","shell.execute_reply.started":"2023-12-07T16:30:22.544126Z","shell.execute_reply":"2023-12-07T16:30:24.544367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-07T16:30:24.546386Z","iopub.execute_input":"2023-12-07T16:30:24.546923Z","iopub.status.idle":"2023-12-07T16:30:24.560800Z","shell.execute_reply.started":"2023-12-07T16:30:24.546881Z","shell.execute_reply":"2023-12-07T16:30:24.559800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pyarrow","metadata":{"execution":{"iopub.status.busy":"2023-12-07T16:30:24.561860Z","iopub.execute_input":"2023-12-07T16:30:24.562247Z","iopub.status.idle":"2023-12-07T16:30:33.347682Z","shell.execute_reply.started":"2023-12-07T16:30:24.562202Z","shell.execute_reply":"2023-12-07T16:30:33.346529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_parquet('submission.parquet', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T16:30:33.349100Z","iopub.execute_input":"2023-12-07T16:30:33.349542Z","iopub.status.idle":"2023-12-07T16:31:30.455416Z","shell.execute_reply.started":"2023-12-07T16:30:33.349487Z","shell.execute_reply":"2023-12-07T16:31:30.454301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}