{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# General\nimport glob\nimport os\nimport pandas as pd\nimport numpy as np\n# import pandas_profiling as pp\n#!pip install  missingno\n# Plotting\nimport matplotlib.pyplot as plt\n\nimport seaborn as sns\nimport plotly.express as px\nimport missingno as msno\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom tensorflow import keras","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-27T10:35:04.779743Z","iopub.execute_input":"2023-10-27T10:35:04.780364Z","iopub.status.idle":"2023-10-27T10:35:17.382837Z","shell.execute_reply.started":"2023-10-27T10:35:04.780317Z","shell.execute_reply":"2023-10-27T10:35:17.381880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_parquet('/kaggle/input/rna-competition/train.parquet')\nprint(train.shape)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:35:18.903309Z","iopub.execute_input":"2023-10-27T10:35:18.904063Z","iopub.status.idle":"2023-10-27T10:35:32.341182Z","shell.execute_reply.started":"2023-10-27T10:35:18.904026Z","shell.execute_reply":"2023-10-27T10:35:32.339902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = (train['SN_filter'].values > 0) \ntrain = train.loc[m].reset_index(drop=True)\nn = (train['reads'].values > 100) \ntrain = train.loc[n].reset_index(drop=True)\ntrain = train.replace (np.nan, 0)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:35:32.343791Z","iopub.execute_input":"2023-10-27T10:35:32.344280Z","iopub.status.idle":"2023-10-27T10:35:35.475392Z","shell.execute_reply.started":"2023-10-27T10:35:32.344238Z","shell.execute_reply":"2023-10-27T10:35:35.474258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['sequence'] = train['sequence'].drop_duplicates()","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:35:35.476930Z","iopub.execute_input":"2023-10-27T10:35:35.477329Z","iopub.status.idle":"2023-10-27T10:35:35.614105Z","shell.execute_reply.started":"2023-10-27T10:35:35.477288Z","shell.execute_reply":"2023-10-27T10:35:35.612895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.dropna()","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:35:35.617137Z","iopub.execute_input":"2023-10-27T10:35:35.617577Z","iopub.status.idle":"2023-10-27T10:35:36.226267Z","shell.execute_reply.started":"2023-10-27T10:35:35.617538Z","shell.execute_reply":"2023-10-27T10:35:36.225097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:35:36.228135Z","iopub.execute_input":"2023-10-27T10:35:36.228559Z","iopub.status.idle":"2023-10-27T10:35:36.236137Z","shell.execute_reply.started":"2023-10-27T10:35:36.228521Z","shell.execute_reply":"2023-10-27T10:35:36.234941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i,j in zip(['A','G','U','C'], ['1','2','3','4']):\n    train['sequence'] = train['sequence'].str.replace(i,j)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:35:36.237748Z","iopub.execute_input":"2023-10-27T10:35:36.238295Z","iopub.status.idle":"2023-10-27T10:35:37.565875Z","shell.execute_reply.started":"2023-10-27T10:35:36.238256Z","shell.execute_reply":"2023-10-27T10:35:37.564965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:35:37.567130Z","iopub.execute_input":"2023-10-27T10:35:37.567461Z","iopub.status.idle":"2023-10-27T10:35:37.650909Z","shell.execute_reply.started":"2023-10-27T10:35:37.567434Z","shell.execute_reply":"2023-10-27T10:35:37.649876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"react_list = []\nreact_err_list = []\nfor i,k in enumerate(train.keys()):\n    if 'reactivity' in k and 'error' not in k:\n        react_list.append(i)\n    elif 'reactivity_error' in k:\n        react_err_list.append(i)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:35:37.652276Z","iopub.execute_input":"2023-10-27T10:35:37.652679Z","iopub.status.idle":"2023-10-27T10:35:37.659077Z","shell.execute_reply.started":"2023-10-27T10:35:37.652643Z","shell.execute_reply":"2023-10-27T10:35:37.658028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs_length = 457\ninputs = []\ntargets = []\n\n# fill list with type of experiment (DMS_MaP or 2A3_MaP)\nexp_type = []\nexp_type_uniques = ['DMS_MaP', '2A3_MaP']\n\n# loop through dataframe\nfor i, row in train.iterrows():\n\n    # get reactivities \n    r = np.array(row[react_list].values).astype('float')\n\n    # get sequence\n    seq = row.sequence#.replace('A', '1').replace('G', '2').replace('U', '3').replace('C', '4')\n    seq = np.array([*seq]).astype('float')\n    \n    # store input values\n    #input = np.zeros(len(r)).astype('int')\n    input = np.zeros(inputs_length).astype('float')\n    input[:len(seq)] = seq\n\n    # store target values\n    target = np.zeros(inputs_length).astype('float')\n    target[:len(r)] = np.nan_to_num(r).astype('float')\n   # wt = np.where(target)[0]\n    #if len(wt)>0:\n       # target = np.roll(target, -wt[0])\n\n    # clip target values between 0 and 1\n    #target[np.where(target<0)] = 0\n    #target[np.where(target>1)] = 1\n\n    # append values\n    exp_type.append(row.experiment_type)\n    inputs.append(input)\n    targets.append(target)\n\n# convert lists to numpy arrays\nexp_type = np.array(exp_type)\ninputs = np.array(inputs)\ntargets = np.array(targets)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:35:37.660524Z","iopub.execute_input":"2023-10-27T10:35:37.660855Z","iopub.status.idle":"2023-10-27T10:38:08.776035Z","shell.execute_reply.started":"2023-10-27T10:35:37.660796Z","shell.execute_reply":"2023-10-27T10:38:08.774789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:38:08.782542Z","iopub.execute_input":"2023-10-27T10:38:08.782923Z","iopub.status.idle":"2023-10-27T10:38:08.790929Z","shell.execute_reply.started":"2023-10-27T10:38:08.782893Z","shell.execute_reply":"2023-10-27T10:38:08.789801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = np.clip(targets,0,1)\nprint(targets.shape)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:38:08.792276Z","iopub.execute_input":"2023-10-27T10:38:08.792616Z","iopub.status.idle":"2023-10-27T10:38:09.016730Z","shell.execute_reply.started":"2023-10-27T10:38:08.792588Z","shell.execute_reply":"2023-10-27T10:38:09.015581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"where_DMS = np.where(exp_type == exp_type_uniques[0])[0]\nif len(where_DMS) > 0:\n    X_DMS=inputs[where_DMS]\n    y_DMS=targets[where_DMS]\n","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:38:09.017960Z","iopub.execute_input":"2023-10-27T10:38:09.018271Z","iopub.status.idle":"2023-10-27T10:38:09.089230Z","shell.execute_reply.started":"2023-10-27T10:38:09.018245Z","shell.execute_reply":"2023-10-27T10:38:09.088333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_DMS, X_test_DMS, y_train_DMS, y_test_DMS = train_test_split(X_DMS, y_DMS, test_size=0.1, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:38:09.090417Z","iopub.execute_input":"2023-10-27T10:38:09.090723Z","iopub.status.idle":"2023-10-27T10:38:09.170899Z","shell.execute_reply.started":"2023-10-27T10:38:09.090697Z","shell.execute_reply":"2023-10-27T10:38:09.170046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nnp.set_printoptions(suppress=True)\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom scipy.io import arff\nfrom sklearn.model_selection import train_test_split\nimport matplotlib\nmatplotlib.rcParams[\"figure.figsize\"] = (6, 4)\nplt.style.use(\"ggplot\")\nimport tensorflow as tf\nfrom tensorflow.keras.models import Model, Sequential\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom tensorflow.keras.metrics import mae\nfrom tensorflow.keras import layers\nfrom tensorflow import keras\nfrom sklearn.metrics import accuracy_score, recall_score, precision_score, confusion_matrix, f1_score, classification_report\nimport os","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:38:09.172261Z","iopub.execute_input":"2023-10-27T10:38:09.172684Z","iopub.status.idle":"2023-10-27T10:38:09.216041Z","shell.execute_reply.started":"2023-10-27T10:38:09.172647Z","shell.execute_reply":"2023-10-27T10:38:09.215240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Building CNN Autoencoder Model\n\nAbout Autoencoders: Learning Efficient Data Representations Autoencoders are a class of neural network architectures commonly used in unsupervised machine learning and deep learning tasks. Their primary purpose is to discover and learn efficient representations of data by encoding it into a lower-dimensional latent space and subsequently decoding it back to its original form. Autoencoders play a crucial role in various applications, such as dimensionality reduction, data denoising, anomaly detection, and generative modeling.\n\nThe core components of an autoencoder consist of an encoder and a decoder. The encoder maps input data to the latent space, while the decoder reconstructs the data from its encoded representation. During training, autoencoders aim to minimize the reconstruction error between the input and the decoded output, which results in the learning of meaningful data representations.\n\nAutoencoders offer a versatile tool for feature extraction, data compression, and more, making them a valuable addition to the toolkit of data scientists and machine learning practitioners.\n","metadata":{}},{"cell_type":"code","source":"from IPython.display import Image\nImage(filename=\"../input/rna-competition/1.png\", width= 900, height=400)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:38:09.217117Z","iopub.execute_input":"2023-10-27T10:38:09.217420Z","iopub.status.idle":"2023-10-27T10:38:09.245282Z","shell.execute_reply.started":"2023-10-27T10:38:09.217394Z","shell.execute_reply":"2023-10-27T10:38:09.244323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class AutoEncoder(Model):\n    def __init__(self, input_dim, latent_dim):\n        super(AutoEncoder, self).__init__()\n        self.input_dim = input_dim\n        self.latent_dim = latent_dim\n\n        self.encoder = tf.keras.Sequential([\n            layers.Input(shape=(input_dim,)),\n            layers.Reshape((input_dim, 1)),  # Reshape to 3D for Conv1D\n            layers.Conv1D(64, 3, strides=1, activation='relu', padding=\"same\"),\n            layers.BatchNormalization(),\n            layers.MaxPooling1D(2, padding=\"same\"),\n            layers.Conv1D(latent_dim, 3, strides=1, activation='relu', padding=\"same\"),\n            layers.BatchNormalization(),\n            layers.MaxPooling1D(2, padding=\"same\"),\n        ])\n\n        self.decoder = tf.keras.Sequential([\n            layers.Conv1D(latent_dim, 3, strides=1, activation='relu', padding=\"same\"),\n            layers.UpSampling1D(2),\n            layers.BatchNormalization(),\n            layers.Conv1D(64, 3, strides=1, activation='relu', padding=\"same\"),\n            layers.UpSampling1D(2),\n            layers.BatchNormalization(),\n            layers.Flatten(),\n            layers.Dense(input_dim)\n        ])\n\n    def call(self, X):\n        encoded = self.encoder(X)\n        decoded = self.decoder(encoded)\n        return decoded\n\n\ninput_dim = X_train_DMS.shape[-1]\nlatent_dim = 32\n\nmodel_DMS = AutoEncoder(input_dim, latent_dim)\nmodel_DMS.build((None, input_dim))\nmodel_DMS.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=0.01), loss=\"mae\")\nmodel_DMS.summary()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:38:09.246402Z","iopub.execute_input":"2023-10-27T10:38:09.246695Z","iopub.status.idle":"2023-10-27T10:38:14.435351Z","shell.execute_reply.started":"2023-10-27T10:38:09.246669Z","shell.execute_reply":"2023-10-27T10:38:14.434025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs = 100\nbatch_size = 64\nearly_stopping = EarlyStopping(patience=10, min_delta=1e-3, monitor=\"val_loss\", restore_best_weights=True)\n\n\nhistory = model_DMS.fit(X_train_DMS, y_train_DMS, epochs=epochs, batch_size=batch_size,\n                    validation_split=0.1, callbacks=[early_stopping])","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:38:14.436596Z","iopub.execute_input":"2023-10-27T10:38:14.436904Z","iopub.status.idle":"2023-10-27T10:40:38.592021Z","shell.execute_reply.started":"2023-10-27T10:38:14.436878Z","shell.execute_reply":"2023-10-27T10:40:38.591077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['loss'], label=\"Training loss\")\nplt.plot(history.history['val_loss'], label=\"Validation loss\", ls=\"--\")\nplt.legend(shadow=True, frameon=True, facecolor=\"inherit\", loc=\"best\", fontsize=9)\nplt.title(\"Training loss\")\nplt.ylabel(\"Loss\")\nplt.xlabel(\"Epoch\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:40:38.593616Z","iopub.execute_input":"2023-10-27T10:40:38.594447Z","iopub.status.idle":"2023-10-27T10:40:38.969546Z","shell.execute_reply.started":"2023-10-27T10:40:38.594406Z","shell.execute_reply":"2023-10-27T10:40:38.968456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_mae = model_DMS.evaluate(X_train_DMS, y_train_DMS, verbose=0)\ntest_mae = model_DMS.evaluate(X_test_DMS, y_test_DMS, verbose=0)\n\nprint(\"Training dataset error: \", train_mae)\nprint(\"Testing dataset error: \", test_mae)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:40:38.970735Z","iopub.execute_input":"2023-10-27T10:40:38.971045Z","iopub.status.idle":"2023-10-27T10:40:44.171419Z","shell.execute_reply.started":"2023-10-27T10:40:38.971018Z","shell.execute_reply":"2023-10-27T10:40:44.170399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_PRED = model_DMS.predict(X_train_DMS)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:40:44.172647Z","iopub.execute_input":"2023-10-27T10:40:44.172967Z","iopub.status.idle":"2023-10-27T10:40:49.293568Z","shell.execute_reply.started":"2023-10-27T10:40:44.172940Z","shell.execute_reply":"2023-10-27T10:40:49.292419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.round(Y_PRED,2)[0] ","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:40:49.295343Z","iopub.execute_input":"2023-10-27T10:40:49.296170Z","iopub.status.idle":"2023-10-27T10:40:49.345948Z","shell.execute_reply.started":"2023-10-27T10:40:49.296141Z","shell.execute_reply":"2023-10-27T10:40:49.344771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train_DMS[0]","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:40:49.347276Z","iopub.execute_input":"2023-10-27T10:40:49.347633Z","iopub.status.idle":"2023-10-27T10:40:49.360634Z","shell.execute_reply.started":"2023-10-27T10:40:49.347604Z","shell.execute_reply":"2023-10-27T10:40:49.359550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_PRED[0]","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:40:49.361934Z","iopub.execute_input":"2023-10-27T10:40:49.362314Z","iopub.status.idle":"2023-10-27T10:40:49.378799Z","shell.execute_reply.started":"2023-10-27T10:40:49.362283Z","shell.execute_reply":"2023-10-27T10:40:49.377704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.models import load_model\n#model.save(\"DMS.h5\")\nmodel_DMS.save(\"DMS\", save_format='tf')","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:40:49.380233Z","iopub.execute_input":"2023-10-27T10:40:49.380552Z","iopub.status.idle":"2023-10-27T10:40:54.113541Z","shell.execute_reply.started":"2023-10-27T10:40:49.380525Z","shell.execute_reply":"2023-10-27T10:40:54.112381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"where_2A3 = np.where(exp_type == exp_type_uniques[1])[0]\nif len(where_2A3) > 0:\n    X_2A3=inputs[where_2A3]\n    y_2A3=targets[where_2A3]","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:40:54.116096Z","iopub.execute_input":"2023-10-27T10:40:54.116494Z","iopub.status.idle":"2023-10-27T10:40:54.576114Z","shell.execute_reply.started":"2023-10-27T10:40:54.116459Z","shell.execute_reply":"2023-10-27T10:40:54.575002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_2A3, X_test_2A3, y_train_2A3, y_test_2A3 = train_test_split(X_2A3, y_2A3, test_size=0.1, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:40:54.577505Z","iopub.execute_input":"2023-10-27T10:40:54.578264Z","iopub.status.idle":"2023-10-27T10:40:55.054727Z","shell.execute_reply.started":"2023-10-27T10:40:54.578223Z","shell.execute_reply":"2023-10-27T10:40:55.053591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_dim = X_train_2A3.shape[-1]\nlatent_dim = 32\n\nmodel_2A3 = AutoEncoder(input_dim, latent_dim)\nmodel_2A3.build((None, input_dim))\nmodel_2A3.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=0.01), loss=\"mae\")\nmodel_2A3.summary()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:40:55.056054Z","iopub.execute_input":"2023-10-27T10:40:55.056359Z","iopub.status.idle":"2023-10-27T10:40:55.385286Z","shell.execute_reply.started":"2023-10-27T10:40:55.056333Z","shell.execute_reply":"2023-10-27T10:40:55.384249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs = 100\nbatch_size = 64\nearly_stopping = EarlyStopping(patience=10, min_delta=1e-3, monitor=\"val_loss\", restore_best_weights=True)\n\n\nhistory = model_2A3.fit(X_train_2A3, y_train_2A3, epochs=epochs, batch_size=batch_size,\n                    validation_split=0.1, callbacks=[early_stopping])","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:40:55.392169Z","iopub.execute_input":"2023-10-27T10:40:55.392925Z","iopub.status.idle":"2023-10-27T10:50:18.809348Z","shell.execute_reply.started":"2023-10-27T10:40:55.392886Z","shell.execute_reply":"2023-10-27T10:50:18.808403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['loss'], label=\"Training loss\")\nplt.plot(history.history['val_loss'], label=\"Validation loss\", ls=\"--\")\nplt.legend(shadow=True, frameon=True, facecolor=\"inherit\", loc=\"best\", fontsize=9)\nplt.title(\"Training loss 2A3_map\")\nplt.ylabel(\"Loss\")\nplt.xlabel(\"Epoch\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:50:18.810739Z","iopub.execute_input":"2023-10-27T10:50:18.811121Z","iopub.status.idle":"2023-10-27T10:50:19.183293Z","shell.execute_reply.started":"2023-10-27T10:50:18.811090Z","shell.execute_reply":"2023-10-27T10:50:19.182241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_mae =model_2A3.evaluate(X_train_2A3, y_train_2A3, verbose=0)\ntest_mae = model_2A3.evaluate(X_test_2A3, y_test_2A3, verbose=0)\n\nprint(\"Training dataset error: \", train_mae)\nprint(\"Testing dataset error: \", test_mae)\nmodel_2A3.save(\"2A3\", save_format='tf')","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:50:19.184717Z","iopub.execute_input":"2023-10-27T10:50:19.185123Z","iopub.status.idle":"2023-10-27T10:50:45.531435Z","shell.execute_reply.started":"2023-10-27T10:50:19.185093Z","shell.execute_reply":"2023-10-27T10:50:45.530338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_filename = '/kaggle/input/rna-competition/test_sequences.csv'\ntest_df = pd.read_csv(test_filename)\n\n# make input numerixc\nfor i,j in zip(['A','G','U','C'], ['1','2','3','4']):\n    test_df['sequence'] = test_df['sequence'].str.replace(i,j)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:50:45.533828Z","iopub.execute_input":"2023-10-27T10:50:45.534207Z","iopub.status.idle":"2023-10-27T10:51:01.113587Z","shell.execute_reply.started":"2023-10-27T10:50:45.534177Z","shell.execute_reply":"2023-10-27T10:51:01.112606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INPUT_SIZE =inputs_length\nimport time\n\nt0,t00 = time.time(),time.time()\n\nCHUNKSIZE = 2_000\nEST_TOT = 269_796_671\n\n# define dataframe with predictions\npred_df = pd.DataFrame(columns=['id', 'reactivity_DMS_MaP', 'reactivity_2A3_MaP'])\n\n\nX = np.zeros((0, INPUT_SIZE)).astype('float16')\nX_temp = np.zeros((1,INPUT_SIZE)).astype('float16')\nid_extr = []\n\n\nI,II = 0,0\nII_end = 5\n\n# loop through dataframe\nfor i, row in test_df.iterrows():\n\n    # get numeric inputs\n    seq = row.sequence\n    seq = [*seq]\n    \n    # fill input values\n    X_temp = 0*X_temp\n    X_temp[0,:len(seq)] = np.array(seq).astype('int')\n    X = np.append(X, X_temp, axis=0)\n    id_extr.append([row.id_min, row.id_max+1])\n    \n    if not (i+1)%CHUNKSIZE:\n        I += CHUNKSIZE*len(seq)\n        II += 1\n        print(\"%i %% : %i / %i\" % (100*I/EST_TOT, I, EST_TOT))#, end='\\r')\n        \n        if II==II_end:\n        \n            # do predictions\n            p_DMS = model_DMS.predict(X)\n            p_2A3 = model_2A3.predict(X)\n\n            # add predictions to dataframe\n            for j in range(len(id_extr)):\n                ids = np.arange(id_extr[j][0], id_extr[j][1])\n                df = pd.DataFrame({'id': ids,\n                                  'reactivity_DMS_MaP': p_DMS[j][:len(ids)],\n                                  'reactivity_2A3_MaP': p_2A3[j][:len(ids)]})\n                pred_df = pd.concat([pred_df, df])\n\n            # make id the index column\n            pred_df = pred_df.set_index('id')\n\n            # save predictions\n            if i+1==II*CHUNKSIZE:\n                print('New predict_submission_new.csv file...')\n                pred_df.to_csv('predict_submission_new.csv', header=pred_df.keys())\n            else:\n                print('Appending to predict_submission_new.csv file...')\n                pred_df.to_csv('predict_submission_new.csv', mode='a', header=False)\n\n            print(pred_df.shape)\n            print('Total/interval Time used: %.1f/ %.1f s' % (time.time()-t00,time.time()-t0))\n            t0 = time.time()\n            print()\n            \n            del X, pred_df, id_extr\n            X = np.zeros((0, INPUT_SIZE)).astype('float16')\n            pred_df = pd.DataFrame(columns=['id', 'reactivity_DMS_MaP', 'reactivity_2A3_MaP'])\n            id_extr = []\n            II = 0\n\nif II!=II_end:\n    p_DMS = model_DMS.predict(X)\n    p_2A3 = model_2A3.predict(X)\n    \n    # add predictions to dataframe\n    for j in range(len(id_extr)):\n        ids = np.arange(id_extr[j][0], id_extr[j][1])\n        df = pd.DataFrame({'id': ids,\n                          'reactivity_DMS_MaP': p_DMS[j][:len(ids)],\n                          'reactivity_2A3_MaP': p_2A3[j][:len(ids)]})\n        pred_df = pd.concat([pred_df, df])\n\n    # make id the index column\n    pred_df = pred_df.set_index('id')\n\n    # save predictions\n    print('Appending to predict_submission_new.csv file...')\n    pred_df.to_csv('predict_submission_new.csv', mode='a', header=False)\n    \nprint('DONE!!')","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:51:01.115019Z","iopub.execute_input":"2023-10-27T10:51:01.115319Z","iopub.status.idle":"2023-10-27T16:16:16.081850Z","shell.execute_reply.started":"2023-10-27T10:51:01.115293Z","shell.execute_reply":"2023-10-27T16:16:16.080891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}