{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n","metadata":{"execution":{"iopub.status.busy":"2023-10-18T07:42:53.471703Z","iopub.execute_input":"2023-10-18T07:42:53.471929Z","iopub.status.idle":"2023-10-18T07:42:53.758317Z","shell.execute_reply.started":"2023-10-18T07:42:53.471901Z","shell.execute_reply":"2023-10-18T07:42:53.757658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-10-18T07:43:00.265291Z","iopub.execute_input":"2023-10-18T07:43:00.265927Z","iopub.status.idle":"2023-10-18T07:43:00.321172Z","shell.execute_reply.started":"2023-10-18T07:43:00.265899Z","shell.execute_reply":"2023-10-18T07:43:00.320587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_full= pd.read_csv('/kaggle/input/stanford-ribonanza-rna-folding/train_data.csv')","metadata":{"execution":{"iopub.status.busy":"2023-10-18T07:05:49.143841Z","iopub.execute_input":"2023-10-18T07:05:49.144385Z","iopub.status.idle":"2023-10-18T07:07:44.173466Z","shell.execute_reply.started":"2023-10-18T07:05:49.144355Z","shell.execute_reply":"2023-10-18T07:07:44.172312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Identify and select columns with 'error' in their name\ncolumns = [col for col in data_full.columns if 'error' in col.lower()]\ndata_full = data_full.drop(columns,axis = 1)","metadata":{"execution":{"iopub.status.busy":"2023-10-18T07:08:40.038027Z","iopub.execute_input":"2023-10-18T07:08:40.038384Z","iopub.status.idle":"2023-10-18T07:08:41.341982Z","shell.execute_reply.started":"2023-10-18T07:08:40.038358Z","shell.execute_reply":"2023-10-18T07:08:41.339994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-10-18T07:10:05.664104Z","iopub.execute_input":"2023-10-18T07:10:05.664495Z","iopub.status.idle":"2023-10-18T07:10:11.290229Z","shell.execute_reply.started":"2023-10-18T07:10:05.664463Z","shell.execute_reply":"2023-10-18T07:10:11.288683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_DMS = np.load('/kaggle/working/X_encoded_DMS.npy')\nX_2A3 = np.load('/kaggle/working/X_encoded_2A3.npy')\ny_2A3 = pd.read_csv('/kaggle/working/y_2A3.csv')\ny_DMS = pd.read_csv('/kaggle/working/y_DMS.csv')","metadata":{"execution":{"iopub.status.busy":"2023-10-18T07:51:12.762654Z","iopub.execute_input":"2023-10-18T07:51:12.763204Z","iopub.status.idle":"2023-10-18T07:51:19.501962Z","shell.execute_reply.started":"2023-10-18T07:51:12.763174Z","shell.execute_reply":"2023-10-18T07:51:19.501302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_2A3 = data_full[data_full.experiment_type == '2A3_MaP']","metadata":{"execution":{"iopub.status.busy":"2023-10-18T07:11:35.093903Z","iopub.execute_input":"2023-10-18T07:11:35.094244Z","iopub.status.idle":"2023-10-18T07:11:35.247572Z","shell.execute_reply.started":"2023-10-18T07:11:35.094204Z","shell.execute_reply":"2023-10-18T07:11:35.246511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_DMS = data_full[data_full.experiment_type == 'DMS_MaP']","metadata":{"execution":{"iopub.status.busy":"2023-10-18T07:11:58.58599Z","iopub.execute_input":"2023-10-18T07:11:58.586348Z","iopub.status.idle":"2023-10-18T07:11:58.737983Z","shell.execute_reply.started":"2023-10-18T07:11:58.58632Z","shell.execute_reply":"2023-10-18T07:11:58.736726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns = [col for col in data_DMS.columns if 'reactivity' in col.lower()]\ny_DMS = data_DMS[columns]","metadata":{"execution":{"iopub.status.busy":"2023-10-18T07:15:32.845719Z","iopub.execute_input":"2023-10-18T07:15:32.846087Z","iopub.status.idle":"2023-10-18T07:15:33.11845Z","shell.execute_reply.started":"2023-10-18T07:15:32.846061Z","shell.execute_reply":"2023-10-18T07:15:33.117298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Fone_hot","metadata":{"execution":{"iopub.status.busy":"2023-10-18T07:15:59.083357Z","iopub.execute_input":"2023-10-18T07:15:59.083759Z","iopub.status.idle":"2023-10-18T07:15:59.089921Z","shell.execute_reply.started":"2023-10-18T07:15:59.08372Z","shell.execute_reply":"2023-10-18T07:15:59.089005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_2A3 = np.load('/kaggle/working/X_encoded_2A3.npy')\nX_DMS = np.load('/kaggle/working/X_encoded_DMS.npy')","metadata":{"execution":{"iopub.status.busy":"2023-10-18T07:02:06.882884Z","iopub.execute_input":"2023-10-18T07:02:06.883202Z","iopub.status.idle":"2023-10-18T07:02:07.172749Z","shell.execute_reply.started":"2023-10-18T07:02:06.883177Z","shell.execute_reply":"2023-10-18T07:02:07.171166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('X_DMS_SHAPE',X_DMS.shape)\nprint('y_DMS_SHAPE',y_DMS.shape)\nprint('X_2A3_SHAPE',X_2A3.shape)\nprint('y_2A3_SHAPE',y_2A3.shape)","metadata":{"execution":{"iopub.status.busy":"2023-10-18T07:51:43.682281Z","iopub.execute_input":"2023-10-18T07:51:43.682555Z","iopub.status.idle":"2023-10-18T07:51:43.689986Z","shell.execute_reply.started":"2023-10-18T07:51:43.682533Z","shell.execute_reply":"2023-10-18T07:51:43.689304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_2A3 = pd.read_csv('/kaggle/working/y_2A3_SN_1.csv')\ny_DMS = pd.read_csv('/kaggle/working/y_DMS_SN_1.csv')","metadata":{"execution":{"iopub.status.busy":"2023-10-18T06:48:23.800121Z","iopub.execute_input":"2023-10-18T06:48:23.800446Z","iopub.status.idle":"2023-10-18T06:48:30.665354Z","shell.execute_reply.started":"2023-10-18T06:48:23.800419Z","shell.execute_reply":"2023-10-18T06:48:30.664288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-10-18T06:48:44.716997Z","iopub.execute_input":"2023-10-18T06:48:44.717345Z","iopub.status.idle":"2023-10-18T06:48:44.809823Z","shell.execute_reply.started":"2023-10-18T06:48:44.717318Z","shell.execute_reply":"2023-10-18T06:48:44.808075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seq_X_2A3 = data.sequence[data.experiment_type == '2A3_MaP']\nseq_X_DMS = data.sequence[data.experiment_type == 'DMS_MaP']","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.416299Z","iopub.status.idle":"2023-10-15T22:04:16.417242Z","shell.execute_reply.started":"2023-10-15T22:04:16.417032Z","shell.execute_reply":"2023-10-15T22:04:16.417055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_2A3.shape","metadata":{"execution":{"iopub.status.busy":"2023-10-18T06:49:55.472616Z","iopub.execute_input":"2023-10-18T06:49:55.472925Z","iopub.status.idle":"2023-10-18T06:49:55.479312Z","shell.execute_reply.started":"2023-10-18T06:49:55.472898Z","shell.execute_reply":"2023-10-18T06:49:55.478694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_2A3 == y_DMS","metadata":{"execution":{"iopub.status.busy":"2023-10-18T06:51:40.578281Z","iopub.execute_input":"2023-10-18T06:51:40.578613Z","iopub.status.idle":"2023-10-18T06:51:40.629092Z","shell.execute_reply.started":"2023-10-18T06:51:40.578592Z","shell.execute_reply":"2023-10-18T06:51:40.6281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_DMS.shape","metadata":{"execution":{"iopub.status.busy":"2023-10-18T06:50:48.437256Z","iopub.execute_input":"2023-10-18T06:50:48.437605Z","iopub.status.idle":"2023-10-18T06:50:48.443284Z","shell.execute_reply.started":"2023-10-18T06:50:48.437582Z","shell.execute_reply":"2023-10-18T06:50:48.442145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_2A3 = data[data.experiment_type == '2A3_MaP']\ncolumns = [col for col in data_2A3.columns if 'reactivity' in col.lower()]\ny_2A3_SN_1 = data[columns]","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:10:24.459937Z","iopub.execute_input":"2023-10-15T22:10:24.46092Z","iopub.status.idle":"2023-10-15T22:10:25.30726Z","shell.execute_reply.started":"2023-10-15T22:10:24.460863Z","shell.execute_reply":"2023-10-15T22:10:25.305695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_DMS = data[data.experiment_type == 'DMS_MaP']\ncolumns = [col for col in data_DMS.columns if 'reactivity' in col.lower()]\ny_DMS_SN_1 = data[columns]","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:14:15.921572Z","iopub.execute_input":"2023-10-15T22:14:15.923014Z","iopub.status.idle":"2023-10-15T22:14:16.874116Z","shell.execute_reply.started":"2023-10-15T22:14:15.922973Z","shell.execute_reply":"2023-10-15T22:14:16.873236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_DMS_SN_1 = y_DMS_SN_1.dropna(axis=1, how='all')\ny_DMS_SN_1","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:14:48.193134Z","iopub.execute_input":"2023-10-15T22:14:48.193607Z","iopub.status.idle":"2023-10-15T22:14:48.96929Z","shell.execute_reply.started":"2023-10-15T22:14:48.193575Z","shell.execute_reply":"2023-10-15T22:14:48.968264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\n# Specify the file path you want to delete\nfile_path = ['/kaggle/working/X_encoded_2A3.npy','/kaggle/working/y_DMS_SN_1.csv','/kaggle/working/X_encoded_DMS.npy','/kaggle/working/y_2A3_SN_1.csv']\n\n# Check if the file exists before attempting to delete it\nfor file_path_line in file_path:\n    if os.path.exists(file_path_line):\n        os.remove(file_path_line)\n        print(f\"The file '{file_path_line}' has been deleted.\")\n    else:\n        print(f\"The file '{file_path_line}' does not exist.\")\n","metadata":{"execution":{"iopub.status.busy":"2023-10-18T07:04:42.706451Z","iopub.execute_input":"2023-10-18T07:04:42.706948Z","iopub.status.idle":"2023-10-18T07:04:43.21385Z","shell.execute_reply.started":"2023-10-18T07:04:42.706908Z","shell.execute_reply":"2023-10-18T07:04:43.212766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_2A3.to_csv('y_2A3.csv', index = False)\ny_DMS.to_csv('y_DMS.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2023-10-18T07:32:24.20389Z","iopub.execute_input":"2023-10-18T07:32:24.20429Z","iopub.status.idle":"2023-10-18T07:33:12.813302Z","shell.execute_reply.started":"2023-10-18T07:32:24.204261Z","shell.execute_reply":"2023-10-18T07:33:12.811916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ny_2A3_SN_1 = y_2A3_SN_1.dropna(axis=1, how='all')\ny_2A3_SN_1","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:12:09.410846Z","iopub.execute_input":"2023-10-15T22:12:09.411337Z","iopub.status.idle":"2023-10-15T22:12:10.090667Z","shell.execute_reply.started":"2023-10-15T22:12:09.411306Z","shell.execute_reply":"2023-10-15T22:12:10.089072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.experiment_type.count","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.422421Z","iopub.status.idle":"2023-10-15T22:04:16.422862Z","shell.execute_reply.started":"2023-10-15T22:04:16.422676Z","shell.execute_reply":"2023-10-15T22:04:16.422694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = data[data.SN_filter == 1]","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.425488Z","iopub.status.idle":"2023-10-15T22:04:16.426017Z","shell.execute_reply.started":"2023-10-15T22:04:16.425802Z","shell.execute_reply":"2023-10-15T22:04:16.425825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.42856Z","iopub.status.idle":"2023-10-15T22:04:16.429163Z","shell.execute_reply.started":"2023-10-15T22:04:16.428938Z","shell.execute_reply":"2023-10-15T22:04:16.428962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.430751Z","iopub.status.idle":"2023-10-15T22:04:16.431215Z","shell.execute_reply.started":"2023-10-15T22:04:16.43101Z","shell.execute_reply":"2023-10-15T22:04:16.43103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.434421Z","iopub.status.idle":"2023-10-15T22:04:16.435013Z","shell.execute_reply.started":"2023-10-15T22:04:16.434774Z","shell.execute_reply":"2023-10-15T22:04:16.4348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import logging\nimport typing\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\n\n\nlogger = logging.getLogger(__name__)\n\n\nURL_PREFIX = \"https://s3.amazonaws.com/songlabdata/proteindata/pytorch-models/\"\nLSTM_PRETRAINED_CONFIG_ARCHIVE_MAP: typing.Dict[str, str] = {}\nLSTM_PRETRAINED_MODEL_ARCHIVE_MAP: typing.Dict[str, str] = {}\n\n\nclass ProteinLSTMConfig(vocab_size,\n                 input_size,\n                 hidden_size,\n                 num_hidden_layers,\n                 hidden_dropout_prob,\n                 initializer_range ,):\n    pretrained_config_archive_map = LSTM_PRETRAINED_CONFIG_ARCHIVE_MAP\n\n    def __init__(self,\n                 vocab_size: int = 30,\n                 input_size: int = 128,\n                 hidden_size: int = 1024,\n                 num_hidden_layers: int = 3,\n                 hidden_dropout_prob: float = 0.1,\n                 initializer_range: float = 0.02,\n                 **kwargs):\n        super().__init__(**kwargs)\n        self.vocab_size = vocab_size\n        self.input_size = input_size\n        self.hidden_size = hidden_size\n        self.num_hidden_layers = num_hidden_layers\n        self.hidden_dropout_prob = hidden_dropout_prob\n        self.initializer_range = initializer_range\n\n\nclass ProteinLSTMLayer(nn.Module):\n\n    def __init__(self, input_size: int, hidden_size: int, dropout: float = 0.):\n        super().__init__()\n        self.dropout = nn.Dropout(dropout)\n        self.lstm = nn.LSTM(input_size, hidden_size, batch_first=True)\n\n    def forward(self, inputs):\n        inputs = self.dropout(inputs)\n        self.lstm.flatten_parameters()\n        return self.lstm(inputs)\n\n\nclass ProteinLSTMPooler(nn.Module):\n    def __init__(self, config):\n        super().__init__()\n        self.scalar_reweighting = nn.Linear(2 * config.num_hidden_layers, 1)\n        self.dense = nn.Linear(config.hidden_size, config.hidden_size)\n        self.activation = nn.Tanh()\n\n    def forward(self, hidden_states):\n        # We \"pool\" the model by simply taking the hidden state corresponding\n        # to the first token.\n        pooled_output = self.scalar_reweighting(hidden_states).squeeze(2)\n        pooled_output = self.dense(pooled_output)\n        pooled_output = self.activation(pooled_output)\n        return pooled_output\n\n\nclass ProteinLSTMEncoder(nn.Module):\n\n    def __init__(self, config: ProteinLSTMConfig):\n        super().__init__()\n        forward_lstm = [ProteinLSTMLayer(config.input_size, config.hidden_size)]\n        reverse_lstm = [ProteinLSTMLayer(config.input_size, config.hidden_size)]\n        for _ in range(config.num_hidden_layers - 1):\n            forward_lstm.append(ProteinLSTMLayer(\n                config.hidden_size, config.hidden_size, config.hidden_dropout_prob))\n            reverse_lstm.append(ProteinLSTMLayer(\n                config.hidden_size, config.hidden_size, config.hidden_dropout_prob))\n        self.forward_lstm = nn.ModuleList(forward_lstm)\n        self.reverse_lstm = nn.ModuleList(reverse_lstm)\n        self.output_hidden_states = config.output_hidden_states\n\n    def forward(self, inputs, input_mask=None):\n        all_forward_pooled = ()\n        all_reverse_pooled = ()\n        all_hidden_states = (inputs,)\n        forward_output = inputs\n        for layer in self.forward_lstm:\n            forward_output, forward_pooled = layer(forward_output)\n            all_forward_pooled = all_forward_pooled + (forward_pooled[0],)\n            all_hidden_states = all_hidden_states + (forward_output,)\n\n        reversed_sequence = self.reverse_sequence(inputs, input_mask)\n        reverse_output = reversed_sequence\n        for layer in self.reverse_lstm:\n            reverse_output, reverse_pooled = layer(reverse_output)\n            all_reverse_pooled = all_reverse_pooled + (reverse_pooled[0],)\n            all_hidden_states = all_hidden_states + (reverse_output,)\n        reverse_output = self.reverse_sequence(reverse_output, input_mask)\n\n        output = torch.cat((forward_output, reverse_output), dim=2)\n        pooled = all_forward_pooled + all_reverse_pooled\n        pooled = torch.stack(pooled, 3).squeeze(0)\n        outputs = (output, pooled)\n        if self.output_hidden_states:\n            outputs = outputs + (all_hidden_states,)\n\n        return outputs  # sequence_embedding, pooled_embedding, (hidden_states)\n\n    def reverse_sequence(self, sequence, input_mask):\n        if input_mask is None:\n            idx = torch.arange(sequence.size(1) - 1, -1, -1)\n            reversed_sequence = sequence.index_select(1, idx, device=sequence.device)\n        else:\n            sequence_lengths = input_mask.sum(1)\n            reversed_sequence = []\n            for seq, seqlen in zip(sequence, sequence_lengths):\n                idx = torch.arange(seqlen - 1, -1, -1, device=seq.device)\n                seq = seq.index_select(0, idx)\n                seq = F.pad(seq, [0, 0, 0, sequence.size(1) - seqlen])\n                reversed_sequence.append(seq)\n            reversed_sequence = torch.stack(reversed_sequence, 0)\n        return reversed_sequence\n\n\nclass ProteinLSTMAbstractModel(ProteinModel):\n\n    config_class = ProteinLSTMConfig\n    pretrained_model_archive_map = LSTM_PRETRAINED_MODEL_ARCHIVE_MAP\n    base_model_prefix = \"lstm\"\n\n    def _init_weights(self, module):\n        \"\"\" Initialize the weights \"\"\"\n        if isinstance(module, (nn.Linear, nn.Embedding)):\n            module.weight.data.normal_(mean=0.0, std=self.config.initializer_range)\n        if isinstance(module, nn.Linear) and module.bias is not None:\n            module.bias.data.zero_()\n\n\n@registry.register_task_model('embed', 'lstm')\nclass ProteinLSTMModel(ProteinLSTMAbstractModel):\n\n    def __init__(self, config: ProteinLSTMConfig):\n        super().__init__(config)\n        self.embed_matrix = nn.Embedding(config.vocab_size, config.input_size)\n        self.encoder = ProteinLSTMEncoder(config)\n        self.pooler = ProteinLSTMPooler(config)\n        self.output_hidden_states = config.output_hidden_states\n        self.init_weights()\n\n    def forward(self, input_ids, input_mask=None):\n        if input_mask is None:\n            input_mask = torch.ones_like(input_ids)\n\n        # fp16 compatibility\n        embedding_output = self.embed_matrix(input_ids)\n        outputs = self.encoder(embedding_output, input_mask=input_mask)\n        sequence_output = outputs[0]\n        pooled_outputs = self.pooler(outputs[1])\n\n        outputs = (sequence_output, pooled_outputs) + outputs[2:]\n        return outputs  # sequence_output, pooled_output, (hidden_states)\n\n\n@registry.register_task_model('language_modeling', 'lstm')\nclass ProteinLSTMForLM(ProteinLSTMAbstractModel):\n\n    def __init__(self, config):\n        super().__init__(config)\n\n        self.lstm = ProteinLSTMModel(config)\n        self.feedforward = nn.Linear(config.hidden_size, config.vocab_size)\n\n        self.init_weights()\n\n    def forward(self,\n                input_ids,\n                input_mask=None,\n                targets=None):\n\n        outputs = self.lstm(input_ids, input_mask=input_mask)\n\n        sequence_output, pooled_output = outputs[:2]\n\n        forward_prediction, reverse_prediction = sequence_output.chunk(2, -1)\n        forward_prediction = F.pad(forward_prediction[:, :-1], [0, 0, 1, 0])\n        reverse_prediction = F.pad(reverse_prediction[:, 1:], [0, 0, 0, 1])\n        prediction_scores = \\\n            self.feedforward(forward_prediction) + self.feedforward(reverse_prediction)\n        prediction_scores = prediction_scores.contiguous()\n\n        # add hidden states and if they are here\n        outputs = (prediction_scores,) + outputs[2:]\n\n        if targets is not None:\n            loss_fct = nn.CrossEntropyLoss(ignore_index=-1)\n            lm_loss = loss_fct(\n                prediction_scores.view(-1, self.config.vocab_size), targets.view(-1))\n            outputs = (lm_loss,) + outputs\n\n        # (loss), prediction_scores, seq_relationship_score, (hidden_states)\n        return outputs\n\n\n@registry.register_task_model('fluorescence', 'lstm')\n@registry.register_task_model('stability', 'lstm')\nclass ProteinLSTMForValuePrediction(ProteinLSTMAbstractModel):\n\n    def __init__(self, config):\n        super().__init__(config)\n\n        self.lstm = ProteinLSTMModel(config)\n        self.predict = ValuePredictionHead(config.hidden_size)\n\n        self.init_weights()\n\n    def forward(self, input_ids, input_mask=None, targets=None):\n\n        outputs = self.lstm(input_ids, input_mask=input_mask)\n\n        sequence_output, pooled_output = outputs[:2]\n        outputs = self.predict(pooled_output, targets) + outputs[2:]\n        # (loss), prediction_scores, (hidden_states)\n        return outputs\n\n\n@registry.register_task_model('remote_homology', 'lstm')\nclass ProteinLSTMForSequenceClassification(ProteinLSTMAbstractModel):\n\n    def __init__(self, config):\n        super().__init__(config)\n\n        self.lstm = ProteinLSTMModel(config)\n        self.classify = SequenceClassificationHead(\n            config.hidden_size, config.num_labels)\n\n        self.init_weights()\n\n    def forward(self, input_ids, input_mask=None, targets=None):\n\n        outputs = self.lstm(input_ids, input_mask=input_mask)\n\n        sequence_output, pooled_output = outputs[:2]\n        outputs = self.classify(pooled_output, targets) + outputs[2:]\n        # (loss), prediction_scores, (hidden_states)\n        return outputs\n\n\n@registry.register_task_model('secondary_structure', 'lstm')\nclass ProteinLSTMForSequenceToSequenceClassification(ProteinLSTMAbstractModel):\n\n    def __init__(self, config):\n        super().__init__(config)\n\n        self.lstm = ProteinLSTMModel(config)\n        self.classify = SequenceToSequenceClassificationHead(\n            config.hidden_size * 2, config.num_labels, ignore_index=-1)\n\n        self.init_weights()\n\n    def forward(self, input_ids, input_mask=None, targets=None):\n\n        outputs = self.lstm(input_ids, input_mask=input_mask)\n\n        sequence_output, pooled_output = outputs[:2]\n        amino_acid_class_scores = self.classify(sequence_output.contiguous())\n\n        # add hidden states and if they are here\n        outputs = (amino_acid_class_scores,) + outputs[2:]\n\n        if targets is not None:\n            loss_fct = nn.CrossEntropyLoss(ignore_index=-1)\n            classification_loss = loss_fct(\n                amino_acid_class_scores.view(-1, self.config.num_labels),\n                targets.view(-1))\n            outputs = (classification_loss,) + outputs\n\n        # (loss), prediction_scores, seq_relationship_score, (hidden_states)\n        return outputs\n\n\n@registry.register_task_model('contact_prediction', 'lstm')\nclass ProteinLSTMForContactPrediction(ProteinLSTMAbstractModel):\n\n    def __init__(self, config):\n        super().__init__(config)\n\n        self.lstm = ProteinLSTMModel(config)\n        self.predict = PairwiseContactPredictionHead(config.hidden_size, ignore_index=-1)\n\n        self.init_weights()\n\n    def forward(self, input_ids, protein_length, input_mask=None, targets=None):\n\n        outputs = self.lstm(input_ids, input_mask=input_mask)\n\n        sequence_output, pooled_output = outputs[:2]\n        outputs = self.predict(sequence_output, protein_length, targets) + outputs[2:]\n        # (loss), prediction_scores, (hidden_states), (attentions)\n        return outputs","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.437932Z","iopub.status.idle":"2023-10-15T22:04:16.438431Z","shell.execute_reply.started":"2023-10-15T22:04:16.438224Z","shell.execute_reply":"2023-10-15T22:04:16.438245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model lstm\n\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Embedding, LSTM, Dense, Dropout, Input, Lambda\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.initializers import TruncatedNormal\n\nclass ProteinLSTMConfig:\n    def __init__(self, vocab_size=30, input_size=128, hidden_size=1024, num_hidden_layers=3,\n                 hidden_dropout_prob=0.1, initializer_range=0.02,output_hidden_states=128):\n        self.vocab_size = vocab_size\n        self.input_size = input_size\n        self.hidden_size = hidden_size\n        self.num_hidden_layers = num_hidden_layers\n        self.hidden_dropout_prob = hidden_dropout_prob\n        self.initializer_range = initializer_range\n        self.output_hidden_states = output_hidden_states\n\nclass ProteinLSTMLayer(tf.keras.layers.Layer):\n    def __init__(self, input_size, hidden_size, dropout=0.):\n        super(ProteinLSTMLayer, self).__init__()\n        self.dropout = Dropout(dropout)\n        self.lstm = LSTM(hidden_size, return_sequences=True, return_state=True)\n\n    def call(self, inputs):\n        inputs = self.dropout(inputs)\n        outputs, _ , _ = self.lstm(inputs)\n        return outputs\n\nclass ProteinLSTMPooler(tf.keras.layers.Layer):\n    def __init__(self, config):\n        super(ProteinLSTMPooler, self).__init__()\n        self.dense = Dense(config.hidden_size, activation='tanh')\n\n    def call(self, hidden_states):\n        pooled_output = self.dense(hidden_states[:, 0, :])\n        return pooled_output\n\nclass ProteinLSTMEncoder(tf.keras.layers.Layer):\n    def __init__(self, config):\n        super(ProteinLSTMEncoder, self).__init__()\n        forward_lstm = [ProteinLSTMLayer(config.input_size, config.hidden_size)]\n        reverse_lstm = [ProteinLSTMLayer(config.input_size, config.hidden_size)]\n        for _ in range(config.num_hidden_layers - 1):\n            forward_lstm.append(ProteinLSTMLayer(config.hidden_size, config.hidden_size,\n                                                 config.hidden_dropout_prob))\n            reverse_lstm.append(ProteinLSTMLayer(config.hidden_size, config.hidden_size,\n                                                 config.hidden_dropout_prob))\n        self.forward_lstm = tf.keras.Sequential(forward_lstm)\n        self.reverse_lstm = tf.keras.Sequential(reverse_lstm)\n        self.output_hidden_states = config.output_hidden_states\n\n    def call(self, inputs, input_mask=None):\n        all_forward_pooled = []\n        all_reverse_pooled = []\n        all_hidden_states = [inputs]\n        forward_output = inputs\n\n        for layer in self.forward_lstm.layers:\n            forward_output = layer(forward_output)\n            all_forward_pooled.append(forward_output[:, 0, :])\n            all_hidden_states.append(forward_output)\n\n        reversed_sequence = self.reverse_sequence(inputs, input_mask)\n        reverse_output = reversed_sequence\n\n        for layer in self.reverse_lstm.layers:\n            reverse_output = layer(reverse_output)\n            all_reverse_pooled.append(reverse_output[:, 0, :])\n            all_hidden_states.append(reverse_output)\n\n        reverse_output = self.reverse_sequence(reverse_output, input_mask)\n        output = tf.concat([forward_output, reverse_output], axis=-1)\n        pooled = tf.stack(all_forward_pooled + all_reverse_pooled, axis=-1)\n        pooled = tf.squeeze(pooled, axis=0)\n        outputs = (output, pooled)\n\n        if self.output_hidden_states:\n            outputs += (all_hidden_states,)\n\n        return outputs\n\n#     def reverse_sequence(self, sequence, input_mask):\n#         if input_mask is None:\n#             reversed_sequence = Lambda(lambda x: tf.reverse(x, axis=[1]))(sequence)\n#         else:\n#             sequence_lengths = tf.math.reduce_sum(input_mask, axis=1)\n#             reversed_sequence = []\n\n#             for seq, seqlen in zip(sequence, sequence_lengths):\n#                 seq = Lambda(lambda x: x[:seqlen])(seq)\n#                 seq = Lambda(lambda x: tf.reverse(x, axis=[0]))(seq)\n#                 seq = Lambda(lambda x: tf.pad(x, [[0, sequence.shape[1] - seqlen], [0, 0]]))(seq)\n#                 reversed_sequence.append(seq)\n\n#             reversed_sequence = tf.stack(reversed_sequence, axis=0)\n\n#         return reversed_sequence\n#     def reverse_sequence(self, sequence, input_mask):\n#         if input_mask is None:\n#             sequence_lengths = tf.reduce_sum(tf.ones_like(sequence), axis=1)\n#         else:\n#             sequence_lengths = tf.reduce_sum(input_mask, axis=1)\n    \n#         reversed_sequence = tf.reverse_sequence(sequence, sequence_lengths, seq_axis=1, batch_axis=0)\n    \n#         return reversed_sequence\n    def reverse_sequence(self, sequence, input_mask):\n        if input_mask is None:\n            sequence_lengths = tf.reduce_sum(tf.ones_like(sequence, dtype=tf.int32), axis=1)\n        else:\n            sequence_lengths = tf.reduce_sum(tf.cast(input_mask, tf.int32), axis=1)\n\n        reversed_sequence = tf.reverse_sequence(sequence, sequence_lengths, seq_axis=1, batch_axis=0)\n\n        return reversed_sequence\n\nclass ProteinLSTMModel(tf.keras.Model):\n    def __init__(self, config):\n        super(ProteinLSTMModel, self).__init__()\n        self.embed_matrix = Embedding(config.vocab_size, config.input_size, embeddings_initializer=TruncatedNormal(stddev=config.initializer_range))\n        self.encoder = ProteinLSTMEncoder(config)\n        self.pooler = ProteinLSTMPooler(config)\n        self.output_hidden_states = config.output_hidden_states\n\n    def call(self, input_ids, input_mask=None):\n        if input_mask is None:\n            input_mask = tf.ones_like(input_ids)\n\n        embedding_output = self.embed_matrix(input_ids)\n        outputs = self.encoder(embedding_output, input_mask=input_mask)\n        sequence_output = outputs[0]\n        pooled_outputs = self.pooler(outputs[1])\n        outputs = (sequence_output, pooled_outputs) + outputs[2:]\n\n        return outputs\n\nconfig = ProteinLSTMConfig()\nmodel = ProteinLSTMModel(config)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.442818Z","iopub.status.idle":"2023-10-15T22:04:16.443371Z","shell.execute_reply.started":"2023-10-15T22:04:16.443143Z","shell.execute_reply":"2023-10-15T22:04:16.443163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\n\n# Define the LSTM layer for the ProteinLSTMModel\nclass ProteinLSTMLayer(nn.Module):\n    def __init__(self, input_size, hidden_size, dropout=0.0):\n        super(ProteinLSTMLayer, self).__init__()\n        self.dropout = nn.Dropout(dropout)\n        self.lstm = nn.LSTM(input_size, hidden_size, batch_first=True)\n\n    def forward(self, inputs):\n        inputs = self.dropout(inputs)\n        self.lstm.flatten_parameters()\n        return self.lstm(inputs)\n\n# Define the pooling layer for the ProteinLSTMModel\nclass ProteinLSTMPooler(nn.Module):\n    def __init__(self, config):\n        super(ProteinLSTMPooler, self).__init__()\n        self.scalar_reweighting = nn.Linear(2 * config.num_hidden_layers, 1)\n        self.dense = nn.Linear(config.hidden_size, config.hidden_size)\n        self.activation = nn.Tanh()\n\n    def forward(self, hidden_states):\n        pooled_output = self.scalar_reweighting(hidden_states).squeeze(2)\n        pooled_output = self.dense(pooled_output)\n        pooled_output = self.activation(pooled_output)\n        return pooled_output\n\n# Define the LSTM encoder for the ProteinLSTMModel\nclass ProteinLSTMEncoder(nn.Module):\n    def __init__(self, config):\n        super(ProteinLSTMEncoder, self).__init__()\n        forward_lstm = [ProteinLSTMLayer(config.input_size, config.hidden_size)]\n        reverse_lstm = [ProteinLSTMLayer(config.input_size, config.hidden_size)]\n        for _ in range(config.num_hidden_layers - 1):\n            forward_lstm.append(ProteinLSTMLayer(config.hidden_size, config.hidden_size, config.hidden_dropout_prob))\n            reverse_lstm.append(ProteinLSTMLayer(config.hidden_size, config.hidden_size, config.hidden_dropout_prob))\n        self.forward_lstm = nn.ModuleList(forward_lstm)\n        self.reverse_lstm = nn.ModuleList(reverse_lstm)\n        self.output_hidden_states = config.output_hidden_states\n\n    def forward(self, inputs, input_mask=None):\n        all_forward_pooled = ()\n        all_reverse_pooled = ()\n        all_hidden_states = (inputs,)\n        forward_output = inputs\n        for layer in self.forward_lstm:\n            forward_output, forward_pooled = layer(forward_output)\n            all_forward_pooled = all_forward_pooled + (forward_pooled[0],)\n            all_hidden_states = all_hidden_states + (forward_output,)\n\n        reversed_sequence = self.reverse_sequence(inputs, input_mask)\n        reverse_output = reversed_sequence\n        for layer in self.reverse_lstm:\n            reverse_output, reverse_pooled = layer(reverse_output)\n            all_reverse_pooled = all_reverse_pooled + (reverse_pooled[0],)\n            all_hidden_states = all_hidden_states + (reverse_output,)\n        reverse_output = self.reverse_sequence(reverse_output, input_mask)\n\n        output = torch.cat((forward_output, reverse_output), dim=2)\n        pooled = all_forward_pooled + all_reverse_pooled\n        pooled = torch.stack(pooled, 3).squeeze(0)\n        outputs = (output, pooled)\n        if self.output_hidden_states:\n            outputs = outputs + (all_hidden_states,)\n\n        return outputs  # sequence_embedding, pooled_embedding, (hidden_states)\n\n    def reverse_sequence(self, sequence, input_mask):\n        if input_mask is None:\n            idx = torch.arange(sequence.size(1) - 1, -1, -1)\n            reversed_sequence = sequence.index_select(1, idx, device=sequence.device)\n        else:\n            sequence_lengths = input_mask.sum(1)\n            reversed_sequence = []\n            for seq, seqlen in zip(sequence, sequence_lengths):\n                idx = torch.arange(seqlen - 1, -1, -1, device=seq.device)\n                seq = seq.index_select(0, idx)\n                seq = F.pad(seq, [0, 0, 0, sequence.size(1) - seqlen])\n                reversed_sequence.append(seq)\n            reversed_sequence = torch.stack(reversed_sequence, 0)\n        return reversed_sequence\n\n# Define the configuration for the ProteinLSTMModel\nclass ProteinLSTMConfig:\n    def __init__(self, vocab_size=30, input_size=128, hidden_size=1024, num_hidden_layers=3,\n                 hidden_dropout_prob=0.1, initializer_range=0.02, output_hidden_states=True):\n        self.vocab_size = vocab_size\n        self.input_size = input_size\n        self.hidden_size = hidden_size\n        self.num_hidden_layers = num_hidden_layers\n        self.hidden_dropout_prob = hidden_dropout_prob\n        self.initializer_range = initializer_range\n        self.output_hidden_states = output_hidden_states\n\n# Define the ProteinLSTMModel\nclass ProteinLSTMModel(nn.Module):\n    def __init__(self, config):\n        super(ProteinLSTMModel, self).__init__()\n        self.embed_matrix = nn.Embedding(config.vocab_size, config.input_size)\n        self.encoder = ProteinLSTMEncoder(config)\n        self.pooler = ProteinLSTMPooler(config)\n        self.output_hidden_states = config.output_hidden_states\n\n    def forward(self, input_ids, input_mask=None):\n        if input_mask is None:\n            input_mask = torch.ones_like(input_ids, dtype=torch.long)\n\n        embedding_output = self.embed_matrix(input_ids)\n        outputs = self.encoder(embedding_output, input_mask=input_mask)\n        sequence_output = outputs[0]\n        pooled_outputs = self.pooler(outputs[1])\n\n        outputs = (sequence_output, pooled_outputs) + outputs[2:]\n        return outputs  # sequence_output, pooled_output, (hidden_states)\n\n# Define your input data here\ninput_data = torch.randint(0, 30, (601354, 139, 4), dtype=torch.long)  # Assuming input_data is of shape (601354, 139, 4)\n\n# Instantiate the model with your desired configuration\nmodel_config = ProteinLSTMConfig()\nmodel = ProteinLSTMModel(model_config)\n\n# Run the model on your input data\nwith torch.no_grad():\n    output = model(input_data)\n\n# 'output' now contains the model's predictions or embeddings\n","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.445154Z","iopub.status.idle":"2023-10-15T22:04:16.445614Z","shell.execute_reply.started":"2023-10-15T22:04:16.445427Z","shell.execute_reply":"2023-10-15T22:04:16.445446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#working\nimport torch\nimport torch.nn as nn\n\n# Define the LSTM layer for the ProteinLSTMModel\nclass ProteinLSTMLayer(nn.Module):\n    def __init__(self, input_size, hidden_size, dropout=0.0):\n        super(ProteinLSTMLayer, self).__init__()\n        self.dropout = nn.Dropout(dropout)\n        self.lstm = nn.LSTM(input_size, hidden_size, batch_first=True)\n\n    def forward(self, inputs):\n        inputs = self.dropout(inputs)\n        return self.lstm(inputs)\n\n# Define the pooling layer for the ProteinLSTMModel\nclass ProteinLSTMPooler(nn.Module):\n    def __init__(self, config):\n        super(ProteinLSTMPooler, self).__init__()\n        self.scalar_reweighting = nn.Linear(2 * config.num_hidden_layers, 1)\n        self.dense = nn.Linear(config.hidden_size, config.hidden_size)\n        self.activation = nn.Tanh()\n\n    def forward(self, hidden_states):\n        pooled_output = self.scalar_reweighting(hidden_states).squeeze(2)\n        pooled_output = self.dense(pooled_output)\n        pooled_output = self.activation(pooled_output)\n        return pooled_output\n\n# Define the LSTM encoder for the ProteinLSTMModel\nclass ProteinLSTMEncoder(nn.Module):\n    def __init__(self, config):\n        super(ProteinLSTMEncoder, self).__init__()\n        forward_lstm = [ProteinLSTMLayer(config.input_size, config.hidden_size)]\n        reverse_lstm = [ProteinLSTMLayer(config.input_size, config.hidden_size)]\n        for _ in range(config.num_hidden_layers - 1):\n            forward_lstm.append(ProteinLSTMLayer(config.hidden_size, config.hidden_size, config.hidden_dropout_prob))\n            reverse_lstm.append(ProteinLSTMLayer(config.hidden_size, config.hidden_size, config.hidden_dropout_prob))\n        self.forward_lstm = nn.ModuleList(forward_lstm)\n        self.reverse_lstm = nn.ModuleList(reverse_lstm)\n        self.output_hidden_states = config.output_hidden_states\n\n    def forward(self, inputs, input_mask=None):\n        all_forward_pooled = []\n        all_reverse_pooled = []\n        all_hidden_states = []\n        forward_output = inputs\n        for layer in self.forward_lstm:\n            forward_output, forward_pooled = layer(forward_output)\n            all_forward_pooled.append(forward_pooled[0])\n            all_hidden_states.append(forward_output)\n\n        reversed_sequence = self.reverse_sequence(inputs, input_mask)\n        reverse_output = reversed_sequence\n        for layer in self.reverse_lstm:\n            reverse_output, reverse_pooled = layer(reverse_output)\n            all_reverse_pooled.append(reverse_pooled[0])\n            all_hidden_states.append(reverse_output)\n        reverse_output = self.reverse_sequence(reverse_output, input_mask)\n\n        output = torch.cat((forward_output, reverse_output), dim=2)\n        pooled = torch.stack(all_forward_pooled + all_reverse_pooled, 3).squeeze(0)\n        outputs = (output, pooled)\n        if self.output_hidden_states:\n            outputs = outputs + (all_hidden_states,)\n\n        return outputs  # sequence_embedding, pooled_embedding, (hidden_states)\n\n    def reverse_sequence(self, sequence, input_mask):\n        if input_mask is None:\n            idx = torch.arange(sequence.size(1) - 1, -1, -1)\n            reversed_sequence = sequence.index_select(1, idx)\n        else:\n            sequence_lengths = input_mask.sum(1)\n            reversed_sequence = []\n            for seq, seqlen in zip(sequence, sequence_lengths):\n                idx = torch.arange(seqlen - 1, -1, -1)\n                seq = seq.index_select(0, idx)\n                seq = F.pad(seq, [0, 0, 0, sequence.size(1) - seqlen])\n                reversed_sequence.append(seq)\n            reversed_sequence = torch.stack(reversed_sequence, 0)\n        return reversed_sequence\n\n# Define the configuration for the ProteinLSTMModel\nclass ProteinLSTMConfig:\n    def __init__(self, vocab_size=30, input_size=128, hidden_size=1024, num_hidden_layers=3,\n                 hidden_dropout_prob=0.1, initializer_range=0.02, output_hidden_states=True):\n        self.vocab_size = vocab_size\n        self.input_size = input_size\n        self.hidden_size = hidden_size\n        self.num_hidden_layers = num_hidden_layers\n        self.hidden_dropout_prob = hidden_dropout_prob\n        self.initializer_range = initializer_range\n        self.output_hidden_states = output_hidden_states\n\n# Define the ProteinLSTMModel\nclass ProteinLSTMModel(nn.Module):\n    def __init__(self, config):\n        super(ProteinLSTMModel, self).__init__()\n        self.embed_matrix = nn.Embedding(config.vocab_size, config.input_size)\n        self.encoder = ProteinLSTMEncoder(config)\n        self.pooler = ProteinLSTMPooler(config)\n        self.output_hidden_states = config.output_hidden_states\n\n    def forward(self, input_ids, input_mask=None):\n        if input_mask is None:\n            input_mask = torch.ones_like(input_ids, dtype=torch.long)\n\n        embedding_output = self.embed_matrix(input_ids)\n        outputs = self.encoder(embedding_output, input_mask=input_mask)\n        sequence_output = outputs[0]\n        pooled_outputs = self.pooler(outputs[1])\n\n        outputs = (sequence_output, pooled_outputs) + outputs[2:]\n        return outputs  # sequence_output, pooled_output, (hidden_states)\n\n# Define your input data here\ninput_data = torch.randint(0, 30, (601354, 139, 4), dtype=torch.long)  # Assuming input_data is of shape (601354, 139, 4)\n\n# Instantiate the model with your desired configuration\nmodel_config = ProteinLSTMConfig()\nmodel = ProteinLSTMModel(model_config)\n\n# Run the model on your input data\nwith torch.no_grad():\n    output = model(input_data)\n\n# 'output' now contains the model's predictions or embeddings\n","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.447444Z","iopub.status.idle":"2023-10-15T22:04:16.447912Z","shell.execute_reply.started":"2023-10-15T22:04:16.447723Z","shell.execute_reply":"2023-10-15T22:04:16.447743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\n\nclass ProteinLSTMConfig:\n    def __init__(self, vocab_size=30, input_size=128, hidden_size=1024, num_hidden_layers=3,\n                 hidden_dropout_prob=0.1, initializer_range=0.02, output_hidden_states=True):\n        self.vocab_size = vocab_size\n        self.input_size = input_size\n        self.hidden_size = hidden_size\n        self.num_hidden_layers = num_hidden_layers\n        self.hidden_dropout_prob = hidden_dropout_prob\n        self.initializer_range = initializer_range\n        self.output_hidden_states = output_hidden_states\n\nclass ProteinLSTMModel(nn.Module):\n    def __init__(self, config):\n        super(ProteinLSTMModel, self).__init__()\n        self.embed_matrix = nn.Embedding(config.vocab_size, config.input_size)\n        self.encoder = ProteinLSTMEncoder(config)\n        self.pooler = ProteinLSTMPooler(config)\n        self.output_hidden_states = config.output_hidden_states\n\n    def forward(self, input_ids, input_mask=None):\n        if input_mask is None:\n            input_mask = torch.ones_like(input_ids)\n\n        embedding_output = self.embed_matrix(input_ids)\n        outputs = self.encoder(embedding_output, input_mask=input_mask)\n        sequence_output = outputs[0]\n        pooled_outputs = self.pooler(outputs[1])\n\n        outputs = (sequence_output, pooled_outputs) + outputs[2:]\n        return outputs  # sequence_output, pooled_output, (hidden_states)\n\n# Define your input data here\ninput_data = torch.randint(0, 30, (601354, 139, 4), dtype=torch.long)  # Assuming input_data is of shape (601354, 139, 4)\n\n# Adjust the vocab_size in the configuration to match your data\nmodel_config = ProteinLSTMConfig(vocab_size=30)  # Adjust vocab_size as needed\n\n# Instantiate the model with the modified configuration\nmodel = ProteinLSTMModel(model_config)\n\n# Run the model on your input data\nwith torch.no_grad():\n    output = model(input_data)\n\n# 'output' now contains the model's predictions or embeddings\n","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.449642Z","iopub.status.idle":"2023-10-15T22:04:16.450098Z","shell.execute_reply.started":"2023-10-15T22:04:16.449917Z","shell.execute_reply":"2023-10-15T22:04:16.449936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\n\nclass ProteinLSTMConfig:\n    def __init__(self, vocab_size=30, input_size=128, hidden_size=1024, num_hidden_layers=3,\n                 hidden_dropout_prob=0.1, initializer_range=0.02, output_hidden_states=True):\n        self.vocab_size = vocab_size\n        self.input_size = input_size\n        self.hidden_size = hidden_size\n        self.num_hidden_layers = num_hidden_layers\n        self.hidden_dropout_prob = hidden_dropout_prob\n        self.initializer_range = initializer_range\n        self.output_hidden_states = output_hidden_states\n\nclass ProteinLSTMModel(nn.Module):\n    def __init__(self, config):\n        super(ProteinLSTMModel, self).__init__()\n        self.embed_matrix = nn.Embedding(config.vocab_size, config.input_size)\n        self.encoder = ProteinLSTMEncoder(config)\n        self.pooler = ProteinLSTMPooler(config)\n        self.output_hidden_states = config.output_hidden_states\n\n    def forward(self, input_ids, input_mask=None):\n        if input_mask is None:\n            input_mask = torch.ones_like(input_ids)\n\n        embedding_output = self.embed_matrix(input_ids)\n        outputs = self.encoder(embedding_output, input_mask=input_mask)\n        sequence_output = outputs[0]\n        pooled_outputs = self.pooler(outputs[1])\n\n        outputs = (sequence_output, pooled_outputs) + outputs[2:]\n        return outputs  # sequence_output, pooled_output, (hidden_states)\n\n# Define your input data here\ninput_data = torch.randint(0, 30, (601354, 139, 4), dtype=torch.long)  # Assuming input_data is of shape (601354, 139, 4)\n\n# Adjust the vocab_size in the configuration to match your data\nmodel_config = ProteinLSTMConfig(vocab_size=30)  # Adjust vocab_size as needed\n\n# Instantiate the model with the modified configuration\nmodel = ProteinLSTMModel(model_config)\n\n# Run the model on your input data\nwith torch.no_grad():\n    output = model(input_data)\n\n# 'output' now contains the model's predictions or embeddings\n","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.452294Z","iopub.status.idle":"2023-10-15T22:04:16.452772Z","shell.execute_reply.started":"2023-10-15T22:04:16.452547Z","shell.execute_reply":"2023-10-15T22:04:16.452566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\nfolder_to_delete = '/kaggle/working/X_encoded_2A3_SN_1.npy'\n\n# Check if the file exists before attempting to delete\nif os.path.exists(folder_to_delete):\n    try:\n        os.remove(folder_to_delete)  # This will delete the file\n        print(f\"File '{folder_to_delete}' has been deleted successfully.\")\n    except OSError as e:\n        print(f\"Error: {e}\")\nelse:\n    print(f\"File '{folder_to_delete}' does not exist.\")\n","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.453885Z","iopub.status.idle":"2023-10-15T22:04:16.454337Z","shell.execute_reply.started":"2023-10-15T22:04:16.454121Z","shell.execute_reply":"2023-10-15T22:04:16.45414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"model.build(input_shape = (139,4))","metadata":{"execution":{"iopub.status.busy":"2023-10-01T21:55:20.736935Z","iopub.execute_input":"2023-10-01T21:55:20.737577Z","iopub.status.idle":"2023-10-01T21:55:21.648478Z","shell.execute_reply.started":"2023-10-01T21:55:20.737532Z","shell.execute_reply":"2023-10-01T21:55:21.646793Z"}}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.build(input_shape = (139,4))","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.455685Z","iopub.status.idle":"2023-10-15T22:04:16.456096Z","shell.execute_reply.started":"2023-10-15T22:04:16.45592Z","shell.execute_reply":"2023-10-15T22:04:16.455938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.458322Z","iopub.status.idle":"2023-10-15T22:04:16.458962Z","shell.execute_reply.started":"2023-10-15T22:04:16.458724Z","shell.execute_reply":"2023-10-15T22:04:16.458751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tape_proteins","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.460857Z","iopub.status.idle":"2023-10-15T22:04:16.461324Z","shell.execute_reply.started":"2023-10-15T22:04:16.461127Z","shell.execute_reply":"2023-10-15T22:04:16.461145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv('/kaggle/working/train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.463175Z","iopub.status.idle":"2023-10-15T22:04:16.463931Z","shell.execute_reply.started":"2023-10-15T22:04:16.463486Z","shell.execute_reply":"2023-10-15T22:04:16.463531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.465402Z","iopub.status.idle":"2023-10-15T22:04:16.465847Z","shell.execute_reply.started":"2023-10-15T22:04:16.465634Z","shell.execute_reply":"2023-10-15T22:04:16.46565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_encoded_2A3 = np.load('/kaggle/working/X_encoded_2A3.npy')","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.468282Z","iopub.status.idle":"2023-10-15T22:04:16.468733Z","shell.execute_reply.started":"2023-10-15T22:04:16.468529Z","shell.execute_reply":"2023-10-15T22:04:16.468546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_encoded_2A3 = X_encoded_2A3[:,27:166,:]","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.470408Z","iopub.status.idle":"2023-10-15T22:04:16.470856Z","shell.execute_reply.started":"2023-10-15T22:04:16.470645Z","shell.execute_reply":"2023-10-15T22:04:16.47068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.savez_compressed('X_encoded_2A3_refined.npz', X_encoded_2A3_refined)","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.471801Z","iopub.status.idle":"2023-10-15T22:04:16.472883Z","shell.execute_reply.started":"2023-10-15T22:04:16.472583Z","shell.execute_reply":"2023-10-15T22:04:16.472608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport zipfile\n\n# Assuming you have the DataFrame y_2A3\n# Define the CSV file name\ncsv_filename = 'y_2A3.csv'\n\n\n# Define the ZIP file name\nzip_filename = 'y_2A3.zip'\n\n# Create a ZIP archive containing the CSV file\nwith zipfile.ZipFile(zip_filename, 'w', zipfile.ZIP_DEFLATED) as zipf:\n    zipf.write(csv_filename, arcname='y_2A3.csv')  # arcname sets the name of the file within the ZIP archive\n","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.474331Z","iopub.status.idle":"2023-10-15T22:04:16.474735Z","shell.execute_reply.started":"2023-10-15T22:04:16.47455Z","shell.execute_reply":"2023-10-15T22:04:16.474566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_2A3.to_csv('y_2A3.csv')","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.476009Z","iopub.status.idle":"2023-10-15T22:04:16.476815Z","shell.execute_reply.started":"2023-10-15T22:04:16.476593Z","shell.execute_reply":"2023-10-15T22:04:16.476614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_encoded_2A3.shape","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.478094Z","iopub.status.idle":"2023-10-15T22:04:16.478455Z","shell.execute_reply.started":"2023-10-15T22:04:16.478292Z","shell.execute_reply":"2023-10-15T22:04:16.478308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a = []\nfor i in range (605394):\n    a.append(len(data['sequence'].loc[i]))","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.480362Z","iopub.status.idle":"2023-10-15T22:04:16.481289Z","shell.execute_reply.started":"2023-10-15T22:04:16.481018Z","shell.execute_reply":"2023-10-15T22:04:16.481045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"min(a)","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.483164Z","iopub.status.idle":"2023-10-15T22:04:16.483597Z","shell.execute_reply.started":"2023-10-15T22:04:16.483408Z","shell.execute_reply":"2023-10-15T22:04:16.483425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_encoded_2A3 = X_encoded_2A3[:, 27:166, :]","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.485626Z","iopub.status.idle":"2023-10-15T22:04:16.486094Z","shell.execute_reply.started":"2023-10-15T22:04:16.485909Z","shell.execute_reply":"2023-10-15T22:04:16.485929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.487339Z","iopub.status.idle":"2023-10-15T22:04:16.487751Z","shell.execute_reply.started":"2023-10-15T22:04:16.487541Z","shell.execute_reply":"2023-10-15T22:04:16.487557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_encoded_2A3.shape","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.489653Z","iopub.status.idle":"2023-10-15T22:04:16.490137Z","shell.execute_reply.started":"2023-10-15T22:04:16.489942Z","shell.execute_reply":"2023-10-15T22:04:16.489961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_2A3 = data[data.experiment_type == '2A3_MaP']","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.492524Z","iopub.status.idle":"2023-10-15T22:04:16.492951Z","shell.execute_reply.started":"2023-10-15T22:04:16.49277Z","shell.execute_reply":"2023-10-15T22:04:16.492788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del(data)","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.494705Z","iopub.status.idle":"2023-10-15T22:04:16.495164Z","shell.execute_reply.started":"2023-10-15T22:04:16.494978Z","shell.execute_reply":"2023-10-15T22:04:16.494996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.496772Z","iopub.status.idle":"2023-10-15T22:04:16.497157Z","shell.execute_reply.started":"2023-10-15T22:04:16.496988Z","shell.execute_reply":"2023-10-15T22:04:16.497005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_2A3.sequence.loc[7][27:166]","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.498626Z","iopub.status.idle":"2023-10-15T22:04:16.499054Z","shell.execute_reply.started":"2023-10-15T22:04:16.498878Z","shell.execute_reply":"2023-10-15T22:04:16.498896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tape-train lstm /kaggle/working/output.fasta /kaggle/working/lstm.npz --from_pretrained --tokenizer iupac ","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.500929Z","iopub.status.idle":"2023-10-15T22:04:16.501335Z","shell.execute_reply.started":"2023-10-15T22:04:16.501155Z","shell.execute_reply":"2023-10-15T22:04:16.501171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tape-train\\\n    --model_config_file /kaggle/working/lstm.json\\\n    --tokenizer iupac \\\n    --batch_size 256 \\\n    --num_train_epochs 10 \\\n    --learning_rate 0.001 \\\n    --gradient_accumulation_steps 4 \\\n    --warmup_steps 1000 \\\n    --output_dir /kaggle/working/\\\n    language_modeling\n    ","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.503121Z","iopub.status.idle":"2023-10-15T22:04:16.503504Z","shell.execute_reply.started":"2023-10-15T22:04:16.503334Z","shell.execute_reply":"2023-10-15T22:04:16.503351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!wget http://s3.amazonaws.com/songlabdata/proteindata/data_pytorch/pfam.tar.gz","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.505267Z","iopub.status.idle":"2023-10-15T22:04:16.505657Z","shell.execute_reply.started":"2023-10-15T22:04:16.505479Z","shell.execute_reply":"2023-10-15T22:04:16.505496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir /kaggle/working/data\n!mv /kaggle/working/pfam.tar.gz /kaggle/working/data/pfam.tar.gz","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.507507Z","iopub.status.idle":"2023-10-15T22:04:16.507976Z","shell.execute_reply.started":"2023-10-15T22:04:16.507778Z","shell.execute_reply":"2023-10-15T22:04:16.507798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('lstm.json','w') as lstm:\n    lstm.write(\"\"\"\n    {\n    \"vocab_size\": 30,\n    \"input_size\": 139,\n    \"hidden_size\": 128,\n    \"num_hidden_layers\": 3,\n    \"hidden_dropout_prob\": 0.5,\n    \"initializer_range\": 0.02\n}\n\n    \"\"\")","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.509542Z","iopub.status.idle":"2023-10-15T22:04:16.510024Z","shell.execute_reply.started":"2023-10-15T22:04:16.509831Z","shell.execute_reply":"2023-10-15T22:04:16.509851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T23:00:54.289202Z","iopub.execute_input":"2023-10-15T23:00:54.289713Z","iopub.status.idle":"2023-10-15T23:00:55.232979Z","shell.execute_reply.started":"2023-10-15T23:00:54.289677Z","shell.execute_reply":"2023-10-15T23:00:55.231162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/stanford-ribonanza-rna-folding/test_sequences.csv')","metadata":{"execution":{"iopub.status.busy":"2023-10-15T23:01:26.285616Z","iopub.execute_input":"2023-10-15T23:01:26.287431Z","iopub.status.idle":"2023-10-15T23:01:35.555336Z","shell.execute_reply.started":"2023-10-15T23:01:26.287372Z","shell.execute_reply":"2023-10-15T23:01:35.553657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n%%time\nimport time\nN = 335823\nL = 206\nX_submit = np.zeros( (N,L*4), np.int8 )\nt0 = time.time()\nfor k in range(N):\n    seq = dft['sequence'].iat[k]\n    L = len(seq)\n    l =  np.array( [rna_onehot_dict[seq[II]] for II in range(L) ] ).ravel()\n    X_submit[k,:len(l)] = l\n    if (k%50_000 == 0): print(k, '%.1f'%(time.time() - t0 ) )\nX_submit        \n\n","metadata":{"execution":{"iopub.status.busy":"2023-10-15T23:21:08.610502Z","iopub.execute_input":"2023-10-15T23:21:08.611065Z","iopub.status.idle":"2023-10-15T23:21:09.445328Z","shell.execute_reply.started":"2023-10-15T23:21:08.611031Z","shell.execute_reply":"2023-10-15T23:21:09.443922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_submit[1]","metadata":{"execution":{"iopub.status.busy":"2023-10-15T23:17:01.10974Z","iopub.execute_input":"2023-10-15T23:17:01.110157Z","iopub.status.idle":"2023-10-15T23:17:01.121332Z","shell.execute_reply.started":"2023-10-15T23:17:01.11013Z","shell.execute_reply":"2023-10-15T23:17:01.119745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.sequence","metadata":{"execution":{"iopub.status.busy":"2023-10-15T23:12:06.49139Z","iopub.execute_input":"2023-10-15T23:12:06.491894Z","iopub.status.idle":"2023-10-15T23:12:06.50261Z","shell.execute_reply.started":"2023-10-15T23:12:06.491863Z","shell.execute_reply":"2023-10-15T23:12:06.50151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tape-train \\\n    -h","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.512306Z","iopub.status.idle":"2023-10-15T22:04:16.512826Z","shell.execute_reply.started":"2023-10-15T22:04:16.512569Z","shell.execute_reply":"2023-10-15T22:04:16.512599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tape-embed --debug","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.514575Z","iopub.status.idle":"2023-10-15T22:04:16.515166Z","shell.execute_reply.started":"2023-10-15T22:04:16.514964Z","shell.execute_reply":"2023-10-15T22:04:16.514985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rna_to_fasta_file(rna_sequence, sequence_name, output_file):\n    with open(output_file, \"w\") as fasta_file:\n        fasta_file.write(f\">{sequence_name}\\n{rna_sequence}\\n\")\n\n# Example usage\nrna_sequence = data_2A3.sequence.loc[7][27:166]\nsequence_name = \"RNA Sequence 1\"\noutput_file = \"output.fasta\"\n\nrna_to_fasta_file(rna_sequence, sequence_name, output_file)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.517365Z","iopub.status.idle":"2023-10-15T22:04:16.517986Z","shell.execute_reply.started":"2023-10-15T22:04:16.51772Z","shell.execute_reply":"2023-10-15T22:04:16.517758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_encoded_2A3[7]","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.519755Z","iopub.status.idle":"2023-10-15T22:04:16.520209Z","shell.execute_reply.started":"2023-10-15T22:04:16.520005Z","shell.execute_reply":"2023-10-15T22:04:16.520025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del(data)","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.521602Z","iopub.status.idle":"2023-10-15T22:04:16.522174Z","shell.execute_reply.started":"2023-10-15T22:04:16.521975Z","shell.execute_reply":"2023-10-15T22:04:16.521996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_2A3 = data[data.experiment_type == '2A3_MaP' ]","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.524711Z","iopub.status.idle":"2023-10-15T22:04:16.525154Z","shell.execute_reply.started":"2023-10-15T22:04:16.524968Z","shell.execute_reply":"2023-10-15T22:04:16.524989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\ndef ONE_HOT(sequence, max_length=206):\n    aa_alphabet = 'AUGC'\n    \n    # Create a mapping from amino acids to integers\n    aa_dict = {aa: i for i, aa in enumerate(aa_alphabet)}\n\n    # Ensure the sequence is no longer than max_length\n    sequence = sequence[:max_length]\n\n    # Convert the sequence to a list of integer indices\n    aa_indices = [aa_dict.get(aa, len(aa_alphabet)) for aa in sequence]\n\n    # Create a one-hot encoding using tf.one_hot\n    one_hot_sequence = tf.one_hot(aa_indices, depth=len(aa_alphabet))\n\n    # Pad the one-hot encoding with zeros to reach max_length\n    pad_length = max_length - tf.shape(one_hot_sequence)[0]\n    padding = tf.zeros((pad_length, len(aa_alphabet)), dtype=tf.float32)\n    one_hot_sequence = tf.concat([one_hot_sequence, padding], axis=0)\n\n    return one_hot_sequence","metadata":{"execution":{"iopub.status.busy":"2023-10-18T07:16:49.624032Z","iopub.execute_input":"2023-10-18T07:16:49.624599Z","iopub.status.idle":"2023-10-18T07:17:01.054216Z","shell.execute_reply.started":"2023-10-18T07:16:49.624566Z","shell.execute_reply":"2023-10-18T07:17:01.053349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\n\n# Load your sequence data as strings\nX = data_DMS.sequence\n\n# Preprocess the sequence data using your ONE_HOT function with tqdm\nX_encoded_2A = []\nfor sequence in tqdm(X, desc=\"Encoding sequences\"):\n    encoded_sequence = ONE_HOT(sequence)\n    X_encoded_2A.append(encoded_sequence)\n\n# Convert the list of encoded sequences to a numpy array\n#import numpy as np\nX_encoded_DMS = np.array(X_encoded_2A)\n\n# Split your data into training and testing sets\n#X_train, X_test, y_train, y_test = train_test_split(X_encoded, y, test_size=0.2, random_state=42)\n\n# ... Rest of your model and training code remains the same ...\n","metadata":{"execution":{"iopub.status.busy":"2023-10-18T07:21:57.617682Z","iopub.execute_input":"2023-10-18T07:21:57.618101Z","iopub.status.idle":"2023-10-18T07:25:55.637463Z","shell.execute_reply.started":"2023-10-18T07:21:57.618075Z","shell.execute_reply":"2023-10-18T07:25:55.636577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.save('X_encoded_DMS.npy',X_encoded_DMS)\nnp.save('X_encoded_2A3.npy',X_encoded_2A3)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-18T07:29:00.652554Z","iopub.execute_input":"2023-10-18T07:29:00.652919Z","iopub.status.idle":"2023-10-18T07:29:03.712219Z","shell.execute_reply.started":"2023-10-18T07:29:00.652891Z","shell.execute_reply":"2023-10-18T07:29:03.711084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_2A3.to_csv('y_2A3.csv'.index = False)\ny_DMS.to_csv('y_DMS.csv'.index = False)","metadata":{"execution":{"iopub.status.busy":"2023-10-18T07:29:48.231022Z","iopub.execute_input":"2023-10-18T07:29:48.231366Z","iopub.status.idle":"2023-10-18T07:29:48.237232Z","shell.execute_reply.started":"2023-10-18T07:29:48.231339Z","shell.execute_reply":"2023-10-18T07:29:48.235937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\n\n# Load your sequence data as strings\nX = seq_X_DMS\n\n# Preprocess the sequence data using your ONE_HOT function with tqdm\nX_encoded_2A = []\nfor sequence in tqdm(X, desc=\"Encoding sequences\"):\n    encoded_sequence = ONE_HOT(sequence)\n    X_encoded_2A.append(encoded_sequence)\n\n# Convert the list of encoded sequences to a numpy array\n#import numpy as np\nX_encoded_DMS = np.array(X_encoded_2A)\n\n# Split your data into training and testing sets\n#X_train, X_test, y_train, y_test = train_test_split(X_encoded, y, test_size=0.2, random_state=42)\n\n# ... Rest of your model and training code remains the same ...\n","metadata":{"execution":{"iopub.status.busy":"2023-10-15T23:29:43.208303Z","iopub.execute_input":"2023-10-15T23:29:43.208969Z","iopub.status.idle":"2023-10-15T23:33:59.984021Z","shell.execute_reply.started":"2023-10-15T23:29:43.208924Z","shell.execute_reply":"2023-10-15T23:33:59.982592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_encoded_DMS.shape","metadata":{"execution":{"iopub.status.busy":"2023-10-15T23:37:45.671655Z","iopub.execute_input":"2023-10-15T23:37:45.672087Z","iopub.status.idle":"2023-10-15T23:37:45.680345Z","shell.execute_reply.started":"2023-10-15T23:37:45.672051Z","shell.execute_reply":"2023-10-15T23:37:45.679074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_encoded_2A3 = X_encoded_2A3[:,27:166,:]","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:56:29.209537Z","iopub.execute_input":"2023-10-15T22:56:29.210053Z","iopub.status.idle":"2023-10-15T22:56:29.216556Z","shell.execute_reply.started":"2023-10-15T22:56:29.210021Z","shell.execute_reply":"2023-10-15T22:56:29.214925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_encoded_DMS = X_encoded_DMS[:,27:166,:]","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:57:57.730111Z","iopub.execute_input":"2023-10-15T22:57:57.730601Z","iopub.status.idle":"2023-10-15T22:57:57.738098Z","shell.execute_reply.started":"2023-10-15T22:57:57.730572Z","shell.execute_reply":"2023-10-15T22:57:57.73649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.save('X_encoded_2A3.npy',X_encoded_2A3)\nnp.save('X_encoded_DMS.npy',X_encoded_DMS)","metadata":{"execution":{"iopub.status.busy":"2023-10-15T23:38:04.846773Z","iopub.execute_input":"2023-10-15T23:38:04.847302Z","iopub.status.idle":"2023-10-15T23:38:07.419362Z","shell.execute_reply.started":"2023-10-15T23:38:04.847265Z","shell.execute_reply":"2023-10-15T23:38:07.41764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rna_onehot_dict = {'A': [1,0,0,0], 'U': [0,1,0,0], 'G':[0,0,1,0], 'C': [0,0,0,1]}","metadata":{"execution":{"iopub.status.busy":"2023-10-15T23:46:40.304651Z","iopub.execute_input":"2023-10-15T23:46:40.305644Z","iopub.status.idle":"2023-10-15T23:46:40.31231Z","shell.execute_reply.started":"2023-10-15T23:46:40.305606Z","shell.execute_reply":"2023-10-15T23:46:40.310544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nN = 335823\nL = 206\nX_submit = np.zeros( (N,L*4), np.int8 )\nt0 = time.time()\nfor k in range(N):\n    seq = test['sequence'].iat[k]\n    L = len(seq)\n    l =  np.array( [rna_onehot_dict[seq[II]] for II in range(L) ] ).ravel()\n    X_submit[k,:len(l)] = l\n    if (k%50_000 == 0): print(k, '%.1f'%(time.time() - t0 ) )\nX_submit     ","metadata":{"execution":{"iopub.status.busy":"2023-10-15T23:46:44.464568Z","iopub.execute_input":"2023-10-15T23:46:44.465027Z","iopub.status.idle":"2023-10-15T23:47:27.90983Z","shell.execute_reply.started":"2023-10-15T23:46:44.464996Z","shell.execute_reply":"2023-10-15T23:47:27.908374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.save('X_submit.npy',X_submit)","metadata":{"execution":{"iopub.status.busy":"2023-10-15T23:48:06.086179Z","iopub.execute_input":"2023-10-15T23:48:06.086816Z","iopub.status.idle":"2023-10-15T23:48:06.459532Z","shell.execute_reply.started":"2023-10-15T23:48:06.086767Z","shell.execute_reply":"2023-10-15T23:48:06.458073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_2A3_SN_1","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:50:23.057955Z","iopub.execute_input":"2023-10-15T22:50:23.058462Z","iopub.status.idle":"2023-10-15T22:50:23.135303Z","shell.execute_reply.started":"2023-10-15T22:50:23.058431Z","shell.execute_reply":"2023-10-15T22:50:23.133574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.loc[['reactivity_23'],1]","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:39:39.336112Z","iopub.execute_input":"2023-10-15T22:39:39.336615Z","iopub.status.idle":"2023-10-15T22:39:39.821859Z","shell.execute_reply.started":"2023-10-15T22:39:39.336585Z","shell.execute_reply":"2023-10-15T22:39:39.82001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_2A3_SN_1","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:53:38.514829Z","iopub.execute_input":"2023-10-15T22:53:38.515354Z","iopub.status.idle":"2023-10-15T22:53:38.563591Z","shell.execute_reply.started":"2023-10-15T22:53:38.515315Z","shell.execute_reply":"2023-10-15T22:53:38.561924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.538179Z","iopub.status.idle":"2023-10-15T22:04:16.538749Z","shell.execute_reply.started":"2023-10-15T22:04:16.538471Z","shell.execute_reply":"2023-10-15T22:04:16.5385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_encoded_2A3 = np.load('/kaggle/working/X_encoded_2A3.npy')","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.540596Z","iopub.status.idle":"2023-10-15T22:04:16.541501Z","shell.execute_reply.started":"2023-10-15T22:04:16.541115Z","shell.execute_reply":"2023-10-15T22:04:16.541147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.543175Z","iopub.status.idle":"2023-10-15T22:04:16.543618Z","shell.execute_reply.started":"2023-10-15T22:04:16.543424Z","shell.execute_reply":"2023-10-15T22:04:16.543445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Identify and select columns with 'error' in their name\ncolumns = [col for col in data_full.columns if 'reactivity' in col.lower()]\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.545812Z","iopub.status.idle":"2023-10-15T22:04:16.546275Z","shell.execute_reply.started":"2023-10-15T22:04:16.546073Z","shell.execute_reply":"2023-10-15T22:04:16.546096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_2A3.reactivity_0029.loc[7]","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.548143Z","iopub.status.idle":"2023-10-15T22:04:16.548558Z","shell.execute_reply.started":"2023-10-15T22:04:16.548377Z","shell.execute_reply":"2023-10-15T22:04:16.548396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.549851Z","iopub.status.idle":"2023-10-15T22:04:16.550273Z","shell.execute_reply.started":"2023-10-15T22:04:16.550091Z","shell.execute_reply":"2023-10-15T22:04:16.550111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del(data)","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.552285Z","iopub.status.idle":"2023-10-15T22:04:16.553114Z","shell.execute_reply.started":"2023-10-15T22:04:16.552753Z","shell.execute_reply":"2023-10-15T22:04:16.552786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_2A3.loc[7]","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.554915Z","iopub.status.idle":"2023-10-15T22:04:16.555511Z","shell.execute_reply.started":"2023-10-15T22:04:16.555214Z","shell.execute_reply":"2023-10-15T22:04:16.555241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del(data_2A3)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.557444Z","iopub.status.idle":"2023-10-15T22:04:16.558044Z","shell.execute_reply.started":"2023-10-15T22:04:16.557763Z","shell.execute_reply":"2023-10-15T22:04:16.55779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = data_2A3.sequence","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.56051Z","iopub.status.idle":"2023-10-15T22:04:16.560967Z","shell.execute_reply.started":"2023-10-15T22:04:16.560775Z","shell.execute_reply":"2023-10-15T22:04:16.560794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_2A3 = fiy_2A3","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.56327Z","iopub.status.idle":"2023-10-15T22:04:16.56393Z","shell.execute_reply.started":"2023-10-15T22:04:16.563599Z","shell.execute_reply":"2023-10-15T22:04:16.563628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.565223Z","iopub.status.idle":"2023-10-15T22:04:16.565824Z","shell.execute_reply.started":"2023-10-15T22:04:16.565521Z","shell.execute_reply":"2023-10-15T22:04:16.565547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_encoded_2A3.shape","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.568158Z","iopub.status.idle":"2023-10-15T22:04:16.568766Z","shell.execute_reply.started":"2023-10-15T22:04:16.568461Z","shell.execute_reply":"2023-10-15T22:04:16.568489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.570918Z","iopub.status.idle":"2023-10-15T22:04:16.571488Z","shell.execute_reply.started":"2023-10-15T22:04:16.571198Z","shell.execute_reply":"2023-10-15T22:04:16.571224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract embeddings from the LSTM layer\nembedding_model = tf.keras.Model(inputs=model.input, outputs=model.layers[0].output)\nembeddings = embedding_model.predict(X_reshaped)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.573093Z","iopub.status.idle":"2023-10-15T22:04:16.573879Z","shell.execute_reply.started":"2023-10-15T22:04:16.573379Z","shell.execute_reply":"2023-10-15T22:04:16.573413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.layers import Embedding, LSTM, Dense\nnum_unique_characters = 4\nembedding_dim = 139\n\n# Assuming X_encoded_2A3 has shape (605394, 139, 4)\nX_reshaped = X_encoded_2A3.transpose(0, 2, 1)\n\n# Define the model\nmodel = tf.keras.Sequential([\n    LSTM(units=139, activation='tanh', input_shape=(4, 139)),  # Adjust input_shape\n    \n     # Adjust the output dimension as needed\n])\n\n# Compile the model with an appropriate loss function and optimizer\nmodel.compile(optimizer='adam', loss='mean_squared_error', metrics=['mae'])\n\n# Train the model on your RNA data\nmodel.fit(X_reshaped, y_2A3, epochs=20, batch_size=512)\n\n# Extract embeddings from the LSTM layer\nembedding_model = tf.keras.Model(inputs=model.input, outputs=model.layers[1].output)\nembeddings = embedding_model.predict(X_reshaped)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.575293Z","iopub.status.idle":"2023-10-15T22:04:16.575871Z","shell.execute_reply.started":"2023-10-15T22:04:16.575569Z","shell.execute_reply":"2023-10-15T22:04:16.575596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.577323Z","iopub.status.idle":"2023-10-15T22:04:16.577896Z","shell.execute_reply.started":"2023-10-15T22:04:16.577591Z","shell.execute_reply":"2023-10-15T22:04:16.577616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embeddings.shape","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.579393Z","iopub.status.idle":"2023-10-15T22:04:16.579974Z","shell.execute_reply.started":"2023-10-15T22:04:16.579687Z","shell.execute_reply":"2023-10-15T22:04:16.579714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_2A3 = y_2A3.fillna(0)","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.582151Z","iopub.status.idle":"2023-10-15T22:04:16.582732Z","shell.execute_reply.started":"2023-10-15T22:04:16.582434Z","shell.execute_reply":"2023-10-15T22:04:16.582461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.layers import Conv1D, Flatten, Dense, LSTM, Input, concatenate\n\n# Define the CNN model\ncnn_model = tf.keras.Sequential([\n    Conv1D(64, 3, activation='relu', padding='same', input_shape=(206, 4)),\n    Flatten(),\n    \n])\n\n# Define the LSTM model\nlstm_model = tf.keras.Sequential([\n    LSTM(units=64, activation='relu', input_shape=(206, 4)),\n    Dense(128, activation='relu')\n])\n\n# Combine both models\ninput_layer = Input(shape=(206, 4))  # Adjust input shape\ncnn_output = cnn_model(input_layer)\nlstm_output = lstm_model(input_layer)  # Use the same input layer for LSTM\ncombined_layer = concatenate([cnn_output, lstm_output])\noutput_layer = Dense(206, activation='linear')(combined_layer)\n\ncombined_model_2A3 = tf.keras.Model(inputs=input_layer, outputs=output_layer)\n\n# Compile the combined model\ncombined_model_2A3.compile(optimizer='adam', loss='mean_squared_error', metrics=['mae'])\n\n# Train the model with input data X_encoded_2A3 and y_2A3\ncombined_model_2A3.fit(X_2A3, y_2A3, epochs=20, batch_size=512)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-18T08:02:59.62222Z","iopub.execute_input":"2023-10-18T08:02:59.622529Z","iopub.status.idle":"2023-10-18T08:29:25.392339Z","shell.execute_reply.started":"2023-10-18T08:02:59.622507Z","shell.execute_reply":"2023-10-18T08:29:25.39155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.layers import Conv1D, Flatten, Dense, LSTM, Input, concatenate\n\n# Define the CNN model\ncnn_model = tf.keras.Sequential([\n    Conv1D(64, 3, activation='relu', padding='same', input_shape=(206, 4)),\n    Flatten(),\n    \n])\n\n# Define the LSTM model\nlstm_model = tf.keras.Sequential([\n    LSTM(units=64, activation='relu', input_shape=(206, 4)),\n    Dense(128, activation='relu')\n])\n\n# Combine both models\ninput_layer = Input(shape=(206, 4))  # Adjust input shape\ncnn_output = cnn_model(input_layer)\nlstm_output = lstm_model(input_layer)  # Use the same input layer for LSTM\ncombined_layer = concatenate([cnn_output, lstm_output])\noutput_layer = Dense(206, activation='linear')(combined_layer)\n\ncombined_model_DMS = tf.keras.Model(inputs=input_layer, outputs=output_layer)\n\n# Compile the combined model\ncombined_model_DMS.compile(optimizer='adam', loss='mean_squared_error', metrics=['mae'])\n\n# Train the model with input data X_encoded_2A3 and y_2A3\ncombined_model_DMS.fit(X_DMS, y_DMS, epochs=20, batch_size=512)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-18T08:32:02.484556Z","iopub.execute_input":"2023-10-18T08:32:02.485126Z","iopub.status.idle":"2023-10-18T08:59:28.375349Z","shell.execute_reply.started":"2023-10-18T08:32:02.485098Z","shell.execute_reply":"2023-10-18T08:59:28.374543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-10-18T09:03:43.594289Z","iopub.execute_input":"2023-10-18T09:03:43.595157Z","iopub.status.idle":"2023-10-18T09:03:43.955132Z","shell.execute_reply.started":"2023-10-18T09:03:43.595124Z","shell.execute_reply":"2023-10-18T09:03:43.954538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_2A3 = y_2A3.fillna(0)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-18T07:59:58.701693Z","iopub.execute_input":"2023-10-18T07:59:58.702269Z","iopub.status.idle":"2023-10-18T07:59:58.843911Z","shell.execute_reply.started":"2023-10-18T07:59:58.702219Z","shell.execute_reply":"2023-10-18T07:59:58.843236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_DMS = y_DMS.fillna(0)","metadata":{"execution":{"iopub.status.busy":"2023-10-18T08:00:34.143496Z","iopub.execute_input":"2023-10-18T08:00:34.1438Z","iopub.status.idle":"2023-10-18T08:00:34.310416Z","shell.execute_reply.started":"2023-10-18T08:00:34.143777Z","shell.execute_reply":"2023-10-18T08:00:34.309644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#LSTM and CNN\n\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Embedding, LSTM, Dense, Conv1D, MaxPooling1D, Flatten, Input, concatenate\nfrom tensorflow.keras.layers import Input, Reshape\n\n# Assuming X_encoded_2A3 has shape (605394, 128)\n# Reshape it to (605394, 128, 1) to match the LSTM input shape\n# Assuming X_encoded_2A3 has shape (605394, 128)\n# Reshape it to (605394, 128, 1) to match the LSTM input shape\nX_reshaped = tf.reshape(X_encoded_2A3, shape=(-1, 128, 1))\n\n\n# Define the CNN model\ncnn_model = Sequential([\n    Conv1D(64, 3, activation='relu', padding='same', input_shape=(128, 1)),\n    Flatten(),\n    Dense(64, activation='relu')\n])\n\n# Define the LSTM model\nlstm_model = Sequential([\n    LSTM(units=64, activation='relu', input_shape=(128,)),\n    Dense(128, activation='relu')\n])\n\n# Combine both models\ncombined_model = Sequential([\n    Input(shape=(128,)),  # Adjust input shape\n    cnn_model,\n    lstm_model,\n    Dense(139, activation='linear')  # Output layer with linear activation (139 outputs)\n])\n\n# Compile the combined model\ncombined_model.compile(optimizer='adam', loss='mean_squared_error', metrics=['mae'])\n\n# Train the model with reshaped input data X_reshaped and y_2A3\ncombined_model.fit(X_reshaped, y_2A3, epochs=20, batch_size=512)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.58842Z","iopub.status.idle":"2023-10-15T22:04:16.589065Z","shell.execute_reply.started":"2023-10-15T22:04:16.58881Z","shell.execute_reply":"2023-10-15T22:04:16.588833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_reshaped = np.expand_dims(X_encoded_2A3, axis=-1)","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.591257Z","iopub.status.idle":"2023-10-15T22:04:16.591859Z","shell.execute_reply.started":"2023-10-15T22:04:16.59159Z","shell.execute_reply":"2023-10-15T22:04:16.591623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#LSTM FOR EMBEDDING\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Embedding, LSTM, Dense\nnum_unique_characters = 4\nembedding_dim = 139\n# Define the model\n# Define the model with the correct input shape\nmodel = tf.keras.Sequential([\n    LSTM(units=64, activation='relu', input_shape=(139, 4)),  # Adjust input_shape\n    Dense(128, activation='relu'),\n    Dense(embedding_dim, activation='linear')  # Adjust the output dimension as needed\n])\n\n\n# Compile the model with an appropriate loss function and optimizer\nmodel.compile(optimizer='adam', loss='mean_squared_error', metrics=['mae'])\n\n# Train the model on your RNA data\nmodel.fit(X_encoded_2A3, y_2A3, epochs=20, batch_size=512)\n\n# Extract embeddings from the LSTM layer\nembedding_model = tf.keras.Model(inputs=model.input, outputs=model.layers[1].output)\nembeddings = embedding_model.predict(X_encoded_2A3)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.59453Z","iopub.status.idle":"2023-10-15T22:04:16.595044Z","shell.execute_reply.started":"2023-10-15T22:04:16.594845Z","shell.execute_reply":"2023-10-15T22:04:16.594863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embedding_model = tf.keras.Model(inputs=model.input, outputs=model.layers[1].output)\nembeddings = embedding_model.predict(X_encoded_2A3)","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.597135Z","iopub.status.idle":"2023-10-15T22:04:16.597612Z","shell.execute_reply.started":"2023-10-15T22:04:16.597418Z","shell.execute_reply":"2023-10-15T22:04:16.597436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X_encoded_2A3, y_2A3, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.599593Z","iopub.status.idle":"2023-10-15T22:04:16.600116Z","shell.execute_reply.started":"2023-10-15T22:04:16.599914Z","shell.execute_reply":"2023-10-15T22:04:16.599934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.save('embeddings.npy',embeddings)","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.60136Z","iopub.status.idle":"2023-10-15T22:04:16.601776Z","shell.execute_reply.started":"2023-10-15T22:04:16.601577Z","shell.execute_reply":"2023-10-15T22:04:16.601593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del(X_encoded_2A3)\ndel(y_2A3)","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.602903Z","iopub.status.idle":"2023-10-15T22:04:16.603268Z","shell.execute_reply.started":"2023-10-15T22:04:16.603104Z","shell.execute_reply":"2023-10-15T22:04:16.60312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.604451Z","iopub.status.idle":"2023-10-15T22:04:16.604841Z","shell.execute_reply.started":"2023-10-15T22:04:16.604643Z","shell.execute_reply":"2023-10-15T22:04:16.604676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test.iloc[1]","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.606573Z","iopub.status.idle":"2023-10-15T22:04:16.607036Z","shell.execute_reply.started":"2023-10-15T22:04:16.606844Z","shell.execute_reply":"2023-10-15T22:04:16.606863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_2A3 = y_2A3.fillna(0)","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.610347Z","iopub.status.idle":"2023-10-15T22:04:16.610862Z","shell.execute_reply.started":"2023-10-15T22:04:16.610604Z","shell.execute_reply":"2023-10-15T22:04:16.610625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = y_train.fillna(0)\n\n# Replace NaN values with zeros in y_test\ny_test = y_test.fillna(0)","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.612947Z","iopub.status.idle":"2023-10-15T22:04:16.613388Z","shell.execute_reply.started":"2023-10-15T22:04:16.613203Z","shell.execute_reply":"2023-10-15T22:04:16.613221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.614801Z","iopub.status.idle":"2023-10-15T22:04:16.615218Z","shell.execute_reply.started":"2023-10-15T22:04:16.615034Z","shell.execute_reply":"2023-10-15T22:04:16.615051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Calculate the number of columns (features) in y_train\nnum_outputs = y_train.shape[-1]\n\n# Add the output layer with the calculated number of units\n\n","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.618145Z","iopub.status.idle":"2023-10-15T22:04:16.618583Z","shell.execute_reply.started":"2023-10-15T22:04:16.618402Z","shell.execute_reply":"2023-10-15T22:04:16.61842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.621154Z","iopub.status.idle":"2023-10-15T22:04:16.621656Z","shell.execute_reply.started":"2023-10-15T22:04:16.62145Z","shell.execute_reply":"2023-10-15T22:04:16.62147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.623307Z","iopub.status.idle":"2023-10-15T22:04:16.623769Z","shell.execute_reply.started":"2023-10-15T22:04:16.62356Z","shell.execute_reply":"2023-10-15T22:04:16.623579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming your array is named encoded_RNAs\nX_train = X_train[:, 27:, :]\nX_test = X_test[:, 27:, :]\n\n# Verify the new shape\nprint(X_train.shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.62498Z","iopub.status.idle":"2023-10-15T22:04:16.625392Z","shell.execute_reply.started":"2023-10-15T22:04:16.625213Z","shell.execute_reply":"2023-10-15T22:04:16.62523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_2A3","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.627459Z","iopub.status.idle":"2023-10-15T22:04:16.627956Z","shell.execute_reply.started":"2023-10-15T22:04:16.627754Z","shell.execute_reply":"2023-10-15T22:04:16.627776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.630842Z","iopub.status.idle":"2023-10-15T22:04:16.631311Z","shell.execute_reply.started":"2023-10-15T22:04:16.631121Z","shell.execute_reply":"2023-10-15T22:04:16.631139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.to_csv('')","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.633644Z","iopub.status.idle":"2023-10-15T22:04:16.63434Z","shell.execute_reply.started":"2023-10-15T22:04:16.633992Z","shell.execute_reply":"2023-10-15T22:04:16.634034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense,Dropout\nfrom tensorflow.keras.optimizers import Adam\n\n# Define the model\nmodel = Sequential([\n    Dense(64, activation='relu', input_shape=(139,)),  # Input layer with 128 features\n    Dense(128, activation='relu'),\n    Dense(256, activation='relu'),\n    Dropout(0.5),\n    Dense(512, activation='relu'),# Hidden layer with 128 units and ReLU activation\n    Dense(139, activation='linear')  # Output layer with linear activation (139 outputs)\n])\n\n# Compile the model\noptimizer = Adam(learning_rate=0.001)  # You can adjust the learning rate\nmodel.compile(optimizer=optimizer, loss='mean_squared_error', metrics=['mean_absolute_error'])\n\n# Print model summary\nmodel.build((None, 139))  # Build the model with a specific input shape\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.636521Z","iopub.status.idle":"2023-10-15T22:04:16.637069Z","shell.execute_reply.started":"2023-10-15T22:04:16.636848Z","shell.execute_reply":"2023-10-15T22:04:16.636871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.638813Z","iopub.status.idle":"2023-10-15T22:04:16.639321Z","shell.execute_reply.started":"2023-10-15T22:04:16.639106Z","shell.execute_reply":"2023-10-15T22:04:16.639126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(embeddings, y_2A3, epochs=10, batch_size=256)  # You can adjust the number of epochs and batch size","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.640881Z","iopub.status.idle":"2023-10-15T22:04:16.641332Z","shell.execute_reply.started":"2023-10-15T22:04:16.641144Z","shell.execute_reply":"2023-10-15T22:04:16.641164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embeddings.shape","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.644119Z","iopub.status.idle":"2023-10-15T22:04:16.644982Z","shell.execute_reply.started":"2023-10-15T22:04:16.644706Z","shell.execute_reply":"2023-10-15T22:04:16.644746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv1D, MaxPooling1D, Flatten, Dense, BatchNormalization, Dropout\nfrom tensorflow.keras.losses import MeanAbsoluteError\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom sklearn.preprocessing import StandardScaler\n\naa_alphabet = 'AUGC'\n\n# Define the model\nmodel = Sequential()\ninput_shape = (16,)\n\n# Add BatchNormalization layer after input\n# model.add(BatchNormalization(input_shape=input_shape))\nConv1D(64, 3, activation='relu', padding='same',input_shape=input_shape)\nmodel.add(Flatten())\nmodel.add(Dense(256, activation='relu'))\n\n\nmodel.add(Dense(256, activation='relu'))\nmodel.add(Dropout(0.5))  # Dropout for regularization\n\nmodel.add(Dense(num_outputs, activation='linear'))\n\n# Build the model explicitly\nmodel.build(input_shape=(None, 139, len(aa_alphabet)))\n\n# Compile the model with MAE as the loss function\noptimizer = Adam(learning_rate=0.001, clipnorm=1.0)\nmodel.compile(optimizer=optimizer, loss=MeanAbsoluteError(), metrics=['mean_absolute_error'])  # Use MAE\n\n# Implement early stoppingmodel.add(Dense(num_outputs, activation='linear'))\nearly_stopping = EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True)\n\n# Train the model with early stopping\n# model.fit(X_train_normalized, y_train, validation_data=(X_test_normalized, y_test), epochs=100, callbacks=[early_stopping])\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.655001Z","iopub.status.idle":"2023-10-15T22:04:16.655636Z","shell.execute_reply.started":"2023-10-15T22:04:16.655371Z","shell.execute_reply":"2023-10-15T22:04:16.655396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(X_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.658118Z","iopub.status.idle":"2023-10-15T22:04:16.658637Z","shell.execute_reply.started":"2023-10-15T22:04:16.658411Z","shell.execute_reply":"2023-10-15T22:04:16.65844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del(model)","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.660459Z","iopub.status.idle":"2023-10-15T22:04:16.660943Z","shell.execute_reply.started":"2023-10-15T22:04:16.660746Z","shell.execute_reply":"2023-10-15T22:04:16.660767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.662244Z","iopub.status.idle":"2023-10-15T22:04:16.662816Z","shell.execute_reply.started":"2023-10-15T22:04:16.662481Z","shell.execute_reply":"2023-10-15T22:04:16.662501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom scipy.sparse import csr_matrix\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Input, Conv1D, Flatten, Dense, BatchNormalization, Dropout, concatenate, Embedding\nfrom tensorflow.keras.losses import MeanAbsoluteError\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom sklearn.preprocessing import StandardScaler\nfrom tensorflow.keras.initializers import HeNormal\n\n# Assuming you have X_train_normalized and y_train as csr_matrix objects\n# Convert csr_matrix to dense NumPy array\n\naa_alphabet = 'AUGC'\ninput_shape = (139, len(aa_alphabet))\n\n# List of kernel sizes for Conv1D layers\nkernel_sizes = [3, 4, 10, 15, 20]\n\n# Initialize a list to store the output of each Conv1D branch\nconv_outputs = []\ninputs = []  # Create a list to hold the input branches\n\n# Define the size of the embedding layer\nembedding_dim = 64  # You can adjust this dimension as needed\n\n# Create an embedding layer\nembedding_layer = Embedding(input_dim=len(aa_alphabet), output_dim=embedding_dim)(input_layer)\n\nfor kernel_size in kernel_sizes:\n    conv_layer = Conv1D(64, kernel_size, activation='relu', padding='same', kernel_initializer=HeNormal())(embedding_layer)\n    conv_outputs.append(conv_layer)\n\n# Concatenate the outputs of all Conv1D branches\nconcatenated = concatenate(conv_outputs)\n\n# Rest of your model architecture...\nflatten_layer = Flatten()(concatenated)\ndense_layer1 = Dense(256, activation='relu')(flatten_layer)\ndropout_layer = Dropout(0.5)(dense_layer1)\nnum_outputs = len(y_train.columns)\noutput_layer = Dense(num_outputs, activation='linear')(dropout_layer)\n\n# Create the model using the Functional API\nmodel = tf.keras.Model(inputs=input_layer, outputs=output_layer)\n\n# Compile the model with MAE as the loss function\noptimizer = Adam(learning_rate=0.001, clipnorm=1.0)\nmodel.compile(optimizer=optimizer, loss=MeanAbsoluteError(), metrics=['mean_absolute_error'])  # Use MAE\n\n# Implement early stopping\nearly_stopping = EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True)\n\n# Train the model with early stopping\n# model.fit([X_train_normalized for _ in range(len(kernel_sizes))], y_train, validation_data=(X_test_normalized, y_test), epochs=100, callbacks=[early_stopping])\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.664851Z","iopub.status.idle":"2023-10-15T22:04:16.665267Z","shell.execute_reply.started":"2023-10-15T22:04:16.665082Z","shell.execute_reply":"2023-10-15T22:04:16.6651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom scipy.sparse import csr_matrix\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv1D, MaxPooling1D, Flatten, Dense, BatchNormalization, Dropout\nfrom tensorflow.keras.losses import MeanAbsoluteError\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom sklearn.preprocessing import StandardScaler\n\n# Assuming you have X_train_normalized and y_train as csr_matrix objects\n# Convert csr_matrix to dense NumPy array\n\naa_alphabet = 'AUGC'\n# Define the model\nmodel = Sequential()\ninput_shape = (139, len(aa_alphabet))\n\n# Rest of your model architecture...\nmodel.add(Conv1D(64, 5, activation='relu', padding='same',input_shape=input_shape))\nmodel.add(MaxPooling1D(2))\nmodel.add(Conv1D(128, 3, activation='relu', padding='same'))\nmodel.add(MaxPooling1D(2))\nmodel.add(Conv1D(256, 3, activation='relu', padding='same'))\nmodel.add(MaxPooling1D(2))\n\nmodel.add(Flatten())\nmodel.add(Dense(256, activation='relu'))\nmodel.add(Dropout(0.5))  # Dropout for regularization\n\nnum_outputs = len(y_train.columns)\nmodel.add(Dense(num_outputs, activation='linear'))\n\n# Compile the model with MAE as the loss function\noptimizer = Adam(learning_rate=0.001, clipnorm=1.0)\nmodel.compile(optimizer=optimizer, loss=MeanAbsoluteError(), metrics=['mean_absolute_error'])  # Use MAE\n\n# Implement early stopping\nearly_stopping = EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True)\n\n# Train the model with early stopping\n#model.fit(X_train_normalized, y_train, validation_data=(X_test_normalized, y_test), epochs=100, callbacks=[early_stopping])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.666625Z","iopub.status.idle":"2023-10-15T22:04:16.667171Z","shell.execute_reply.started":"2023-10-15T22:04:16.666988Z","shell.execute_reply":"2023-10-15T22:04:16.667007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.668614Z","iopub.status.idle":"2023-10-15T22:04:16.669157Z","shell.execute_reply.started":"2023-10-15T22:04:16.668956Z","shell.execute_reply":"2023-10-15T22:04:16.668974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.670298Z","iopub.status.idle":"2023-10-15T22:04:16.670744Z","shell.execute_reply.started":"2023-10-15T22:04:16.670511Z","shell.execute_reply":"2023-10-15T22:04:16.670529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.672542Z","iopub.status.idle":"2023-10-15T22:04:16.673174Z","shell.execute_reply.started":"2023-10-15T22:04:16.672931Z","shell.execute_reply":"2023-10-15T22:04:16.672961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train\n","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.676379Z","iopub.status.idle":"2023-10-15T22:04:16.676899Z","shell.execute_reply.started":"2023-10-15T22:04:16.676682Z","shell.execute_reply":"2023-10-15T22:04:16.676715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del(X_train,X_test,y_train,y_test)","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.6784Z","iopub.status.idle":"2023-10-15T22:04:16.67888Z","shell.execute_reply.started":"2023-10-15T22:04:16.678646Z","shell.execute_reply":"2023-10-15T22:04:16.678697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.681889Z","iopub.status.idle":"2023-10-15T22:04:16.68237Z","shell.execute_reply.started":"2023-10-15T22:04:16.682169Z","shell.execute_reply":"2023-10-15T22:04:16.682189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming you have X_train_normalized and y_train as NumPy arrays\n# Convert csr_matrix to dense NumPy array\n\n# Create a list of input data tensors, one for each kernel size\n\n# Train the model with early stopping\nmodel.fit(X_train, y_train, validation_data=(X_test, y_test), epochs=100, callbacks=[early_stopping], batch_size=512)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.683877Z","iopub.status.idle":"2023-10-15T22:04:16.68433Z","shell.execute_reply.started":"2023-10-15T22:04:16.684135Z","shell.execute_reply":"2023-10-15T22:04:16.684155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.get_weights()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.6895Z","iopub.status.idle":"2023-10-15T22:04:16.690264Z","shell.execute_reply.started":"2023-10-15T22:04:16.68992Z","shell.execute_reply":"2023-10-15T22:04:16.689962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pydot\nfrom tensorflow.keras.utils import plot_model\n\n# Assuming you already have the 'model' defined\n\n# Generate the schematic diagram\nplot_model(combined_model, to_file='simplified_model.png', show_shapes=True, show_layer_names=True)\n\n# This will save the diagram as 'simplified_model.png' in your current directory\n","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.692594Z","iopub.status.idle":"2023-10-15T22:04:16.693278Z","shell.execute_reply.started":"2023-10-15T22:04:16.69296Z","shell.execute_reply":"2023-10-15T22:04:16.69299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.695755Z","iopub.status.idle":"2023-10-15T22:04:16.696421Z","shell.execute_reply.started":"2023-10-15T22:04:16.696067Z","shell.execute_reply":"2023-10-15T22:04:16.696117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del(model)","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.698428Z","iopub.status.idle":"2023-10-15T22:04:16.69904Z","shell.execute_reply.started":"2023-10-15T22:04:16.698744Z","shell.execute_reply":"2023-10-15T22:04:16.698772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape = (X_train.shape[0],X_train.shape[1]*X_train.shape[2])\nX_test.shape = (X_test.shape[0],X_test.shape[1]*X_test.shape[2])","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.701414Z","iopub.status.idle":"2023-10-15T22:04:16.70204Z","shell.execute_reply.started":"2023-10-15T22:04:16.701734Z","shell.execute_reply":"2023-10-15T22:04:16.701763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.linear_model import Lasso\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.model_selection import train_test_split\n\n# Assuming you have X_train_normalized and y_train as NumPy arrays\n# If you haven't already, convert csr_matrix to dense NumPy array\n# You can also split your data into training and testing sets\n\n\n# Create and train a Ridge regression model\nridge_model = Lasso(alpha=1)  # You can adjust the alpha (regularization strength) as needed\nridge_model.fit(X_train, y_train)\n\n# Make predictions on the test data\ny_pred = ridge_model.predict(X_test)\n\n# Evaluate the model's performance\nmae = mean_absolute_error(y_test, y_pred)\nprint(f\"Mean Absolute Error: {mae}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.70518Z","iopub.status.idle":"2023-10-15T22:04:16.705888Z","shell.execute_reply.started":"2023-10-15T22:04:16.705528Z","shell.execute_reply":"2023-10-15T22:04:16.705563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(X_train,y_train,batch_size = 512)","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.708113Z","iopub.status.idle":"2023-10-15T22:04:16.708766Z","shell.execute_reply.started":"2023-10-15T22:04:16.708443Z","shell.execute_reply":"2023-10-15T22:04:16.708472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport numpy as np\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv1D, MaxPooling1D, Flatten, Dense, BatchNormalization, Dropout\nfrom tensorflow.keras.losses import MeanAbsoluteError\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom sklearn.preprocessing import StandardScaler\naa_alphabet = 'AUGC'\n# Convert your NumPy arrays to TensorFlow tensors\nX_train = tf.convert_to_tensor(X_train, dtype=tf.float32)\nX_test = tf.convert_to_tensor(X_test, dtype=tf.float32)\ny_train = tf.convert_to_tensor(y_train, dtype=tf.float32)\ny_test = tf.convert_to_tensor(y_test, dtype=tf.float32)\n\n# Define the model\nmodel = Sequential()\ninput_shape = (206, len(aa_alphabet))\n\n# Add BatchNormalization layer after input\n#model.add(BatchNormalization(input_shape=input_shape))\n\nmodel.add(Conv1D(64, 5, activation='relu', padding='same',input_shape=input_shape))\nmodel.add(MaxPooling1D(2))\nmodel.add(Conv1D(128, 3, activation='relu', padding='same'))\nmodel.add(MaxPooling1D(2))\nmodel.add(Conv1D(256, 3, activation='relu', padding='same'))\nmodel.add(MaxPooling1D(2))\n\nmodel.add(Flatten())\nmodel.add(Dense(256, activation='relu'))\nmodel.add(Dropout(0.5))  # Dropout for regularization\n\nnum_outputs = len(y_train.columns)\nmodel.add(Dense(num_outputs, activation='linear'))\n\n# Compile the model with MAE as the loss function\noptimizer = Adam(learning_rate=0.001, clipnorm=1.0)\nmodel.compile(optimizer=optimizer, loss=MeanAbsoluteError(), metrics=['mean_absolute_error'])  # Use MAE\n\n# Implement early stopping\nearly_stopping = EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True)\n\n# Train the model with early stopping\n#model.fit(X_train_normalized, y_train, validation_data=(X_test_normalized, y_test), epochs=100, callbacks=[early_stopping])\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-15T22:04:16.710644Z","iopub.status.idle":"2023-10-15T22:04:16.711269Z","shell.execute_reply.started":"2023-10-15T22:04:16.710958Z","shell.execute_reply":"2023-10-15T22:04:16.710987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nN = 269796671\ndf_submit = pd.DataFrame(index = range(N) )\ndf_submit.index.name = 'id'\ndf_submit['reactivity_DMS_MaP'] = np.zeros( N, dtype = np.float16 )\ndf_submit['reactivity_2A3_MaP'] = np.zeros( N, dtype = np.float16 )\nprint(df_submit.values.nbytes/1e6 )","metadata":{"execution":{"iopub.status.busy":"2023-10-18T09:06:17.334193Z","iopub.execute_input":"2023-10-18T09:06:17.334807Z","iopub.status.idle":"2023-10-18T09:06:18.404935Z","shell.execute_reply.started":"2023-10-18T09:06:17.33478Z","shell.execute_reply":"2023-10-18T09:06:18.404274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_submit = np.load('/kaggle/working/X_submit.npy')","metadata":{"execution":{"iopub.status.busy":"2023-10-18T09:07:27.945942Z","iopub.execute_input":"2023-10-18T09:07:27.946515Z","iopub.status.idle":"2023-10-18T09:07:28.039579Z","shell.execute_reply.started":"2023-10-18T09:07:27.946483Z","shell.execute_reply":"2023-10-18T09:07:28.038822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-10-18T09:07:37.393625Z","iopub.execute_input":"2023-10-18T09:07:37.394164Z","iopub.status.idle":"2023-10-18T09:07:37.814215Z","shell.execute_reply.started":"2023-10-18T09:07:37.394138Z","shell.execute_reply":"2023-10-18T09:07:37.813561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_submit.shape","metadata":{"execution":{"iopub.status.busy":"2023-10-18T09:11:27.383217Z","iopub.execute_input":"2023-10-18T09:11:27.384004Z","iopub.status.idle":"2023-10-18T09:11:27.387991Z","shell.execute_reply.started":"2023-10-18T09:11:27.383975Z","shell.execute_reply":"2023-10-18T09:11:27.387303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_submit_reshaped = X_submit.reshape(-1, 206, 4)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-18T09:15:11.136873Z","iopub.execute_input":"2023-10-18T09:15:11.137674Z","iopub.status.idle":"2023-10-18T09:15:11.141319Z","shell.execute_reply.started":"2023-10-18T09:15:11.13764Z","shell.execute_reply":"2023-10-18T09:15:11.14068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_DMS_pred = combined_model_DMS.predict(X_submit_reshaped)\nY_2A3_pred = combined_model_2A3.predict(X_submit_reshaped)\nprint(Y_DMS_pred.shape, Y_2A3_pred.shape)\nY_DMS_pred[:5,26:30]","metadata":{"execution":{"iopub.status.busy":"2023-10-18T09:16:02.282493Z","iopub.execute_input":"2023-10-18T09:16:02.282964Z","iopub.status.idle":"2023-10-18T09:24:31.647273Z","shell.execute_reply.started":"2023-10-18T09:16:02.282938Z","shell.execute_reply":"2023-10-18T09:24:31.646596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nNN = Y_DMS_pred.shape[0] * Y_DMS_pred.shape[1] \ndf_submit.loc[:(NN-1),'reactivity_DMS_MaP'] = Y_DMS_pred.ravel()\ndf_submit.loc[:(NN-1),'reactivity_2A3_MaP'] = Y_2A3_pred.ravel()","metadata":{"execution":{"iopub.status.busy":"2023-10-18T09:29:13.962784Z","iopub.execute_input":"2023-10-18T09:29:13.963197Z","iopub.status.idle":"2023-10-18T09:29:18.391207Z","shell.execute_reply.started":"2023-10-18T09:29:13.963169Z","shell.execute_reply":"2023-10-18T09:29:18.390545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display( df_submit.iloc[:NN,:].describe() ) ","metadata":{"execution":{"iopub.status.busy":"2023-10-18T09:30:44.601435Z","iopub.execute_input":"2023-10-18T09:30:44.601991Z","iopub.status.idle":"2023-10-18T09:30:49.923318Z","shell.execute_reply.started":"2023-10-18T09:30:44.60196Z","shell.execute_reply":"2023-10-18T09:30:49.922566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-10-18T09:32:09.212747Z","iopub.execute_input":"2023-10-18T09:32:09.213283Z","iopub.status.idle":"2023-10-18T09:32:09.576768Z","shell.execute_reply.started":"2023-10-18T09:32:09.213254Z","shell.execute_reply":"2023-10-18T09:32:09.576088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf_submit.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-10-18T09:32:21.767372Z","iopub.execute_input":"2023-10-18T09:32:21.768154Z","iopub.status.idle":"2023-10-18T09:41:25.682714Z","shell.execute_reply.started":"2023-10-18T09:32:21.768124Z","shell.execute_reply":"2023-10-18T09:41:25.681925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}