{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":51294,"databundleVersionId":6923401,"sourceType":"competition"}],"dockerImageVersionId":30558,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport csv\nimport tensorflow as tf","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-07T22:33:43.081908Z","iopub.execute_input":"2023-12-07T22:33:43.082400Z","iopub.status.idle":"2023-12-07T22:33:53.664340Z","shell.execute_reply.started":"2023-12-07T22:33:43.082352Z","shell.execute_reply":"2023-12-07T22:33:53.663178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INPUT_SHAPE = 1830 #result_horizontal.shape[1]\nOUTPUT_SHAPE = 457 #df_concatenated.shape[1]\n#print(INPUT_SHAPE)\n#print(OUTPUT_SHAPE)\n\n# Creating the neural network model\nmodel_1 = tf.keras.Sequential([\n    tf.keras.layers.Dense(1536, activation='relu', input_shape=(INPUT_SHAPE,)),  # Input layer with 10 features\n    \n    #tf.keras.layers.Dense(32, activation='relu'),                   # Hidden layer with 128 units\n    tf.keras.layers.Dropout(0.3),\n    #tf.keras.layers.Dense(1024, activation='relu'),\n    tf.keras.layers.Dense(1024, activation=tf.keras.layers.LeakyReLU(alpha=0.3)),\n    tf.keras.layers.Dropout(0.2),\n    tf.keras.layers.Dense(512, activation=tf.keras.layers.LeakyReLU(alpha=0.2)),\n    #tf.keras.layers.Dropout(0.3),\n    \n    tf.keras.layers.Dense(OUTPUT_SHAPE)                                  # Output layer with 5 units (for 5 output columns)\n])\n\nprint(model_1)\n\n# Tuning the learning rate\nlearning_rate = 0.001  # Tuned learning rate\noptimizer = tf.keras.optimizers.Adam(learning_rate=learning_rate)\n\n# Compile the model\n#model.compile(optimizer='adam', loss='mean_squared_error', metrics=['accuracy'])\nmodel_1.compile(optimizer=optimizer, loss='mean_squared_error', metrics=['accuracy'])\n\n\n# Train the model\n####model_1.fit(result_horizontal, df_concatenated, epochs=20, batch_size=335616)  # Adjust epochs and batch_size as needed\n#model_1.fit(result_horizontal, df_concatenated, epochs=20, batch_size=limiting_num_to_train)  # Adjust epochs and batch_size as needed\n\n# 20 Epoch decisions\n# Before leaky Rely it was 0.0289\n# With Leaky relu it is 0.0376 and alpha at 0.1\n# With Leaky relu it is 0.0361 and alpha at 0.2\n#Reverting back the alpha value to 0.1 and changing the last activation into softmax worsened the output to 0.0099\n# 640 and 0.4 and 0.4 were changed from 512 & 0.3 & 0.3  output is 0.0347\n# From the 0.0376 config, changed thelast layet to leaky relu as well 0.0367\n# Changed both alphas to 0.2 from 0.1 getting 0.0407\n#Shifted alphas to 0.3 from 0.2 and changed the dropouts to 0.4 from 0.3 == 0.0405\n# Added another leaky ReLU layer of 768 with 1024 changed to 1280 and an additionat dropout layer sandwiched == 0.0298\n# Redid whatever I did just above and change dthe dropouts to 0.5 from 0.4 == 0.0320\n# Reset to the fest found value of 0.0407 settings and added only an additional droppout layer just before the OutPut layer  == 0.0277\n# Reset to best  == 0.0409","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def one_hot_encode(sequence, max_length=457): # As per additional notes here (https://www.kaggle.com/competitions/stanford-ribonanza-rna-folding/data) this is the maximum\n    mapping = {'A': [1, 0, 0, 0], 'U': [0, 1, 0, 0], 'G': [0, 0, 1, 0], 'C': [0, 0, 0, 1]}\n    # Initialize an empty list to store the encoded sequence\n    encoded_sequence = []\n    for nucleotide in sequence[:max_length]:\n        encoded_sequence.extend(mapping[nucleotide])\n    # Pad with null values if the sequence is shorter than the maximum length\n    while len(encoded_sequence) < max_length * 4:\n        #encoded_sequence.extend([np.nan, np.nan, np.nan, np.nan])\n        encoded_sequence.extend([0, 0, 0, 0])\n    return encoded_sequence[:max_length * 4]\n\n\n#counterp=0\n\nfor train_data_df in pd.read_csv('/kaggle/input/stanford-ribonanza-rna-folding/train_data_QUICK_START.csv', chunksize=50000):\n    \n    # Dropping specific columns\n    columns_to_drop = ['sequence_id','sequence','experiment_type','dataset_name']\n\n\n    df_after_drop = train_data_df.drop(columns_to_drop, axis=1)\n\n    # Calculating the midpoint of the columns\n    midpoint = len(df_after_drop.columns) // 2\n\n    # Dropping the later half of the columns as reactivity Error is unnecessary\n    df_after_drop = df_after_drop.iloc[:, :midpoint]\n\n\n    # Replace NaN values with 0\n    df_after_drop = df_after_drop.fillna(0)\n\n\n    # Number of columns needed to reach about 457 columns\n    total_columns_needed = 457 # As per additional notes here (https://www.kaggle.com/competitions/stanford-ribonanza-rna-folding/data) this is the maximum\n\n    # Calculate the number of new columns to be added\n    a = len(df_after_drop.columns)\n    new_columns_count = total_columns_needed - a\n\n\n    # Create a DataFrame with zeros\n    zero_df = pd.DataFrame(0.0, index=df_after_drop.index, columns=[f'reactivity_{i:04d}' for i in range(a, a + new_columns_count)])\n\n\n\n    # Concatenate the zero DataFrame with the original DataFrame\n    df_concatenated = pd.concat([df_after_drop, zero_df], axis=1)\n\n    # Displaying the resulting DataFrame\n    #print(df_concatenated)\n\n\n    #df_concatenated = df_concatenated.head(limiting_num_to_train)\n    #df_concatenated\n\n    # Function to one-hot encode a single RNA sequence and pad it to a maximum length\n\n\n    # Create an empty list to store one-hot encoded sequences\n    one_hot_sequences = []\n\n    # Apply the function to the sequence column to create the one-hot encoded sequences\n    #counter = 0\n    for index, row in train_data_df.iterrows():\n        encoded_sequence = one_hot_encode(row['sequence'])\n        one_hot_sequences.append(encoded_sequence)\n        #counter+=1\n        #if counter >= limiting_num_to_train:\n            #break\n\n    # Create a new DataFrame from the list of one-hot encoded sequences\n    one_hot_df = pd.DataFrame(one_hot_sequences)\n\n    \n    \n    \n    \n    \n   \n\n    # Creating one-hot encoded columns\n    one_hot_encoded = pd.get_dummies(train_data_df['experiment_type'], prefix='One_Hot').astype(int)\n\n    # Displaying the resulting dataframe with the one-hot encoded columns\n    #print(one_hot_encoded)\n\n    # Extracting the first 10 rows\n    #one_hot_encoded = one_hot_encoded.head(limiting_num_to_train)\n    #one_hot_encoded\n    \n    \n    # Reset the index of df1 before concatenating\n    one_hot_encoded.reset_index(drop=True, inplace=True)\n    # Reset the index of df2 before concatenating\n    one_hot_df.reset_index(drop=True, inplace=True)\n    \n    result_horizontal = pd.concat([one_hot_encoded, one_hot_df], axis=1)\n    \n    \n    model_1.fit(result_horizontal, df_concatenated, epochs=20, batch_size=10000)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T23:56:39.875265Z","iopub.execute_input":"2023-12-07T23:56:39.875716Z","iopub.status.idle":"2023-12-07T23:59:57.753890Z","shell.execute_reply.started":"2023-12-07T23:56:39.875681Z","shell.execute_reply":"2023-12-07T23:59:57.752995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_data_df\ndel df_after_drop\ndel zero_df\ndel df_concatenated\ndel one_hot_sequences\ndel one_hot_df\ndel one_hot_encoded\ndel result_horizontal","metadata":{"execution":{"iopub.status.busy":"2023-12-08T00:00:01.271719Z","iopub.execute_input":"2023-12-08T00:00:01.272129Z","iopub.status.idle":"2023-12-08T00:00:01.622127Z","shell.execute_reply.started":"2023-12-08T00:00:01.272096Z","shell.execute_reply":"2023-12-08T00:00:01.620885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#limiting_num_to_train= 83904 #35616","metadata":{"execution":{"iopub.status.busy":"2023-12-07T22:33:53.665532Z","iopub.execute_input":"2023-12-07T22:33:53.666465Z","iopub.status.idle":"2023-12-07T22:33:53.670981Z","shell.execute_reply.started":"2023-12-07T22:33:53.666432Z","shell.execute_reply":"2023-12-07T22:33:53.670134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_data_df = pd.read_csv('/kaggle/input/stanford-ribonanza-rna-folding/train_data_QUICK_START.csv')\n\n# largest_string = len(max(train_data_df['sequence'], key=len))\n# smallest_string= len(min(train_data_df['sequence'], key=len))\n\n# print(largest_string, smallest_string)\n\n# train_data_df","metadata":{"execution":{"iopub.status.busy":"2023-12-07T22:33:53.673625Z","iopub.execute_input":"2023-12-07T22:33:53.674022Z","iopub.status.idle":"2023-12-07T22:34:13.748189Z","shell.execute_reply.started":"2023-12-07T22:33:53.673992Z","shell.execute_reply":"2023-12-07T22:34:13.747117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Dropping specific columns\n# columns_to_drop = ['sequence_id','sequence','experiment_type','dataset_name']\n\n\n# df_after_drop = train_data_df.drop(columns_to_drop, axis=1)\n\n# # Calculating the midpoint of the columns\n# midpoint = len(df_after_drop.columns) // 2\n\n# # Dropping the later half of the columns\n# df_after_drop = df_after_drop.iloc[:, :midpoint]\n\n\n# # Replace NaN values with 0\n# df_after_drop = df_after_drop.fillna(0)\n\n\n# # Number of columns needed to reach about 470 columns\n# total_columns_needed = 457 # As per additional notes here (https://www.kaggle.com/competitions/stanford-ribonanza-rna-folding/data) this is the maximum\n\n# # Calculate the number of new columns to be added\n# a = len(df_after_drop.columns)\n# new_columns_count = total_columns_needed - a\n\n\n# # Create a DataFrame with zeros\n# zero_df = pd.DataFrame(0.0, index=df_after_drop.index, columns=[f'reactivity_{i:04d}' for i in range(a, a + new_columns_count)])\n\n\n\n# # Concatenate the zero DataFrame with the original DataFrame\n# df_concatenated = pd.concat([df_after_drop, zero_df], axis=1)\n\n# # Displaying the resulting DataFrame\n# #print(df_concatenated)\n\n\n# df_concatenated = df_concatenated.head(limiting_num_to_train)\n# df_concatenated","metadata":{"execution":{"iopub.status.busy":"2023-12-07T22:34:13.749569Z","iopub.execute_input":"2023-12-07T22:34:13.750184Z","iopub.status.idle":"2023-12-07T22:34:15.844986Z","shell.execute_reply.started":"2023-12-07T22:34:13.750148Z","shell.execute_reply":"2023-12-07T22:34:15.843933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pandas as pd\n# import numpy as np\n\n# # Sample data for the dataframe\n# data = {\n#     'RNA_sequence': [\n#         \"GGGAACGACUCGAGUAGAGUCGAAAAAGAUCGC\",\n#         \"UUGCAUCGGAUAGCUGAAGUCAGCUAGCAGUC\",\n#         \"AGUCGAUAGCUAGCUAGCGAUAGCUAGCAGUC\",\n#         \"CGAUAGCUAGCAGUCGAUCGAUCGUAUCGAUC\"\n#     ]\n# }\n\n# # Creating the dataframe\n# df = pd.DataFrame(data)\n\n# # Function to one-hot encode a single RNA sequence\n# def one_hot_encode(sequence):\n#     mapping = {'A': [1, 0, 0, 0], 'U': [0, 1, 0, 0], 'G': [0, 0, 1, 0], 'C': [0, 0, 0, 1]}\n#     # Initialize an empty list to store the encoded sequence\n#     encoded_sequence = []\n#     for nucleotide in sequence:\n#         encoded_sequence.append(mapping[nucleotide])\n#     return encoded_sequence\n\n# # Create an empty DataFrame to store the one-hot encoded sequences\n# one_hot_df = pd.DataFrame()\n\n# # Apply the function to the sequence column to create the one-hot encoded sequences directly in one_hot_df\n# for index, row in df.iterrows():\n#     encoded_sequence = one_hot_encode(row['RNA_sequence'])\n#     # Append the encoded sequence as columns to the one_hot_df\n#     one_hot_df = pd.concat([one_hot_df, pd.DataFrame(encoded_sequence)], axis=1)\n\n# # Transpose the resulting dataframe for the correct orientation\n# one_hot_df = one_hot_df.T.reset_index(drop=True)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-07T22:34:15.846510Z","iopub.execute_input":"2023-12-07T22:34:15.846943Z","iopub.status.idle":"2023-12-07T22:34:15.853039Z","shell.execute_reply.started":"2023-12-07T22:34:15.846904Z","shell.execute_reply":"2023-12-07T22:34:15.852024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pandas as pd\n\n# # Sample data for the dataframe\n# data = {\n#     'RNA_sequence': [\n#         \"GGGAACGACUCGAGUAGAGUCGAAAAAGAUCGC\",\n#         \"UUGCAUCGGAUAGCUGAAGUCAGCUAGCAGUC\",\n#         \"AGUCGAUAGCUAGCUAGCGAUAGCUAGCAGUC\",\n#         \"CGAUAGCUAGCAGUCGAUCGAUCGUAUCGAUC\"\n#     ]\n# }\n\n# # Creating the dataframe\n# df = pd.DataFrame(data)\n\n# # Function to one-hot encode a single RNA sequence\n# def one_hot_encode(sequence):\n#     mapping = {'A': [1, 0, 0, 0], 'U': [0, 1, 0, 0], 'G': [0, 0, 1, 0], 'C': [0, 0, 0, 1]}\n#     # Initialize an empty list to store the encoded sequence\n#     encoded_sequence = []\n#     for nucleotide in sequence:\n#         encoded_sequence.extend(mapping[nucleotide])\n#     return encoded_sequence\n\n# # Create an empty list to store one-hot encoded sequences\n# one_hot_sequences = []\n\n# # Apply the function to the sequence column to create the one-hot encoded sequences\n# for index, row in df.iterrows():\n#     encoded_sequence = one_hot_encode(row['RNA_sequence'])\n#     one_hot_sequences.append(encoded_sequence)\n\n# # Create a new DataFrame from the list of one-hot encoded sequences\n# one_hot_df = pd.DataFrame(one_hot_sequences)\n\n# # Show the resulting DataFrame with the one-hot encoded sequences stitched into a single row per sequence\n# print(one_hot_df)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-07T22:34:15.855221Z","iopub.execute_input":"2023-12-07T22:34:15.855713Z","iopub.status.idle":"2023-12-07T22:34:15.868089Z","shell.execute_reply.started":"2023-12-07T22:34:15.855672Z","shell.execute_reply":"2023-12-07T22:34:15.866956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Function to one-hot encode a single RNA sequence and pad it to a maximum length\n# def one_hot_encode(sequence, max_length=457): # As per additional notes here (https://www.kaggle.com/competitions/stanford-ribonanza-rna-folding/data) this is the maximum\n#     mapping = {'A': [1, 0, 0, 0], 'U': [0, 1, 0, 0], 'G': [0, 0, 1, 0], 'C': [0, 0, 0, 1]}\n#     # Initialize an empty list to store the encoded sequence\n#     encoded_sequence = []\n#     for nucleotide in sequence[:max_length]:\n#         encoded_sequence.extend(mapping[nucleotide])\n#     # Pad with null values if the sequence is shorter than the maximum length\n#     while len(encoded_sequence) < max_length * 4:\n#         #encoded_sequence.extend([np.nan, np.nan, np.nan, np.nan])\n#         encoded_sequence.extend([0, 0, 0, 0])\n#     return encoded_sequence[:max_length * 4]\n\n# # Create an empty list to store one-hot encoded sequences\n# one_hot_sequences = []\n\n# # Apply the function to the sequence column to create the one-hot encoded sequences\n# counter = 0\n# for index, row in train_data_df.iterrows():\n#     encoded_sequence = one_hot_encode(row['sequence'])\n#     one_hot_sequences.append(encoded_sequence)\n#     counter+=1\n#     if counter >= limiting_num_to_train:\n#         break\n\n# # Create a new DataFrame from the list of one-hot encoded sequences\n# one_hot_df = pd.DataFrame(one_hot_sequences)\n\n# # Show the resulting DataFrame with the one-hot encoded sequences padded up to the maximum length\n# one_hot_df","metadata":{"execution":{"iopub.status.busy":"2023-12-07T22:34:15.869679Z","iopub.execute_input":"2023-12-07T22:34:15.870101Z","iopub.status.idle":"2023-12-07T22:35:39.774366Z","shell.execute_reply.started":"2023-12-07T22:34:15.870063Z","shell.execute_reply":"2023-12-07T22:35:39.773332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # data = {\n# #     'Original_Column': ['A', 'B', 'A', 'B', 'B', 'A']\n# # }\n\n# # # Creating the dataframe\n# # df = pd.DataFrame(data)\n\n# # Creating one-hot encoded columns\n# one_hot_encoded = pd.get_dummies(train_data_df['experiment_type'], prefix='One_Hot').astype(int)\n\n# # Displaying the resulting dataframe with the one-hot encoded columns\n# #print(one_hot_encoded)\n\n# # Extracting the first 10 rows\n# one_hot_encoded = one_hot_encoded.head(limiting_num_to_train)\n# one_hot_encoded","metadata":{"execution":{"iopub.status.busy":"2023-12-07T22:35:39.777936Z","iopub.execute_input":"2023-12-07T22:35:39.778250Z","iopub.status.idle":"2023-12-07T22:35:39.832804Z","shell.execute_reply.started":"2023-12-07T22:35:39.778222Z","shell.execute_reply":"2023-12-07T22:35:39.831786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# result_horizontal = pd.concat([one_hot_encoded, one_hot_df], axis=1)\n# result_horizontal","metadata":{"execution":{"iopub.status.busy":"2023-12-07T22:35:39.834101Z","iopub.execute_input":"2023-12-07T22:35:39.834430Z","iopub.status.idle":"2023-12-07T22:35:40.365714Z","shell.execute_reply.started":"2023-12-07T22:35:39.834402Z","shell.execute_reply":"2023-12-07T22:35:40.364585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# result_horizontal.shape[1] # Should be 1830 with no NaN in the cells","metadata":{"execution":{"iopub.status.busy":"2023-12-07T22:35:40.367362Z","iopub.execute_input":"2023-12-07T22:35:40.367829Z","iopub.status.idle":"2023-12-07T22:35:40.375155Z","shell.execute_reply.started":"2023-12-07T22:35:40.367789Z","shell.execute_reply":"2023-12-07T22:35:40.374052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"# INPUT_SHAPE = result_horizontal.shape[1]\n# OUTPUT_SHAPE = df_concatenated.shape[1]\n# print(INPUT_SHAPE)\n# print(OUTPUT_SHAPE)\n\n# # Creating the neural network model\n# model_1 = tf.keras.Sequential([\n#     tf.keras.layers.Dense(1536, activation='relu', input_shape=(INPUT_SHAPE,)),  # Input layer with 10 features\n    \n#     #tf.keras.layers.Dense(32, activation='relu'),                   # Hidden layer with 128 units\n#     tf.keras.layers.Dropout(0.3),\n#     #tf.keras.layers.Dense(1024, activation='relu'),\n#     tf.keras.layers.Dense(1024, activation=tf.keras.layers.LeakyReLU(alpha=0.2)),\n#     tf.keras.layers.Dropout(0.3),\n#     tf.keras.layers.Dense(512, activation=tf.keras.layers.LeakyReLU(alpha=0.2)),\n#     #tf.keras.layers.Dropout(0.3),\n    \n#     tf.keras.layers.Dense(OUTPUT_SHAPE)                                  # Output layer with 5 units (for 5 output columns)\n# ])\n\n# print(model_1)\n\n# # Tuning the learning rate\n# learning_rate = 0.001  # Tuned learning rate\n# optimizer = tf.keras.optimizers.Adam(learning_rate=learning_rate)\n\n# # Compile the model\n# #model.compile(optimizer='adam', loss='mean_squared_error', metrics=['accuracy'])\n# model_1.compile(optimizer=optimizer, loss='mean_squared_error', metrics=['accuracy'])\n\n\n# # Train the model\n# #model_1.fit(result_horizontal, df_concatenated, epochs=20, batch_size=335616)  # Adjust epochs and batch_size as needed\n# model_1.fit(result_horizontal, df_concatenated, epochs=20, batch_size=limiting_num_to_train)  # Adjust epochs and batch_size as needed\n\n# # 20 Epoch decisions\n# # Before leaky Rely it was 0.0289\n# # With Leaky relu it is 0.0376 and alpha at 0.1\n# # With Leaky relu it is 0.0361 and alpha at 0.2\n# #Reverting back the alpha value to 0.1 and changing the last activation into softmax worsened the output to 0.0099\n# # 640 and 0.4 and 0.4 were changed from 512 & 0.3 & 0.3  output is 0.0347\n# # From the 0.0376 config, changed thelast layet to leaky relu as well 0.0367\n# # Changed both alphas to 0.2 from 0.1 getting 0.0407\n# #Shifted alphas to 0.3 from 0.2 and changed the dropouts to 0.4 from 0.3 == 0.0405\n# # Added another leaky ReLU layer of 768 with 1024 changed to 1280 and an additionat dropout layer sandwiched == 0.0298\n# # Redid whatever I did just above and change dthe dropouts to 0.5 from 0.4 == 0.0320\n# # Reset to the fest found value of 0.0407 settings and added only an additional droppout layer just before the OutPut layer  == 0.0277\n# # Reset to best  == 0.0409","metadata":{"execution":{"iopub.status.busy":"2023-12-07T22:35:40.376785Z","iopub.execute_input":"2023-12-07T22:35:40.377193Z","iopub.status.idle":"2023-12-07T22:43:12.037583Z","shell.execute_reply.started":"2023-12-07T22:35:40.377156Z","shell.execute_reply":"2023-12-07T22:43:12.035455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# INPUT_SHAPE = result_horizontal.shape[1]\n# OUTPUT_SHAPE = df_concatenated.shape[1]\n# print(INPUT_SHAPE)\n# print(OUTPUT_SHAPE)\n\n# # Creating the neural network model\n# model_0 = tf.keras.Sequential([\n#     tf.keras.layers.Dense(1500, activation='relu', input_shape=(INPUT_SHAPE,)),  # Input layer with 10 features\n#     tf.keras.layers.Dropout(0.5),\n#     tf.keras.layers.Dense(1000, activation='relu'),\n#     #tf.keras.layers.Dense(32, activation='relu'),                   # Hidden layer with 128 units\n#     tf.keras.layers.Dropout(0.5),\n#     tf.keras.layers.Dense(500, activation='relu'),                    # Hidden layer with 64 units\n#     tf.keras.layers.Dense(OUTPUT_SHAPE)                                  # Output layer with 5 units (for 5 output columns)\n# ])\n\n# print(model_0)\n\n\n# # Compile the model\n# model_0.compile(optimizer='adam', loss='mean_squared_error', metrics=['accuracy'])\n\n# # Train the model\n# model_0.fit(result_horizontal, df_concatenated, epochs=200, batch_size=35616)  # Adjust epochs and batch_size as needed","metadata":{"execution":{"iopub.status.busy":"2023-12-07T03:38:53.679287Z","iopub.execute_input":"2023-12-07T03:38:53.679818Z","iopub.status.idle":"2023-12-07T03:38:53.686600Z","shell.execute_reply.started":"2023-12-07T03:38:53.679782Z","shell.execute_reply":"2023-12-07T03:38:53.684756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Testing","metadata":{}},{"cell_type":"code","source":"chunk_size = 10000 #34457*3*13 = ##num_of_unique_seq_to_test = 1343823","metadata":{"execution":{"iopub.status.busy":"2023-12-07T22:45:48.512774Z","iopub.execute_input":"2023-12-07T22:45:48.513391Z","iopub.status.idle":"2023-12-07T22:45:48.520647Z","shell.execute_reply.started":"2023-12-07T22:45:48.513335Z","shell.execute_reply":"2023-12-07T22:45:48.519532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test_sequences_df = pd.read_csv('/kaggle/input/stanford-ribonanza-rna-folding/test_sequences.csv')\n#test_sequences_df","metadata":{"execution":{"iopub.status.busy":"2023-12-07T22:45:51.476080Z","iopub.execute_input":"2023-12-07T22:45:51.476720Z","iopub.status.idle":"2023-12-07T22:45:52.322455Z","shell.execute_reply.started":"2023-12-07T22:45:51.476677Z","shell.execute_reply":"2023-12-07T22:45:52.320388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # For the 2A3\n\n# # creating the test dataset\n# # Function to one-hot encode a single RNA sequence and pad it to a maximum length\n# def test_one_hot_encode(sequence, max_length=457): # As per additional notes here (https://www.kaggle.com/competitions/stanford-ribonanza-rna-folding/data) this is the maximum\n#     mapping = {'A': [1, 0, 0, 0], 'U': [0, 1, 0, 0], 'G': [0, 0, 1, 0], 'C': [0, 0, 0, 1]}\n#     # Initialize an empty list to store the encoded sequence\n#     encoded_sequence = []\n#     for nucleotide in sequence[:max_length]:\n#         encoded_sequence.extend(mapping[nucleotide])\n#     # Pad with null values if the sequence is shorter than the maximum length\n#     while len(encoded_sequence) < max_length * 4:\n#         #encoded_sequence.extend([np.nan, np.nan, np.nan, np.nan])\n#         encoded_sequence.extend([0, 0, 0, 0])\n#     return encoded_sequence[:max_length * 4]\n\n# # Create an empty list to store one-hot encoded sequences\n# #test_one_hot_sequences = []\n\n# # Apply the function to the sequence column to create the one-hot encoded sequences\n# submission_array_2A3 = []\n# for index, row in test_sequences_df.iterrows():\n#     encoded_sequence = test_one_hot_encode(row['sequence'])\n#     #test_one_hot_sequences.append(encoded_sequence)\n#     #counter+=1\n    \n#     encoded_sequence.insert(0, 0) # Inserting the DMS experimentation one-hot at the beginning\n#     encoded_sequence.insert(0, 1) # Inserting the 2A3 experimentation one-hot at the beginning (DMS shifts to 2nd place)\n#     encoded_sequence = np.array(encoded_sequence)\n#     encoded_sequence = encoded_sequence.reshape(1, len(encoded_sequence))\n    \n#     #print(len(encoded_sequence))\n#     #print(type(encoded_sequence))\n    \n#     #Predictions_2A3 = model_1.predict(encoded_sequence)\n    \n#     Predictions_2A3 = model_1.predict(encoded_sequence)\n#     #print(Predictions_2A3)\n    \n    \n#     start=row[\"id_min\"]\n#     stop=row[\"id_max\"]\n#     dura = stop-start\n    \n    \n#     #submission_array_DMS.extend(Predictions_DMS[i][0:dura])\n#     submission_array_2A3.extend(Predictions_2A3[0][0:dura+1])\n    \n    \n\n    \n    \n#     break\n\n# # Create a new DataFrame from the list of one-hot encoded sequences\n# #test_one_hot_df = pd.DataFrame(test_one_hot_sequences)\n\n# # Show the resulting DataFrame with the one-hot encoded sequences padded up to the maximum length\n# #test_one_hot_df\n\n# len(submission_array_2A3)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T04:07:21.819204Z","iopub.execute_input":"2023-12-07T04:07:21.819627Z","iopub.status.idle":"2023-12-07T04:07:22.290700Z","shell.execute_reply.started":"2023-12-07T04:07:21.819596Z","shell.execute_reply":"2023-12-07T04:07:22.289280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # For the 2A3\n\n# # creating the test dataset\n# # Function to one-hot encode a single RNA sequence and pad it to a maximum length\n# def test_one_hot_encode(sequence, max_length=457): # As per additional notes here (https://www.kaggle.com/competitions/stanford-ribonanza-rna-folding/data) this is the maximum\n#     mapping = {'A': [1, 0, 0, 0], 'U': [0, 1, 0, 0], 'G': [0, 0, 1, 0], 'C': [0, 0, 0, 1]}\n#     # Initialize an empty list to store the encoded sequence\n#     encoded_sequence = []\n#     for nucleotide in sequence[:max_length]:\n#         encoded_sequence.extend(mapping[nucleotide])\n#     # Pad with null values if the sequence is shorter than the maximum length\n#     while len(encoded_sequence) < max_length * 4:\n#         #encoded_sequence.extend([np.nan, np.nan, np.nan, np.nan])\n#         encoded_sequence.extend([0, 0, 0, 0])\n#     return encoded_sequence[:max_length * 4]\n\n# # Create an empty list to store one-hot encoded sequences\n# #test_one_hot_sequences = []\n\n# # Apply the function to the sequence column to create the one-hot encoded sequences\n# submission_array_2A3 = []\n\n# chunk_size=50000\n# for i in range(0, len(test_sequences_df), chunk_size):\n#     chunk = df.iloc[i:i+chunk_size]\n#     # Process each chunk (chunk is a DataFrame)\n#     # Perform operations or analysis on each chunk\n#     print(chunk.head()) \n    \n    \n# for index, row in test_sequences_df.iterrows():\n#     encoded_sequence = test_one_hot_encode(row['sequence'])\n#     #test_one_hot_sequences.append(encoded_sequence)\n#     #counter+=1\n    \n#     encoded_sequence.insert(0, 0) # Inserting the DMS experimentation one-hot at the beginning\n#     encoded_sequence.insert(0, 1) # Inserting the 2A3 experimentation one-hot at the beginning (DMS shifts to 2nd place)\n#     encoded_sequence = np.array(encoded_sequence)\n#     encoded_sequence = encoded_sequence.reshape(1, len(encoded_sequence))\n    \n#     #print(len(encoded_sequence))\n#     #print(type(encoded_sequence))\n    \n#     #Predictions_2A3 = model_1.predict(encoded_sequence)\n    \n#     Predictions_2A3 = model_1.predict(encoded_sequence)\n#     #print(Predictions_2A3)\n    \n    \n#     start=row[\"id_min\"]\n#     stop=row[\"id_max\"]\n#     dura = stop-start\n    \n    \n#     #submission_array_DMS.extend(Predictions_DMS[i][0:dura])\n#     submission_array_2A3.extend(Predictions_2A3[0][0:dura+1])\n    \n    \n\n    \n    \n#     break\n\n# # Create a new DataFrame from the list of one-hot encoded sequences\n# #test_one_hot_df = pd.DataFrame(test_one_hot_sequences)\n\n# # Show the resulting DataFrame with the one-hot encoded sequences padded up to the maximum length\n# #test_one_hot_df\n\n# len(submission_array_2A3)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-12-07T22:46:00.442032Z","iopub.execute_input":"2023-12-07T22:46:00.442851Z","iopub.status.idle":"2023-12-07T22:46:01.480252Z","shell.execute_reply.started":"2023-12-07T22:46:00.442809Z","shell.execute_reply":"2023-12-07T22:46:01.478823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def test_one_hot_encode(sequence, max_length=457): # As per additional notes here (https://www.kaggle.com/competitions/stanford-ribonanza-rna-folding/data) this is the maximum\n#     mapping = {'A': [1, 0, 0, 0], 'U': [0, 1, 0, 0], 'G': [0, 0, 1, 0], 'C': [0, 0, 0, 1]}\n#     # Initialize an empty list to store the encoded sequence\n#     encoded_sequence = []\n#     for nucleotide in sequence[:max_length]:\n#         encoded_sequence.extend(mapping[nucleotide])\n#     # Pad with null values if the sequence is shorter than the maximum length\n#     while len(encoded_sequence) < max_length * 4:\n#         #encoded_sequence.extend([np.nan, np.nan, np.nan, np.nan])\n#         encoded_sequence.extend([0, 0, 0, 0])\n#     return encoded_sequence[:max_length * 4]\n\n\n\n# submission_array_DMS = []\n# submission_array_2A3 = []\n\n\n# for test_sequences_df in pd.read_csv('/kaggle/input/stanford-ribonanza-rna-folding/test_sequences.csv', chunksize=chunk_size):\n\n#     # Create an empty list to store one-hot encoded sequences\n#     test_one_hot_sequences = []\n\n#     # Apply the function to the sequence column to create the one-hot encoded sequences\n    \n#     for index, row in test_sequences_df.iterrows():\n#         encoded_sequence = test_one_hot_encode(row['sequence'])\n#         test_one_hot_sequences.append(encoded_sequence)\n        \n\n#     # Create a new DataFrame from the list of one-hot encoded sequences\n#     test_one_hot_df = pd.DataFrame(test_one_hot_sequences)\n#     #print(test_one_hot_df)\n\n#     # Show the resulting DataFrame with the one-hot encoded sequences padded up to the maximum length\n#     #test_one_hot_df\n    \n#     test_sequences_ids_df = test_sequences_df.drop([\"sequence_id\",\"sequence\",\"future\"],axis=1)\n#     #print(test_sequences_ids_df)\n    \n\n#     #test_sequences_ids_df = test_sequences_ids_df.head(num_of_unique_seq_to_test)\n    \n    \n#     test_one_hot_2A3_df = test_one_hot_df.copy()\n#     test_one_hot_2A3_df.insert(0, 'One_Hot_DMS_MaP', 0)\n#     test_one_hot_2A3_df.insert(0, 'One_Hot_2A3_MaP', 1)\n    \n    \n#     test_one_hot_DMS_df = test_one_hot_df.copy()\n#     test_one_hot_DMS_df.insert(0, 'One_Hot_DMS_MaP', 1)\n#     test_one_hot_DMS_df.insert(0, 'One_Hot_2A3_MaP', 0)\n    \n    \n#     Predictions_2A3 = model_1.predict(test_one_hot_2A3_df)\n#     Predictions_DMS = model_1.predict(test_one_hot_DMS_df)\n    \n    \n    \n#     #print(Predictions_DMS)\n#     #print(Predictions_2A3)\n    \n    \n#     counter=0\n#     for i,row in test_sequences_df.iterrows():\n#         #print(row)\n\n#         start=row[\"id_min\"]\n#         stop=row[\"id_max\"]\n#         dura = stop-start+1\n\n\n#         submission_array_DMS.extend(Predictions_DMS[counter][0:dura])\n#         submission_array_2A3.extend(Predictions_2A3[counter][0:dura])\n        \n#         counter+=1\n\n    \n    \n#     print(len(submission_array_DMS))\n#     print(len(submission_array_2A3))\n#     #break","metadata":{"execution":{"iopub.status.busy":"2023-12-07T12:38:31.521308Z","iopub.execute_input":"2023-12-07T12:38:31.521741Z","iopub.status.idle":"2023-12-07T13:33:00.351166Z","shell.execute_reply.started":"2023-12-07T12:38:31.521708Z","shell.execute_reply":"2023-12-07T13:33:00.349953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Predictions_DMS[0][0:dura]\n#indexes = list(range(len(submission_array_DMS)))\n#indexes = list(range(269796671))\n","metadata":{"execution":{"iopub.status.busy":"2023-12-07T13:42:28.648421Z","iopub.execute_input":"2023-12-07T13:42:28.649051Z","iopub.status.idle":"2023-12-07T13:42:40.243575Z","shell.execute_reply.started":"2023-12-07T13:42:28.648994Z","shell.execute_reply":"2023-12-07T13:42:40.242478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import csv\n\n# Define the file path\nfile_path = 'submission.csv'\n\n# Open the file in 'append' mode (or 'write' mode if the file doesn't exist)\nwith open(file_path, 'w', newline='') as file:\n    writer = csv.writer(file)\n    writer.writerow(['id','reactivity_DMS_MaP', 'reactivity_2A3_MaP'])","metadata":{"execution":{"iopub.status.busy":"2023-12-07T23:19:02.652520Z","iopub.execute_input":"2023-12-07T23:19:02.652934Z","iopub.status.idle":"2023-12-07T23:19:02.949661Z","shell.execute_reply.started":"2023-12-07T23:19:02.652898Z","shell.execute_reply":"2023-12-07T23:19:02.948773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def test_one_hot_encode(sequence, max_length=457): # As per additional notes here (https://www.kaggle.com/competitions/stanford-ribonanza-rna-folding/data) this is the maximum\n    mapping = {'A': [1, 0, 0, 0], 'U': [0, 1, 0, 0], 'G': [0, 0, 1, 0], 'C': [0, 0, 0, 1]}\n    # Initialize an empty list to store the encoded sequence\n    encoded_sequence = []\n    for nucleotide in sequence[:max_length]:\n        encoded_sequence.extend(mapping[nucleotide])\n    # Pad with null values if the sequence is shorter than the maximum length\n    while len(encoded_sequence) < max_length * 4:\n        #encoded_sequence.extend([np.nan, np.nan, np.nan, np.nan])\n        encoded_sequence.extend([0, 0, 0, 0])\n    return encoded_sequence[:max_length * 4]\n\n\n\n#submission_array_DMS = []\n#submission_array_2A3 = []\n\n\nfor test_sequences_df in pd.read_csv('/kaggle/input/stanford-ribonanza-rna-folding/test_sequences.csv', chunksize=chunk_size):\n\n    # Create an empty list to store one-hot encoded sequences\n    test_one_hot_sequences = []\n\n    # Apply the function to the sequence column to create the one-hot encoded sequences\n    \n    for index, row in test_sequences_df.iterrows():\n        encoded_sequence = test_one_hot_encode(row['sequence'])\n        test_one_hot_sequences.append(encoded_sequence)\n        \n\n    # Create a new DataFrame from the list of one-hot encoded sequences\n    test_one_hot_df = pd.DataFrame(test_one_hot_sequences)\n    #print(test_one_hot_df)\n\n    # Show the resulting DataFrame with the one-hot encoded sequences padded up to the maximum length\n    #test_one_hot_df\n    \n    test_sequences_ids_df = test_sequences_df.drop([\"sequence_id\",\"sequence\",\"future\"],axis=1)\n    #print(test_sequences_ids_df)\n    \n\n    #test_sequences_ids_df = test_sequences_ids_df.head(num_of_unique_seq_to_test)\n    \n    \n    test_one_hot_2A3_df = test_one_hot_df.copy()\n    test_one_hot_2A3_df.insert(0, 'One_Hot_DMS_MaP', 0)\n    test_one_hot_2A3_df.insert(0, 'One_Hot_2A3_MaP', 1)\n    \n    \n    test_one_hot_DMS_df = test_one_hot_df.copy()\n    test_one_hot_DMS_df.insert(0, 'One_Hot_DMS_MaP', 1)\n    test_one_hot_DMS_df.insert(0, 'One_Hot_2A3_MaP', 0)\n    \n    \n    Predictions_2A3 = model_1.predict(test_one_hot_2A3_df)\n    Predictions_DMS = model_1.predict(test_one_hot_DMS_df)\n    \n    \n    \n    #print(Predictions_DMS)\n    #print(Predictions_2A3)\n    \n    \n    counter=0\n    for i,row in test_sequences_df.iterrows():\n        #print(row)\n\n        start=row[\"id_min\"]\n        stop=row[\"id_max\"]\n        dura = stop-start+1\n\n\n        # submission_array_DMS.extend(Predictions_DMS[counter][0:dura])\n        # submission_array_2A3.extend(Predictions_2A3[counter][0:dura])\n        #indexes = list(range(len(submission_array_DMS)))\n        \n        a = list(range(start,stop+1))\n        b = Predictions_DMS[counter][0:dura]\n        c = Predictions_2A3[counter][0:dura]\n        \n        \n        counter+=1\n        \n        \n        # import csv\n\n        # Define the file path\n        file_path = '/kaggle/working/submission.csv'\n\n        # Open the file in 'append' mode (or 'write' mode if the file doesn't exist)\n        with open(file_path, 'a', newline='') as file:\n            writer = csv.writer(file)\n            \n            d=zip(a,b,c)\n            writer.writerows(d)\n\n    \n    print(stop)\n    \n    #print(len(submission_array_DMS))\n    #print(len(submission_array_2A3))\n","metadata":{"execution":{"iopub.status.busy":"2023-12-07T23:19:08.809371Z","iopub.execute_input":"2023-12-07T23:19:08.810036Z","iopub.status.idle":"2023-12-07T23:20:09.219589Z","shell.execute_reply.started":"2023-12-07T23:19:08.810000Z","shell.execute_reply":"2023-12-07T23:20:09.218397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Define the file path\n# file_path = 'output.csv'\n\n# # Open the file in 'append' mode (or 'write' mode if the file doesn't exist)\n# with open(file_path, 'a', newline='') as file:\n#     writer = csv.writer(file)\n    \n#     # Write header if needed (if it's the first iteration)\n#     # writer.writerow(['Name', 'Age', 'City'])  # Uncomment to write header\n    \n#     # Write rows iteratively\n#     for row in data:\n#         writer.writerow(row)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-12-07T13:56:04.482226Z","iopub.execute_input":"2023-12-07T13:56:04.482709Z","iopub.status.idle":"2023-12-07T13:56:04.492515Z","shell.execute_reply.started":"2023-12-07T13:56:04.482674Z","shell.execute_reply":"2023-12-07T13:56:04.490901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# file_path = '/kaggle/working/submission.csv'\n\n\n\n# # Assuming you have a list of data that you want to write to a CSV file\n# data = [\n#     ['Name', 'Age', 'City'],\n#     ['Alice', 25, 'New York'],\n#     ['Bob', 30, 'San Francisco'],\n#     # More data...\n# ]\n\n\n# # Open the file in 'append' mode (or 'write' mode if the file doesn't exist)\n# with open(file_path, 'a', newline='') as file:\n#     writer = csv.writer(file)\n#     pass\n\n#     # Write rows iteratively\n#     for row in data:\n#         writer.writerow(row)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T13:58:27.659921Z","iopub.execute_input":"2023-12-07T13:58:27.660409Z","iopub.status.idle":"2023-12-07T13:58:27.669252Z","shell.execute_reply.started":"2023-12-07T13:58:27.660373Z","shell.execute_reply":"2023-12-07T13:58:27.667702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Assuming you have a list of data that you want to write to a CSV file\n# data = [\n#     ['Name', 'Age', 'City'],\n#     ['Alice', 25, 'New York'],\n#     ['Bob', 30, 'San Francisco'],\n#     # More data...\n# ]\n\n# for row in data:\n#     print(row)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T14:01:44.993118Z","iopub.execute_input":"2023-12-07T14:01:44.993757Z","iopub.status.idle":"2023-12-07T14:01:45.003769Z","shell.execute_reply.started":"2023-12-07T14:01:44.993698Z","shell.execute_reply":"2023-12-07T14:01:45.002081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# # Zip the arrays together to create rows\n# rows = zip(indexes,submission_array_DMS,submission_array_2A3)\n\n# # Specify the file name for the CSV\n# file_name = 'submission.csv'\n\n# # Write the data to a CSV file\n# with open(file_name, 'w', newline='') as csvfile:\n#     csv_writer = csv.writer(csvfile)\n#     csv_writer.writerow(['id','reactivity_DMS_MaP', 'reactivity_2A3_MaP'])  # Write header\n#     csv_writer.writerows(rows)  # Write rows","metadata":{"execution":{"iopub.status.busy":"2023-12-07T06:39:44.218483Z","iopub.status.idle":"2023-12-07T06:39:44.219884Z","shell.execute_reply.started":"2023-12-07T06:39:44.219515Z","shell.execute_reply":"2023-12-07T06:39:44.219554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_sequences_ids_df = test_sequences_df.drop([\"sequence_id\",\"sequence\",\"future\"],axis=1)\n\n# test_sequences_ids_df = test_sequences_ids_df.head(num_of_unique_seq_to_test)\n# test_sequences_ids_df\n\n# #test_sequences_ids_vertically_doubled_df = pd.concat([test_sequences_ids_df, test_sequences_ids_df], axis=0)\n\n# #test_sequences_ids_vertically_doubled_df = test_sequences_ids_vertically_doubled_df.reset_index(drop=True)\n# #test_sequences_ids_vertically_doubled_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_one_hot_df_doubled_vertically = pd.concat([test_one_hot_df, test_one_hot_df], axis=0)\n\n# # Resetting the row index\n# test_one_hot_df_doubled_vertically = test_one_hot_df_doubled_vertically.reset_index(drop=True)\n\n# test_one_hot_df_doubled_vertically","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Create an array of desired length with alternating 0's and 1's\n\n# alternate_column = np.zeros(num_of_unique_seq_to_test)  # Initialize an array of zeros\n\n# # Set alternate elements to 1\n# alternate_column[1::2] = 1\n\n# # Create a DataFrame with a column of alternating 0's and 1's\n# df_2A3 = pd.DataFrame({'One_Hot_2A3_MaP': alternate_column})\n\n\n# # Create an array of desired length with alternating 0's and 1's\n# length = 30  # Change this to the desired length of the column\n# alternate_column = np.zeros(length)  # Initialize an array of zeros\n\n# # Set alternate elements to 1\n# alternate_column[0::2] = 1\n\n# # Create a DataFrame with a column of alternating 0's and 1's\n# df_DMS = pd.DataFrame({'One_Hot_DMS_MaP': alternate_column})\n\n\n# # Concatenate the zero DataFrame with the original DataFrame\n# test_df_concatenated = pd.concat([df_2A3, df_DMS], axis=1)\n# #print(test_df_concatenated)\n\n# full_test_df = pd.concat([test_df_concatenated, test_one_hot_df_doubled_vertically_reset], axis=1)\n# print(full_test_df)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_one_hot_2A3_df = test_one_hot_df.copy()\n# test_one_hot_2A3_df.insert(0, 'One_Hot_DMS_MaP', 0)\n# test_one_hot_2A3_df.insert(0, 'One_Hot_2A3_MaP', 1)\n\n# test_one_hot_2A3_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_one_hot_DMS_df = test_one_hot_df.copy()\n\n# test_one_hot_DMS_df.insert(0, 'One_Hot_DMS_MaP', 1)\n# test_one_hot_DMS_df.insert(0, 'One_Hot_2A3_MaP', 0)\n\n# test_one_hot_DMS_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predictions_2A3 = model_1.predict(test_one_hot_2A3_df)\n# Predictions_2A3","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predictions_DMS = model_1.predict(test_one_hot_DMS_df)\n# Predictions_DMS","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # print(len(Predictions_DMS))\n# # print(len(Predictions_DMS[24]))\n# # print(len(Predictions_DMS[0]))\n\n# print(Predictions_DMS[0][456])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission_array_DMS = []\n# submission_array_2A3 = []\n\n\n# counter = 0\n# for i,row in test_sequences_df.iterrows():\n#     #print(row)\n    \n#     start=row[\"id_min\"]\n#     stop=row[\"id_max\"]\n#     dura = stop-start+1\n       \n    \n#     submission_array_DMS.extend(Predictions_DMS[i][0:dura])\n#     submission_array_2A3.extend(Predictions_2A3[i][0:dura])\n    \n#     #print(meaningful)\n#     counter += 1\n#     if counter >= num_of_unique_seq_to_test:\n#         break\n\n# print(len(submission_array_DMS))\n# print(len(submission_array_2A3))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# indexes = list(range(len(submission_array_DMS)))\n\n# # Zip the arrays together to create rows\n# rows = zip(indexes,submission_array_DMS,submission_array_2A3)\n\n# # Specify the file name for the CSV\n# file_name = 'submission.csv'\n\n# # Write the data to a CSV file\n# with open(file_name, 'w', newline='') as csvfile:\n#     csv_writer = csv.writer(csvfile)\n#     csv_writer.writerow(['id','reactivity_DMS_MaP', 'reactivity_2A3_MaP'])  # Write header\n#     csv_writer.writerows(rows)  # Write rows","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ample_submission = pd.read_csv('/kaggle/input/stanford-ribonanza-rna-folding/sample_submission.csv')\n# ample_submission","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# # Sample arrays\n# names = ['Alice', 'Bob', 'Charlie']\n# ages = [25, 30, 35]\n# cities = ['New York', 'San Francisco', 'Los Angeles']\n\n# # Zip the arrays together to create rows\n# rows = zip(names, ages, cities)\n\n# # Specify the file name for the CSV\n# file_name = 'output_file.csv'\n\n# # Write the data to a CSV file\n# with open(file_name, 'w', newline='') as csvfile:\n#     csv_writer = csv.writer(csvfile)\n#     csv_writer.writerow(['Name', 'Age', 'City'])  # Write header\n#     csv_writer.writerows(rows)  # Write rows","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training the Mirror Image (anti-sequence) as well","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}