{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":51294,"databundleVersionId":6923401,"sourceType":"competition"},{"sourceId":6953689,"sourceType":"datasetVersion","datasetId":3993925},{"sourceId":6953783,"sourceType":"datasetVersion","datasetId":3993971},{"sourceId":6957292,"sourceType":"datasetVersion","datasetId":3996236},{"sourceId":6964730,"sourceType":"datasetVersion","datasetId":4001149},{"sourceId":6966232,"sourceType":"datasetVersion","datasetId":4002189}],"dockerImageVersionId":30579,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from keras.models import load_model\nmodel_DMS_MaP = load_model(\"/kaggle/working/modeltrainDMS_2x1024.h5\")\nmodel_DMS_MaP.summary()\ninput_layer_size = model_DMS_MaP.layers[0].input_shape[0]\nprint(\"Input layer size:\", input_layer_size)","metadata":{"_uuid":"09c114a9-a5f8-4858-b19f-a15a00c26fed","_cell_guid":"7f9c7d44-1dfa-4fdc-a142-b86c10e977b1","collapsed":false,"execution":{"iopub.status.busy":"2023-11-16T17:43:08.026953Z","iopub.execute_input":"2023-11-16T17:43:08.027866Z","iopub.status.idle":"2023-11-16T17:43:08.106702Z","shell.execute_reply.started":"2023-11-16T17:43:08.027828Z","shell.execute_reply":"2023-11-16T17:43:08.105398Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = \"/kaggle/input/stanford-ribonanza-rna-folding/train_data.csv\"\npathdm = \"/kaggle/working/traindms.csv\"\npath2a = \"/kaggle/working/train2a3.csv\"\n\nf = open(path, \"r\");\nline = f.readline()\n\nfd = open(pathdm, \"w\")\nf2 = open(path2a, \"w\")\nfd.write(line)\nf2.write(line)\n\nline = f.readline()\n\nI = 0\nwhile line != \"\":\n\n    lines = line\n    line=line.strip()\n    sequence = line.split(',')\n\n    if int(sequence[6]) == 1:\n        if sequence[2] == \"DMS_MaP\":\n            fd.write(lines)\n        elif sequence[2] == \"2A3_MaP\":\n            f2.write(lines)\n\n\n    line = f.readline()\n\n    I+=1\n    if (I+1)%100000 == 0:\n        print(I)","metadata":{"_uuid":"714186bb-8247-495e-be74-3c7a678bbf78","_cell_guid":"2948754e-9dfe-47d3-8553-cd4bba3388d8","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-11-18T10:40:28.459782Z","iopub.execute_input":"2023-11-18T10:40:28.460296Z","iopub.status.idle":"2023-11-18T10:41:23.627640Z","shell.execute_reply.started":"2023-11-18T10:40:28.460199Z","shell.execute_reply":"2023-11-18T10:41:23.626631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, LSTM, Input, Bidirectional, GRU, Embedding, GlobalMaxPooling1D\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.layers import BatchNormalization,Dropout\nfrom tensorflow.keras.initializers import VarianceScaling\nfrom tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.layers import Activation\nfrom itertools import cycle, islice\nfrom tensorflow.keras.regularizers import l2\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.preprocessing.sequence import TimeseriesGenerator\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\"\"\"\nimport os\ndirectory_path = '/kaggle/working/'\n\nfor dirname, _, filenames in os.walk(directory_path):\n    for filename in filenames:\n        file_path = os.path.join(dirname, filename)\n        print(f\"Deleting file: {file_path}\")\n        os.remove(file_path)\n\n\"\"\"\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\npath = \"/kaggle/input/stanford-ribonanza-rna-folding/train_data.csv\"\npathdm = \"/kaggle/working/traindms.csv\"\npath2a = \"/kaggle/working/train2a3.csv\"\n\n\"\"\"\nstrategy = 0\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Device:', tpu.master())\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nexcept:\n    strategy = tf.distribute.get_strategy()\nprint('Number of replicas:', strategy.num_replicas_in_sync)\n\"\"\"\n\n# Load the training data from the CSV file\nfile_path = path2a\ndata = pd.read_csv(file_path)\n\n# Replace empty values between commas with NaN\ndata.replace(r'^\\s*$', np.nan, regex=True, inplace=True)\n# Fill NaN values with 0 or any other appropriate value\ndata.fillna(0, inplace=True)\n\n# Define a mapping for nucleotides to numerical values\nnucleotide_mapping = {'A': 0.25, 'C': 0.5, 'G': 0.75, 'U': 1}\n\n# Convert the \"sequence\" column to numerical values\ndata['sequence'] = data['sequence'].apply(lambda x: [nucleotide_mapping[nt] for nt in x])\nsequence_array = pad_sequences(data['sequence'], maxlen=457, padding='post', truncating='post', dtype='float32', value=0.0)\n\nreactivity_columns = [f\"reactivity_{i+1:04}\" for i in range(206)]\nreactivity_data = data[reactivity_columns]\nreactivity_array = pad_sequences(reactivity_data.values, maxlen=457, padding='post', truncating='post', dtype='float32', value=0.0)\n\n\nclipped_data = np.clip(reactivity_array, 0, 1)\n# Assuming reactivity_array is your target variable\ny = clipped_data  # Your target variable, you need to define or load it\n\nprint(\"label ready...\")\n\n# Split the data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(sequence_array, y, test_size=0.2, random_state=42)\n\n\"\"\"\n# Assuming y_train is a 2D array with shape (number_of_samples, number_of_features)\n# Randomly select 50,000 indices\nindices_to_augment = np.random.choice(y_train.shape[0], size=50000, replace=False)\n\n# Generate random noise within the specified range\nnoise_range = (-0.01, 0.01)\nnoise = np.random.uniform(noise_range[0], noise_range[1], size=(50000, y_train.shape[1]))\n\n# Add noise to the selected samples in y_train\ny_train_augmented = y_train.copy()\ny_train_augmented[indices_to_augment] += noise\n\n# Clip the values to ensure they stay within a reasonable range\ny_train_augmented = np.clip(y_train_augmented, 0, 1)  # Assuming the values are in the range [0, 1]\n\n# Concatenate the noisy samples to X_train and y_train\nX_train = np.vstack((X_train, X_train[indices_to_augment]))\ny_train = np.vstack((y_train, y_train_augmented[indices_to_augment]))\n\"\"\"\n\nprint(X_train.shape, y_train.shape)\nprint(X_test.shape, y_test.shape)\n\nprint(\"Create model...\")\n#with strategy.scope():\nmodel = Sequential()\n\nmodel.add(Input(shape=(457,)))  # Input layer with 457 units\n\nmodel.add(Dense(units=2048, activation='relu', kernel_regularizer=l2(0.01)))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.5))\n\nmodel.add(Dense(units=2048, activation='relu', kernel_regularizer=l2(0.01)))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.5))\n\nmodel.add(Dense(units=457, activation='sigmoid'))  # Output layer with 206 units and linear activation\n\n\noptimizer = Adam(learning_rate=0.001)\n# Compile the model\n#loss_fn = tf.keras.losses.MeanAbsoluteError()\n\nmodel.compile(optimizer=optimizer, loss='mse',  metrics=['accuracy'])\n\n#model.build(input_shape=(457, 1))\n# Display the model summary\nmodel.summary()\n\n# Train the model\nmodel.fit(X_train, y_train, epochs=5, batch_size=100, validation_data=(X_test, y_test))\n\n# Evaluate the model on the test set\nloss = model.evaluate(X_test, y_test)\nprint(f'Test Loss: {loss}')\n\nmodel.save('/kaggle/working/modeltrain2a3_7777.h5')","metadata":{"_uuid":"8e6692a7-8755-4fbd-b2f7-43482000b086","_cell_guid":"a893528d-3ae5-4209-852f-bc0605fa2ecf","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-11-18T21:24:40.036811Z","iopub.execute_input":"2023-11-18T21:24:40.037540Z","iopub.status.idle":"2023-11-18T21:36:14.944584Z","shell.execute_reply.started":"2023-11-18T21:24:40.037435Z","shell.execute_reply":"2023-11-18T21:36:14.942499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport csv  \nimport os\nfrom keras.models import load_model\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.preprocessing.text import Tokenizer\n\n\n# Load test sequences\ntest_data = pd.read_csv(\"/kaggle/input/stanford-ribonanza-rna-folding/test_sequences.csv\")\nprint(\"loaded csv test\")\n\n# Preprocess test sequences\nnucleotide_mapping = {'A': 0.25, 'C': 0.5, 'G': 0.75, 'U': 1}\ntest_data['sequence'] = test_data['sequence'].apply(lambda x: [nucleotide_mapping[nt] for nt in x])\n\n# Recreate the tokenizer\n#tokenizer = Tokenizer(char_level=True)\n#tokenizer.fit_on_texts(test_data['sequence'])\n\n# Tokenize the test sequences\n#numerical_sequences = tokenizer.texts_to_sequences(test_data['sequence'])\n\n#padded_sequences = pad_sequences(numerical_sequences, maxlen=457, padding='post', truncating='post', dtype='float32', value=0.0)\n\n\n\n\n# Assuming you have two models for reactivity_DMS_MaP and reactivity_2A3_MaP named model_dms and model_2a3\n# You need to adjust these names based on your actual model variable names\nmodel_DMS_MaP = load_model(\"/kaggle/working/modeltrainDMS_7777.h5\")\nmodel_2A3_MaP = load_model(\"/kaggle/working/modeltrain2a3_7777.h5\")\n\n#os.remove(\"/kaggle/working/modeltrainDMS_7777.h5\")\n#os.remove(\"/kaggle/working/modeltrain2a3_7777.h5\")\nprint(\"loaded model\")\n\n# Create submission file\n# Create submission file\ncsv_file_path  = \"/kaggle/working/submission.csv\"\nheader_written = False\n\n# Check if the file exists, if yes, delete it\nif os.path.exists(csv_file_path):\n    #os.remove(csv_file_path)\n    exit(1)\n\nprint(\"Start...\")\ndata_to_concat = []\nid_counter = 0\n\nbatch_size = 1000  # Adjust the batch size according to your needs\nnum_batches = len(test_data) // batch_size + 1\n\nwith open(csv_file_path, mode='a', newline='') as csv_file:\n    writer = csv.writer(csv_file)\n    if not header_written:\n        writer.writerow([\"id\", \"reactivity_DMS_MaP\", \"reactivity_2A3_MaP\"])\n        header_written = True\n\n    for batch_num in range(num_batches):\n        start_idx = batch_num * batch_size\n        end_idx = min((batch_num + 1) * batch_size, len(test_data))\n        batch_data = test_data.iloc[start_idx:end_idx]\n\n        current_sequences = pad_sequences(batch_data['sequence'].tolist(), maxlen=457, padding='post', truncating='post', dtype='float32', value=0.0)\n        #current_sequences = padded_sequences[start_idx:end_idx]\n\n\n\n        predictions_DMS_MaP = model_DMS_MaP.predict(current_sequences, batch_size=1000, verbose=0)\n        predictions_2A3_MaP = model_2A3_MaP.predict(current_sequences, batch_size=1000, verbose=0)\n\n        print(\"batch \" + str(batch_num) + \" / \" + str(num_batches))\n\n        for i in range(len(batch_data)):\n            sequence_id = batch_data.iloc[i][\"sequence_id\"]\n            sequence = batch_data.iloc[i][\"sequence\"]\n            future = batch_data.iloc[i][\"future\"]\n\n            for j in range(len(sequence)):\n                id_value = id_counter\n                id_counter += 1\n                reactivity_DMS_MaP = predictions_DMS_MaP[i, j]\n                reactivity_2A3_MaP = predictions_2A3_MaP[i, j]\n\n                # Append to the submission CSV file\n\n                writer.writerow([id_value, reactivity_DMS_MaP, reactivity_2A3_MaP])\n\n\n\nprint(\"Prediction complete.\")","metadata":{"_uuid":"1e340ccc-7bf4-48db-9aac-a4af6aca9716","_cell_guid":"945a8b64-23dd-4699-8594-61302bd37ed6","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-11-18T21:37:49.511498Z","iopub.execute_input":"2023-11-18T21:37:49.511877Z","iopub.status.idle":"2023-11-18T22:00:53.970465Z","shell.execute_reply.started":"2023-11-18T21:37:49.511847Z","shell.execute_reply":"2023-11-18T22:00:53.968412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\ndirectory_path = '/kaggle/working/'\n\"\"\"\nfor dirname, _, filenames in os.walk(directory_path):\n    for filename in filenames:\n        file_path = os.path.join(dirname, filename)\n        print(f\"Deleting file: {file_path}\")\n        if filename != \"traindms.csv\" and filename != \"train2a3.csv\" and filename != \"submission.csv\":\n            os.remove(file_path)\n\"\"\"\n#####os.remove('/kaggle/working/submission.csv')","metadata":{"_uuid":"770d7bfa-bf8c-43ca-88a8-f4c685b12b10","_cell_guid":"41fba00f-a69c-4c6a-8ce2-0a687e301564","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-11-18T20:34:38.307012Z","iopub.execute_input":"2023-11-18T20:34:38.307466Z","iopub.status.idle":"2023-11-18T20:34:38.802654Z","shell.execute_reply.started":"2023-11-18T20:34:38.307434Z","shell.execute_reply":"2023-11-18T20:34:38.801116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\n\npathdm = \"/kaggle/working/traindms.csv\"\npath2a = \"/kaggle/working/train2a3.csv\"\npath = \"/kaggle/input/stanford-ribonanza-rna-folding/train_data.csv\"\n\n# Load the training data from the CSV file\nfile_path = path2a\ndata = pd.read_csv(file_path)\n\n# Replace empty values between commas with NaN\ndata.replace(r'^\\s*$', np.nan, regex=True, inplace=True)\n# Fill NaN values with 0 or any other appropriate value\ndata.fillna(0, inplace=True)\n\nreactivity_columns = [f\"reactivity_{i+1:04}\" for i in range(206)]\nreactivity_data = data[reactivity_columns]\nreactivity_array = pad_sequences(reactivity_data.values, maxlen=457, padding='post', truncating='post', dtype='float32', value=0.0)\n\n# Calculate min and max values\nmin_value = np.min(reactivity_array)\nmax_value = np.mean(reactivity_array)\n\nprint(\"min : {}, max: {}\".format(min_value, max_value))","metadata":{"_uuid":"1ade1154-a78f-447b-805c-2fc2dc22d4a7","_cell_guid":"3f84a3ac-c1a5-41d9-83ac-a1722f11ecdc","collapsed":false,"execution":{"iopub.status.busy":"2023-11-16T17:43:09.421181Z","iopub.status.idle":"2023-11-16T17:43:09.421549Z","shell.execute_reply.started":"2023-11-16T17:43:09.421376Z","shell.execute_reply":"2023-11-16T17:43:09.421393Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"hey\")","metadata":{"_uuid":"8c4832e3-0cad-4d2a-8a4a-f0a3632d6b4c","_cell_guid":"c89e9282-56db-47de-8bca-d09ddb672296","collapsed":false,"execution":{"iopub.status.busy":"2023-11-16T17:43:09.423425Z","iopub.status.idle":"2023-11-16T17:43:09.423744Z","shell.execute_reply.started":"2023-11-16T17:43:09.423583Z","shell.execute_reply":"2023-11-16T17:43:09.423598Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]}]}