{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11512973,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-24T21:43:00.605248Z","iopub.execute_input":"2025-03-24T21:43:00.605786Z","iopub.status.idle":"2025-03-24T21:43:01.25004Z","shell.execute_reply.started":"2025-03-24T21:43:00.605745Z","shell.execute_reply":"2025-03-24T21:43:01.248869Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Input, Dense, Dropout, LayerNormalization\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.optimizers import Adam\nfrom sklearn.model_selection import train_test_split\n!pip install Bio\nfrom Bio.Seq import Seq\nfrom Bio import SeqIO\n!pip install RNA\nimport RNA\n\nclass PositionalEncoding(tf.keras.layers.Layer):\n    def __init__(self, position, d_model):\n        super(PositionalEncoding, self).__init__()\n        self.pos_encoding = self.positional_encoding(position, d_model)\n        \n    def get_angles(self, position, i, d_model):\n        angles = 1 / tf.pow(10000, (2 * (i // 2)) / tf.cast(d_model, tf.float32))\n        return position * angles\n    \n    def positional_encoding(self, position, d_model):\n        angle_rads = self.get_angles(\n            position=tf.range(position, dtype=tf.float32)[:, tf.newaxis],\n            i=tf.range(d_model, dtype=tf.float32)[tf.newaxis, :],\n            d_model=d_model)\n        \n        # Apply sin to even indices\n        sines = tf.math.sin(angle_rads[:, 0::2])\n        # Apply cos to odd indices\n        cosines = tf.math.cos(angle_rads[:, 1::2])\n        \n        pos_encoding = tf.concat([sines, cosines], axis=-1)\n        pos_encoding = pos_encoding[tf.newaxis, ...]\n        return tf.cast(pos_encoding, tf.float32)\n    \n    def call(self, inputs):\n        return inputs + self.pos_encoding[:, :tf.shape(inputs)[1], :]\n\ndef transformer_encoder(inputs, head_size, num_heads, ff_dim, dropout=0):\n    # Normalization and Attention\n    x = LayerNormalization(epsilon=1e-6)(inputs)\n    x = tf.keras.layers.MultiHeadAttention(\n        key_dim=head_size, num_heads=num_heads, dropout=dropout)(x, x)\n    x = Dropout(dropout)(x)\n    res = x + inputs\n    \n    # Feed Forward Part\n    x = LayerNormalization(epsilon=1e-6)(res)\n    x = Dense(ff_dim, activation=\"relu\")(x)\n    x = Dropout(dropout)(x)\n    x = Dense(inputs.shape[-1])(x)\n    return x + res\n\ndef build_model(input_shape, head_size, num_heads, ff_dim, num_transformer_blocks, mlp_units, dropout=0):\n    inputs = Input(shape=input_shape)\n    x = PositionalEncoding(input_shape[0], input_shape[1])(inputs)\n    \n    for _ in range(num_transformer_blocks):\n        x = transformer_encoder(x, head_size, num_heads, ff_dim, dropout)\n    \n    x = tf.keras.layers.GlobalAveragePooling1D()(x)\n    for dim in mlp_units:\n        x = Dense(dim, activation=\"relu\")(x)\n        x = Dropout(dropout)(x)\n    \n    # Output 3D coordinates (x,y,z) for each nucleotide\n    outputs = Dense(input_shape[0]*3, activation=\"linear\")(x)\n    outputs = tf.keras.layers.Reshape((input_shape[0], 3))(outputs)\n    \n    return Model(inputs, outputs)\n\ndef preprocess_sequence(sequence):\n    # Convert RNA sequence to one-hot encoding\n    mapping = {'A': [1,0,0,0], 'U': [0,1,0,0], \n               'G': [0,0,1,0], 'C': [0,0,0,1]}\n    return np.array([mapping.get(nuc, [0,0,0,0]) for nuc in sequence])\n\ndef load_data(fasta_file, pdb_dir):\n    sequences = []\n    structures = []\n    \n    for record in SeqIO.parse(fasta_file, \"fasta\"):\n        seq = str(record.seq)\n        pdb_file = f\"{pdb_dir}/{record.id}.pdb\"\n        \n        # Load structure from PDB file (simplified)\n        # In practice, you'd parse the actual 3D coordinates\n        structure = np.random.rand(len(seq), 3)  # Placeholder\n        \n        sequences.append(preprocess_sequence(seq))\n        structures.append(structure)\n    \n    return np.array(sequences), np.array(structures)\n\ndef main():\n    # Hyperparameters\n    SEQ_LENGTH = 100  # Max sequence length (pad shorter sequences)\n    EMBED_DIM = 64\n    HEAD_SIZE = 64\n    NUM_HEADS = 4\n    FF_DIM = 128\n    NUM_TRANSFORMER_BLOCKS = 4\n    MLP_UNITS = [256, 128]\n    DROPOUT = 0.1\n    BATCH_SIZE = 32\n    EPOCHS = 50\n    \n    # Load data (replace with actual data paths)\n    X, y = load_data(\"rna_sequences.fasta\", \"pdb_files\")\n    \n    # Pad sequences to uniform length\n    X = tf.keras.preprocessing.sequence.pad_sequences(X, maxlen=SEQ_LENGTH, padding='post')\n    y = tf.keras.preprocessing.sequence.pad_sequences(y, maxlen=SEQ_LENGTH, padding='post')\n    \n    # Split data\n    X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2)\n    \n    # Build model\n    model = build_model(\n        input_shape=(SEQ_LENGTH, 4),  # 4 nucleotides\n        head_size=HEAD_SIZE,\n        num_heads=NUM_HEADS,\n        ff_dim=FF_DIM,\n        num_transformer_blocks=NUM_TRANSFORMER_BLOCKS,\n        mlp_units=MLP_UNITS,\n        dropout=DROPOUT)\n    \n    model.compile(\n        optimizer=Adam(learning_rate=1e-4),\n        loss='mse',  # Mean squared error for coordinates prediction\n        metrics=['mae'])\n    \n    # Train model\n    history = model.fit(\n        X_train, y_train,\n        validation_data=(X_val, y_val),\n        epochs=EPOCHS,\n        batch_size=BATCH_SIZE)\n    \n    # Save model\n    model.save(\"rna_structure_predictor.h5\")\n    \n    # Evaluate\n    loss, mae = model.evaluate(X_val, y_val)\n    print(f\"Validation MAE: {mae:.4f} Å\")\n    \n    # Example prediction\n    test_seq = preprocess_sequence(\"GGGAAACCCUUU\")\n    test_seq = np.array([test_seq])\n    test_seq = tf.keras.preprocessing.sequence.pad_sequences(test_seq, maxlen=SEQ_LENGTH, padding='post')\n    pred_structure = model.predict(test_seq)[0]\n    print(\"Predicted structure shape:\", pred_structure.shape)\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T21:22:11.192622Z","iopub.execute_input":"2025-03-24T21:22:11.192982Z","iopub.status.idle":"2025-03-24T21:22:20.400905Z","shell.execute_reply.started":"2025-03-24T21:22:11.192955Z","shell.execute_reply":"2025-03-24T21:22:20.399151Z"}},"outputs":[],"execution_count":null}]}