{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11512973,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-12T05:05:49.417594Z","iopub.execute_input":"2025-03-12T05:05:49.417944Z","iopub.status.idle":"2025-03-12T05:05:51.042554Z","shell.execute_reply.started":"2025-03-12T05:05:49.417906Z","shell.execute_reply":"2025-03-12T05:05:51.041425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T04:29:57.896107Z","iopub.execute_input":"2025-03-23T04:29:57.896398Z","iopub.status.idle":"2025-03-23T04:30:18.105224Z","shell.execute_reply.started":"2025-03-23T04:29:57.896364Z","shell.execute_reply":"2025-03-23T04:30:18.104108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sequences = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_sequences.csv\")\ntrain_labels = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_labels.csv\")\nvalidation_sequences = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/validation_sequences.csv\")\nvalidation_labels = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/validation_labels.csv\")\ntest_sequences = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/test_sequences.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T04:30:48.075510Z","iopub.execute_input":"2025-03-23T04:30:48.075863Z","iopub.status.idle":"2025-03-23T04:30:48.372219Z","shell.execute_reply.started":"2025-03-23T04:30:48.075836Z","shell.execute_reply":"2025-03-23T04:30:48.371384Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Train Sequences:\", train_sequences.shape)\nprint(\"Train Labels:\", train_labels.shape)\nprint(\"Validation Sequences:\", validation_sequences.shape)\nprint(\"Test Sequences:\", test_sequences.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T04:30:51.341085Z","iopub.execute_input":"2025-03-23T04:30:51.341425Z","iopub.status.idle":"2025-03-23T04:30:51.348649Z","shell.execute_reply.started":"2025-03-23T04:30:51.341384Z","shell.execute_reply":"2025-03-23T04:30:51.347857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sequences.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T04:36:12.206547Z","iopub.execute_input":"2025-03-23T04:36:12.206959Z","iopub.status.idle":"2025-03-23T04:36:12.241081Z","shell.execute_reply.started":"2025-03-23T04:36:12.206927Z","shell.execute_reply":"2025-03-23T04:36:12.240217Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_labels.info()\nvalidation_sequences.info()\ntest_sequences.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T04:37:36.852942Z","iopub.execute_input":"2025-03-23T04:37:36.853277Z","iopub.status.idle":"2025-03-23T04:37:36.888848Z","shell.execute_reply.started":"2025-03-23T04:37:36.853251Z","shell.execute_reply":"2025-03-23T04:37:36.887596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Encoding RNA Sequences\n# Create a dictionary to map RNA bases to numeric values\nrna_vocab = {'A': 1, 'U': 2, 'G': 3, 'C': 4}  \n\ndef encode_sequence(seq):\n    # Convert each RNA character in the sequence to its corresponding numeric value\n    return [rna_vocab[char] for char in seq if char in rna_vocab]  # Ensure only valid characters are processed","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T04:46:07.015921Z","iopub.execute_input":"2025-03-23T04:46:07.016288Z","iopub.status.idle":"2025-03-23T04:46:07.021264Z","shell.execute_reply.started":"2025-03-23T04:46:07.016257Z","shell.execute_reply":"2025-03-23T04:46:07.020266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Apply encoding to all sequences\ntrain_sequences['encoded'] = train_sequences['sequence'].apply(encode_sequence)\nvalidation_sequences['encoded'] = validation_sequences['sequence'].apply(encode_sequence)\ntest_sequences['encoded'] = test_sequences['sequence'].apply(encode_sequence)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T04:46:21.324895Z","iopub.execute_input":"2025-03-23T04:46:21.325251Z","iopub.status.idle":"2025-03-23T04:46:21.348375Z","shell.execute_reply.started":"2025-03-23T04:46:21.325223Z","shell.execute_reply":"2025-03-23T04:46:21.347505Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Padding sequences to ensure uniform length across all samples\nmax_len = max(train_sequences['encoded'].apply(len))  # Determine the max sequence length\ntrain_sequences_padded = pad_sequences(train_sequences['encoded'], maxlen=max_len, padding='post')\nvalidation_sequences_padded = pad_sequences(validation_sequences['encoded'], maxlen=max_len, padding='post')\ntest_sequences_padded = pad_sequences(test_sequences['encoded'], maxlen=max_len, padding='post')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T04:46:34.123082Z","iopub.execute_input":"2025-03-23T04:46:34.123493Z","iopub.status.idle":"2025-03-23T04:46:34.147608Z","shell.execute_reply.started":"2025-03-23T04:46:34.123462Z","shell.execute_reply":"2025-03-23T04:46:34.146566Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Ensure labels are in NumPy array format, excluding the ID column\nif isinstance(train_labels, pd.DataFrame):\n    train_labels = train_labels.iloc[:, 1:].values  # Convert DataFrame to NumPy array\n\nif isinstance(validation_labels, pd.DataFrame):\n    validation_labels = validation_labels.iloc[:, 1:].values  # Convert DataFrame to NumPy array\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T04:48:08.152006Z","iopub.execute_input":"2025-03-23T04:48:08.152313Z","iopub.status.idle":"2025-03-23T04:48:08.157309Z","shell.execute_reply.started":"2025-03-23T04:48:08.152290Z","shell.execute_reply":"2025-03-23T04:48:08.156259Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}