{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11553390,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-27T15:08:23.496176Z","iopub.execute_input":"2025-03-27T15:08:23.496689Z","iopub.status.idle":"2025-03-27T15:08:24.805997Z","shell.execute_reply.started":"2025-03-27T15:08:23.496648Z","shell.execute_reply":"2025-03-27T15:08:24.804917Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#In this notebook, I use a Hidden Markov Model (HMM) to extract statistical features from RNA sequences. These features — including entropy, log-likelihood, and dominant hidden state — can help improve model understanding of sequence dynamics.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T15:08:24.807503Z","iopub.execute_input":"2025-03-27T15:08:24.808061Z","iopub.status.idle":"2025-03-27T15:08:24.812073Z","shell.execute_reply.started":"2025-03-27T15:08:24.808031Z","shell.execute_reply":"2025-03-27T15:08:24.810986Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install pgmpy hmmlearn","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T15:08:55.749474Z","iopub.execute_input":"2025-03-27T15:08:55.749810Z","iopub.status.idle":"2025-03-27T15:09:03.166751Z","shell.execute_reply.started":"2025-03-27T15:08:55.749785Z","shell.execute_reply":"2025-03-27T15:09:03.165501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from hmmlearn import hmm\nfrom scipy.stats import entropy\nimport numpy as np\nimport pandas as pd\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T15:09:18.019423Z","iopub.execute_input":"2025-03-27T15:09:18.020020Z","iopub.status.idle":"2025-03-27T15:09:19.273048Z","shell.execute_reply.started":"2025-03-27T15:09:18.019964Z","shell.execute_reply":"2025-03-27T15:09:19.271827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Valid bases and encoding\nvalid_nucleotides = ['A', 'C', 'G', 'U']\nnucleotide_to_int = {n: i for i, n in enumerate(valid_nucleotides)}\n\n# Load data (adjust path if running locally or on Kaggle)\ndf = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_sequences.csv\")\n\n# Clean and encode\ndef clean_and_encode(seq):\n    return [nucleotide_to_int[n] for n in str(seq) if n in nucleotide_to_int]\n\nencoded_sequences = [clean_and_encode(seq) for seq in df[\"sequence\"]]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T15:09:21.204551Z","iopub.execute_input":"2025-03-27T15:09:21.205106Z","iopub.status.idle":"2025-03-27T15:09:21.299214Z","shell.execute_reply.started":"2025-03-27T15:09:21.205075Z","shell.execute_reply":"2025-03-27T15:09:21.298240Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Filter out short sequences\nsequences = [seq for seq in encoded_sequences if len(seq) > 1]\nX = np.concatenate([np.array(seq).reshape(-1, 1) for seq in sequences])\nlengths = [len(seq) for seq in sequences]\n\n# Train a 4-state Multinomial HMM\nmodel = hmm.MultinomialHMM(n_components=4, n_iter=100, random_state=42)\nmodel.fit(X, lengths)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T15:09:25.264320Z","iopub.execute_input":"2025-03-27T15:09:25.264691Z","iopub.status.idle":"2025-03-27T15:09:26.646719Z","shell.execute_reply.started":"2025-03-27T15:09:25.264663Z","shell.execute_reply":"2025-03-27T15:09:26.645433Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"hmm_features = []\n\nfor seq in encoded_sequences:\n    if len(seq) < 2:\n        hmm_features.append({\n            \"hmm_entropy\": np.nan,\n            \"hmm_log_prob\": np.nan,\n            \"hmm_main_state\": np.nan\n        })\n        continue\n\n    X_seq = np.array(seq).reshape(-1, 1)\n    \n    # Decode sequence via Viterbi\n    logprob, hidden_states = model.decode(X_seq, algorithm=\"viterbi\")\n    state_counts = np.bincount(hidden_states, minlength=model.n_components)\n    \n    # Compute features\n    state_probs = state_counts / state_counts.sum()\n    hmm_entropy = entropy(state_probs)\n    main_state = np.argmax(state_counts)\n    \n    hmm_features.append({\n        \"hmm_entropy\": hmm_entropy,\n        \"hmm_log_prob\": logprob,\n        \"hmm_main_state\": main_state\n    })\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T15:09:30.118436Z","iopub.execute_input":"2025-03-27T15:09:30.118831Z","iopub.status.idle":"2025-03-27T15:09:31.091613Z","shell.execute_reply.started":"2025-03-27T15:09:30.118802Z","shell.execute_reply":"2025-03-27T15:09:31.090386Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create DataFrame and join with original\nfeatures_df = pd.DataFrame(hmm_features)\ndf = pd.concat([df, features_df], axis=1)\n# Save new version (if needed locally)\ndf.to_csv(\"train_sequences_with_hmm_features.csv\", index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T15:09:33.502132Z","iopub.execute_input":"2025-03-27T15:09:33.502539Z","iopub.status.idle":"2025-03-27T15:09:33.594391Z","shell.execute_reply.started":"2025-03-27T15:09:33.502506Z","shell.execute_reply":"2025-03-27T15:09:33.593249Z"}},"outputs":[],"execution_count":null}]}