{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":12024591,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# First Notebook for the Stanford RNA Kaggle\n\n## Exploratory Data Analysis of the RNA Sequence\n\n### Looking at the Histograms of the sequence Lengths and the occurences of the 4 Letters (the four nitrogenous bases are adenine, guanine, uracil and cytosine)","metadata":{"execution":{"iopub.status.busy":"2025-04-28T11:49:45.239912Z","iopub.execute_input":"2025-04-28T11:49:45.240249Z","iopub.status.idle":"2025-04-28T11:49:45.244823Z","shell.execute_reply.started":"2025-04-28T11:49:45.240221Z","shell.execute_reply":"2025-04-28T11:49:45.244001Z"}}},{"cell_type":"code","source":"### Loading Relevant Packages\n\nimport os\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:44.016104Z","iopub.execute_input":"2025-04-28T16:23:44.016477Z","iopub.status.idle":"2025-04-28T16:23:44.021370Z","shell.execute_reply.started":"2025-04-28T16:23:44.016427Z","shell.execute_reply":"2025-04-28T16:23:44.020340Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"current_directory = os.getcwd()\nprint(current_directory)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:44.022681Z","iopub.execute_input":"2025-04-28T16:23:44.022992Z","iopub.status.idle":"2025-04-28T16:23:44.044326Z","shell.execute_reply.started":"2025-04-28T16:23:44.022913Z","shell.execute_reply":"2025-04-28T16:23:44.043533Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path = \"/kaggle/input/stanford-rna-3d-folding/\"\nos.chdir(path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:44.045933Z","iopub.execute_input":"2025-04-28T16:23:44.046230Z","iopub.status.idle":"2025-04-28T16:23:44.062866Z","shell.execute_reply.started":"2025-04-28T16:23:44.046198Z","shell.execute_reply":"2025-04-28T16:23:44.061823Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sequence_raw = pd.read_csv(\"train_sequences.csv\")\ntrain_sequence_raw.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:44.063890Z","iopub.execute_input":"2025-04-28T16:23:44.064209Z","iopub.status.idle":"2025-04-28T16:23:44.128471Z","shell.execute_reply.started":"2025-04-28T16:23:44.064182Z","shell.execute_reply":"2025-04-28T16:23:44.127510Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sequence_raw.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:44.130416Z","iopub.execute_input":"2025-04-28T16:23:44.130741Z","iopub.status.idle":"2025-04-28T16:23:44.140499Z","shell.execute_reply.started":"2025-04-28T16:23:44.130718Z","shell.execute_reply":"2025-04-28T16:23:44.139690Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### The key in the training data seems to be the sequence of nitrogenous bases\n\n### Let's focus on that for now.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:44.141442Z","iopub.execute_input":"2025-04-28T16:23:44.141737Z","iopub.status.idle":"2025-04-28T16:23:44.155927Z","shell.execute_reply.started":"2025-04-28T16:23:44.141712Z","shell.execute_reply":"2025-04-28T16:23:44.154825Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sequence_train_raw = train_sequence_raw[\"sequence\"]\nsequence_train_raw[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:44.156908Z","iopub.execute_input":"2025-04-28T16:23:44.157185Z","iopub.status.idle":"2025-04-28T16:23:44.173449Z","shell.execute_reply.started":"2025-04-28T16:23:44.157165Z","shell.execute_reply":"2025-04-28T16:23:44.172450Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sequence_train_length = sequence_train_raw.apply(lambda x: len(x))\nprint(np.mean(sequence_train_length))\nprint(np.median(sequence_train_length))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:44.174445Z","iopub.execute_input":"2025-04-28T16:23:44.175354Z","iopub.status.idle":"2025-04-28T16:23:44.193147Z","shell.execute_reply.started":"2025-04-28T16:23:44.175316Z","shell.execute_reply":"2025-04-28T16:23:44.192165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### The mean length is about 162 nitrogenous bases.\n\n### The median length is about 40 nitrogenous bases.\n\n### The distribution is very skewed. Let's have a look.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:44.194231Z","iopub.execute_input":"2025-04-28T16:23:44.194578Z","iopub.status.idle":"2025-04-28T16:23:44.208868Z","shell.execute_reply.started":"2025-04-28T16:23:44.194555Z","shell.execute_reply":"2025-04-28T16:23:44.207976Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generate a Histogram of the RNA Chain Length measured in number of nitrogenous bases\n\n# This histogram has all of the lenghts included\n\ndata = sequence_train_length\n\n# Define colors for the histogram bars\ncolors = plt.cm.plasma(np.linspace(0, 1, 15))  # 15 colors from 'plasma' colormap\n\n# Create the histogram\nfig, ax = plt.subplots(figsize=(10, 6))\n\n# 'bins' controls the number of bars\nn, bins, patches = ax.hist(data, bins=30, edgecolor='black')\n\n# Color each bar individually\nfor patch, color in zip(patches, colors):\n    patch.set_facecolor(color)\n\n# Set y-axis to log scale\nax.set_yscale('log')\n\n# Fancy styling\nax.set_title('Histogram (Log Scale)', fontsize=18)\nax.set_xlabel('Value', fontsize=14)\nax.set_ylabel('Frequency (log scale)', fontsize=14)\nax.grid(True, which=\"both\", linestyle='--', alpha=0.6)\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:44.210055Z","iopub.execute_input":"2025-04-28T16:23:44.210390Z","iopub.status.idle":"2025-04-28T16:23:44.662482Z","shell.execute_reply.started":"2025-04-28T16:23:44.210365Z","shell.execute_reply":"2025-04-28T16:23:44.661568Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generate a Histogram of the RNA Chain Length measured in number of nitrogenous bases\n\n# This histogram has only lengths of 1,000 or less\n\ndata = [x for x in sequence_train_length if x <= 1000]\n\n# Define colors for the histogram bars\ncolors = plt.cm.plasma(np.linspace(0, 1, 15))  # 15 colors from 'plasma' colormap\n\n# Create the histogram\nfig, ax = plt.subplots(figsize=(12, 6))\n\n# 'bins' controls the number of bars\nn, bins, patches = ax.hist(data, bins=30, edgecolor='black')\n\n# Color each bar individually\nfor patch, color in zip(patches, colors):\n    patch.set_facecolor(color)\n\n# Set y-axis to log scale\nax.set_yscale('log')\n\n# Fancy styling\nax.set_title('Histogram (Log Scale)', fontsize=18)\nax.set_xlabel('Value', fontsize=14)\nax.set_ylabel('Frequency (log scale)', fontsize=14)\nax.grid(True, which=\"both\", linestyle='--', alpha=0.6)\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:44.665343Z","iopub.execute_input":"2025-04-28T16:23:44.665701Z","iopub.status.idle":"2025-04-28T16:23:45.088513Z","shell.execute_reply.started":"2025-04-28T16:23:44.665681Z","shell.execute_reply":"2025-04-28T16:23:45.087536Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Let's do some statistics on the occurences of nitrogenous bases","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:45.089376Z","iopub.execute_input":"2025-04-28T16:23:45.089654Z","iopub.status.idle":"2025-04-28T16:23:45.093639Z","shell.execute_reply.started":"2025-04-28T16:23:45.089634Z","shell.execute_reply":"2025-04-28T16:23:45.092739Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# I am going to concatenate all of the sequences together using this format\n\n# I will ignore the ends for now and treat everything as a long continous chain to run some statisics.\n\nsequence_train_concatenated = sequence_train_raw[0] + sequence_train_raw[1]\nsequence_train_concatenated","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:45.094656Z","iopub.execute_input":"2025-04-28T16:23:45.094953Z","iopub.status.idle":"2025-04-28T16:23:45.112424Z","shell.execute_reply.started":"2025-04-28T16:23:45.094932Z","shell.execute_reply":"2025-04-28T16:23:45.111544Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Here is the concatenation of all of the training set\n\nsequence_train_concat = list()\nfor i in sequence_train_raw:\n    sequence_train_concat += i","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:45.113512Z","iopub.execute_input":"2025-04-28T16:23:45.114132Z","iopub.status.idle":"2025-04-28T16:23:45.133131Z","shell.execute_reply.started":"2025-04-28T16:23:45.114103Z","shell.execute_reply":"2025-04-28T16:23:45.131924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check what are the unique elements in the concatenation\n\nset(sequence_train_concat)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:45.133987Z","iopub.execute_input":"2025-04-28T16:23:45.134254Z","iopub.status.idle":"2025-04-28T16:23:45.154385Z","shell.execute_reply.started":"2025-04-28T16:23:45.134224Z","shell.execute_reply":"2025-04-28T16:23:45.153383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### I observe that there are the four nitrogenous bases (adenine, guanine, uracil and cytosine)\n\n### However, there are also '-' or 'X' which occur and I will intentionally ignore for now","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:45.155549Z","iopub.execute_input":"2025-04-28T16:23:45.155890Z","iopub.status.idle":"2025-04-28T16:23:45.168240Z","shell.execute_reply.started":"2025-04-28T16:23:45.155863Z","shell.execute_reply":"2025-04-28T16:23:45.167332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sequence_train_concat_df = pd.DataFrame(sequence_train_concat)\nsequence_train_concat_df.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:45.169126Z","iopub.execute_input":"2025-04-28T16:23:45.169405Z","iopub.status.idle":"2025-04-28T16:23:45.211025Z","shell.execute_reply.started":"2025-04-28T16:23:45.169384Z","shell.execute_reply":"2025-04-28T16:23:45.210110Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sequence_train_concat_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:45.212068Z","iopub.execute_input":"2025-04-28T16:23:45.212432Z","iopub.status.idle":"2025-04-28T16:23:45.220240Z","shell.execute_reply.started":"2025-04-28T16:23:45.212397Z","shell.execute_reply":"2025-04-28T16:23:45.219365Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Let's now do some statistics on the four nitrogenous bases (adenine, guanine, uracil and cytosine)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:45.221063Z","iopub.execute_input":"2025-04-28T16:23:45.221326Z","iopub.status.idle":"2025-04-28T16:23:45.238649Z","shell.execute_reply.started":"2025-04-28T16:23:45.221305Z","shell.execute_reply":"2025-04-28T16:23:45.237732Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from collections import Counter\n\n# Example DataFrame\ndf = sequence_train_concat_df\n\n# Function to count letters in a text\ndef count_letters(text):\n    return Counter(c for c in text if c.isalpha())  # only letters\n\n# Apply the function to the whole column\nletter_counts = df.apply(count_letters)\n\n# Sum up all the counters\ntotal_counts = sum(letter_counts, Counter())\n\n# Convert to a DataFrame\nletter_df = pd.DataFrame.from_dict(total_counts, orient='index', columns=['count'])\n\n# Pivot-style: Sort by letter\nletter_df_1L = letter_df.sort_values(by='count', ascending=False)\n\nprint(letter_df_1L)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:45.239544Z","iopub.execute_input":"2025-04-28T16:23:45.239844Z","iopub.status.idle":"2025-04-28T16:23:45.287038Z","shell.execute_reply.started":"2025-04-28T16:23:45.239817Z","shell.execute_reply":"2025-04-28T16:23:45.286294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### What is the probability for each of the nitrogenous base (adenine, guanine, uracil and cytosine)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:45.287862Z","iopub.execute_input":"2025-04-28T16:23:45.288215Z","iopub.status.idle":"2025-04-28T16:23:45.292104Z","shell.execute_reply.started":"2025-04-28T16:23:45.288194Z","shell.execute_reply":"2025-04-28T16:23:45.291347Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"prob_letter_1L = 100 * letter_df/(sum(letter_df_1L[\"count\"]))\nprob_letter_1L = prob_letter_1L.sort_values(by='count', ascending=False)\nprob_letter_1L","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:45.293246Z","iopub.execute_input":"2025-04-28T16:23:45.293589Z","iopub.status.idle":"2025-04-28T16:23:45.315337Z","shell.execute_reply.started":"2025-04-28T16:23:45.293552Z","shell.execute_reply":"2025-04-28T16:23:45.314349Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## So Guanine is the most frequent and Uracil is the least frequent.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:45.316307Z","iopub.execute_input":"2025-04-28T16:23:45.316673Z","iopub.status.idle":"2025-04-28T16:23:45.332556Z","shell.execute_reply.started":"2025-04-28T16:23:45.316639Z","shell.execute_reply":"2025-04-28T16:23:45.331444Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Let's have a histogram for this\n\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom collections import Counter\n\n# Example DataFrame (with fewer letters for demonstration)\ndf = sequence_train_concat_df\n\n# Function to count letters\ndef count_letters(text):\n    return Counter(c for c in text if c.isalpha())\n\n# Count letters\nletter_counts = df.apply(count_letters)\ntotal_counts = sum(letter_counts, Counter())\nletter_df = pd.DataFrame.from_dict(total_counts, orient='index', columns=['count'])\nletter_df = letter_df.sort_values(by='count', ascending=False)\n\n# Define 4 distinct colors manually\ncolors = ['#1f77b4', '#ff7f0e', '#2ca02c', '#d62728']  # blue, orange, green, red\n\n# Assign colors cyclically\nbar_colors = [colors[i % len(colors)] for i in range(len(letter_df))]\n\n# Plotting\nfig, ax = plt.subplots(figsize=(8, 5))\n\nbars = ax.bar(letter_df.index, letter_df['count'], color=bar_colors, edgecolor='black')\n\n# Styling\nax.set_xlabel('Nitrogenous Base', fontsize=14)\nax.set_ylabel('Count', fontsize=14)\nax.set_title('1-Letter Frequency Histogram', fontsize=18)\nax.grid(axis='y', linestyle='--', alpha=0.7)\n\n# Add value labels on top\nfor bar in bars:\n    height = bar.get_height()\n\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:45.333652Z","iopub.execute_input":"2025-04-28T16:23:45.333930Z","iopub.status.idle":"2025-04-28T16:23:45.566138Z","shell.execute_reply.started":"2025-04-28T16:23:45.333910Z","shell.execute_reply":"2025-04-28T16:23:45.565268Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Let's go up in complexity and try to do some statistics on 2 consecutive nitrogenous bases or 2-letters","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:45.567276Z","iopub.execute_input":"2025-04-28T16:23:45.567598Z","iopub.status.idle":"2025-04-28T16:23:45.571706Z","shell.execute_reply.started":"2025-04-28T16:23:45.567570Z","shell.execute_reply":"2025-04-28T16:23:45.570826Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Let's first contatenate the entire dataset to build a 2-Letter dataset\n\nsequence_train_concat_2 = [sequence_train_concat[i] + sequence_train_concat[i+1] for i in range(len(sequence_train_concat) - 1)]\nsequence_train_concat_2[:10]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:45.572511Z","iopub.execute_input":"2025-04-28T16:23:45.572742Z","iopub.status.idle":"2025-04-28T16:23:45.614861Z","shell.execute_reply.started":"2025-04-28T16:23:45.572724Z","shell.execute_reply":"2025-04-28T16:23:45.613912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Let's run some statistics on the 2 consecutive nitrogenous bases or 2-letters dataset\n\nfrom collections import Counter\n\n# Example DataFrame\ndf = pd.DataFrame(sequence_train_concat_2)\n\n# Function to count letters in a text\ndef count_letters(text):\n    return Counter(c for c in text if c.isalpha())  # only letters\n\n# Apply the function to the whole column\nletter_counts = df.apply(count_letters)\n\n# Sum up all the counters\ntotal_counts = sum(letter_counts, Counter())\n\n# Convert to a DataFrame\nletter_df = pd.DataFrame.from_dict(total_counts, orient='index', columns=['count'])\n\n# Pivot-style: Sort by letter\nletter_df = letter_df.sort_index()\n\nletter_df = letter_df.sort_values(by='count', ascending=False)\n\n# Pivot-style: Sort by letter\nletter_df_L2 = letter_df.sort_values(by='count', ascending=False)\n\nprint(letter_df_L2)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:45.615937Z","iopub.execute_input":"2025-04-28T16:23:45.616247Z","iopub.status.idle":"2025-04-28T16:23:45.668351Z","shell.execute_reply.started":"2025-04-28T16:23:45.616226Z","shell.execute_reply":"2025-04-28T16:23:45.667316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Here are the probabilities for the 2 consecutive nitrogenous bases or 2-letters dataset\n\nprob_letter_df_L2 = 100 * letter_df/(sum(letter_df_L2[\"count\"]))\nprob_letter_df_L2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:45.669378Z","iopub.execute_input":"2025-04-28T16:23:45.669681Z","iopub.status.idle":"2025-04-28T16:23:45.679102Z","shell.execute_reply.started":"2025-04-28T16:23:45.669660Z","shell.execute_reply":"2025-04-28T16:23:45.678349Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## So the most common pattern is to have 2 consecutive guanines.\n\n## The next most common pattern is to have a guanine next to a cytosine. ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:45.684798Z","iopub.execute_input":"2025-04-28T16:23:45.685087Z","iopub.status.idle":"2025-04-28T16:23:45.694816Z","shell.execute_reply.started":"2025-04-28T16:23:45.685066Z","shell.execute_reply":"2025-04-28T16:23:45.693958Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Let's generate a Histogram for 2 consecutive nitrogenous bases or 2-letters dataset\n\n\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom collections import Counter\n\n# Example DataFrame (with fewer letters for demonstration)\ndf = pd.DataFrame(sequence_train_concat_2)\n\n# Function to count letters\ndef count_letters(text):\n    return Counter(c for c in text if c.isalpha())\n\n# Count letters\nletter_counts = df.apply(count_letters)\ntotal_counts = sum(letter_counts, Counter())\nletter_df = pd.DataFrame.from_dict(total_counts, orient='index', columns=['count'])\nletter_df = letter_df.sort_index()\n\n# Define 4 distinct colors manually\ncolors = ['#1f77b4', '#ff7f0e', '#2ca02c', '#d62728']  # blue, orange, green, red\n\n# Assign colors cyclically\nbar_colors = [colors[i % len(colors)] for i in range(len(letter_df))]\n\n# Plotting\nfig, ax = plt.subplots(figsize=(8, 5))\n\nbars = ax.bar(letter_df.index, letter_df['count'], color=bar_colors, edgecolor='black')\n\n# Styling\nax.set_xlabel('2 Consecutive Nitrogenous Bases', fontsize=14)\nax.set_ylabel('Count', fontsize=14)\nax.set_title('2 Letter Frequency Histogram', fontsize=18)\nax.grid(axis='y', linestyle='--', alpha=0.7)\n\n# Add value labels on top\nfor bar in bars:\n    height = bar.get_height()\n\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:45.695856Z","iopub.execute_input":"2025-04-28T16:23:45.696182Z","iopub.status.idle":"2025-04-28T16:23:46.015801Z","shell.execute_reply.started":"2025-04-28T16:23:45.696158Z","shell.execute_reply":"2025-04-28T16:23:46.014830Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Let's go up in complexity and try to do some statistics on 3 consecutive nitrogenous bases or 3-letters","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:46.016837Z","iopub.execute_input":"2025-04-28T16:23:46.017103Z","iopub.status.idle":"2025-04-28T16:23:46.021524Z","shell.execute_reply.started":"2025-04-28T16:23:46.017076Z","shell.execute_reply":"2025-04-28T16:23:46.020521Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Let's build the 3-letter dataset using the same approach\n\nsequence_train_concat_3 = [sequence_train_concat[i] + sequence_train_concat[i+1] + sequence_train_concat[i+2] for i in range(len(sequence_train_concat) - 2)]\nsequence_train_concat_3[:10]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:46.022506Z","iopub.execute_input":"2025-04-28T16:23:46.022840Z","iopub.status.idle":"2025-04-28T16:23:46.075152Z","shell.execute_reply.started":"2025-04-28T16:23:46.022819Z","shell.execute_reply":"2025-04-28T16:23:46.074372Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Let's do the counting\n\nfrom collections import Counter\n\n# Example DataFrame\ndf = pd.DataFrame(sequence_train_concat_3)\n\n# Function to count letters in a text\ndef count_letters(text):\n    return Counter(c for c in text if c.isalpha())  # only letters\n\n# Apply the function to the whole column\nletter_counts = df.apply(count_letters)\n\n# Sum up all the counters\ntotal_counts = sum(letter_counts, Counter())\n\n# Convert to a DataFrame\nletter_df_3L = pd.DataFrame.from_dict(total_counts, orient='index', columns=['count'])\n\n# Pivot-style: Sort by letter\nletter_df_3L = letter_df_3L.sort_values(by='count', ascending=False)\n\npd.set_option('display.max_rows', None) \n\nprint(letter_df_3L)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:46.076164Z","iopub.execute_input":"2025-04-28T16:23:46.076516Z","iopub.status.idle":"2025-04-28T16:23:46.129172Z","shell.execute_reply.started":"2025-04-28T16:23:46.076486Z","shell.execute_reply":"2025-04-28T16:23:46.128303Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Let's calculate the probabilities\n\nprob_letter_df_L3 = 100 * letter_df_3L/(sum(letter_df_3L[\"count\"]))\n\nprob_letter_df_L3 = prob_letter_df_L3.sort_values(by='count', ascending=False)\n\nprob_letter_df_L3","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:46.130121Z","iopub.execute_input":"2025-04-28T16:23:46.130438Z","iopub.status.idle":"2025-04-28T16:23:46.142619Z","shell.execute_reply.started":"2025-04-28T16:23:46.130410Z","shell.execute_reply":"2025-04-28T16:23:46.141664Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## So the most common pattern is to have 3 consecutive guanines.\n\n## The next most common pattern is to have a cytosine next to 2 consecutive guanines.\n\n## The next most common patter is to have two consecutive cytosine next to one guanine.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:46.143708Z","iopub.execute_input":"2025-04-28T16:23:46.144031Z","iopub.status.idle":"2025-04-28T16:23:46.158135Z","shell.execute_reply.started":"2025-04-28T16:23:46.144002Z","shell.execute_reply":"2025-04-28T16:23:46.157111Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Let's generate a histogram for the 3 consecutive nitrogenous bases or 3-letters\n\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom collections import Counter\n\n# Example DataFrame (with fewer letters for demonstration)\ndf = pd.DataFrame(sequence_train_concat_3)\n\n# Function to count letters\ndef count_letters(text):\n    return Counter(c for c in text if c.isalpha())\n\n# Count letters\nletter_counts = df.apply(count_letters)\ntotal_counts = sum(letter_counts, Counter())\nletter_df = pd.DataFrame.from_dict(total_counts, orient='index', columns=['count'])\nletter_df = letter_df.sort_index()\n\n# Define 4 distinct colors manually\ncolors = ['#1f77b4', '#ff7f0e', '#2ca02c', '#d62728']  # blue, orange, green, red\n\n# Assign colors cyclically\nbar_colors = [colors[i % len(colors)] for i in range(len(letter_df))]\n\n# Plotting\nfig, ax = plt.subplots(figsize=(12, 5))\n\nbars = ax.bar(letter_df.index, letter_df['count'], color=bar_colors, edgecolor='black')\n\n# Styling\nax.set_xlabel('3 consecutive nitrogenous bases', fontsize=14)\nax.set_ylabel('Count', fontsize=14)\nax.set_title('3_Letter Frequency Histogram', fontsize=18)\nax.grid(axis='y', linestyle='--', alpha=0.7)\n\n# Add value labels on top\nfor bar in bars:\n    height = bar.get_height()\n\nplt.xticks(rotation=45)\n\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:46.159186Z","iopub.execute_input":"2025-04-28T16:23:46.159501Z","iopub.status.idle":"2025-04-28T16:23:46.846382Z","shell.execute_reply.started":"2025-04-28T16:23:46.159447Z","shell.execute_reply":"2025-04-28T16:23:46.845495Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Let's build a simple machine learning model to predict the next consecutive nitrogenous bases.\n\nIn other words, let's build a model that takes as input two consecutive nitrogenous bases and tries to guess the next base.\n","metadata":{"execution":{"iopub.status.busy":"2025-04-28T16:16:09.692187Z","iopub.execute_input":"2025-04-28T16:16:09.692474Z","iopub.status.idle":"2025-04-28T16:16:09.697898Z","shell.execute_reply.started":"2025-04-28T16:16:09.692426Z","shell.execute_reply":"2025-04-28T16:16:09.696144Z"}}},{"cell_type":"code","source":"### Let's build the necessary dataset for this\n\nL2_Letters = [x[0] + x[1] for x in sequence_train_concat_3]\n\nL_first_Letter = [x[0] for x in sequence_train_concat_3]\n\nL_last_Letter = [x[1] for x in sequence_train_concat_3]\n\nL_3_label = [x[2] for x in sequence_train_concat_3]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:46.847239Z","iopub.execute_input":"2025-04-28T16:23:46.847497Z","iopub.status.idle":"2025-04-28T16:23:46.888753Z","shell.execute_reply.started":"2025-04-28T16:23:46.847448Z","shell.execute_reply":"2025-04-28T16:23:46.887829Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_2_Letters = pd.DataFrame(L2_Letters)\ndf_2_Letters.columns = [\"feature_1\"]\n\ndf_L_first_Letter = pd.DataFrame(L_first_Letter)\ndf_L_first_Letter.columns = [\"feature_2\"]\n\ndf_L_last_Letter = pd.DataFrame(L_last_Letter)\ndf_L_last_Letter.columns = [\"feature_3\"]\n\ndf_3_label = pd.DataFrame(L_3_label)\ndf_3_label.columns = [\"label\"]\n\ndf = pd.concat([df_2_Letters, df_L_first_Letter, df_L_last_Letter,  df_3_label], axis = 1)\ndf.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:46.889938Z","iopub.execute_input":"2025-04-28T16:23:46.890855Z","iopub.status.idle":"2025-04-28T16:23:47.006811Z","shell.execute_reply.started":"2025-04-28T16:23:46.890829Z","shell.execute_reply":"2025-04-28T16:23:47.006038Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Now that we have the right dataset\n\n### Let's build a simple machine learning model using Naive Bayes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:47.008029Z","iopub.execute_input":"2025-04-28T16:23:47.008353Z","iopub.status.idle":"2025-04-28T16:23:47.011925Z","shell.execute_reply.started":"2025-04-28T16:23:47.008323Z","shell.execute_reply":"2025-04-28T16:23:47.011173Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.naive_bayes import CategoricalNB\nfrom sklearn.metrics import accuracy_score\n\n# Example dataset with categorical features\n\n# Separate features and target\nX = df[['feature_1', 'feature_2', 'feature_3']]  # input features\ny = df['label']                   # target label\n\n# Encode the categorical features\nencoder = OrdinalEncoder()\nX_encoded = encoder.fit_transform(X)\n\n# Split into train and test sets\nX_train, X_test, y_train, y_test = train_test_split(X_encoded, y, test_size=0.2, random_state=42)\n\n# Train Categorical Naive Bayes model\nmodel = CategoricalNB()\nmodel.fit(X_train, y_train)\n\n# Optional: evaluate model\ny_pred = model.predict(X_test)\nprint(\"Test accuracy:\", accuracy_score(y_test, y_pred))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:47.012671Z","iopub.execute_input":"2025-04-28T16:23:47.012948Z","iopub.status.idle":"2025-04-28T16:23:47.507037Z","shell.execute_reply.started":"2025-04-28T16:23:47.012921Z","shell.execute_reply":"2025-04-28T16:23:47.506271Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### The model accuracy is low. ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:47.507904Z","iopub.execute_input":"2025-04-28T16:23:47.508164Z","iopub.status.idle":"2025-04-28T16:23:47.512233Z","shell.execute_reply.started":"2025-04-28T16:23:47.508143Z","shell.execute_reply":"2025-04-28T16:23:47.511500Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Let's look at the confusion matrix\n\nfrom sklearn.datasets import make_classification\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import ConfusionMatrixDisplay\n\n\n_ = ConfusionMatrixDisplay.from_estimator(model, X_test, y_test)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:47.513177Z","iopub.execute_input":"2025-04-28T16:23:47.513424Z","iopub.status.idle":"2025-04-28T16:23:47.908670Z","shell.execute_reply.started":"2025-04-28T16:23:47.513404Z","shell.execute_reply":"2025-04-28T16:23:47.907778Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Let's see how this model compares against always predicting 'G\", the most common one.\n\nfrom sklearn.dummy import DummyClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\n\n# Example data\nX = df[['feature_1', 'feature_2', 'feature_3']]  # input features\ny = df['label']                   # target label\n\n# Split into train/test\nX_train, X_test, y_train, y_test = train_test_split(X, y, random_state=42)\n\n# Create a baseline model\nbaseline_model = DummyClassifier(strategy=\"most_frequent\")\n\n# Train the model\nbaseline_model.fit(X_train, y_train)\n\n# Predict\ny_pred = baseline_model.predict(X_test)\n\n# Evaluate\nprint(\"Predictions:\", y_pred)\nprint(\"Accuracy:\", accuracy_score(y_test, y_pred))\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:47.909683Z","iopub.execute_input":"2025-04-28T16:23:47.910077Z","iopub.status.idle":"2025-04-28T16:23:48.019272Z","shell.execute_reply.started":"2025-04-28T16:23:47.910041Z","shell.execute_reply":"2025-04-28T16:23:48.018296Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### The Baseline model with the Guanine prediction is very similar to the Naive Bayes model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:48.020366Z","iopub.execute_input":"2025-04-28T16:23:48.020686Z","iopub.status.idle":"2025-04-28T16:23:48.025021Z","shell.execute_reply.started":"2025-04-28T16:23:48.020657Z","shell.execute_reply":"2025-04-28T16:23:48.024254Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Let's look at the confusion matrix\n\n### Let's look at the confusion matrix\n\nfrom sklearn.datasets import make_classification\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import ConfusionMatrixDisplay\n\n\n_ = ConfusionMatrixDisplay.from_estimator(baseline_model, X_test, y_test)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:48.025939Z","iopub.execute_input":"2025-04-28T16:23:48.026149Z","iopub.status.idle":"2025-04-28T16:23:48.570328Z","shell.execute_reply.started":"2025-04-28T16:23:48.026133Z","shell.execute_reply":"2025-04-28T16:23:48.569526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Let's try the same approach with xgboost","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:48.571239Z","iopub.execute_input":"2025-04-28T16:23:48.571577Z","iopub.status.idle":"2025-04-28T16:23:48.575154Z","shell.execute_reply.started":"2025-04-28T16:23:48.571544Z","shell.execute_reply":"2025-04-28T16:23:48.574490Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import OrdinalEncoder, LabelEncoder\nfrom xgboost import XGBClassifier\nfrom sklearn.metrics import accuracy_score\n\n# Example dataset with categorical features and categorical target\n\n# Separate features and target\n\nX = df[['feature_1', 'feature_2', 'feature_3']]\ny = df['label']\n\n# Encode categorical features\nfeature_encoder = OrdinalEncoder()\nX_encoded = feature_encoder.fit_transform(X)\n\n# Encode target labels\nlabel_encoder = LabelEncoder()\ny_encoded = label_encoder.fit_transform(y)\n\n# Split into train and test sets\nX_train, X_test, y_train, y_test = train_test_split(X_encoded, y_encoded, test_size=0.2, random_state=42)\n\n# Initialize and train the XGBoost classifier\nmodel = XGBClassifier(use_label_encoder=False, eval_metric='logloss')\nmodel.fit(X_train, y_train)\n\n# Optional: evaluate model\ny_pred_encoded = model.predict(X_test)\ny_pred = label_encoder.inverse_transform(y_pred_encoded)\ny_test_decoded = label_encoder.inverse_transform(y_test)\n\nprint(\"Test accuracy:\", accuracy_score(y_test_decoded, y_pred))\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:48.576216Z","iopub.execute_input":"2025-04-28T16:23:48.576601Z","iopub.status.idle":"2025-04-28T16:23:51.273129Z","shell.execute_reply.started":"2025-04-28T16:23:48.576567Z","shell.execute_reply":"2025-04-28T16:23:51.272415Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### The Baseline model with the Guanine prediction is very similar to the XGBoost","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:51.274012Z","iopub.execute_input":"2025-04-28T16:23:51.274295Z","iopub.status.idle":"2025-04-28T16:23:51.278367Z","shell.execute_reply.started":"2025-04-28T16:23:51.274267Z","shell.execute_reply":"2025-04-28T16:23:51.277650Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Let's look at the confusion matrix\n\nfrom sklearn.datasets import make_classification\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import ConfusionMatrixDisplay\n\n\n_ = ConfusionMatrixDisplay.from_estimator(model, X_test, y_test)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:51.279313Z","iopub.execute_input":"2025-04-28T16:23:51.279600Z","iopub.status.idle":"2025-04-28T16:23:51.699120Z","shell.execute_reply.started":"2025-04-28T16:23:51.279580Z","shell.execute_reply":"2025-04-28T16:23:51.698234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T16:23:51.699933Z","iopub.execute_input":"2025-04-28T16:23:51.700243Z","iopub.status.idle":"2025-04-28T16:23:51.705033Z","shell.execute_reply.started":"2025-04-28T16:23:51.700221Z","shell.execute_reply":"2025-04-28T16:23:51.704053Z"}},"outputs":[],"execution_count":null}]}