{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":12024591,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Second Notebook for the Stanford RNA Kaggle\n\n## Exploratory Data Analysis of the RNA Sequence\n\n### Looking at the distances and the angles between the C-1 atoms of consecutive nitrogenous bases\n","metadata":{}},{"cell_type":"code","source":"### Loading Relevant Packages\n\nimport os\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom matplotlib import colormaps","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:19:35.949189Z","iopub.execute_input":"2025-04-30T20:19:35.949534Z","iopub.status.idle":"2025-04-30T20:19:35.954382Z","shell.execute_reply.started":"2025-04-30T20:19:35.949484Z","shell.execute_reply":"2025-04-30T20:19:35.953468Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path = \"/kaggle/input/stanford-rna-3d-folding/\"\nos.chdir(path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:19:35.955803Z","iopub.execute_input":"2025-04-30T20:19:35.956110Z","iopub.status.idle":"2025-04-30T20:19:35.978444Z","shell.execute_reply.started":"2025-04-30T20:19:35.956087Z","shell.execute_reply":"2025-04-30T20:19:35.977557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_label_raw = pd.read_csv(\"train_labels.csv\")\ntrain_label_raw.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:19:35.979388Z","iopub.execute_input":"2025-04-30T20:19:35.979784Z","iopub.status.idle":"2025-04-30T20:19:36.247630Z","shell.execute_reply.started":"2025-04-30T20:19:35.979751Z","shell.execute_reply":"2025-04-30T20:19:36.246622Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_label_raw.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:19:36.249292Z","iopub.execute_input":"2025-04-30T20:19:36.249581Z","iopub.status.idle":"2025-04-30T20:19:36.290374Z","shell.execute_reply.started":"2025-04-30T20:19:36.249559Z","shell.execute_reply":"2025-04-30T20:19:36.289281Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_label_raw.tail()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:19:36.291421Z","iopub.execute_input":"2025-04-30T20:19:36.291752Z","iopub.status.idle":"2025-04-30T20:19:36.303722Z","shell.execute_reply.started":"2025-04-30T20:19:36.291721Z","shell.execute_reply":"2025-04-30T20:19:36.302706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Let's focus on calculating the Euclidean distances between 2 consecutive C-1 atoms\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:19:36.304686Z","iopub.execute_input":"2025-04-30T20:19:36.304934Z","iopub.status.idle":"2025-04-30T20:19:36.322846Z","shell.execute_reply.started":"2025-04-30T20:19:36.304913Z","shell.execute_reply":"2025-04-30T20:19:36.321441Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Define a function\n\ndef euclidean_distance(df):\n\n    j = 0\n\n    x1_first = df[\"x_1\"].iloc[j]\n    y1_first = df[\"y_1\"].iloc[j]\n    z1_first = df[\"z_1\"].iloc[j]\n    \n    j = 1\n    \n    x1_second = df[\"x_1\"].iloc[j]\n    y1_second = df[\"y_1\"].iloc[j]\n    z1_second = df[\"z_1\"].iloc[j]\n    \n    d_Euclidean = np.sqrt(np.power( (x1_second - x1_first), 2) + np.power( (y1_second - y1_first), 2) + np.power( (z1_second - z1_first), 2) )\n\n    return d_Euclidean\n\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:19:36.323807Z","iopub.execute_input":"2025-04-30T20:19:36.324329Z","iopub.status.idle":"2025-04-30T20:19:36.346718Z","shell.execute_reply.started":"2025-04-30T20:19:36.324293Z","shell.execute_reply":"2025-04-30T20:19:36.345290Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"euclidean_distance(train_label_raw)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:19:36.347897Z","iopub.execute_input":"2025-04-30T20:19:36.348267Z","iopub.status.idle":"2025-04-30T20:19:36.372388Z","shell.execute_reply.started":"2025-04-30T20:19:36.348237Z","shell.execute_reply":"2025-04-30T20:19:36.371396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def two_letter_identification(df, i):\n    return df[\"resname\"][i] + df[\"resname\"][i+1]\n\ntwo_letter_identification(train_label_raw, 0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:19:36.375038Z","iopub.execute_input":"2025-04-30T20:19:36.375342Z","iopub.status.idle":"2025-04-30T20:19:36.401437Z","shell.execute_reply.started":"2025-04-30T20:19:36.375317Z","shell.execute_reply":"2025-04-30T20:19:36.399977Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"L_index = list()\nL_2_Letters = list()\nL_2_distance = list()\nL_data_distance = list()\n\n\n# Go through the entire dataset row by row\nfor i in range(0, len(train_label_raw) - 1):\n\n    # Check that the 2 consecutive C-1 atoms are part of the same RNA chain\n    if train_label_raw[\"resid\"].iloc[i] == (train_label_raw[\"resid\"].iloc[(i+1)] - 1):\n\n        # Store the index\n        L_index.append(i)\n\n        # Store the 2 consecutive letters \n        L_2_Letters.append(two_letter_identification(train_label_raw[i:(i+2)], i))\n\n        # Store the Euclidean distance between the 2 consecutive letters\n        L_2_distance.append(euclidean_distance(train_label_raw[i:(i+2)]))\n\n        # Make a list of list containing the index, the 2 consecutive letters and the Euclidean distance\n        L_data_distance.append([i, two_letter_identification(train_label_raw[i:(i+2)], i), euclidean_distance(train_label_raw[i:(i+2)])])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:19:36.402188Z","iopub.execute_input":"2025-04-30T20:19:36.402447Z","iopub.status.idle":"2025-04-30T20:20:37.668866Z","shell.execute_reply.started":"2025-04-30T20:19:36.402426Z","shell.execute_reply":"2025-04-30T20:20:37.668019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Check the list of list\n\nL_data_distance[:10]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:20:37.670005Z","iopub.execute_input":"2025-04-30T20:20:37.670651Z","iopub.status.idle":"2025-04-30T20:20:37.678275Z","shell.execute_reply.started":"2025-04-30T20:20:37.670486Z","shell.execute_reply":"2025-04-30T20:20:37.677343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Create a dataframe\n\ndf_data_distances_raw = pd.DataFrame(L_data_distance)\ndf_data_distances_raw.columns = [\"index\",\"pair\",\"distance\"]\ndf_data_distances_raw","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:20:37.679273Z","iopub.execute_input":"2025-04-30T20:20:37.679609Z","iopub.status.idle":"2025-04-30T20:20:37.789547Z","shell.execute_reply.started":"2025-04-30T20:20:37.679581Z","shell.execute_reply":"2025-04-30T20:20:37.788338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Remove the NaN\n\ndf_data_distances = df_data_distances_raw[df_data_distances_raw[\"distance\"] == df_data_distances_raw[\"distance\"]]\ndf_data_distances","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:20:37.790662Z","iopub.execute_input":"2025-04-30T20:20:37.790977Z","iopub.status.idle":"2025-04-30T20:20:37.811696Z","shell.execute_reply.started":"2025-04-30T20:20:37.790949Z","shell.execute_reply":"2025-04-30T20:20:37.810555Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate the mean distance in Angstroms between pairs\n\ngrouped = df_data_distances.groupby(\"pair\")[\"distance\"].mean(numeric_only=True)\ngrouped = grouped.sort_values(ascending=False)\ngrouped","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:20:37.812724Z","iopub.execute_input":"2025-04-30T20:20:37.813032Z","iopub.status.idle":"2025-04-30T20:20:37.842391Z","shell.execute_reply.started":"2025-04-30T20:20:37.813006Z","shell.execute_reply":"2025-04-30T20:20:37.841405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot the mean distance in Angstroms between pairs\n\nplt.figure(figsize=(12, 6))\nnum_pairs = len(grouped)\ncmap = colormaps['Blues']\ncolors = [cmap(0.6 + 0.4 * i / max(num_pairs - 1, 1)) for i in range(num_pairs)]  # dark blues only\n\n# Plot\nplt.figure(figsize=(12, 6))\nplt.bar(grouped.index, grouped.values, color=colors)\n\n# Styling\nplt.title(\"Average Distance by Pair\", fontsize=16)\nplt.xlabel(\"Pair\", fontsize=12)\nplt.ylabel(\"Average Distance\", fontsize=12)\nplt.xticks(rotation=45, ha='right')\nplt.grid(axis='y', linestyle='--', alpha=0.7)\nplt.tight_layout()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:20:37.843406Z","iopub.execute_input":"2025-04-30T20:20:37.843750Z","iopub.status.idle":"2025-04-30T20:20:38.159780Z","shell.execute_reply.started":"2025-04-30T20:20:37.843727Z","shell.execute_reply":"2025-04-30T20:20:38.158843Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot the mean standard deviation for the distance (Angstroms) between pairs\n\ndf_data_distances.groupby(\"pair\")[\"distance\"].std(numeric_only=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:20:38.160757Z","iopub.execute_input":"2025-04-30T20:20:38.161004Z","iopub.status.idle":"2025-04-30T20:20:38.193221Z","shell.execute_reply.started":"2025-04-30T20:20:38.160982Z","shell.execute_reply":"2025-04-30T20:20:38.192207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nfrom matplotlib import colormaps\n\n# Calculate mean and standard deviation\nmean_dist = df_data_distances.groupby(\"pair\")[\"distance\"].mean(numeric_only=True)\nstd_dist = df_data_distances.groupby(\"pair\")[\"distance\"].std(numeric_only=True)\n\n# Sort by mean for consistency\nmean_dist = mean_dist.sort_values(ascending=False)\nstd_dist = std_dist[mean_dist.index]  # reindex to match order\n\n# Set up distinct colors\nnum_pairs = len(mean_dist)\ncmap = colormaps['Blues']\ncolors = [cmap(0.6 + 0.4 * i / max(num_pairs - 1, 1)) for i in range(num_pairs)]  # dark blues only\n\n# Plot\nplt.figure(figsize=(12, 6))\nplt.bar(mean_dist.index, mean_dist.values, yerr=std_dist.values, color=colors, capsize=5)\n\n# Styling\nplt.title(\"Average Distance by Pair (with Standard Deviation)\", fontsize=16)\nplt.xlabel(\"Pair\", fontsize=12)\nplt.ylabel(\"Average Distance\", fontsize=12)\nplt.xticks(rotation=45, ha='right')\nplt.grid(axis='y', linestyle='--', alpha=0.7)\nplt.tight_layout()\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:20:38.194252Z","iopub.execute_input":"2025-04-30T20:20:38.194511Z","iopub.status.idle":"2025-04-30T20:20:38.548751Z","shell.execute_reply.started":"2025-04-30T20:20:38.194470Z","shell.execute_reply":"2025-04-30T20:20:38.547643Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate a histogram of the distances for each pair\n\nimport matplotlib.pyplot as plt\nimport numpy as np\n\n# Define number of bins\nnum_bins = 35\n\n# Compute global min and max for distances\nxmin = df_data_distances[\"distance\"].min()\nxmax = df_data_distances[\"distance\"].max()\n\n# Generate consistent bin edges\nbins = np.linspace(xmin, xmax, num_bins + 1)\n\n# Loop through each pair\nfor pair in df_data_distances[\"pair\"].unique():\n    subset = df_data_distances[df_data_distances[\"pair\"] == pair][\"distance\"].dropna()\n    \n    if subset.empty:\n        continue\n\n    plt.figure(figsize=(8, 4))\n    plt.hist(subset, bins=bins, color='skyblue', edgecolor='black')\n    plt.xlim(xmin, xmax)\n    plt.title(f\"Histogram of Distances for Pair: {pair}\")\n    plt.xlabel(\"Distance\")\n    plt.ylabel(\"Frequency\")\n    plt.grid(axis='y', linestyle='--', alpha=0.6)\n    plt.tight_layout()\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:20:38.549705Z","iopub.execute_input":"2025-04-30T20:20:38.549998Z","iopub.status.idle":"2025-04-30T20:20:43.749478Z","shell.execute_reply.started":"2025-04-30T20:20:38.549979Z","shell.execute_reply":"2025-04-30T20:20:43.748392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### These are very asymetric distriubtions. Some of them appear to have very large distances on the tails.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:20:43.750557Z","iopub.execute_input":"2025-04-30T20:20:43.750892Z","iopub.status.idle":"2025-04-30T20:20:43.756338Z","shell.execute_reply.started":"2025-04-30T20:20:43.750863Z","shell.execute_reply":"2025-04-30T20:20:43.754980Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_data_distances[df_data_distances[\"distance\"] >20]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:20:43.757467Z","iopub.execute_input":"2025-04-30T20:20:43.757763Z","iopub.status.idle":"2025-04-30T20:20:43.788477Z","shell.execute_reply.started":"2025-04-30T20:20:43.757741Z","shell.execute_reply":"2025-04-30T20:20:43.787229Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# This sounds off. Either a data problem, an experimental problem or an exception.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:20:43.789739Z","iopub.execute_input":"2025-04-30T20:20:43.790118Z","iopub.status.idle":"2025-04-30T20:20:43.798945Z","shell.execute_reply.started":"2025-04-30T20:20:43.790086Z","shell.execute_reply":"2025-04-30T20:20:43.798038Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# There are some large distances of over 20 Angstroms. This is surprising. Let's have a sanity chec on one","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:20:43.800087Z","iopub.execute_input":"2025-04-30T20:20:43.800851Z","iopub.status.idle":"2025-04-30T20:20:43.817091Z","shell.execute_reply.started":"2025-04-30T20:20:43.800813Z","shell.execute_reply":"2025-04-30T20:20:43.815981Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_label_raw.iloc[12182:12186]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:20:43.818006Z","iopub.execute_input":"2025-04-30T20:20:43.818249Z","iopub.status.idle":"2025-04-30T20:20:43.843295Z","shell.execute_reply.started":"2025-04-30T20:20:43.818229Z","shell.execute_reply":"2025-04-30T20:20:43.842450Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Yes, there is a large jump from 12183 to 12184.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:20:43.847022Z","iopub.execute_input":"2025-04-30T20:20:43.847333Z","iopub.status.idle":"2025-04-30T20:20:43.862128Z","shell.execute_reply.started":"2025-04-30T20:20:43.847311Z","shell.execute_reply":"2025-04-30T20:20:43.861163Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Now, let's focus on the angles (theta) formed by the 3 consecutive C-1 atoms","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:20:43.862994Z","iopub.execute_input":"2025-04-30T20:20:43.863240Z","iopub.status.idle":"2025-04-30T20:20:43.881817Z","shell.execute_reply.started":"2025-04-30T20:20:43.863219Z","shell.execute_reply":"2025-04-30T20:20:43.880935Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Define a function\n\ndef angle_theta(df):\n\n    j = 0\n\n    x1_first = df[\"x_1\"].iloc[j]\n    y1_first = df[\"y_1\"].iloc[j]\n    z1_first = df[\"z_1\"].iloc[j]\n    \n    j = 1\n    \n    x1_second = df[\"x_1\"].iloc[j]\n    y1_second = df[\"y_1\"].iloc[j]\n    z1_second = df[\"z_1\"].iloc[j]\n\n    j = 2\n\n    x1_third = df[\"x_1\"].iloc[j]\n    y1_third = df[\"y_1\"].iloc[j]\n    z1_third = df[\"z_1\"].iloc[j]\n    \n        \n    # Define vectors to the 3 consecutive C-1 atoms\n    v_c1 = np.array([x1_first, y1_first, z1_first])\n    v_c2 = np.array([x1_second, y1_second, z1_second])\n    v_c3 = np.array([x1_third, y1_third, z1_third])\n    \n    # vector connecting the C-1 and C-2 as well as the vector connecting C-3 and C-2\n    v1 = v_c1 - v_c2\n    v2 = v_c3 - v_c2\n    \n    # Dot product between the two vectors\n    dot_product = np.dot(v1, v2)\n    \n    # Norms (magnitudes)\n    norm_v1 = np.linalg.norm(v1)\n    norm_v2 = np.linalg.norm(v2)\n    \n    # Angle in radians\n    theta_rad = np.arccos(dot_product / (norm_v1 * norm_v2))\n    \n    # Angle in degrees\n    theta_deg = np.degrees(theta_rad)\n    \n    return theta_deg\n    \n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:20:43.882774Z","iopub.execute_input":"2025-04-30T20:20:43.883109Z","iopub.status.idle":"2025-04-30T20:20:43.903208Z","shell.execute_reply.started":"2025-04-30T20:20:43.883048Z","shell.execute_reply":"2025-04-30T20:20:43.902167Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"angle_theta(train_label_raw)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:20:43.904257Z","iopub.execute_input":"2025-04-30T20:20:43.904539Z","iopub.status.idle":"2025-04-30T20:20:43.931087Z","shell.execute_reply.started":"2025-04-30T20:20:43.904512Z","shell.execute_reply":"2025-04-30T20:20:43.930161Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def three_letter_identification(df, i):\n    return df[\"resname\"][i] + df[\"resname\"][i+1] + df[\"resname\"][i+2]\n\nthree_letter_identification(train_label_raw, 0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:20:43.932124Z","iopub.execute_input":"2025-04-30T20:20:43.932404Z","iopub.status.idle":"2025-04-30T20:20:43.956605Z","shell.execute_reply.started":"2025-04-30T20:20:43.932375Z","shell.execute_reply":"2025-04-30T20:20:43.955453Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"L_index = list()\nL_3_Letters = list()\nL_angle = list()\nL_data_angle = list()\n\n\n# Go through the entire dataset row by row\nfor i in range(0, len(train_label_raw) - 2):\n\n    # Check that the 3 consecutive C-1 atoms are part of the same RNA chain\n    condition_1 = train_label_raw[\"resid\"].iloc[i] == (train_label_raw[\"resid\"].iloc[(i+1)] - 1)\n    condition_2 = train_label_raw[\"resid\"].iloc[(i+1)] == (train_label_raw[\"resid\"].iloc[(i+2)] - 1)\n    \n    if (condition_1 and condition_2):\n\n        # Store the index\n        L_index.append(i)\n\n        # Store the 3 consecutive letters \n        L_3_Letters.append(three_letter_identification(train_label_raw[i:(i+3)], i))\n\n        # Store the Euclidean distance between the 3 consecutive letters\n        L_angle.append(angle_theta(train_label_raw[i:(i+3)]))\n\n        # Make a list of list containing the index, the 2 consecutive letters and the Euclidean distance\n        L_data_angle.append([i, three_letter_identification(train_label_raw[i:(i+3)], i), angle_theta(train_label_raw[i:(i+3)])])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:20:43.957752Z","iopub.execute_input":"2025-04-30T20:20:43.958113Z","iopub.status.idle":"2025-04-30T20:22:01.309440Z","shell.execute_reply.started":"2025-04-30T20:20:43.958091Z","shell.execute_reply":"2025-04-30T20:22:01.308339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"L_data_angle[:10]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:22:01.310441Z","iopub.execute_input":"2025-04-30T20:22:01.310736Z","iopub.status.idle":"2025-04-30T20:22:01.317276Z","shell.execute_reply.started":"2025-04-30T20:22:01.310713Z","shell.execute_reply":"2025-04-30T20:22:01.316355Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Create a dataframe\n\ndf_data_angles_raw = pd.DataFrame(L_data_angle)\ndf_data_angles_raw.columns = [\"index\",\"3-Letters\",\"angle\"]\ndf_data_angles_raw","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:22:01.318783Z","iopub.execute_input":"2025-04-30T20:22:01.319246Z","iopub.status.idle":"2025-04-30T20:22:01.454925Z","shell.execute_reply.started":"2025-04-30T20:22:01.319214Z","shell.execute_reply":"2025-04-30T20:22:01.453434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Remove the NaN\n\ndf_data_angles = df_data_angles_raw[df_data_angles_raw[\"angle\"] == df_data_angles_raw[\"angle\"]]\ndf_data_angles","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:22:01.456369Z","iopub.execute_input":"2025-04-30T20:22:01.456721Z","iopub.status.idle":"2025-04-30T20:22:01.489580Z","shell.execute_reply.started":"2025-04-30T20:22:01.456698Z","shell.execute_reply":"2025-04-30T20:22:01.487935Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"grouped = df_data_angles.groupby(\"3-Letters\")[\"angle\"].mean(numeric_only=True)\ngrouped = grouped.sort_values(ascending=False)\ngrouped","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:22:01.491198Z","iopub.execute_input":"2025-04-30T20:22:01.492058Z","iopub.status.idle":"2025-04-30T20:22:01.534259Z","shell.execute_reply.started":"2025-04-30T20:22:01.492029Z","shell.execute_reply":"2025-04-30T20:22:01.533275Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\nnum_pairs = len(grouped)\ncmap = colormaps['Blues']\ncolors = [cmap(0.6 + 0.4 * i / max(num_pairs - 1, 1)) for i in range(num_pairs)]  # dark blues only\n\n# Plot\nplt.figure(figsize=(12, 6))\nplt.bar(grouped.index, grouped.values, color=colors)\n\n# Styling\nplt.title(\"Average Angle (degrees) between 3 consecutive bases\", fontsize=16)\nplt.xlabel(\"3 consecutive bases\", fontsize=12)\nplt.ylabel(\"Average Angle (degrees)\", fontsize=12)\nplt.xticks(rotation=45, ha='right')\nplt.grid(axis='y', linestyle='--', alpha=0.7)\nplt.tight_layout()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:22:01.535455Z","iopub.execute_input":"2025-04-30T20:22:01.535904Z","iopub.status.idle":"2025-04-30T20:22:02.257466Z","shell.execute_reply.started":"2025-04-30T20:22:01.535873Z","shell.execute_reply":"2025-04-30T20:22:02.256479Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# There is a distribution of angles. Most of the angles are between 100 and 160 degres.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:22:02.258773Z","iopub.execute_input":"2025-04-30T20:22:02.259116Z","iopub.status.idle":"2025-04-30T20:22:02.264598Z","shell.execute_reply.started":"2025-04-30T20:22:02.259085Z","shell.execute_reply":"2025-04-30T20:22:02.263581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nfrom matplotlib import colormaps\n\n# Calculate mean and standard deviation\nmean_dist = df_data_angles.groupby(\"3-Letters\")[\"angle\"].mean(numeric_only=True)\nstd_dist = df_data_angles.groupby(\"3-Letters\")[\"angle\"].std(numeric_only=True)\n\n# Sort by mean for consistency\nmean_dist = mean_dist.sort_values(ascending=False)\nstd_dist = std_dist[mean_dist.index]  # reindex to match order\n\n# Set up distinct colors\nnum_pairs = len(mean_dist)\ncmap = colormaps['Blues']\ncolors = [cmap(0.6 + 0.4 * i / max(num_pairs - 1, 1)) for i in range(num_pairs)]  # dark blues only\n\n# Plot\nplt.figure(figsize=(12, 6))\nplt.bar(mean_dist.index, mean_dist.values, yerr=std_dist.values, color=colors, capsize=5)\n\n# Styling\nplt.title(\"Average Angle (degrees) between 3 consecutive bases\", fontsize=16)\nplt.xlabel(\"3 consecutive bases\", fontsize=12)\nplt.ylabel(\"Average Angle (degrees)\", fontsize=12)\nplt.xticks(rotation=45, ha='right')\nplt.grid(axis='y', linestyle='--', alpha=0.7)\nplt.tight_layout()\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:22:02.265485Z","iopub.execute_input":"2025-04-30T20:22:02.265804Z","iopub.status.idle":"2025-04-30T20:22:02.966889Z","shell.execute_reply.started":"2025-04-30T20:22:02.265775Z","shell.execute_reply":"2025-04-30T20:22:02.966000Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\n# Define number of bins\nnum_bins = 35\n\n# Compute global min and max for distances\nxmin = df_data_angles[\"angle\"].min()\nxmax = df_data_angles[\"angle\"].max()\n\n# Generate consistent bin edges\nbins = np.linspace(xmin, xmax, num_bins + 1)\n\n# Loop through each pair\nfor pair in df_data_angles[\"3-Letters\"].unique():\n    subset = df_data_angles[df_data_angles[\"3-Letters\"] == pair][\"angle\"].dropna()\n    \n    if subset.empty:\n        continue\n\n    plt.figure(figsize=(8, 4))\n    plt.hist(subset, bins=bins, color='skyblue', edgecolor='black')\n    plt.xlim(xmin, xmax)\n    plt.title(f\"Histogram of Angles (degrees) for 3 consecutive C-1 atoms for a specific sequence: {pair}\")\n    plt.xlabel(\"Angle (degrees)\")\n    plt.ylabel(\"Frequency\")\n    plt.grid(axis='y', linestyle='--', alpha=0.6)\n    plt.tight_layout()\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:22:02.968207Z","iopub.execute_input":"2025-04-30T20:22:02.968572Z","iopub.status.idle":"2025-04-30T20:22:23.683642Z","shell.execute_reply.started":"2025-04-30T20:22:02.968541Z","shell.execute_reply":"2025-04-30T20:22:23.682681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# The specific consecutive sequence of 3 nitrogenous bases leads to a distribution of angles.\n# The distributions are often left-skewed. The mean, median and mode changes depending on the specific 3 nitrogenous base pairs.\n# It is interesting that most histogram have on peak like the GGG histogram.\n# However, there are histogram with two peaks like the GAA histogram.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T20:22:23.684858Z","iopub.execute_input":"2025-04-30T20:22:23.685155Z","iopub.status.idle":"2025-04-30T20:22:23.689927Z","shell.execute_reply.started":"2025-04-30T20:22:23.685127Z","shell.execute_reply":"2025-04-30T20:22:23.688989Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}