{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nplt.style.use('fivethirtyeight')\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2023-10-12T12:03:53.213995Z","iopub.execute_input":"2023-10-12T12:03:53.214346Z","iopub.status.idle":"2023-10-12T12:03:53.220250Z","shell.execute_reply.started":"2023-10-12T12:03:53.214320Z","shell.execute_reply":"2023-10-12T12:03:53.219293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/stanford-ribonanza-rna-folding/train_data.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-12T12:03:55.280879Z","iopub.execute_input":"2023-10-12T12:03:55.281360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe().style.background_gradient(cmap='summer')","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:00:54.803935Z","iopub.execute_input":"2023-10-12T11:00:54.804407Z","iopub.status.idle":"2023-10-12T11:01:25.745368Z","shell.execute_reply.started":"2023-10-12T11:00:54.804368Z","shell.execute_reply":"2023-10-12T11:01:25.744367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_counts = df['sequence'].apply(lambda x: pd.Series(list(x)).value_counts()).sum()\nbase_counts.plot(kind='bar')\nplt.xlabel('Base', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.ylabel('Count', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.title('Base Composition', fontsize = 14, fontweight = 'bold', color = 'darkgreen')\n\n# Save the plot\nplt.savefig('Base Composition.png')\n\n# Show the plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:01:36.863914Z","iopub.execute_input":"2023-10-12T11:01:36.864410Z","iopub.status.idle":"2023-10-12T11:15:04.092677Z","shell.execute_reply.started":"2023-10-12T11:01:36.864378Z","shell.execute_reply":"2023-10-12T11:15:04.091531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sequence_lengths = df['sequence'].apply(len)\nsns.histplot(sequence_lengths, kde=True)\nplt.xlabel('Sequence Length', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.ylabel('Frequency', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.title('Sequence Length Distribution', fontsize = 14, fontweight = 'bold', color = 'darkgreen')\n\n# Save the plot\nplt.savefig('Sequence Length Distribution.png')\n\n# Show the plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:15:21.485590Z","iopub.execute_input":"2023-10-12T11:15:21.485967Z","iopub.status.idle":"2023-10-12T11:15:29.515521Z","shell.execute_reply.started":"2023-10-12T11:15:21.485940Z","shell.execute_reply":"2023-10-12T11:15:29.514710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"experiment_type_counts = df['experiment_type'].value_counts()\nplt.pie(experiment_type_counts, labels=experiment_type_counts.index, autopct='%1.1f%%')\nplt.title('Experiment Type Distribution', fontsize = 14, fontweight = 'bold', color = 'darkgreen')\n\n# Save the plot\nplt.savefig('Experiment Type Distribution.png')\n\n# Show the plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:15:35.097298Z","iopub.execute_input":"2023-10-12T11:15:35.097630Z","iopub.status.idle":"2023-10-12T11:15:35.364058Z","shell.execute_reply.started":"2023-10-12T11:15:35.097605Z","shell.execute_reply":"2023-10-12T11:15:35.362709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.scatter(df['signal_to_noise'], df['reads'])\nplt.xlabel('Signal to Noise', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.ylabel('Reads', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.title('Signal to Noise vs Reads', fontsize = 14, fontweight = 'bold', color = 'darkgreen')\n\n# Save the plot\nplt.savefig('Signal to Noise vs Reads.png')\n\n# Show the plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:15:44.979596Z","iopub.execute_input":"2023-10-12T11:15:44.980007Z","iopub.status.idle":"2023-10-12T11:15:54.225852Z","shell.execute_reply.started":"2023-10-12T11:15:44.979976Z","shell.execute_reply":"2023-10-12T11:15:54.224959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Select only the first 50 reactivity columns\nreactivity_columns = [col for col in df.columns if col.startswith('reactivity')][:50]\nreactivity_data = df[reactivity_columns]\n\n# Convert the reactivity data to a numpy array\nreactivity_array = reactivity_data.values\n\n# Create a heatmap\nplt.figure(figsize=(10, 8))\nsns.heatmap(reactivity_array, cmap='viridis', cbar=True, xticklabels=20, yticklabels=False)\nplt.xlabel('Position', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.ylabel('Sample', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.title('Reactivity Heatmap (First 50 Columns)', fontsize = 14, fontweight = 'bold', color = 'darkgreen')\n\n# Save the plot\nplt.savefig('Reactivity Heatmap (First 50 Columns).png')\n\n# Show the plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:16:02.195514Z","iopub.execute_input":"2023-10-12T11:16:02.195864Z","iopub.status.idle":"2023-10-12T11:18:27.344687Z","shell.execute_reply.started":"2023-10-12T11:16:02.195839Z","shell.execute_reply":"2023-10-12T11:18:27.343801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_reactivity = reactivity_data.mean(axis=0)\nstd_reactivity = reactivity_data.std(axis=0)\n\nx = np.arange(len(mean_reactivity))\nplt.errorbar(x, mean_reactivity, yerr=std_reactivity, fmt='o')\nplt.xlabel('Position', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.ylabel('Mean Reactivity', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.title('Reactivity with Error Bars', fontsize = 14, fontweight = 'bold', color = 'darkgreen')\n\n# Save the plot\nplt.savefig('Reactivity with Error Bars.png')\n\n# Show the plot\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:18:29.243428Z","iopub.execute_input":"2023-10-12T11:18:29.243749Z","iopub.status.idle":"2023-10-12T11:18:31.108522Z","shell.execute_reply.started":"2023-10-12T11:18:29.243714Z","shell.execute_reply":"2023-10-12T11:18:31.107647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_reactivity = reactivity_data.mean(axis=0)\nstd_reactivity = reactivity_data.std(axis=0)\n\nx = np.arange(len(mean_reactivity))\nplt.errorbar(x, mean_reactivity, yerr=std_reactivity, fmt='o')\nplt.xlabel('Position', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.ylabel('Mean Reactivity', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.title('Reactivity with Error Bars', fontsize = 14, fontweight = 'bold', color = 'darkgreen')\n\n# Save the plot\nplt.savefig('Reactivity with Error Bars.png')\n\n# Show the plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:18:47.750398Z","iopub.execute_input":"2023-10-12T11:18:47.750778Z","iopub.status.idle":"2023-10-12T11:18:49.642981Z","shell.execute_reply.started":"2023-10-12T11:18:47.750749Z","shell.execute_reply":"2023-10-12T11:18:49.641702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def sensitivity(true_positives, false_negatives):\n    return true_positives / (true_positives + false_negatives)\n\ntrue_positives = 100  \nfalse_negatives = 20  \n\nprint(f\"True Positives: {true_positives}\")\nprint(f\"False Negatives: {false_negatives}\")\n\nsensitivity_score = sensitivity(true_positives, false_negatives)\nprint(f\"Sensitivity: {sensitivity_score}\")","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:19:10.265412Z","iopub.execute_input":"2023-10-12T11:19:10.265779Z","iopub.status.idle":"2023-10-12T11:19:10.272257Z","shell.execute_reply.started":"2023-10-12T11:19:10.265752Z","shell.execute_reply":"2023-10-12T11:19:10.270810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ppv(true_positives, false_positives):\n    return true_positives / (true_positives + false_positives)\n\ntrue_positives = 100  \nfalse_positives = 30  \n\nppv_score = ppv(true_positives, false_positives)\nprint(f\"Positive Predictive Value (PPV): {ppv_score}\")","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:19:22.696297Z","iopub.execute_input":"2023-10-12T11:19:22.696629Z","iopub.status.idle":"2023-10-12T11:19:22.703177Z","shell.execute_reply.started":"2023-10-12T11:19:22.696605Z","shell.execute_reply":"2023-10-12T11:19:22.701876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mcc(true_positives, true_negatives, false_positives, false_negatives):\n    numerator = (true_positives * true_negatives) - (false_positives * false_negatives)\n    denominator = ((true_positives + false_positives) * (true_positives + false_negatives) * \n                   (true_negatives + false_positives) * (true_negatives + false_negatives)) ** 0.5\n    return numerator / denominator if denominator != 0 else 0\n\n\ntrue_positives = 100  \ntrue_negatives = 50   \nfalse_positives = 30  \nfalse_negatives = 20  \n\nmcc_score = mcc(true_positives, true_negatives, false_positives, false_negatives)\nprint(f\"Matthews Correlation Coefficient (MCC): {mcc_score}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:19:46.167505Z","iopub.execute_input":"2023-10-12T11:19:46.167865Z","iopub.status.idle":"2023-10-12T11:19:46.175499Z","shell.execute_reply.started":"2023-10-12T11:19:46.167838Z","shell.execute_reply":"2023-10-12T11:19:46.174335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def f_measure(true_positives, false_positives, false_negatives):\n    precision = ppv(true_positives, false_positives)\n    recall = sensitivity(true_positives, false_negatives)\n    return 2 * (precision * recall) / (precision + recall) if (precision + recall) != 0 else 0\n\ntrue_positives = 100  \nfalse_positives = 30  \nfalse_negatives = 20  \n\nf_measure_score = f_measure(true_positives, false_positives, false_negatives)\nprint(f\"F-measure (F1 Score): {f_measure_score}\")","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:20:04.827317Z","iopub.execute_input":"2023-10-12T11:20:04.827717Z","iopub.status.idle":"2023-10-12T11:20:04.834641Z","shell.execute_reply.started":"2023-10-12T11:20:04.827678Z","shell.execute_reply":"2023-10-12T11:20:04.833553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\n%pip install arnie","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:20:22.017398Z","iopub.execute_input":"2023-10-12T11:20:22.017795Z","iopub.status.idle":"2023-10-12T11:20:33.901807Z","shell.execute_reply.started":"2023-10-12T11:20:22.017767Z","shell.execute_reply":"2023-10-12T11:20:33.900260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\n%pip install draw_rna","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:21:24.760184Z","iopub.execute_input":"2023-10-12T11:21:24.760573Z","iopub.status.idle":"2023-10-12T11:21:36.482525Z","shell.execute_reply.started":"2023-10-12T11:21:24.760541Z","shell.execute_reply":"2023-10-12T11:21:36.481354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# install eternafold\n!conda config --set auto_update_conda false\n!conda install -c bioconda eternafold --yes","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:21:45.275161Z","iopub.execute_input":"2023-10-12T11:21:45.275984Z","iopub.status.idle":"2023-10-12T11:23:09.440341Z","shell.execute_reply.started":"2023-10-12T11:21:45.275937Z","shell.execute_reply":"2023-10-12T11:23:09.439102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%env ETERNAFOLD_PATH=/opt/conda/bin/eternafold-bin\n%env ETERNAFOLD_PARAMETERS=/opt/conda/lib/eternafold-lib/parameters/EternaFoldParams.v1","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:23:16.463858Z","iopub.execute_input":"2023-10-12T11:23:16.464231Z","iopub.status.idle":"2023-10-12T11:23:16.472107Z","shell.execute_reply.started":"2023-10-12T11:23:16.464200Z","shell.execute_reply":"2023-10-12T11:23:16.470967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  df is your DataFrame\nsequences = df['sequence'].tolist()\nsequences","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:23:25.538939Z","iopub.execute_input":"2023-10-12T11:23:25.539288Z","iopub.status.idle":"2023-10-12T11:23:25.769819Z","shell.execute_reply.started":"2023-10-12T11:23:25.539262Z","shell.execute_reply":"2023-10-12T11:23:25.768675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from arnie.mfe import mfe\nsequence =\"GGGAACGACUCGAGUAGAGUCGAAAACGUACGUGGAGACACGUACGACGAGACUUCGGUCUCGAAUAGCUCGAACCGGUGCCGAGCGCGCACGAGGCGUCGGUGGUCGGCGAGCCCACGGACGCAAAAACUCGUGCGUAAAUAAAUAGCAUUGGAAGUGACGUCGACCGUUCGCGGUCGACGUCACAAAAGAAACAACAACAACAAC\"\nstructure = mfe(sequence,package=\"eternafold\")\nprint(structure)","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:23:49.663244Z","iopub.execute_input":"2023-10-12T11:23:49.663643Z","iopub.status.idle":"2023-10-12T11:23:49.819505Z","shell.execute_reply.started":"2023-10-12T11:23:49.663612Z","shell.execute_reply":"2023-10-12T11:23:49.818807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from draw_rna.ipynb_draw import draw_struct\ndraw_struct(sequence, structure)","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:24:00.131077Z","iopub.execute_input":"2023-10-12T11:24:00.131417Z","iopub.status.idle":"2023-10-12T11:24:02.584499Z","shell.execute_reply.started":"2023-10-12T11:24:00.131394Z","shell.execute_reply":"2023-10-12T11:24:02.583787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from arnie.mfe import mfe\nsequence = \"GGGAACGACTCGAGTAGAGTCGAAAAATAAGAGTGATTGGCGTCCGTACGTACCCTTTCTACTCTCAAACTCTTGTTAGTTTAAATCTAATCTAAACTTTATAAACGGCACTTCCTGTGTGTCCATGCCCGTGGGCTTGGTCTTGTCATAGTGCTGACATTTGTGGTTCCTTGGTTTTTGTTCTCTGCCAGTGACGTGTCCATTCGGCGCCAGCAGCCCACCCATAGGTTGCATAATGGCAAAGATGGGCAAATACGGTCTCGGCTTCAAATGGGCCCCAGAATTTCCATGGATGCTTCCGAACGCATCGGAGAAGTTGGGTAGCCCTGAGAGGTCAGAGGAGGATGGGTTTTGCCCCTCTGCTGCGCAAGAACCAAAAACTAAAGGAAAAACTTTGATTAATCACGTGAGGGTGGGGATCCTGTTCGCAGGATCCAAAAGAAACAACAACAACAAC\"\nstructure = mfe(sequence,package=\"eternafold\")\nprint(structure)","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:24:25.468515Z","iopub.execute_input":"2023-10-12T11:24:25.469065Z","iopub.status.idle":"2023-10-12T11:24:26.653193Z","shell.execute_reply.started":"2023-10-12T11:24:25.469028Z","shell.execute_reply":"2023-10-12T11:24:26.652143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from draw_rna.ipynb_draw import draw_struct\ndraw_struct(sequence, structure)","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:24:37.107219Z","iopub.execute_input":"2023-10-12T11:24:37.107589Z","iopub.status.idle":"2023-10-12T11:24:42.521361Z","shell.execute_reply.started":"2023-10-12T11:24:37.107561Z","shell.execute_reply":"2023-10-12T11:24:42.520188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from arnie.bpps import bpps\nbpps(sequence,package=\"eternafold\")","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:24:47.894641Z","iopub.execute_input":"2023-10-12T11:24:47.895103Z","iopub.status.idle":"2023-10-12T11:24:49.094184Z","shell.execute_reply.started":"2023-10-12T11:24:47.895073Z","shell.execute_reply":"2023-10-12T11:24:49.093035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Example of saving BPPs to a CSV file\nimport pandas as pd\n\nbpps_data = bpps(sequence, package=\"eternafold\")\nbpps_df = pd.DataFrame(bpps_data)\nbpps_df.to_csv('bpps_data.csv')","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:24:56.429286Z","iopub.execute_input":"2023-10-12T11:24:56.430214Z","iopub.status.idle":"2023-10-12T11:24:57.731815Z","shell.execute_reply.started":"2023-10-12T11:24:56.430183Z","shell.execute_reply":"2023-10-12T11:24:57.730881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nThis script calculates base pair probabilities (BPPs) for a given RNA sequence.\n\"\"\"\n\nfrom arnie.bpps import bpps\n\ndef calculate_bpps(sequence):\n    \"\"\"\n    Calculate BPPs for a given RNA sequence.\n    \n    Args:\n        sequence (str): The RNA sequence.\n        \n    Returns:\n        dict: A dictionary containing BPPs.\n    \"\"\"\n    return bpps(sequence, package=\"eternafold\")\n\n# Usage\nbpps_data = calculate_bpps(sequence)\n\nbpps_data\n","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:25:09.699534Z","iopub.execute_input":"2023-10-12T11:25:09.699879Z","iopub.status.idle":"2023-10-12T11:25:10.881759Z","shell.execute_reply.started":"2023-10-12T11:25:09.699854Z","shell.execute_reply":"2023-10-12T11:25:10.880758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop columns with NaN values (if any)\nbpps_df_cleaned = bpps_df.dropna(axis=1, how='any')\n\n# Calculate Mean BPPs for Each Position\nmean_bpps = bpps_df_cleaned.mean()\n\n# Plot Mean BPPs\nplt.figure(figsize=(10, 6))\nplt.plot(mean_bpps)\nplt.title('Mean Base Pair Probabilities', fontsize = 14, fontweight = 'bold', color = 'darkgreen')\nplt.xlabel('Position', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.ylabel('Mean BPP', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.savefig('Mean Base Pair Probabilities.png')\nplt.show()\n\n# Calculate Total BPP for Each Sequence\ntotal_bpps = bpps_df_cleaned.sum(axis=1)\n\n# 4. Plot Total BPP Distribution\nplt.figure(figsize=(10, 6))\nplt.hist(total_bpps, bins=30, color='skyblue', edgecolor='black')\nplt.title('Total Base Pair Probabilities Distribution', fontsize = 14, fontweight = 'bold', color = 'darkgreen')\nplt.xlabel('Total BPP', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.ylabel('Frequency', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.savefig('Total Base Pair Probabilities Distribution.png')\nplt.show()\n\n\n# Visualize BPPs Heatmap for a Few Sequences\nsample_sequences = bpps_df.sample(n=5, random_state=42)\n\nplt.figure(figsize=(10, 6))\nplt.imshow(sample_sequences.values, cmap='viridis', aspect='auto')\nplt.colorbar(label='BPP')\nplt.title('Base Pair Probabilities Heatmap', fontsize = 14, fontweight = 'bold', color = 'darkgreen')\nplt.xlabel('Position', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.ylabel('Sample Sequence', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.savefig('Base Pair Probabilities Heatmap.png')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:25:42.018611Z","iopub.execute_input":"2023-10-12T11:25:42.019021Z","iopub.status.idle":"2023-10-12T11:25:43.480291Z","shell.execute_reply.started":"2023-10-12T11:25:42.018993Z","shell.execute_reply":"2023-10-12T11:25:43.479299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.graph_objects as go\n\n# Define the performance measures\nlabels = ['Sensitivity', 'Positive Predictive Value (PPV)', 'Matthews Correlation Coefficient (MCC)', 'F-measure (F1 Score)']\nscores = [sensitivity_score, ppv_score, mcc_score, f_measure_score]\n\n# Create a horizontal bar plot\nfig = go.Figure(data=[go.Bar(\n    y=labels,\n    x=scores,\n    orientation='h',\n    marker=dict(color=['blue', 'green', 'red', 'purple'])\n)])\n\nfig.update_layout(\n    title='Performance Measures',\n    xaxis_title='Score',\n    yaxis_title='Metric',\n    yaxis=dict(autorange='reversed')\n)\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:25:59.215443Z","iopub.execute_input":"2023-10-12T11:25:59.215821Z","iopub.status.idle":"2023-10-12T11:25:59.652628Z","shell.execute_reply.started":"2023-10-12T11:25:59.215794Z","shell.execute_reply":"2023-10-12T11:25:59.651684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_data = pd.read_csv('/kaggle/input/stanford-ribonanza-rna-folding/sample_submission.csv')\nsubmission_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:26:34.290232Z","iopub.execute_input":"2023-10-12T11:26:34.290583Z","iopub.status.idle":"2023-10-12T11:27:53.718796Z","shell.execute_reply.started":"2023-10-12T11:26:34.290559Z","shell.execute_reply":"2023-10-12T11:27:53.717551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = pd.DataFrame(submission_data)\n\n# Save the DataFrame to a CSV file\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:27:59.124003Z","iopub.execute_input":"2023-10-12T11:27:59.124359Z","iopub.status.idle":"2023-10-12T11:33:56.905419Z","shell.execute_reply.started":"2023-10-12T11:27:59.124333Z","shell.execute_reply":"2023-10-12T11:33:56.904526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df","metadata":{"execution":{"iopub.status.busy":"2023-10-12T11:34:02.660253Z","iopub.execute_input":"2023-10-12T11:34:02.660631Z","iopub.status.idle":"2023-10-12T11:34:02.673056Z","shell.execute_reply.started":"2023-10-12T11:34:02.660600Z","shell.execute_reply":"2023-10-12T11:34:02.671900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}