{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<br>\n\n<center><img src=\"https://i.ibb.co/YP1rtFp/Screenshot-2023-09-10-at-12-22-25-PM.png\" alt=\"Competition Banner Image\" style=\"width: 100%;\"></center>\n\n<br style=\"margin: 15px;\">\n\n<h2 style=\"text-align: center; font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-transform: none; letter-spacing: 2px; color: #DC6FAF; background-color: #ffffff;\">\n    <span style=\"text-decoration: underline;\">\n        <font color=#FD3792>L</font>ET'S \n        <font color=#FD3792>L</font>EARN \n        <font color=#FD3792>T</font>OGETHER !\n    </span><br><br><br style=\"margin: 15px;\">\n<span style=\"font-size: 18px; letter-spacing: 1px;\">\n    <font color=#FD3792>E</font>XPLAIN IT \n    <font color=#FD3792>L</font>IKE \n    <font color=#FD3792>I</font>'M\n    <font color=#FD3792>5</font> \n    <font color=#FD3792>&nbsp;&nbsp;(ELI5)</font>\n<br style=\"margin: 18px;\">+<br style=\"margin: 18px;\">\n    <font color=#FD3792>E</font>XPLORATORY\n    <font color=#FD3792>D</font>ATA\n    <font color=#FD3792>A</font>NALYSIS \n    <font color=#FD3792>&nbsp;&nbsp;(EDA)</font>\n<br style=\"margin: 18px;\">+<br style=\"margin: 18px;\">\n    <font color=#FD3792>B</font>ASELINE\n<br style=\"margin: 15px;\">\n</span><br style=\"margin: 15px;\"></h2>\n\n<p style=\"text-align: center; font-family: Verdana; font-size: 12px; font-style: normal; font-weight: bold; text-decoration: None; text-transform: none; letter-spacing: 1px; color: black; background-color: #ffffff;\">CREATED BY: DARIEN SCHETTLER</p>\n\n<br>\n\n<sub><b>🎨 Font colour choices were inspired by the Barbie movie. (<font color=\"#FD3792\">#FD3792</font> & <font color=\"#DC6FAF\">#1bd19d</font>) 🎨<br>If the colours cause issues when viewing, please let me know and I can edit it to make it more accessible. Thanks!</b></sub>\n<br>\n\n---\n\n<br>\n\n<center><div class=\"alert alert-block alert-info\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 18px;\">⚠️ &nbsp; NOTE &nbsp; ⚠️</b><br><br><b>This notebook's primary purpose is to educate through exploration of this competition's data, the background information, and other notebooks and discussion posts that help in building a complete understanding.</b><br><br>Kaggle user's Notebooks and Discussions that I include will be listed below!<br style=\"margin: 15px;\"><b>Please upvote the original authors/contributors.</b><br><br>\n    \n<table class=\"alert alert-block alert-info\" style=\"text-align:center;\">\n<thead>\n  <tr>\n    <th>&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;Contributor&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;</th>\n    <th>&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;Notebook&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;</th>\n  </tr>\n</thead>\n<tbody>\n  <tr>\n      <td><b><a href=\"https://www.kaggle.com/hiramcho\">Hiram Coria</a></b></td>\n      <td><b><a href=\"https://www.kaggle.com/code/hiramcho/ribonanza-rna-visualization#Plot-2D-RNA-Structure\">Ribonanza: RNA Visualization</a></b></td>  \n  </tr>\n  <tr>\n      <td><b><a href=\"https://www.kaggle.com/iafoss\">iafoss</a></b></td>\n      <td><b><a href=\"https://www.kaggle.com/code/iafoss/rna-starter-0-186-lb\">RNA starter [0.186 LB]</a></b></td>\n  </tr>\n  <tr>\n    <td>placeholder 3</td>\n    <td>notebook 3</td>\n  </tr>\n</tbody>\n</table>\n    \n<br>\n    \n</div></center>\n\n\n\n<center><div class=\"alert alert-block alert-danger\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 18px;\">🛑 &nbsp; WARNING:</b><br><br><b>THIS IS A WORK IN PROGRESS</b><br>\n</div></center>\n\n\n<center><div class=\"alert alert-block alert-warning\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 18px;\">👏 &nbsp; IF YOU FORK THIS OR FIND THIS HELPFUL &nbsp; 👏</b><br><br><b style=\"font-size: 22px; color: darkorange\">PLEASE UPVOTE!</b><br><br>This was a lot of work for me and it makes me feel appreciated when others like my work. 😅\n</div></center>\n\n\n","metadata":{}},{"cell_type":"markdown","source":"<p id=\"toc\"></p>\n\n<br><br>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #DC6FAF; background-color: #ffffff;\">TABLE OF CONTENTS</h1>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: black; background-color: #ffffff;\"><a href=\"#introduction\" style=\"color: #FD3792;\">1&nbsp;&nbsp;&nbsp;&nbsp;INTRODUCTION & JUSTIFICATION</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: black; background-color: #ffffff;\"><a href=\"#competition_background_information\" style=\"color: #FD3792;\">2&nbsp;&nbsp;&nbsp;&nbsp;HOST PROVIDED COMPETITION INFORMATION</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: black; background-color: #ffffff;\"><a href=\"#my_understanding\" style=\"color: #FD3792;\">3&nbsp;&nbsp;&nbsp;&nbsp;MY UNDERSTANDING OF THE COMPETITION</a></h3>\n\n---\n\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: black; background-color: #ffffff;\"><a href=\"#background_information_glossary\" style=\"color: #FD3792;\">4&nbsp;&nbsp;&nbsp;&nbsp;BACKGROUND INFORMATION GLOSSARY</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: black; background-color: #ffffff;\"><a href=\"#imports\" style=\"color: #FD3792;\">5&nbsp;&nbsp;&nbsp;&nbsp;IMPORTS</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: black; background-color: #ffffff;\"><a href=\"#setup\" style=\"color: #FD3792;\">6&nbsp;&nbsp;&nbsp;&nbsp;SETUP & HELPER FUNCTIONS</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: black; background-color: #ffffff;\"><a href=\"#data_exploration\" style=\"color: #FD3792;\">7&nbsp;&nbsp;&nbsp;&nbsp;DATA EXPLORATION</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: black; background-color: #ffffff;\"><a href=\"#tbd\" style=\"color: #FD3792;\">8&nbsp;&nbsp;&nbsp;&nbsp;TBD</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: black; background-color: #ffffff;\"><a href=\"#next_steps\" style=\"color: #FD3792;\">9&nbsp;&nbsp;&nbsp;&nbsp;NEXT STEPS</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: black; background-color: #ffffff;\"><a href=\"#baseline\" style=\"color: #FD3792;\">10&nbsp;&nbsp;&nbsp;&nbsp;BASELINE</a></h3>\n\n---","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<a id=\"introduction\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #DC6FAF;\" id=\"introduction\">1&nbsp;&nbsp;INTRODUCTION & JUSTIFICATION&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\" style=\"color: #FD3792;\">&#10514;</a></h1>\n\n<br>\n\nThis notebook aims to provide an educational walkthrough following my personal journey of understanding. This will hopefully bring me from completely ignorant of the background and science for this competition to a place where I feel ready to tackle the challenge in earnest!<br style=\"margin: 15px;\">\n\n<b>I hope you can join me and we can learn together!</b>\n\n> <i><b>NOTE:</b> Unlike my previous notebooks that were primarily Exploratory Data Analyses (EDAs) or Educational (ELI5) or Baselines... this notebook will merge those two educational pursuits into a single unified resource.</i>","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DC6FAF; background-color: #ffffff;\">1.1 <b>WHAT</b> IS THIS?</h3>\n\n---\n\nThis notebook will attempt to do two (maybe three) things:\n1. \"<b>E</b>xplain It <b>L</b>ike <b>I</b>'m <b>F</b>ive\". \n    * Where <b>\"IT\"</b> refers to everything you need to know to understand this competition (data, background information, etc)\n    * Note that while I aim to honour the five-year old part of this statement, it mostly used because it is a catchy term, and I will just be <b>explaining things as simply as possible</b>\n2. \"<b>E</b>xploratory <b>D</b>ata <b>A</b>nalysis\"\n    * I will perform a traditional Exploratory Data Analysis (<b>EDA</b>), which is a critical early step in tackling any data science problem!\n    * I see this as a secondary and complimentary piece to the <b>ELI5</b> piece, and as such, this part of the notebook comes AFTER the <b>ELI5</b> part.\n3. <b>B</b>aseline <sub><i>(optional)</i></sub>\n    * If I think I have time and if it's appropriate, I will finish this notebook out with a final section that performs a baseline submission resulting in a valid leaderboard submission.\n    \n<sup><b>REMINDER:</b> If I have cited other authors, please don't forget to upvote them. I'll reiterate their works below but it can be found above too!</sup>\n\n<table class=\"alert alert-block alert-info\" style=\"text-align:center;\">\n<thead>\n  <tr>\n    <th>&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;Contributor&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;</th>\n    <th>&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;Notebook&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;</th>\n  </tr>\n</thead>\n<tbody>\n  <tr>\n      <td><b><a href=\"https://www.kaggle.com/hiramcho\">Hiram Coria</a></b></td>\n      <td><b><a href=\"https://www.kaggle.com/code/hiramcho/ribonanza-rna-visualization#Plot-2D-RNA-Structure\">Ribonanza: RNA Visualization</a></b></td>  \n  </tr>\n  <tr>\n      <td><b><a href=\"https://www.kaggle.com/iafoss\">iafoss</a></b></td>\n      <td><b><a href=\"https://www.kaggle.com/code/iafoss/rna-starter-0-186-lb\">RNA starter [0.186 LB]</a></b></td>\n  </tr>\n  <tr>\n    <td>placeholder 3</td>\n    <td>notebook 3</td>\n  </tr>\n</tbody>\n</table>\n    ","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DC6FAF; background-color: #ffffff;\">1.2 <b>WHY</b> IS THIS?</h3>\n\n---\n\n<b>I wanted to share my learning journey with others as this type of work has resonated well in the past</b>. \n* Note that I make these type of notebooks primarily for myself, however, I am finding it brings value when I share them with others. As such I plan to continue this format in the future.\n* By upvoting/sharing this work you are supporting me and encouraging me to spend more time on similar endeavours in the future!","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DC6FAF; background-color: #ffffff;\">1.3 <b>WHO</b> IS THIS FOR?</h3>\n\n---\n\nThe primary purpose of this notebook is to educate <b style=\"color: #FD3792\">MYSELF</b>, however, my review/learning might be beneficial to others. Hence this notebook.\n\nPlease understand that not everyone <b>needs</b> nor <b>wants</b> to read a notebook like this. That being said the following are some good reasons why you may wish to read on:\n* <b>If you want to learn with me!</b>\n* If you want to learn more about the <b>background information</b> (ELI5)\n* If you want to learn more about the <b>competition specifics</b> (ELI5)\n* If you want to learn more about the <b>competition data</b> (EDA)\n* If you have a good understanding of the <b>competition specifics</b> but want to doublecheck against my understanding or reinforce your own understanding (ELI5)\n* If you have a good understanding of the <b>background information</b> but want to doublecheck against my understanding or reinforce your own understanding (ELI5)\n* If you have a good understanding of the <b>competition data</b> but want to doublecheck against my understanding or reinforce your own understanding (EDA)\n* If you want to see some <b>baseline approaches</b> implemented (baseline)","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DC6FAF; background-color: #ffffff;\">1.4 <b>HOW</b> WILL THIS WORK?</h3>\n\n---\n\n1. I'm going to assemble some markdown cells (like this one) at the beginning of the notebook to go over the competition details\n2. General background information mostly expressed as a glossary with lots of visuals (this will include a table to track my learning).\n3. I will perform a reasonably in-depth Exploratory Data Analysis\n4. I will create one or more baseline submissions (including)","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<a id=\"competition_background_information\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #DC6FAF;\" id=\"competition_background_information\">2&nbsp;&nbsp;HOST PROVIDED COMPETITION INFORMATION&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\" style=\"color: #FD3792;\">&#10514;</a></h1>\n\n<br>\n\n**NOTE***\n* The first sections are dedicated to exploring the basic resources provided by the host.\n* The final sections are dedicated to my understanding of the competition","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DC6FAF; background-color: #ffffff;\">2.1 PRIMARY TASK DESCRIPTION</h3>\n\n---\n\nThe goal of this competition is to <b>predict the reactivity (DMS, and 2A3) at every position for a particular RNA sequence</b>.\n* To do this we need to create a model that understands the structure of RNA and as such can infer the reactivity at a particular nucleotide\n* The model will leverage millions of RNA sequences along with the labelled reactivity data (and additional supplemental data).\n* The model will be required to output two float values (***regression***) for each **`id`** in the test set.\n    * **reactivity_DMS_MaP** – A value from **0-1** measuring reactivity to the **DMS** chemical, with 1 being the most reactive and 0 being the least reactive. *[A float]*\n    * **reactivity_2A3_MaP** – A value from **0-1** measuring reactivity to the **2A3** chemical, with 1 being the most reactive and 0 being the least reactive. *[A float]*\n    \n<i><b>Note:</b> MaP refers to <b>M</b>ut<b>a</b>tional <b>P</b>rofiling not <b>M</b>ean <b>a</b>verage <b>P</b>recision. <b>M</b>ut<b>a</b>tional <b>P</b>rofiling experiments are used in the determination of the reactivity at a particular nucleotide location.</i>","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DC6FAF; background-color: #ffffff;\">2.2 <b>BASIC</b> HOST PROVIDED BACKGROUND INFO AND CONTEXT</h3>\n\n---\n\n<div style=\"line-height: 1.0\"><b style=\"font-size: 11px;\">This section is kept intentionally brief and only reflects the host provided information. The concepts that help better define the context will be explored in the GLOSSARY BACKGROUND INFORMATION section of this notebook. Some terms are linked to their glossary entries for convenience.</b></div><br>\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">DATA DESCRIPTION</b>\n\nWithout being able to understand how RNA molecules fold, we are missing a deeper understanding of how nature works, how life began, and how we can design better medicines and biological approaches to grand problems like climate change.\n\nIn this competition, <b>your goal is to predict <mark>how an RNA sequence reacts to two specific chemical modifiers: DMS and 2A3</mark></b>. You can obtain this reactivity data through a <b>M</b>ut<b>A</b>tional <b>P</b>rofiling (<b>MaP</b>) experiment which is then analyzed via high-throughput sequencing. <b><mark>Positions in the RNA sequence that remain unmodified (protected from chemical modification)</mark></b> are more likely involved in forming base pairs or other structural elements.\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">CONTEXT DESCRIPTION</b>\n\n<b>R</b>ibo<b>N</b>ucleic <b>A</b>cid (<b>RNA</b>) is essential for most biological functions. A better understanding of <b><mark>how to manipulate RNA</mark></b> could help usher in an age of programmable medicine, including first cures for pancreatic cancer and Alzheimer’s disease as well as much-needed antibiotics and new biotechnology approaches for climate change. But first, researchers must <b><mark>better understand each RNA molecule's structure</mark></b>, an ideal problem for data science.\n\nRecent <b>efforts to predict RNA structure have run into a number of challenges:</b>\n1. Paucity (small or insufficient quantities or amounts) of training data <mark>(lack of data)</mark>\n2. Lack of intellectual and computational power <mark>(lack of compute)</mark>\n3. Difficulties in rigorously splitting training and test data. <mark>(lack of cross validation rigor)</mark>\n\nCan a Kaggle competition close these gaps?\n\nTowards sourcing a large and diverse collection of RNA molecules for data acquisition, the host <b><a style=\"color: #FD3792;\" href=\"https://daslab.stanford.edu/\">Das laboratory (Stanford)</a></b> brings together scientists and gamers to solve puzzles and invent medicine in the <b><a style=\"color: #FD3792;\" href=\"https://eternagame.org/\">Eterna project</a></b>. The <b><a style=\"color: #FD3792;\" href=\"https://eternagame.org/\">Eterna</a></b> community has previously unlocked new scientific principles, designed <b><a style=\"color: #FD3792;\" href=\"https://www.ncbi.nlm.nih.gov/pmc/articles/PMC9170038/\">thermodynamically-optimized riboswitches</a></b>, and <b><a style=\"color: #FD3792;\" href=\"https://www.nature.com/articles/s42256-022-00571-8\">revealed RNA degradation patterns for improving the shelf life of mRNA vaccines</a></b>, which formed the basis of the <b><a style=\"color: #FD3792;\" href=\"https://www.kaggle.com/competitions/stanford-covid-vaccine\">Kaggle OpenVaccine challenge</a></b>. Now the <b><a style=\"color: #FD3792;\" href=\"https://eternagame.org/\">Eterna project</a> is creating diverse sequences that are predicted to form complex structures.</b>\n\n<mark>The data for this new <b>Ribonanza</b> Competition are experimental measurements of the chemical reactivity at each position of an RNA molecule</mark>. These data are <b><mark>exquisitely sensitive to the structure – or multiple structures – that each RNA forms</mark></b> in the test tube. An algorithm that could perfectly predict these chemical reactivities would need to have an implicit ‘understanding’ of RNA structure to do so. Such an <b>oracle could be then utilized to predictively model structures of novel RNA molecules. As in the OpenVaccine competition, the majority of the private leaderboard data will be collected in parallel with your prediction efforts -- no one will know the answers until the competition closes!\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">IMPACT DECLARATION</b>\n\nTo reiterate the potential impact of your participation, an accurate model that solves the RNA structure prediction problem could be a game changer for medical researchers who are trying to identify unique RNA-based drug targets in the many bacterial, viral, neurological, and cancer genes that remain undruggable at the protein level. In addition, accurate RNA structure prediction is needed to predictively design RNA-based medicines such as mRNA vaccines and CRISPR gene therapeutics that promise to treat nearly all human disease. More than its medical implications, RNA molecules underlie and can even dominate core biological processes for all of life, including the very first forms of life on Earth and the plants and marine organisms that fix most of the carbon on our planet now. A full understanding of life requires a full, predictive understanding of RNA.\n\nIn addition to being made available as open source code for the entire science and medicine communities, winners’ innovations will be featured in a keynote talk by Rhiju Das at the flagship conference in machine learning in structure biology at NeurIPS in December 2023.\n\nIt’s possible that every biologist and biotechnologist in the world could one day use your algorithmic solution.\n    \n","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DC6FAF; background-color: #ffffff;\">2.3 EVALUATION INFORMATION</h3>\n\n---\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">GENERAL EVALUATION INFORMATION</b>\n\nSubmissions are scored using <b><a style=\"color: #FD3792;\" href=\"https://en.wikipedia.org/wiki/Mean_absolute_error\">Mean Absolute Error (MAE)</a></b>:\n\n$$\n\\text{MAE} = \\frac{1}{N}\\sum_{i=1}^{N} \\left| y_{i} - \\widehat{y}_{i} \\right|\n$$\n\n**WHERE:** \n* $N$ is the number of scored ground truth values\n* $y$ and $\\widehat{y}$ are the actual and predicted values, respectively. \n* The $y$ values will be clipped between 0 and 1 before calculating <b>MAE</b>\n    * $y_{i}=max(min(y^{RAW}_{i},1.0),0.0)$\n    * where $y^{RAW}$ are the raw data values\n\n\nEach RNA sequence has some number of nucleotides (positions). For each position there will be two ground truth values, corresponding to reactivity determined from two kinds of chemical mapping experiments, **DMS_MaP and 2A3_MaP**.\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">SUBMISSION FILE INFORMATION</b>\n\nFor each **sequence** in the test set, you must predict **target reactivities** for each sequence position **id**, **two per row**. \n* i.e. If the sum of the lengths of all 1,343,824 sequences in the test set is 269,796,671, then you should make 2 × 269,796,670 = 539,593,340 predictions.\n\nThe submission file should contain a header and have the following format:\n\n```\nid,reactivity_DMS_MaP,reactivity_2A3_MaP\n0,0.1,0.3\n1,0.3,0.2\n2,0.5,0.4\netc.\n```\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; color: red; letter-spacing: 2px;\">IS THIS A CODE COMPETITION ??</b>\n\n<b style=\"color: green; font-size: 24px;\">NO</b>\n\n> This competition provides the entirety of the test set. You will be submitting your final predictions for all the test data and some portion of that will be used for public LB evaluation while the rest will be held out for private LB evaluation.","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DC6FAF; background-color: #ffffff;\">2.4 DATASET INFORMATION</h3>\n\n---\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">HIGH LEVEL DATA SUMMARY</b>\n\nAt the beginning of the competition, Stanford scientists have data on <b><mark>1,118,513</mark> RNA sequences of lengths ranging from <mark>115 to 206.</mark></b>\n\nWe have <b><mark>split out 311,935 of these 1,118,513 sequences for a public test set</mark></b> to allow for continuous evaluation through the competition, on the Public Leaderboard. This set has been additionally <b>filtered to ensure high signal-to-noise and read coverage</b>. The remaining <b>806,578 sequences for which we have data are in train_data.csv</b>. We note that <b>37,828 of the test sequences, derived from the RFAM database, are identical or near-identical to the train set</b>; for these cases the leaderboard test set contains higher signal-to-noise measurements than the train_data and serve as a test of <b>model ability to 'denoise' chemical mapping data</b>.\n\nOur final and most important scoring <b><mark>(the Private Leaderboard) involves 1,031,888 sequences</mark></b>. Within this set, the majority (1,008,000 RNAs) will be <b>experimentally synthesized and profiled after the Kaggle competition begins</b>, to help ensure rigor. Furthermore, these test sequences will have <b><mark>lengths ranging from 207 to 457 bases -- we intentionally chose a different length distribution</mark></b> compared to the train data and Public Leaderboard, to help <b>test the generality of your models</b>. As those profiles are collected, they will again be filtered for acceptable signal to noise before computing your final Private Leaderboard scores.\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">DATA FILE DESCRIPTIONS</b>\n\ntrain_data.csv - the training data\ntest_sequences.csv - the test set sequences, without any columns associated with the ground truth.\nsample_submission.csv - a sample submission file in the correct format\n\n<b><code>[train_data|test_sequences].csv</code> columns:</b>\n* **`id` (integer)**\n    * Integer (0,1,…) that identifies each sequence position in the sample submission.\n* **`id_min, id_max` (integer)**\n    * Minimum and maximum id values for each test sequence.\n* **`sequence_id` (string)**\n    * An arbitrary identifier like `8cdfeef00` for each sequence.\n* **`sequence` (string)**\n    * Describes the RNA sequence, a combination of A, G, U, and C for each sample. \n    * Should be **115 to 457 characters long**.\n* **`experiment_type` (string)**\n    * Either `DMS_MaP` or `2A3_MaP` to describe the type of chemical mapping experiment that was used to generate each profile. \n    * References: <b><a style=\"color: #FD3792;\" href=\"https://doi.org/10.1093/nar/gkad522\">DMS</a></b>, <b><a style=\"color: #FD3792;\" href=\"https://academic.oup.com/nar/article/49/6/e34/6062772\">2A3</a></b>.\n* **`dataset_name` (string)**\n    * Name of high throughput sequencing dataset from which the reactivity profile was extracted.\n* **`reads` (integer)**\n    * Number of reads in the high throughput sequencing experiment that were assigned to the RNA sequence, and whose mutations were tabulated to compile the reactivity profile.\n* **`signal_to_noise` (float)**\n    * Signal/noise value for the profile\n    * Defined as: **`mean(measurement value over probed nts) / mean(statistical error in measurement value over probed nts)`**.\n* **`SN_filter` (Boolean)**\n    * 0 **or** 1 depending on whether the profile has `signal_to_noise > 1.0` and `reads > 100`. \n    * <b>For evaluation, only sequences whose DMS_MaP and 2A3_MaP profiles both pass this filter will be used to score submissions.</b>\n* **`reactivity_0001, reactivity_0002,…` (float)**\n    * An array of floating-point numbers of the train data, should have the same length as the RNA sequence, which defines the reactivity profile for the RNA. \n    * <b>Notes on null values</b>\n        * For sequences shorter than the maximum RNA length, positions that go beyond the sequence length have null. \n        * Several positions early and late in the sequence also cannot be probed due to technical reasons, and their reactivity values are null.\n* **`reactivity_error_0001, reactivity_error_0002,…` (float)**\n    * An array of floating-point numbers, should have the same length as the corresponding `reactivity_*` columns, calculated errors in experimental values obtained in reactivity derived from counting statistics in the high-throughput sequencing experiment.\n* **`reactivity_DMS_MaP, reactivity_2A3_MaP` (float)**\n    * Sample submission values.\n* **`future` (Boolean)**\n    * Sequences whose data will be collected after the start of the competition (but before final scoring) are labeled as 1.\n\n<br>\n\n<b><code>sample_submission.csv</code> description</b>\n* An example submission with the correct columns and properly ordered event IDs. \n\n<br>\n\n<b>Additional Folders</b>\n* **`sequence_libraries`**\n    * all 2,150,401 sequences, with titles, in FASTA format, grouped into the actual collections that were synthesized together. \n    * Note: some files list DNA not RNA sequences (T instead of U).\n* **`supplementary_silico_predictions`**\n    * CSV files of predicted RNA secondary structures in dot-parentheses notations. \n    * Note: not all packages were run for all sequences due to computational expense of prediction packages that are able to predict pseudoknots (non-nested pairings).\n* **`eterna_openknot_metadata`**\n    * Contains TSV files with the following columns:\n        * id\n        * name (author)\n        * title (author-provided design title)\n        * body (description)\n        * sequence for Eterna OpenKnot sequences\n    * Note: some files have additional tabs in title or description that may affect reading.\n* **`Ribonanza_bpp_files`**\n    * TXT files of base pair probabilities from LinearPartition-EternaFold package for train and test sequences. \n    * Indexed by `sequence_id`. \n    * Note: this package simulates RNA secondary structure ensembles without pseudoknots or other tertiary structure features.","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DC6FAF; background-color: #ffffff;\">2.5 SUPPLEMENTARY INFORMATION</h3>\n\n---\n\n<ul>\n  <li>\n    <b>How to think about RNA structure</b>\n    <ul>\n      <li>\n        <a rel=\"noreferrer nofollow\" target=\"_blank\" style=\"color: #FD3792;\" href=\"https://www.pnas.org/doi/10.1073/pnas.2112677119\">A perspective from domain experts</a>\n      </li>\n    </ul>\n  </li>\n  <li>\n    <b>What's the state-of-the-art in RNA 3D structure prediction?</b>\n    <ul>\n      <li><a rel=\"noreferrer nofollow\" target=\"_blank\" style=\"color: #FD3792;\" href=\"https://www.biorxiv.org/content/10.1101/2023.04.25.538330v2\">Evaluation of CASP15 competition by host Rhiju Das and colleagues</a></li>\n      <li><a rel=\"noreferrer nofollow\" target=\"_blank\" style=\"color: #FD3792;\" href=\"https://www.youtube.com/watch?v=oe-w1Xx1p1g\">Summary video from CASP15</a></li>\n    </ul>\n  </li>\n  <li>\n    <b>Getting half the way to 3D structure ensembles: modeling RNA \"secondary\" structure</b>\n    <ul>\n      <li><a rel=\"noreferrer nofollow\" target=\"_blank\" style=\"color: #FD3792;\" href=\"https://www.nature.com/articles/s41592-022-01605-0\">EternaBench: Summary and ranking of secondary structure algorithms, including use of chemical mapping data</a></li>\n    </ul>\n  </li>\n  <li>\n    <b>Eterna's OpenKnot challenge, the biggest single source of the Ribonanza RNA sequences</b>\n    <ul>\n      <li><a rel=\"noreferrer nofollow\" target=\"_blank\" style=\"color: #FD3792;\" href=\"https://eternagame.org/challenges/11843006\">The OpenKnot challenge page</a></li>\n      <li><a rel=\"noreferrer nofollow\" target=\"_blank\" style=\"color: #FD3792;\" href=\"https://www.youtube.com/watch?v=PDI7wsdjtt0\">OpenKnot update video from earlier in the year</a></li>\n    </ul>\n  </li>\n  <li>\n    <b>How a prior Eterna/Kaggle collaboration led to improved technologies for mRNA medicines</b>\n    <ul>\n      <li><a rel=\"noreferrer nofollow\" target=\"_blank\" style=\"color: #FD3792;\" href=\"https://www.nature.com/articles/s42256-022-00571-8\">Paper on dual crowdsourcing and prior winners' models</a></li>\n    </ul>\n  </li>\n</ul>\n","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DC6FAF; background-color: #ffffff;\">2.6 COMPETITION HOST INFORMATION</h3>\n\n---\n\n<table>\n<tbody>\n<tr>\n<td><img style=\"width:250px\" src=\"https://storage.googleapis.com/kaggle-media/competitions/Stanford/logo_eterna_reverse.png\"></td>\n<td><a rel=\"noreferrer nofollow\" target=\"_blank\" style=\"color: #FD3792;\" href=\"https://eternagame.org/about\">EteRNA</a> teaches about RNA and the different shapes it takes. You are are given control over RNA bonds (ACGU), which directly affect the RNA molecule's shape. Changing proteins in a chain impacts the natural shape that the chain would take with the goal of matching the shape of the puzzle.</td>\n</tr>\n<tr>\n<td><img style=\"width:200px\" src=\"https://storage.googleapis.com/kaggle-media/competitions/Stanford/logo_das.png\"></td>\n    <td>The <a rel=\"noreferrer nofollow\" target=\"_blank\" style=\"color: #FD3792;\" href=\"https://daslab.stanford.edu/\">DAS Lab</a> seeks an agile and predictive understanding of how RNAs code for information processing and replication in living systems. They are creating new computational and chemical tools to enable the precise modeling and design of these RNAs.</td>\n</tr>\n<tr>\n</tbody>\n</table>","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<a id=\"my_understanding\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #DC6FAF;\" id=\"my_understanding\">3&nbsp;&nbsp;MY UNDERSTANDING OF THE COMPETITION&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\" style=\"color: #FD3792;\">&#10514;</a></h1>","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DC6FAF; background-color: #ffffff;\">3.1 BASICS - ELI5+</h3>\n\n---\n\nChemical reactivity in the context of <b><a style=\"color: #FD3792;\" href=\"https://academic.oup.com/nar/article/51/16/8744/7201944?login=false\">DMS</a></b> (dimethyl sulfate) and <b><a style=\"color: #FD3792;\" href=\"https://academic.oup.com/nar/article/49/6/e34/6062772\">2A3</a></b> (2-aminopyridine-3-carboxylic acid imidazolide) mapping experiments provides a way to probe the structure of RNA molecules. Here's how it works:\n\nIn an RNA molecule, some regions are more \"exposed\" and some are more \"hidden\" due to the way the molecule folds into its unique 3D structure. The exposed regions are more likely to react with chemicals like DMS or 2A3, while the hidden regions are generally less reactive.\n\nDMS and 2A3 are chemicals that can bind to certain positions on the RNA molecule. The reactivity measured using these chemicals serves as an indirect indicator of the molecule's structure. If a certain position on the RNA is highly reactive to DMS or 2A3, it's likely that this position is more exposed in the structure. Conversely, if it's less reactive, that position is probably more hidden or protected within the folded structure.\n\nBy measuring the chemical reactivity of each position within the RNA molecule using DMS and 2A3, scientists can gather data that is sensitive to the RNA's 3D structure. A machine learning model that can accurately predict these reactivity values essentially \"understands\" the likely 3D structure of the RNA molecule, which is why this reactivity data is so important for the competition.","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DC6FAF; background-color: #ffffff;\">3.2 MY INTERNAL Q&A</h3>\n\n---\n\nThis will be a list where there is a question/thought followed by my learnings about that thing...\n\n<br>\n\n---\n\n<b style=\"font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 1px;\">What are the similarities between DMS and 2A3?</b>\n<ul>\n    <li>Both reagents provide a way to gain more precise insights into RNA structures by measuring reactivity at various positions on the RNA strand.</li>\n    <li>The advances in these reagents make them particularly powerful tools for understanding RNA structure in both in vitro and in vivo settings, which is why they are important for the Kaggle competition aimed at predicting RNA structure.</li>\n</ul>\n\n---\n\n<b style=\"font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 1px;\">What are the differences between DMS and 2A3?</b>\n<ul>\n  <li>\n    <b>For 2A3:</b>\n    <ul>\n      <li>It is particularly effective for probing RNA structure in living cells, especially in bacteria.</li>\n      <li>It outperforms other SHAPE (Selective 2'-Hydroxyl Acylation analyzed by Primer Extension) reagents like NAI in accuracy when used for in vivo studies.</li>\n      <li>2A3 is very efficient and accurate, making it a likely successor to conventional SHAPE reagents for both in vitro and in vivo RNA structural probing.</li>\n    </ul>\n  </li>\n  <li>\n    <b>For DMS:</b>\n    <ul>\n      <li>It traditionally probed only adenine and cytosine nucleobases but has been modified to probe all four types of nucleotides, including in cells.</li>\n      <li>Four-base DMS reactivities provide greater structural information than the previous two-base DMS and SHAPE strategies.</li>\n      <li>The improved DMS mutational profiling (MaP) strategy enhances the accuracy of RNA structural modeling.</li>\n    </ul>\n  </li>\n</ul>\n\n---\n\n<b style=\"font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 1px;\">What is 'SHAPE'?</b>\n<ul>\n  <li>Information for this answer comes from the following sources:\n    <ul>\n      <li><b><a style=\"color:#FD3792;\" href=\"https://academic.oup.com/nar/article/49/6/e34/6062772\">https://academic.oup.com/nar/article/49/6/e34/6062772</a></b></li>\n    </ul>\n  </li>\n  <li>SHAPE stands for <b><font color=\"red\">Selective</font> <font color=\"green\">2'-Hydroxyl Acylation</font> analyzed by <font color=\"blue\">Primer Extension</font></b>.</li>\n  <li>To better understand what it is and what it does, let's break it down into its parts:\n    <ul>\n      <li><b><font color=\"red\">Selective:</font></b>\n        <ul>\n          <li>The term (in this case) indicates that the method targets specific positions on the RNA strand, namely those that are more reactive because they are more exposed in the RNA's folded structure.</li>\n        </ul>\n      </li>\n      <li><b><font color=\"green\">2'-Hydroxyl Acylation:</font></b>\n        <ul>\n          <li>This is a chemical reaction where a reagent (like 2A3 or NAI - NOT DMS... that's different!) interacts specifically with the RNA molecule at the 2'-hydroxyl group.</li>\n          <li>In RNA, the sugar component of the ribonucleotide has a hydroxyl group at the 2' position, making it a target for chemical modification.</li>\n        </ul>\n      </li>\n      <li><b><font color=\"blue\">Primer Extension:</font></b>\n        <ul>\n          <li>After the acylation reaction, a primer (a short strand of RNA or DNA) is used to initiate an extension reaction.</li>\n          <li>The extension reaction is stopped wherever the modification has occurred, essentially creating a \"map\" of where the RNA has been modified.</li>\n        </ul>\n      </li>\n    </ul>\n  </li>\n  <li>The SHAPE method measures the flexibility or \"reactivity\" at each nucleotide position in an RNA strand.\n    <ul>\n      <li>The idea is that positions in the RNA strand that are not base-paired (and therefore more exposed and flexible) will react more with the SHAPE reagent (2A3).</li>\n      <li>Those that are base-paired will be less reactive because they are less accessible to the reagent due to the RNA's 3D folded structure.</li>\n    </ul>\n  </li>\n  <li>SHAPE provides valuable data about which parts of an RNA molecule are more or less likely to be involved in base-pairing and hence gives insights into the likely secondary and even tertiary structure of the RNA.\n    <ul>\n      <li>This is particularly useful for predicting RNA structure in a way that can be validated both in-vitro (outside a living organism, usually in a lab setting) and in-vivo (inside a living organism).</li>\n    </ul>\n  </li>\n</ul>\n\n \n---\n\n<b style=\"font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 1px;\">Can DMS be used as a reagent for SHAPE?</b>\n<ul>\n    <li><b>No.</b> SHAPE and DMS are two distinct methods for probing RNA structure.</li>\n    <li>In SHAPE, reagents like 2A3 target the RNA's sugar components to gauge their flexibility. This helps identify which parts of the RNA are more exposed in its folded structure.</li>\n    <li>DMS, on the other hand, reacts with specific nucleobases in the RNA, offering insights into how these bases are exposed or paired. \n    <li>While both reagents provide valuable data on RNA structure, they work through different mechanisms and focus on different aspects of the RNA molecule.</li>\n    <li><b><i>SIDE NOTE:</i></b> The differences make them complimentary methodologies and together they can provide information about an RNA that is superior to either method in isolation.</li>\n</ul>\n\n---\n\n<b style=\"font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 1px;\">What are 'reads' and why are they important (a column in our dataset)</b>\n<ul>\n    <li>Reads are the number of reads in the sequencing experiment related to a specific RNA sequence</li>\n    <li>It could be construed that the higher the number, the more reliable the data might be.</li>\n</ul>\n\n---\n\n<b style=\"font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 1px;\">What is a <i>pseudoknot</i> and why do I care?</b>\n<ul>\n    <li>In RNA, a pseudoknot happens when a loop gets tangled up with another part it's not usually supposed to. This twisty knot can make RNA do special things, but it also makes it tricky to study!\n    <li> An analogy to help understand would be trying to tie your shoelaces with a bow... but instead of doing the usual bow, you make another knot using parts of the loops/bow you've already made. This extra knot on top of the usual one is like a pseudoknot in RNA.\n    <li>Pseudoknots are biologically important and can influence the function of the RNA molecule (and they can act as a catalyst or in other functionally important ways!). They play a crucial role in various cellular activities, including gene regulation, protein synthesis, and even in the life cycle of some viruses. However, <b>pseudoknots make computational prediction of RNA structure more challenging because most conventional algorithms are not designed to handle these non-nested interactions.</b></li>\n</ul>\n\n<!-- signal_to_noise: A floating-point number that indicates the quality of the measurements. Higher values suggest better data quality.\n\nSN_filter: A Boolean value that filters the data based on signal_to_noise and reads. For scoring purposes, only entries with SN_filter set to 1 (which means they pass the quality criteria) will be used. -->","metadata":{}},{"cell_type":"markdown","source":"TABLE TBD\n\n<!-- <br>\n\n<table style=\"width: 100%; background-color: #FFFFFF; border-collapse: collapse; border-width: 2px; border-color: #1CDBA5; border-style: solid; color: #000000;\">\n  <thead style=\"background-color: #5CDB9F;\">\n    <tr>\n      <th style=\"border-width: 2px; border-color: #1CDBA5; border-style: solid; padding: 12px; font-size: 14px;\"><u>QUESTIONS</u></th>\n      <th style=\"border-width: 2px; border-color: #1CDBA5; border-style: solid; padding: 12px;  font-size: 14px;\"><u>ANSWERS</u></th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <td style=\"border-width: 2px; border-color: #1CDBA5; border-style: solid; padding: 6px;\">Why do we care about Neutrinos?</td>\n      <td style=\"border-width: 2px; border-color: #1CDBA5; border-style: solid; padding: 6px;\"><b><i>... To Be Expanded (placeholder content) ...</i></b><ul>\n          <li>Neutrinos, “invented” to balance a physics equation, have grown to fascinate astrophysicists, galactic voyeurs seeking signals from astonishingly energetic structures and events in the deep universe. The direction and energy of neutrinos from each source should offer clues about the origin<ul>\n              <li><b>Gamma Ray Burst:</b> In a couple of dozen seconds, these gargantuan gamma-ray sources can send out as much energy as our sun will during its entire life. The bursts, billions of light years distant, may result from the collapse of a massive star, but a paper from the IceCube group will soon question whether they are major neutrino sources <a href=\"https://neutrinos.fnal.gov/types/energies/#:~:text=The%20energy%20of%20a%20neutrino,will%20create%20more%20energetic%20neutrinos.\"><b>[ref]</b></a></li>\n              <li><b>Active Galactic Nucleus:</b> This stormy region around a black hole emits huge amounts of energy but is shrouded by gas and dust. Active galactic nuclei are astonishingly bright source of microwave, infrared, visible, ultraviolet and gamma radiation, and likely neutrinos as well.</li>\n              <li><b>Supernova:</b> The explosion of a dying star occurs when gravity overwhelms the outward pressure from nuclear fusion. The last nearby supernova, in 1987, energized astronomers and caused a 10-second burst of neutrinos that lent credibility to neutrino science.</li>\n              <li><b>Neutron Star:</b> This relic of a supernova is composed of pure neutrons, which don’t repel each other. Therefore, neutron stars are rather dense: a teaspoonful probably weighs several billion tons. Neutron stars start life at about 10 11° C to 10 12° C, but quickly radiate away energy via an intense blast of neutrinos and electromagnetic radiation.</li>\n              </ul></li>\n          </ul>\n      </td>\n    </tr>\n    <tr>\n      <td style=\"border-width: 2px; border-color: #1CDBA5; border-style: solid; padding: 6px;\">If Neutrino's are 'Ghost Particles' and they don't interact with anything, how can we possibly 'detect' them?</td>\n      <td style=\"border-width: 2px; border-color: #1CDBA5; border-style: solid; padding: 6px;\"><b><i>... To Be Expanded (placeholder content) ...</i></b><ul>\n          <li>The tiny percentage of neutrinos that interact with atomic nuclei in the ice produce muons</li>\n          <li>These muons create Cherenkov Radiation/Light when they interact with matter.\n          <li>The neutrino cross section is a measure of how likely the neutrino is to be stopped by regular matter. <b>The higher energy a neutrino has, the more likely it is to interact.</b></li>\n          <li><b>HOST: </b>A neutrino interaction (deep inelastic scattering) will usually create a number of (charged) particles</li>\n          <li><b>HOST: </b>Cherenkov radiation –  When charged particles through travel a medium faster light, they emit radiation than Cherenkov. This UV/blue light is the same as can be seen in nuclear reactors</li>\n          </ul>\n        </td>\n    </tr>\n    <tr>\n      <td style=\"border-width: 2px; border-color: #1CDBA5; border-style: solid; padding: 6px;\">Why the South Pole?</td>\n      <td style=\"border-width: 2px; border-color: #1CDBA5; border-style: solid; padding: 6px;\"><b><i>... To Be Expanded (placeholder content) ...</i></b><ul>\n          <li>Cosmic rays are deflected at the North Pole but they are detectable at the South Pole</li>\n          <li>The super clear ice thing!! (light propgates almost 200m vs. 2m in distilled water)</li>\n          </ul>\n        </td>\n    </tr>\n    <tr>\n      <td style=\"border-width: 2px; border-color: #1CDBA5; border-style: solid; padding: 6px;\">What does it mean when we say a Neutrino has energy?</td>\n      <td style=\"border-width: 2px; border-color: #1CDBA5; border-style: solid; padding: 6px;\"><ul>\n          <li>the total energy of the particle as: $E^2 = p^2c^2 + m^2c^4$</li>\n          <li>$p$ in the above equation is the relativistic expression for the momentum: $p = \\frac{mv}{\\sqrt{1 - v^2/c^2}}$</li>\n          <li>Neutrinos have a non-zero rest mass (the $m$ in the above equation) but this mass is very small and we don't know its values for the three types of neutrino. At the moment we only have upper limits.</li>\n          <li>The non-zero rest mass means there will be a minimum energy the neutrinos can have of $mc^2$.</li>\n          <li>Because mass of a neutrino appears constant within a slim margin of error, the energy a neutrino has is controlled by the originating event that created that neutrino. A small event (beta decay) will produce lower energy neutrinos, while a larger event (supernova) will produce high energy events.</li>\n          <li>This image shows the approximations for neutrino energies associated with various cosmological events. Note that the neutrino cross section (on the y axis with $mb$ units) is a measure of how likely the neutrino is to be stopped by regular matter ($mb$ stands for <a href=\"https://www.wikiwand.com/en/Barn_(unit)\"><b>millibarn</b></a> and is equal to an area that is $10^{-31}m^2$). i.e. The higher energy a neutrino has, the more likely it is to interact:<br><br><img src=\"https://neutrinos.fnal.gov/wp-content/uploads/2018/04/neutrino-energy-scale-web.jpg\"></li>\n          </ul>\n        </td>\n    </tr>\n    <tr>\n      <td style=\"border-width: 2px; border-color: #1CDBA5; border-style: solid; padding: 6px;\">\n          What are the different types of \"event\" energy signatures that are possible?\n      </td>\n      <td style=\"border-width: 2px; border-color: #1CDBA5; border-style: solid; padding: 6px;\">\n          Neutrinos produce one of two topologically distinct signatures: <b>tracks</b> and <b>cascades</b>\n          <img src=\"https://i.ibb.co/h8N1sZW/Screenshot-2023-01-29-at-8-28-18-PM.png\">\n          <ul>\n              <li><b>Tracks</b> occur when a neutrino collides with matter in or near IceCube, resulting in a <mark><b>high-energy muon</b></mark> that travels a long distance, leaving an elongated “track” of signals in its wake. This is what's known as a Charged Current (CC) interaction and for it to leave a track like this it is imperative that the Neutrino be a Muon type Neutrino.</li>\n              <li><b>Cascades</b> are produced with all Neutral Current (NC) or non-Muon Neutrino CC interactions. These types of interactions yield hadronic and electromagnetic showers that typically range less than 20m (Aartsen et al. 2014a), with 90% of the light emitted within 4m of the shower maximum (Radel & Wiebusch 2013)—a short distance compared to the scattering and absorption lengths of light in the ice (Aartsen et al. 2013b) as well as the spacing of the PMTs. <b>These showers produce a nearly spherically symmetric cascade signature in light.</b></li>\n              <li>\n                  <b>Cascades are more difficult to reconstruct than tracks</b>, which are usually used in searches for astrophysical neutrino sources, but they have their own advantages, including providing a better measurement of neutrino energy.</li>\n              <li>One final note about <b>Cascades</b> is that they can also appear in pairs connected sometimes by a track... this is known as a <b><mark>DOUBLE BANG</mark>. This happens if the incident neutrino is energetic enough, the heavy neutrino may travel some distance before decaying.</b></li>\n              <li>Other names for more common patterns of tracks and cascades exist, but will be ignored for now</li>\n          </ul>\n      </td>\n    </tr>\n    <tr>\n      <td style=\"border-width: 2px; border-color: #1CDBA5; border-style: solid; padding: 6px;\">What are the different flavours of Neutrinos and do we care?</td>\n      <td style=\"border-width: 2px; border-color: #1CDBA5; border-style: solid; padding: 6px;\"><b>TBD – Not checked or formatted...</b>–– Neutrinos have a peculiar property of changing their “flavour” while traveling. For example, an initial electron neutrino (𝜈 neutrino ( 𝜈e ) can become a muon 𝜇 ) or a tau neutrino ( after a travelling a distance)</td>\n    </tr>\n    <tr>\n      <td style=\"border-width: 2px; border-color: #1CDBA5; border-style: solid; padding: 6px;\">Q</td>\n      <td style=\"border-width: 2px; border-color: #1CDBA5; border-style: solid; padding: 6px;\">A</td>\n    </tr>\n    <tr>\n      <td style=\"border-width: 2px; border-color: #1CDBA5; border-style: solid; padding: 6px;\">Q</td>\n      <td style=\"border-width: 2px; border-color: #1CDBA5; border-style: solid; padding: 6px;\">A</td>\n    </tr>\n    <tr>\n      <td style=\"border-width: 2px; border-color: #1CDBA5; border-style: solid; padding: 6px;\">Q</td>\n      <td style=\"border-width: 2px; border-color: #1CDBA5; border-style: solid; padding: 6px;\">A</td>\n    </tr>\n    <tr>\n      <td style=\"border-width: 2px; border-color: #1CDBA5; border-style: solid; padding: 6px;\">Q</td>\n      <td style=\"border-width: 2px; border-color: #1CDBA5; border-style: solid; padding: 6px;\">A</td>\n    </tr>\n  </tbody>\n</table> -->","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<a id=\"background_information_glossary\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #DC6FAF;\" id=\"background_information_glossary\">4&nbsp;&nbsp;BACKGROUND INFORMATION GLOSSARY&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\" style=\"color: #FD3792;\">&#10514;</a></h1>\n\n<b style=\"font-size: 20px;\"><font color=\"red\">WIP ... STAY TUNED</font></b>\n<br><br>\n\n---\n\n<br>\n\n<a style=\"color: #FD3792;\" href=\"#DMS\"><b style=\"font-size: 18px;\">DMS</b></a>\n<ul>\n    <li><a href=\"#dms_definition\">Pointform Definition</a></li>\n    <li><a href=\"#dms_eli5\">Competition (ELI5) Definition</a></li>\n    <li><a href=\"#dms_visual\">Visual Definition/Helpers</a></li>\n</ul>\n\n<a style=\"color: #FD3792;\" href=\"#2A3\"><b style=\"font-size: 18px;\">2A3</b></a>\n<ul>\n    <li><a href=\"#2a3_definition\">Pointform Definition</a></li>\n    <li><a href=\"#2a3_eli5\">Competition (ELI5) Definition</a></li>\n    <li><a href=\"#2a3_visual\">Visual Definition/Helpers</a></li>\n</ul>\n\n<a style=\"color: #FD3792;\" href=\"#tbd\"><b style=\"font-size: 18px;\">RNA STUFF SOON!</b></a>\n<ul>\n    <li><a href=\"#tbd_definition\">Pointform Definition</a></li>\n    <li><a href=\"#tbd_eli5\">Competition (ELI5) Definition</a></li>\n    <li><a href=\"#tbd_visual\">Visual Definition/Helpers</a></li>\n</ul>\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<br><a id=\"DMS\"></a><br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 16px; text-transform: uppercase;\"></b>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DC6FAF; background-color: #ffffff;\">4.1 DIMETHYL SULFATE (DMS)</h3>\n\n---\n\n\n<a id=\"dms_definition\"></a><br><br>\n\n<b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">POINTFORM DEFINITION</b>\n* <b style=\"color: red;\">WIP – More coming</b>\n\n**REFERENCES**\n* TBD\n\n<a id=\"dms_eli5\"></a><br><br>\n<b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">ELI5 COMPETITION DEFINITION</b>\n* <b style=\"color: red;\">TBD</b>\n\n<a id=\"dms_visual\"></a><br><br>\n\n<b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">EXPLAIN IT WITH PICTURES</b>\n* <b style=\"color: red;\">TBD</b>\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<br><a id=\"2A3\"></a><br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 16px; text-transform: uppercase;\"></b>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DC6FAF; background-color: #ffffff;\">4.2 2-AMINOPYRIDINE-3-CARBOXYLIC ACID IMIDAZOLIDE (2A3)</h3>\n\n---\n\n\n<a id=\"2a3_definition\"></a><br><br>\n\n<b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">POINTFORM DEFINITION</b>\n* <b style=\"color: red;\">WIP – More coming</b>\n\n**REFERENCES**\n* TBD\n\n<a id=\"2a3_eli5\"></a><br><br>\n<b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">ELI5 COMPETITION DEFINITION</b>\n* <b style=\"color: red;\">TBD</b>\n\n<a id=\"2a3_visual\"></a><br><br>\n\n<b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">EXPLAIN IT WITH PICTURES</b>\n* <b style=\"color: red;\">TBD</b>","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<a id=\"imports\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #DC6FAF;\" id=\"imports\">5&nbsp;&nbsp;IMPORTS&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\" style=\"color: #FD3792;\">&#10514;</a></h1>\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">COMPETITION SPECIFIC LIBRARIES</b>\n\n* **`arnie`**\n    * a helpful utility library that simplifies interacting with various secondary structure prediction packages.\n    * **Arnie** needs at least one secondary structure predictor. Below are some examples:\n        * **EternaFold**: Specialized algorithm developed for RNA secondary structure prediction. EternaFold aims to integrate human intuition and computational power to predict RNA secondary structures more accurately.\n        * **ViennaRNA**: Provides various algorithms and utilities for predicting RNA secondary structures.\n        * **Mfold**: Another popular tool used for RNA secondary structure prediction.\n        * **RNAstructure**: A software package that predicts RNA secondary structure.\n        * **NUPACK**: Focuses on the analysis and prediction of nucleic acid secondary structures.\n        * **RNAfold**: Part of the ViennaRNA package, specifically used for RNA secondary structure prediction.\n* **`eternafold`**\n    * Eternafold is a leading prediction package that was trained using sequences collected via the citizen science game Eterna. \n    * In fact, Eterna players provided many of the sequences in the data for this competition.\n* **`forgi` (stands for Finding of RNA Graph Information.)**\n    * Provides various utilities for representing, manipulating, and visualizing RNA structures.\n    * In forgi, RNA secondary structures are represented as graphs, making it easier to perform operations that would be complicated in sequence or matrix representations. \n    * These graphs can be straightforward, representing just the base pairs, or more complex, representing multi-loop or pseudoknot structures.\n* **`viennarna`**\n    * ViennaRNA is a package that includes a collection of algorithms and utilities for RNA secondary structure prediction.\n    * Part of the package, **RNAfold**, is specifically used for RNA secondary structure prediction, and there are other tools for things like comparison and visualization of RNA structures. \n    * This is installed via bioconda...\n    * **`bioconda`**\n        * A distribution of bioinformatics software for the Conda package manager.\n        * **Bioconda** makes it easy to install bioinformatics software and their dependencies, including many of the RNA secondary structure prediction packages mentioned above.\n","metadata":{}},{"cell_type":"code","source":"print(\"\\n... PIP INSTALLS STARTING ...\\n\")\n# Install Arnie\n!pip install -q --upgrade arnie \n\n# Install Eternafold; not available via conda-forge quite yet, so we install from the pipeline build artifacts directly\n# !conda install eternafold\n!wget -q https://artprodeus21.artifacts.visualstudio.com/A910fa339-c7c2-46e8-a579-7ea247548706/84710dde-1620-425b-80d0-4cf5baca359d/_apis/artifact/cGlwZWxpbmVhcnRpZmFjdDovL2NvbmRhLWZvcmdlL3Byb2plY3RJZC84NDcxMGRkZS0xNjIwLTQyNWItODBkMC00Y2Y1YmFjYTM1OWQvYnVpbGRJZC83NzI5MTAvYXJ0aWZhY3ROYW1lL2NvbmRhX3BrZ3NfbGludXg1/content?format=zip\n!unzip -q content\\?format\\=zip\n!conda install -y -q conda_pkgs_linux/eternafold-1.3.1-h00ab1b0_0.conda\n\n# Ordinarily, the EternaFold conda package will automatically set necessary environment variables, \n# but Kaggle's conda install works a little differently. \n#\n# Let's set them manually here using %env.\n%env ETERNAFOLD_PATH=/opt/conda/bin/eternafold-bin\n%env ETERNAFOLD_PARAMETERS=/opt/conda/lib/eternafold-lib/parameters/EternaFoldParams.v1\n\n# Install forgi - a library for working with and visualizing RNA structures\n!pip install -q --upgrade forgi\n\n# Install viennarna\n!conda install -y -q -c bioconda viennarna\n\n# Install draw_rna - a library from DaS for visualizing RNA structures (works with Arnie)\n!pip install -q git+https://github.com/DasLab/draw_rna.git\n\nprint(\"\\n... PIP INSTALLS COMPLETE ...\\n\")\n\nprint(\"\\n... IMPORTS STARTING ...\\n\")\nprint(\"\\n\\tVERSION INFORMATION\")\n\n# Competition Specific Imports\nfrom arnie.mfe import mfe   # mfe(sequence,  package=\"eternafold\") - or the package of your choice ()\nfrom arnie.bpps import bpps # bpps(sequence, package=\"eternafold\") - or the package of your choice\nimport forgi.visual.mplotlib as fvm\nimport forgi.graph.bulge_graph as fgb\nimport draw_rna\n\n# Machine Learning and Data Science Imports (basics)\nimport tensorflow as tf; print(f\"\\t\\t– TENSORFLOW VERSION: {tf.__version__}\");\nimport pandas as pd; pd.options.mode.chained_assignment = None; pd.set_option('display.max_columns', None);\nimport numpy as np; print(f\"\\t\\t– NUMPY VERSION: {np.__version__}\");\nimport sklearn; print(f\"\\t\\t– SKLEARN VERSION: {sklearn.__version__}\");\n\n# Built-In Imports (mostly don't worry about these)\nfrom kaggle_datasets import KaggleDatasets\nfrom collections import Counter\nfrom datetime import datetime\nfrom zipfile import ZipFile\nfrom glob import glob\nimport Levenshtein\nimport subprocess\nimport warnings\nimport requests\nimport hashlib\nimport imageio\nimport IPython\nimport sklearn\nimport urllib\nimport zipfile\nimport pickle\nimport random\nimport shutil\nimport string\nimport json\nimport math\nimport time\nimport gzip\nimport ast\nimport sys\nimport io\nimport os\nimport gc\nimport re\n\n# Visualization Imports (overkill)\nfrom matplotlib.colors import ListedColormap, LinearSegmentedColormap\nfrom matplotlib.patches import Rectangle\nimport matplotlib.patches as patches\nimport plotly.graph_objects as go\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm; tqdm.pandas();\nimport plotly.express as px\nimport seaborn as sns\nfrom PIL import Image, ImageEnhance; Image.MAX_IMAGE_PIXELS = 5_000_000_000;\nimport matplotlib; print(f\"\\t\\t– MATPLOTLIB VERSION: {matplotlib.__version__}\");\nfrom matplotlib import animation, rc; rc('animation', html='jshtml')\nimport plotly\nimport PIL\nimport cv2\n\nimport plotly.io as pio\nprint(pio.renderers)\n\ndef check_nvidia_gpu():\n    try:\n        subprocess.check_output('nvidia-smi')\n        print('\\n... NVIDIA GPU DETECTED ...\\n')\n        return True\n    except Exception: # this command not being found can raise quite a few different errors depending on the configuration\n        print('\\n... NO NVIDIA GPU DETECTED ...\\n')\n        return False\n\ndef seed_it_all(seed=7):\n    \"\"\" Attempt to be Reproducible \"\"\"\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n\nseed_it_all()\n\n# Check if GPU attached and imports RAPIDS if so.\nHAS_GPU = check_nvidia_gpu()\nif HAS_GPU: import cudf, cuml, cupy\n\nprint(\"\\n\\n... IMPORTS COMPLETE ...\\n\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-12T13:54:14.716757Z","iopub.execute_input":"2023-09-12T13:54:14.717235Z","iopub.status.idle":"2023-09-12T13:57:24.999474Z","shell.execute_reply.started":"2023-09-12T13:54:14.717183Z","shell.execute_reply":"2023-09-12T13:57:24.998453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<a id=\"setup\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #DC6FAF;\" id=\"setup\">6&nbsp;&nbsp;SETUP AND HELPER FUNCTIONS&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\" style=\"color: #FD3792;\">&#10514;</a></h1>","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DC6FAF; background-color: #ffffff;\">6.1 HELPER FUNCTIONS</h3>\n\n---\n\n<br>","metadata":{}},{"cell_type":"code","source":"def flatten_l_o_l(nested_list):\n    \"\"\" Flatten a list of lists \"\"\"\n    return [item for sublist in nested_list for item in sublist]\n\ndef print_ln(symbol=\"-\", line_len=110):\n    \"\"\" Prints a line for asthetics \"\"\"\n    print(symbol*line_len)\n    \ndef downcast_df(df):\n    \"\"\" Simple function that takes any column represented as 64 bits and casts to 32 \"\"\"\n    for col in df.columns:\n        df[col] = df[col].astype(df[col].dtype.name.replace('64', '32'))\n    return df\n\ndef get_helper_data_files(_train_df):\n    \"\"\" \n    This will downcast the training data (use all of it not a subsample).\n    We then split the downcast data based on their respective experimental methods (2A3-SHAPE and DMS).\n    It will then split out the reactivity and reactivity_error columns as numpy arrays\n    It will then save everything (csv for the dataframes and npz for the reactivity values)\n    \n    This data is available as a public dataset \n        --> https://www.kaggle.com/datasets/dschettler8845/sr-rna-folding-helpers\n    \"\"\"\n    _train_df = downcast_df(_train_df)\n    \n    train_dms_df = _train_df[_train_df.experiment_type==\"DMS_MaP\"].reset_index(drop=True)\n    train_2a3_df = _train_df[_train_df.experiment_type==\"2A3_MaP\"].reset_index(drop=True)\n    \n    train_dms_reactivity_np = train_dms_df[[_c for _c in train_dms_df.columns if _c.startswith(\"reactivity_0\")]].to_numpy()\n    train_dms_reactivity_error_np = train_dms_df[[_c for _c in train_dms_df.columns if _c.startswith(\"reactivity_error_0\")]].to_numpy()\n    np.save(\"train_dms_reactivity_fp32\", train_dms_reactivity_np.astype(np.float32))\n    np.save(\"train_dms_reactivity_error_fp32\", train_dms_reactivity_error_np.astype(np.float32))\n\n    train_2a3_reactivity_np = train_2a3_df[[_c for _c in train_2a3_df.columns if _c.startswith(\"reactivity_0\")]].to_numpy()\n    train_2a3_reactivity_error_np = train_2a3_df[[_c for _c in train_2a3_df.columns if _c.startswith(\"reactivity_error_0\")]].to_numpy()\n    np.save(\"train_2a3_reactivity_fp32\", train_2a3_reactivity_np.astype(np.float32))\n    np.save(\"train_2a3_reactivity_error_fp32\", train_2a3_reactivity_error_np.astype(np.float32))\n\n    train_dms_df[[_c for _c in train_dms_df.columns if not _c.startswith(\"reactivity\")]].to_csv(\"train_dms_wo_reactivity.csv\", index=False)\n    train_2a3_df[[_c for _c in train_2a3_df.columns if not _c.startswith(\"reactivity\")]].to_csv(\"train_2a3_wo_reactivity.csv\", index=False)","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-09-12T13:57:25.001704Z","iopub.execute_input":"2023-09-12T13:57:25.002700Z","iopub.status.idle":"2023-09-12T13:57:25.014075Z","shell.execute_reply.started":"2023-09-12T13:57:25.002656Z","shell.execute_reply":"2023-09-12T13:57:25.013053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DC6FAF; background-color: #ffffff;\">6.2 LOAD THE DATA</h3>\n\n---\n\nThe full data is huge and we will be loading the metadata separately later anyway... so let's just load a portion (10K)","metadata":{}},{"cell_type":"code","source":"# Define the path to theflags root data directory\nUSE_CUDF = False\nDATA_DIR = \"/kaggle/input/stanford-ribonanza-rna-folding\"\n\n# DEBUG to only open a fraction\nDEBUG = True\n\n# Define the paths to the batches of train and test data respectively\nTRAIN_CSV_PATH = os.path.join(DATA_DIR, \"train_data.csv\")\nTEST_CSV_PATH  = os.path.join(DATA_DIR, \"test_sequences.csv\")\nSS_CSV_PATH  = os.path.join(DATA_DIR, \"sample_submission.csv\")\n\n# These columns are NaN for the entire train dataset\ntrain_nan_cols = ['reactivity_0001', 'reactivity_0002', 'reactivity_0003', 'reactivity_0004', 'reactivity_0005', 'reactivity_0006', 'reactivity_0007', 'reactivity_0008', 'reactivity_0009', 'reactivity_0010', 'reactivity_0011', 'reactivity_0012', 'reactivity_0013', 'reactivity_0014', 'reactivity_0015', 'reactivity_0016', 'reactivity_0017', 'reactivity_0018', 'reactivity_0019', 'reactivity_0020', 'reactivity_0021', 'reactivity_0022', 'reactivity_0023', 'reactivity_0024', 'reactivity_0025', 'reactivity_0026', 'reactivity_0166', 'reactivity_0167', 'reactivity_0168', 'reactivity_0169', 'reactivity_0170', 'reactivity_error_0001', 'reactivity_error_0002', 'reactivity_error_0003', 'reactivity_error_0004', 'reactivity_error_0005', 'reactivity_error_0006', 'reactivity_error_0007', 'reactivity_error_0008', 'reactivity_error_0009', 'reactivity_error_0010', 'reactivity_error_0011', 'reactivity_error_0012', 'reactivity_error_0013', 'reactivity_error_0014', 'reactivity_error_0015', 'reactivity_error_0016', 'reactivity_error_0017', 'reactivity_error_0018', 'reactivity_error_0019', 'reactivity_error_0020', 'reactivity_error_0021', 'reactivity_error_0022', 'reactivity_error_0023', 'reactivity_error_0024', 'reactivity_error_0025', 'reactivity_error_0026', 'reactivity_error_0166', 'reactivity_error_0167', 'reactivity_error_0168', 'reactivity_error_0169', 'reactivity_error_0170', 'reactivity_0171', 'reactivity_0172', 'reactivity_0173', 'reactivity_0174', 'reactivity_0175', 'reactivity_0176', 'reactivity_0177', 'reactivity_error_0171', 'reactivity_error_0172', 'reactivity_error_0173', 'reactivity_error_0174', 'reactivity_error_0175', 'reactivity_error_0176', 'reactivity_error_0177', 'reactivity_0178', 'reactivity_0179', 'reactivity_0180', 'reactivity_0181', 'reactivity_0182', 'reactivity_0183', 'reactivity_0184', 'reactivity_0185', 'reactivity_0186', 'reactivity_0187', 'reactivity_0188', 'reactivity_0189', 'reactivity_0190', 'reactivity_0191', 'reactivity_0192', 'reactivity_0193', 'reactivity_0194', 'reactivity_0195', 'reactivity_0196', 'reactivity_0197', 'reactivity_0198', 'reactivity_0199', 'reactivity_0200', 'reactivity_0201', 'reactivity_0202', 'reactivity_0203', 'reactivity_0204', 'reactivity_0205', 'reactivity_0206', 'reactivity_error_0178', 'reactivity_error_0179', 'reactivity_error_0180', 'reactivity_error_0181', 'reactivity_error_0182', 'reactivity_error_0183', 'reactivity_error_0184', 'reactivity_error_0185', 'reactivity_error_0186', 'reactivity_error_0187', 'reactivity_error_0188', 'reactivity_error_0189', 'reactivity_error_0190', 'reactivity_error_0191', 'reactivity_error_0192', 'reactivity_error_0193', 'reactivity_error_0194', 'reactivity_error_0195', 'reactivity_error_0196', 'reactivity_error_0197', 'reactivity_error_0198', 'reactivity_error_0199', 'reactivity_error_0200', 'reactivity_error_0201', 'reactivity_error_0202', 'reactivity_error_0203', 'reactivity_error_0204', 'reactivity_error_0205', 'reactivity_error_0206']\n\nprint(\"\\n... BASIC DATA SETUP STARTING ...\\n\")\nprint(\"\\n\\n... LOAD TRAIN DATAFRAME FROM CSV FILE ...\\n\")\nif (USE_CUDF and HAS_GPU):\n    raise NotImplementedError()\n    # train_df = cudf.read_csv(TRAIN_CSV_PATH)\nelse:\n    train_df = pd.read_csv(TRAIN_CSV_PATH, nrows=10_000 if DEBUG else None)\ntrain_df = downcast_df(train_df)\n\nprint(\"\\n\\n... LOAD TEST META DATAFRAME FROM CSV FILE ...\\n\")\nif (USE_CUDF and HAS_GPU):\n    raise NotImplementedError()\n    # test_df = cudf.read_csv(TEST_CSV_PATH)\nelse:\n    test_df = pd.read_csv(TEST_CSV_PATH, nrows=10_000 if DEBUG else None)\ndisplay(test_df)\n\nprint(\"\\n\\n... LOAD SAMPLE SUBMISSION DATAFRAME FROM CSV FILE ...\\n\")\nss_df = pd.read_csv(SS_CSV_PATH, nrows=10 if DEBUG else None)\ndisplay(ss_df)\n\n# Cleanup just in case....\ngc.collect(); gc.collect(); gc.collect();\n\nprint(\"\\n\\n\\n... BASIC DATA SETUP FINISHED ...\\n\\n\")","metadata":{"execution":{"iopub.status.busy":"2023-09-12T13:57:25.015629Z","iopub.execute_input":"2023-09-12T13:57:25.016037Z","iopub.status.idle":"2023-09-12T13:57:26.569734Z","shell.execute_reply.started":"2023-09-12T13:57:25.015982Z","shell.execute_reply":"2023-09-12T13:57:26.568928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DC6FAF; background-color: #ffffff;\">6.3 SUPERFICIAL EXAMINATION OF THE DATA</h3>\n\n---\n\nWe will use **`.describe`** and **`.info`** to quickly see information about NaN frequency, datatype, etc.\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">TRAIN DATA OBSERVATIONS</b>\n* Many of the columns appear to be completely composed of NaN values\n    * This is confirmed when viewing the full data as well (although to a lesser extent) and I have added the columns that are always missing as a list of strings above title `train_nan_cols` (see full list below)\n* While negative reactivity values are possible they are abnormal. All **predicted** reactivity should be between 0-1.\n* Reactivity values by column appear to have different means and standard deviations... we will examine in more detail later.\n* There are non missing values for **`sequence`**, **`experiment_type`**, **`dataset_name`**, **`reads`**, **`signal_to_noise`** or **`SN_filter`**\n* **`experiment_type`** is either **`2A3_MaP`** or **`DMS_MaP`** for the respective approaches. (I think we should approach these separately and make two models)\n* **`dataset_name`** looks to not be missing any data and contains approximately 40-50 unique strings. (to be confirmed below)\n   \n```python\ntrain_nan_cols = [\n    'reactivity_0001', 'reactivity_0002', 'reactivity_0003', 'reactivity_0004', 'reactivity_0005', 'reactivity_0006', 'reactivity_0007',\n    'reactivity_0008', 'reactivity_0009', 'reactivity_0010', 'reactivity_0011', 'reactivity_0012', 'reactivity_0013', 'reactivity_0014',\n    'reactivity_0015', 'reactivity_0016', 'reactivity_0017', 'reactivity_0018', 'reactivity_0019', 'reactivity_0020', 'reactivity_0021',\n    'reactivity_0022', 'reactivity_0023', 'reactivity_0024', 'reactivity_0025', 'reactivity_0026', 'reactivity_0166', 'reactivity_0167',\n    'reactivity_0168', 'reactivity_0169', 'reactivity_0170', 'reactivity_0171', 'reactivity_0172', 'reactivity_0173', 'reactivity_0174',\n    'reactivity_0175', 'reactivity_0176', 'reactivity_0177', 'reactivity_0178', 'reactivity_0179', 'reactivity_0180', 'reactivity_0181',\n    'reactivity_0182', 'reactivity_0183', 'reactivity_0184', 'reactivity_0185', 'reactivity_0186', 'reactivity_0187', 'reactivity_0188',\n    'reactivity_0189', 'reactivity_0190', 'reactivity_0191', 'reactivity_0192', 'reactivity_0193', 'reactivity_0194', 'reactivity_0195',\n    'reactivity_0196', 'reactivity_0197', 'reactivity_0198', 'reactivity_0199', 'reactivity_0200', 'reactivity_0201', 'reactivity_0202',\n    'reactivity_0203', 'reactivity_0204', 'reactivity_0205', 'reactivity_0206', 'reactivity_error_0001', 'reactivity_error_0002',\n    'reactivity_error_0003', 'reactivity_error_0004', 'reactivity_error_0005', 'reactivity_error_0006', 'reactivity_error_0007',\n    'reactivity_error_0008', 'reactivity_error_0009', 'reactivity_error_0010', 'reactivity_error_0011', 'reactivity_error_0012',\n    'reactivity_error_0013', 'reactivity_error_0014', 'reactivity_error_0015', 'reactivity_error_0016', 'reactivity_error_0017',\n    'reactivity_error_0018', 'reactivity_error_0019', 'reactivity_error_0020', 'reactivity_error_0021', 'reactivity_error_0022',\n    'reactivity_error_0023', 'reactivity_error_0024', 'reactivity_error_0025', 'reactivity_error_0026', 'reactivity_error_0166',\n    'reactivity_error_0167', 'reactivity_error_0168', 'reactivity_error_0169', 'reactivity_error_0170', 'reactivity_error_0171',\n    'reactivity_error_0172', 'reactivity_error_0173', 'reactivity_error_0174', 'reactivity_error_0175', 'reactivity_error_0176',\n    'reactivity_error_0177', 'reactivity_error_0178', 'reactivity_error_0179', 'reactivity_error_0180', 'reactivity_error_0181',\n    'reactivity_error_0182', 'reactivity_error_0183', 'reactivity_error_0184', 'reactivity_error_0185', 'reactivity_error_0186',\n    'reactivity_error_0187', 'reactivity_error_0188', 'reactivity_error_0189', 'reactivity_error_0190', 'reactivity_error_0191',\n    'reactivity_error_0192', 'reactivity_error_0193', 'reactivity_error_0194', 'reactivity_error_0195', 'reactivity_error_0196',\n    'reactivity_error_0197', 'reactivity_error_0198', 'reactivity_error_0199', 'reactivity_error_0200', 'reactivity_error_0201',\n    'reactivity_error_0202', 'reactivity_error_0203', 'reactivity_error_0204', 'reactivity_error_0205', 'reactivity_error_0206'\n]\n```\n\n<br>\n\n<b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">TEST DATA OBSERVATIONS</b>\n* No missing data\n* Nothing else of note at this level","metadata":{}},{"cell_type":"code","source":"print(\"\\n... TRAIN DATA DESCRIPTION ...\\n\")\ndisplay(train_df.describe())\n\nprint(\"\\n... TRAIN DATA COLUMN INFO ...\\n\")\ntrain_df.info()\n\nprint(\"\\n... TEST DATA DESCRIPTION ...\\n\")\ndisplay(test_df.describe())\n\nprint(\"\\n... TEST DATA COLUMN INFO ...\\n\")\ntest_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-09-12T13:57:26.571818Z","iopub.execute_input":"2023-09-12T13:57:26.572751Z","iopub.status.idle":"2023-09-12T13:57:27.844816Z","shell.execute_reply.started":"2023-09-12T13:57:26.572721Z","shell.execute_reply":"2023-09-12T13:57:27.843985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DC6FAF; background-color: #ffffff;\">6.4 LOAD SUPPLEMENTARY METADATA PART IN FULL</h3>\n\n---\n\nWe previously split the metadata (non-reactivity) part of the dataframe into it's own dataset. Let's load this so we can examine the columns one at a time in their entirety before moving on to examining the reactivity data.","metadata":{}},{"cell_type":"code","source":"HELPER_DATASET_DIR = \"/kaggle/input/sr-rna-folding-helpers\"\ntrain_meta_dms = downcast_df(pd.read_csv(os.path.join(HELPER_DATASET_DIR, \"train_dms_wo_reactivity.csv\"))).copy()\ntrain_meta_2a3 = downcast_df(pd.read_csv(os.path.join(HELPER_DATASET_DIR, \"train_2a3_wo_reactivity.csv\"))).copy()\n\nprint(\"\\n... TRAIN METADATA FOR DMS ...\\n\")\ndisplay(train_meta_dms.head())\n\nprint(\"\\n... TRAIN METADATA FOR 2A3 ...\\n\")\ndisplay(train_meta_2a3.head())","metadata":{"execution":{"iopub.status.busy":"2023-09-12T13:57:27.845946Z","iopub.execute_input":"2023-09-12T13:57:27.846341Z","iopub.status.idle":"2023-09-12T13:57:37.678518Z","shell.execute_reply.started":"2023-09-12T13:57:27.846315Z","shell.execute_reply":"2023-09-12T13:57:37.677789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DC6FAF; background-color: #ffffff;\">6.5 ADD HELPFUL COLUMNS</h3>\n\n---\n\nLet's add some helpful columnar information to support our current datasets\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">NEW COLUMNS</b>\n1. **`seq_len`** - Sequence Length\n    * Useful for ordering/filtering based on sequence length and comparing distributions","metadata":{}},{"cell_type":"code","source":"# 1. Add the sequence length column\ntrain_df['seq_len'] = train_df['sequence'].apply(len)\ntrain_meta_dms['seq_len'] = train_meta_dms['sequence'].apply(len)\ntrain_meta_2a3['seq_len'] = train_meta_2a3['sequence'].apply(len)\ntest_df['seq_len'] = test_df['sequence'].apply(len)\n\n\n##### See the updated dataframes ...\nprint(\"\\n... TRAIN DATAFRAME - UPDATED ...\\n\")\ndisplay(train_df.head(3))\n\nprint(\"\\n... TEST DATAFRAME - UPDATED ...\\n\")\ndisplay(test_df.head(3))\n\nprint(\"\\n... TRAIN METADATA FOR DMS - UPDATED ...\\n\")\ndisplay(train_meta_dms.head(3))\n\nprint(\"\\n... TRAIN METADATA FOR 2A3 - UPDATED ...\\n\")\ndisplay(train_meta_2a3.head(3))","metadata":{"execution":{"iopub.status.busy":"2023-09-12T13:57:37.679840Z","iopub.execute_input":"2023-09-12T13:57:37.680332Z","iopub.status.idle":"2023-09-12T13:57:38.417730Z","shell.execute_reply.started":"2023-09-12T13:57:37.680301Z","shell.execute_reply":"2023-09-12T13:57:38.416744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<a id=\"data_exploration\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #DC6FAF;\" id=\"data_exploration\">7&nbsp;&nbsp;DATA EXPLORATION&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\" style=\"color: #FD3792;\">&#10514;</a></h1>\n\nWe will do columnar exploration of the respective DMS and 2A3 metadata dataframes prior to investigating the reactivity values themselves. We will also use this opportunity to plot some examples (when examining sequence).\n\nNote we grab a few helpful constants here before we start too...","metadata":{}},{"cell_type":"code","source":"N_2A3_TRAIN = len(train_meta_2a3)\nN_DMS_TRAIN = len(train_meta_dms)\nN_TRAIN = N_2A3_TRAIN+N_DMS_TRAIN\nN_TEST = len(test_df) # not really though... cause DEBUG...\n\nprint(\"\\n... Training Data Statistics ...\")\nprint(f\"\\tNumber of 2A3 Training Examples   : {N_2A3_TRAIN:,}\")\nprint(f\"\\tNumber of DMS Training Examples   : {N_DMS_TRAIN:,}\")\nprint(f\"\\tTotal Number of Training Examples : {N_TRAIN:,}\")\n\nprint(\"\\n... Test Data Statistics (Will be wrong if debug is on) ...\")\nprint(f\"\\tNumber of Test Examples : {N_TEST:,}\")","metadata":{"execution":{"iopub.status.busy":"2023-09-12T13:57:38.419080Z","iopub.execute_input":"2023-09-12T13:57:38.419398Z","iopub.status.idle":"2023-09-12T13:57:38.426271Z","shell.execute_reply.started":"2023-09-12T13:57:38.419371Z","shell.execute_reply":"2023-09-12T13:57:38.425261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DC6FAF; background-color: #ffffff;\">7.1 SEQUENCE ID</h3>\n\n---\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">HOST DESCRIPTION</b>\n\n* **`sequence_id`** - (string) An arbitrary identifier like `8cdfeef00` for each sequence.\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">MY OBSERVATIONS</b>\n* I initially thought each sequence would be unique in our dataset. This is untrue.\n    * There are over 20,000 repeats per testing type (40,000 total) in the training dataset\n    * This corresponds to the following distribution of repeats by sequence (each testing type is identical so 2x for total training data)\n        * <b>2729</b> sequences are <b>repeated 5 times</b> for each testing method (2x) --> impacts 13645 rows per test (27290 total)\n        * <b>2173</b> sequences are <b>repeated 3 times</b> for each testing method (2x) --> impacts 6519 rows per test (13038 total)\n        * <b>5</b> sequences are <b>repeated 2 times</b> for each testing method (2x) --> impacts 10 rows per test (20 total)","metadata":{}},{"cell_type":"code","source":"repeated_dms_sequence_ids = {k:v for k,v in train_meta_dms.sequence_id.value_counts().items() if v!=1}\nrepeated_2a3_sequence_ids = {k:v for k,v in train_meta_2a3.sequence_id.value_counts().items() if v!=1}\n\nprint(\"\\n... REPEATED DMS SEQUENCE IDS ...\")\nprint(f\"\\tNumber of repeated ids   : {len(repeated_dms_sequence_ids):,}\")\nprint(f\"\\tTotal repeated sequences : {sum(list(repeated_dms_sequence_ids.values())):,}\")\n\nprint(\"\\n... REPEATED 2A3 SEQUENCE IDS ...\")\nprint(f\"\\t- Number of repeated ids  : {len(repeated_2a3_sequence_ids):,}\")\nprint(f\"\\t- Total repeated sequences : {sum(list(repeated_2a3_sequence_ids.values())):,}\")\n\nprint(\"\\n\\n... EXAMPLE WITH SEQUENCE ID '1728e0d67d7c' BELOW ...\\n\")\nprint(\"\\n... TRAIN METADATA FOR DMS ('1728e0d67d7c')...\")\ndisplay(train_meta_dms[train_meta_dms.sequence_id==\"1728e0d67d7c\"])\n\nprint(\"\\n... TRAIN METADATA FOR 2A3 ('1728e0d67d7c')...\")\ndisplay(train_meta_2a3[train_meta_2a3.sequence_id==\"1728e0d67d7c\"])","metadata":{"execution":{"iopub.status.busy":"2023-09-12T13:57:38.427429Z","iopub.execute_input":"2023-09-12T13:57:38.427722Z","iopub.status.idle":"2023-09-12T13:57:40.619587Z","shell.execute_reply.started":"2023-09-12T13:57:38.427697Z","shell.execute_reply":"2023-09-12T13:57:40.618589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DC6FAF; background-color: #ffffff;\">7.2 SEQUENCE</h3>\n\n---\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">HOST DESCRIPTION</b>\n\n* **`sequence`** - (string) \n    * Describes the RNA sequence, a combination of A, G, U, and C for each sample. \n    * Should be 115 to 457 characters long.\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">MY OBSERVATIONS</b>\n* There will be the same number of repeats per test as observed in the sequence ID.\n* Just some viz below... I am taking a break before investigating this further as I want to write a separate viz notebook first...\n    * Much of the initial inspiration is taken from <b><a href=\"https://www.kaggle.com/code/hiramcho/ribonanza-rna-visualization#Plot-2D-RNA-Structure\">This Visualization Starter Notebook</a></b>.\n    * I plan to create my own detailed visualization notebook to teach myself more about all of this and will incorporate that at a later time...","metadata":{}},{"cell_type":"code","source":"def plot_rna(sequence, structure=None, visualizer=\"forgi\", viz_kwargs=None):\n    from ipynb.draw import draw_struct\n    if structure is None:\n        structure=mfe(sequence, package=\"eternafold\")\n    \n    if viz_kwargs is None:\n        viz_kwargs = {}\n    \n    if visualizer==\"das\":\n        draw_struct(sequence, structure, **viz_kwargs)\n        \n    elif visualizer==\"forgi\":\n        plt.figure(figsize=(12,10))\n        bg = fgb.BulgeGraph.from_fasta_text(f'>rna1\\n{structure}\\n{sequence}')[0]\n        fvm.plot_rna(bg, lighten=0.25, text_kwargs={\"fontweight\":None})\n        plt.show()\n    \ndemo_sequence = train_meta_2a3[train_meta_2a3.sequence_id==\"8cdfeef009ea\"].sequence.values[0]\n\n# Normal forgi version\nplot_rna(demo_sequence, visualizer=\"forgi\")\n\n# Normal DaS version\nplot_rna(demo_sequence, visualizer=\"das\")\n\n# Line version\nplot_rna(demo_sequence, visualizer=\"das\", viz_kwargs={\"line\":True})\n\n# Large version\nplot_rna(demo_sequence, visualizer=\"das\", viz_kwargs={\"large_mode\":True})","metadata":{"execution":{"iopub.status.busy":"2023-09-12T14:03:03.015282Z","iopub.execute_input":"2023-09-12T14:03:03.015764Z","iopub.status.idle":"2023-09-12T14:03:10.719335Z","shell.execute_reply.started":"2023-09-12T14:03:03.015725Z","shell.execute_reply":"2023-09-12T14:03:10.718108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bp_matrix = bpps(demo_sequence, package=\"eternafold\")\nplt.figure(figsize=(10,10))\nplt.title('Base-Pairing Probability NxN Array', fontweight=\"bold\")\nplt.xlabel(\"Sequence Position\", fontweight=\"bold\")\nplt.ylabel(\"Sequence Position\", fontweight=\"bold\")\nplt.imshow(bp_matrix, origin='lower', cmap='magma')\nplt.colorbar(shrink=0.8)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-12T14:03:32.049315Z","iopub.execute_input":"2023-09-12T14:03:32.050352Z","iopub.status.idle":"2023-09-12T14:03:32.555962Z","shell.execute_reply.started":"2023-09-12T14:03:32.050312Z","shell.execute_reply":"2023-09-12T14:03:32.555212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"p_unp_vec = 1 - np.sum(bp_matrix, axis=0)\n\nplt.figure(figsize=(13,5))\nplt.plot(p_unp_vec)\nplt.title('Probability that Nucleotide is Unpaired', fontweight=\"bold\")\nplt.xlabel(\"Sequence Position\", fontweight=\"bold\")\nplt.ylabel('p(unpaired)', fontweight=\"bold\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-12T14:03:35.655863Z","iopub.execute_input":"2023-09-12T14:03:35.656988Z","iopub.status.idle":"2023-09-12T14:03:35.934194Z","shell.execute_reply.started":"2023-09-12T14:03:35.656946Z","shell.execute_reply":"2023-09-12T14:03:35.933330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DC6FAF; background-color: #ffffff;\">7.3 SEQUENCE LENGTH</h3>\n\n---\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">HOST DESCRIPTION</b>\n\nSequences should be 115 to 457 characters long.\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">MY OBSERVATIONS</b>\n* There are only 5 different sequence lengths in the training dataset and the vast majority are of length 177 (over 95%)\n    * Sequence length of **177** occurs **784,177** per method (DMS, 2A3)\n    * Sequence length of **170** occurs **15,000** per method (DMS, 2A3)\n    * Sequence length of **115** occurs **13,645** per method (DMS, 2A3)\n    * Sequence length of **155** occurs **6,519** per method (DMS, 2A3)\n    * Sequence length of **206** occurs **2,499** per method (DMS, 2A3)","metadata":{}},{"cell_type":"code","source":"print(train_meta_2a3.seq_len.value_counts())\nfig = px.pie(train_meta_2a3, \"seq_len\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-12T14:03:41.927026Z","iopub.execute_input":"2023-09-12T14:03:41.927710Z","iopub.status.idle":"2023-09-12T14:03:43.303542Z","shell.execute_reply.started":"2023-09-12T14:03:41.927674Z","shell.execute_reply":"2023-09-12T14:03:43.300501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DC6FAF; background-color: #ffffff;\">7.4 DATASET NAME</h3>\n\n---\n\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">HOST DESCRIPTION</b>\n\n**`dataset_name`** (string):\n    * name of high throughput sequencing dataset from which the reactivity profile was extracted\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">MY OBSERVATIONS</b>\n* Unequal distribution across datasets (to be expected)\n* More investigation into how the dataset informs on other properties will be required","metadata":{}},{"cell_type":"code","source":"fig = px.bar(train_meta_2a3.dataset_name.value_counts().to_frame().reset_index(), \"dataset_name\", \"count\", color=\"dataset_name\", height=1000, title=\"<b>Distribution of Samples by Dataset</b>\")\nfig.update_layout(showlegend=False)\nfig.update_xaxes(tickangle=45, showgrid=True, title='')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-12T14:17:38.006213Z","iopub.execute_input":"2023-09-12T14:17:38.006641Z","iopub.status.idle":"2023-09-12T14:17:39.010207Z","shell.execute_reply.started":"2023-09-12T14:17:38.006609Z","shell.execute_reply":"2023-09-12T14:17:39.009063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<center><font size=60px color=\"red\"><b>WIP</b><br>More work coming over the next week ... <br>releasing as a placeholder for now<br><b>WIP</b></font></center>","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}