{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport math\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\n\nimport time\nimport concurrent.futures\n\nfrom script_cfg import *\nfrom script_train import *\nfrom script_bpp import *","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","scrolled":true,"execution":{"iopub.status.busy":"2023-10-24T15:50:18.343656Z","iopub.execute_input":"2023-10-24T15:50:18.344837Z","iopub.status.idle":"2023-10-24T15:50:21.566067Z","shell.execute_reply.started":"2023-10-24T15:50:18.344794Z","shell.execute_reply":"2023-10-24T15:50:21.565167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <font color='blue'>Data Overview</font>","metadata":{}},{"cell_type":"markdown","source":"Train is made of 1.6 millions lines, with, for each:\n- Sequence and sequence_id\n- Experiment type: 2A3_MaP or DMS_MaP, equally distributed\n- Reactivity and reactivity error, for the considered experiment type, and for each base (1 column per base)\n- Reads, signal_to_noise and SN Filter\n- Dataset_name","metadata":{}},{"cell_type":"markdown","source":"## Train Data","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(CFG.TRAIN, dtype=CFG.DTYPES) # 70 seconds to load\n\ntrain_dms = train[ train.experiment_type == 'DMS_MaP' ]\ntrain_2a3 = train[ train.experiment_type == '2A3_MaP' ]\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-24T15:53:38.758829Z","iopub.execute_input":"2023-10-24T15:53:38.761067Z","iopub.status.idle":"2023-10-24T15:55:35.014337Z","shell.execute_reply.started":"2023-10-24T15:53:38.760978Z","shell.execute_reply":"2023-10-24T15:55:35.012820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Sequences\n\nSequences are a succession of 4 bases: 'A', 'U', 'C', 'G'. \n\nThe vast majority of sequences are 177 lenghts and more precisely 177: 1568354, 170: 30000, 115: 27290, 155: 13038, 206: 4998. \n\nIn the test dataset, the sequence will be longer, from 2xx to 457 lenght\n\nEach sequence has at least 2 lines, for 2A3 and DMS experiment. 1.8% of sequences ID has multiples lines for the same experiment","metadata":{}},{"cell_type":"code","source":"train['sequence_len'] = train.sequence.apply(lambda row: len(row))\ntrain[['sequence_len']].describe()","metadata":{"execution":{"iopub.status.busy":"2023-10-22T12:22:35.983099Z","iopub.execute_input":"2023-10-22T12:22:35.983849Z","iopub.status.idle":"2023-10-22T12:22:36.774029Z","shell.execute_reply.started":"2023-10-22T12:22:35.983806Z","shell.execute_reply":"2023-10-22T12:22:36.773306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Signal to Noise and Reads\n\nSF_filter = 1 corresponds to Signal to noise > 1 and reads > 100. Only SF = 1 will be used for final scoring.\n\nMore reads and better signal to noise on the DMS experiment than on the 2A3 experiment.\n\nTotal: SN_filter\n- False    1205763\n- True      437917 >> 26%\n\nDMS: SN_filter\n- False    594915\n- True     226925 >> 27%\n\n2A3: SN_filter\n- False    610848\n- True     210992 >> 25%","metadata":{}},{"cell_type":"code","source":"train.SN_filter.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-10-22T12:22:39.080241Z","iopub.execute_input":"2023-10-22T12:22:39.081194Z","iopub.status.idle":"2023-10-22T12:22:39.104618Z","shell.execute_reply.started":"2023-10-22T12:22:39.081160Z","shell.execute_reply":"2023-10-22T12:22:39.103530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[['reads', 'signal_to_noise']].describe()","metadata":{"execution":{"iopub.status.busy":"2023-10-22T12:22:39.903513Z","iopub.execute_input":"2023-10-22T12:22:39.903865Z","iopub.status.idle":"2023-10-22T12:22:40.191111Z","shell.execute_reply.started":"2023-10-22T12:22:39.903838Z","shell.execute_reply":"2023-10-22T12:22:40.190402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dms[['reads', 'signal_to_noise']].describe()","metadata":{"execution":{"iopub.status.busy":"2023-10-22T12:22:40.701329Z","iopub.execute_input":"2023-10-22T12:22:40.701982Z","iopub.status.idle":"2023-10-22T12:22:40.848705Z","shell.execute_reply.started":"2023-10-22T12:22:40.701937Z","shell.execute_reply":"2023-10-22T12:22:40.847570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_2a3[['reads', 'signal_to_noise']].describe()","metadata":{"execution":{"iopub.status.busy":"2023-10-22T12:22:41.352794Z","iopub.execute_input":"2023-10-22T12:22:41.353325Z","iopub.status.idle":"2023-10-22T12:22:41.498973Z","shell.execute_reply.started":"2023-10-22T12:22:41.353286Z","shell.execute_reply":"2023-10-22T12:22:41.498242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_reads_SN(train_dms.sample(n=10000, axis=0)) ","metadata":{"execution":{"iopub.status.busy":"2023-10-22T12:22:42.225275Z","iopub.execute_input":"2023-10-22T12:22:42.225637Z","iopub.status.idle":"2023-10-22T12:22:43.324639Z","shell.execute_reply.started":"2023-10-22T12:22:42.225611Z","shell.execute_reply":"2023-10-22T12:22:43.323711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_reads_SN(train_2a3.sample(n=10000, axis=0))","metadata":{"execution":{"iopub.status.busy":"2023-10-22T12:22:44.767961Z","iopub.execute_input":"2023-10-22T12:22:44.769168Z","iopub.status.idle":"2023-10-22T12:22:45.815743Z","shell.execute_reply.started":"2023-10-22T12:22:44.769129Z","shell.execute_reply":"2023-10-22T12:22:45.814926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Reactivity and reactivity error\n\nThere is a symetry between Reactivity and Reactivity Error. 2 * 134 = 278 reactivity columns are not empty (over 2*206). ~ 200 columns have values for almost all lines. ~ 80 columns have almost no values.\n\nEmpty values mean no result, either for technical reason (extremities), or because seq length too small\n\nMaP refers to Mutational Profiling experiments, which are used in the determination of the reactivity at a particular nucleotide location.\nTo add noise and help better generalization\n\nThere are significant outliers in the values, both for Reactivity and Reactivity Error. The 2A3 experiments seem to have more outliers than the DMS experimentn and less outliers are observed when SN filter = True","metadata":{}},{"cell_type":"code","source":"train[CFG.COLS_REACT].notna().sum().values","metadata":{"execution":{"iopub.status.busy":"2023-10-22T12:22:47.788307Z","iopub.execute_input":"2023-10-22T12:22:47.789273Z","iopub.status.idle":"2023-10-22T12:22:48.991822Z","shell.execute_reply.started":"2023-10-22T12:22:47.789238Z","shell.execute_reply":"2023-10-22T12:22:48.990879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[CFG.COLS_REACT[30:34] + CFG.COLS_REACT_ERR[30:34]].describe()","metadata":{"execution":{"iopub.status.busy":"2023-10-22T12:22:50.123179Z","iopub.execute_input":"2023-10-22T12:22:50.124173Z","iopub.status.idle":"2023-10-22T12:22:50.779836Z","shell.execute_reply.started":"2023-10-22T12:22:50.124139Z","shell.execute_reply":"2023-10-22T12:22:50.779104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot_dist(train.sample(10000), CFG.COLS_REACT[30:34], 'Reactivity distribution')\n# plot_dist(train.sample(10000), CFG.COLS_REACT_ERR[30:34], 'Reactivity Error distribution')\n\n# plot_dist(train_dms.sample(10000), CFG.COLS_REACT[38:44], 'DMS Reactivity distribution')\n# plot_dist(train_2a3.sample(10000), CFG.COLS_REACT[38:44], '2A3 Reactivity distribution')\n\n# plot_dist(train_dms.sample(10000), CFG.COLS_REACT_ERR[38:44], 'DMS Reactivity distribution')\n# plot_dist(train_2a3.sample(10000), CFG.COLS_REACT_ERR[38:44], '2A3 Reactivity distribution')\n\nplot_dist(train[ train.SN_filter == True ], CFG.COLS_REACT[38:44], 'Reactivity distribution - SN filter = True')\nplot_dist(train[ train.SN_filter == False ], CFG.COLS_REACT[38:44], 'Reactivity distribution - SN filter = False')","metadata":{"execution":{"iopub.status.busy":"2023-10-22T12:22:53.797287Z","iopub.execute_input":"2023-10-22T12:22:53.799657Z","iopub.status.idle":"2023-10-22T12:23:42.626952Z","shell.execute_reply.started":"2023-10-22T12:22:53.799621Z","shell.execute_reply":"2023-10-22T12:23:42.626101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_reactivity(train_dms[train_dms.SN_filter == True].sample(n=10, axis=0), err=False, title='DMS Reactivity - SN True')\nplot_reactivity(train_dms[train_dms.SN_filter == True].sample(n=10, axis=0), err=True, title='DMS Reactivity Error - SN True')","metadata":{"execution":{"iopub.status.busy":"2023-10-22T12:23:46.414750Z","iopub.execute_input":"2023-10-22T12:23:46.415144Z","iopub.status.idle":"2023-10-22T12:23:50.795352Z","shell.execute_reply.started":"2023-10-22T12:23:46.415113Z","shell.execute_reply":"2023-10-22T12:23:50.794347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_reactivity(train_2a3[train_2a3.SN_filter == True].sample(n=10, axis=0), err=False, title='2A3 Reactivity - SN True')\nplot_reactivity(train_2a3[train_2a3.SN_filter == True].sample(n=10, axis=0), err=True, title='2A3 Reactivity Error - SN True')","metadata":{"execution":{"iopub.status.busy":"2023-10-22T12:23:50.797548Z","iopub.execute_input":"2023-10-22T12:23:50.797949Z","iopub.status.idle":"2023-10-22T12:23:55.258024Z","shell.execute_reply.started":"2023-10-22T12:23:50.797914Z","shell.execute_reply":"2023-10-22T12:23:55.256810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Dataset\n\nData come from 46 different datasets, with the following distribution\n- DasLabBigLib_OneMil_OpenKnot_Round_2_train_DMS is the biggest dataset with 8% of the entries\n- 18 datasets (PK50, PK90 and SL5) are the smallest with 2729 entries each, which represent 3% of the entries","metadata":{}},{"cell_type":"code","source":"plot_hist_datasets_size(figsize=(9, 2))","metadata":{"execution":{"iopub.status.busy":"2023-10-22T12:27:07.293208Z","iopub.execute_input":"2023-10-22T12:27:07.293563Z","iopub.status.idle":"2023-10-22T12:27:08.119762Z","shell.execute_reply.started":"2023-10-22T12:27:07.293533Z","shell.execute_reply":"2023-10-22T12:27:08.118248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in list(TRAIN.DATASET.keys())[0:5]:\n    plot_dist(train[ (train.dataset_name == dataset) & (train.SN_filter == True) ], CFG.COLS_REACT[38:44], f\"Reactivity distribution - SN True - {dataset} - {TRAIN.DATASET[dataset]}\")","metadata":{"execution":{"iopub.status.busy":"2023-10-22T12:27:30.297760Z","iopub.execute_input":"2023-10-22T12:27:30.298396Z","iopub.status.idle":"2023-10-22T12:27:49.741223Z","shell.execute_reply.started":"2023-10-22T12:27:30.298364Z","shell.execute_reply":"2023-10-22T12:27:49.740051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in list(TRAIN.DATASET.keys())[0:5]:\n    plot_reactivity(train[ (train.dataset_name == dataset) & (train.SN_filter == True) ], err=False, title=f\"Reactivity - SN True - {dataset} - {TRAIN.DATASET[dataset]}\")","metadata":{"execution":{"iopub.status.busy":"2023-10-22T12:27:58.533037Z","iopub.execute_input":"2023-10-22T12:27:58.533986Z","iopub.status.idle":"2023-10-22T12:28:09.238380Z","shell.execute_reply.started":"2023-10-22T12:27:58.533951Z","shell.execute_reply":"2023-10-22T12:28:09.237222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Tests data\n\nThere are 1.3 M lines in test_sequences. There are 2 * 270 M predictions to perform. So, an avg of 201 base per sequence","metadata":{}},{"cell_type":"code","source":"submission_part, test_part = get_sample_test_part()","metadata":{"execution":{"iopub.status.busy":"2023-10-22T12:29:12.387320Z","iopub.execute_input":"2023-10-22T12:29:12.387685Z","iopub.status.idle":"2023-10-22T12:29:16.644573Z","shell.execute_reply.started":"2023-10-22T12:29:12.387657Z","shell.execute_reply":"2023-10-22T12:29:16.643435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_part.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-10-22T12:29:00.620540Z","iopub.execute_input":"2023-10-22T12:29:00.620955Z","iopub.status.idle":"2023-10-22T12:29:00.633126Z","shell.execute_reply.started":"2023-10-22T12:29:00.620923Z","shell.execute_reply":"2023-10-22T12:29:00.631765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_part.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-10-22T12:29:17.091508Z","iopub.execute_input":"2023-10-22T12:29:17.091918Z","iopub.status.idle":"2023-10-22T12:29:17.102159Z","shell.execute_reply.started":"2023-10-22T12:29:17.091884Z","shell.execute_reply":"2023-10-22T12:29:17.101263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### BPP FILES - Watson-Crick base pairs proba\n\nThe BPP contains 1 file per sequence id\n\nEach file lists position pairing with a non-zero Watson-Crick base pair probabilities by the LinearPartition-EternaFold package","metadata":{}},{"cell_type":"code","source":"try: wc_proba_stats_partial = pd.read_csv('/kaggle/working/wc_proba_stats_partial.csv') # Does not necessary correspond to train_partial\nexcept:\n    wc_proba_stats_partial = get_wc_proba_stats(train_partial, 1500, 2000) # Estimation is 1000s for 500 sequence_id\n    wc_proba_stats_partial.to_csv('/kaggle/working/wc_proba_stats_partial.csv')\n\ncols_ = ['pairs', 'pairs_pct', 'pairs_low_proba_pct', 'mean_proba', 'median_proba', 'pairs_unusual_pct_overpair']\nplot_wc_proba_stats(wc_proba_stats_partial, cols_)","metadata":{"execution":{"iopub.status.busy":"2023-10-22T12:29:28.387469Z","iopub.execute_input":"2023-10-22T12:29:28.387854Z","iopub.status.idle":"2023-10-22T12:29:31.195408Z","shell.execute_reply.started":"2023-10-22T12:29:28.387826Z","shell.execute_reply":"2023-10-22T12:29:31.194393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Sequences Librairy\n\nTotal number of sequences in the library (Train + Test): 2 150 401","metadata":{}},{"cell_type":"code","source":"from Bio import SeqIO\n\nfor k in CFG.SEQ_LIB_DICT.keys():\n    fasta = SeqIO.parse(open(CFG.SEQ_LIB_DICT[k]), 'fasta')\n    sequences = [seq for seq in fasta]\n    print(f\"--------------------{k} - {len(sequences) / 1000} k sequences\")\n    print(sequences[0].id , sequences[0].name , sequences[0].description , sequences[0].seq)\n    print('')","metadata":{"execution":{"iopub.status.busy":"2023-10-22T12:30:26.888295Z","iopub.execute_input":"2023-10-22T12:30:26.889225Z","iopub.status.idle":"2023-10-22T12:31:01.683370Z","shell.execute_reply.started":"2023-10-22T12:30:26.889187Z","shell.execute_reply":"2023-10-22T12:31:01.682218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Supplementary Pred","metadata":{}},{"cell_type":"code","source":"pd.read_csv(CFG.SUPP['PK90']).head(1)","metadata":{"execution":{"iopub.status.busy":"2023-10-22T12:31:21.466510Z","iopub.execute_input":"2023-10-22T12:31:21.467575Z","iopub.status.idle":"2023-10-22T12:31:21.660965Z","shell.execute_reply.started":"2023-10-22T12:31:21.467535Z","shell.execute_reply":"2023-10-22T12:31:21.660049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <font color='blue'>Arnie & EternalFold</font>\n\n- **Arnie**, a helpful utility library that simplifies interacting with various secondary structure prediction packages. Accessibility and Reactivity of Nucleotides for Information Extraction\n- **dot-bracket notation**: . is an unpaired base and () represent two paired bases\n- **draw_rna**, a Das Lab tool that let's us plot RNA structures in 2D\n\nSecondary structure predictor:\n- **Eternafold** is a leading prediction package. It simulates RNA secondary structure ensembles without pseudoknots or other tertiary structure features. It is also limited to Watson-Crick base pairs, even though other kinds of RNA interactions are known to form.\n-**ViennaRNA**: Provides various algorithms and utilities for predicting RNA secondary structures.\n-**Mfold**: Another popular tool used for RNA secondary structure prediction.\n-**RNAstructure**: A software package that predicts RNA secondary structure.\n-**NUPACK**: Focuses on the analysis and prediction of nucleic acid secondary structures.\n-**RNAfold**: Part of the ViennaRNA package, specifically used for RNA secondary structure prediction.\n\n**forgi** (stands for Finding of RNA Graph Information.) Provides various utilities for representing, manipulating, and visualizing RNA structures.\nIn forgi, RNA secondary structures are represented as **graphs**, making it easier to perform operations that would be complicated in sequence or matrix representations. These graphs can be straightforward, representing just the base pairs, or more complex, representing multi-loop or pseudoknot structures.\n\n**ViennaRNA** is a package that includes a collection of algorithms and utilities for RNA secondary structure prediction. Part of the package, RNAfold, is specifically used for RNA secondary structure prediction, and there are other tools for things like comparison and visualization of RNA structures.\n\nThis is installed via bioconda...\n**bioconda** A distribution of bioinformatics software for the Conda package manager. Bioconda makes it easy to install bioinformatics software and their dependencies, including many of the RNA secondary structure prediction packages mentioned above.","metadata":{}},{"cell_type":"markdown","source":"## Install and Import","metadata":{"execution":{"iopub.status.busy":"2023-09-29T13:47:06.771993Z","iopub.execute_input":"2023-09-29T13:47:06.772506Z","iopub.status.idle":"2023-09-29T13:47:06.782875Z","shell.execute_reply.started":"2023-09-29T13:47:06.772471Z","shell.execute_reply":"2023-09-29T13:47:06.781487Z"}}},{"cell_type":"code","source":"# Install Arnie\n!pip install -q arnie draw_rna\n\n# Install Eternafold\n!conda config --set auto_update_conda false\n!conda install -c bioconda eternafold --yes\n\n# Ordinarily, the EternaFold conda package will automatically set necessary environment variables, but Kaggle's conda install works a little differently. Let's set them manually here using %env.\n%env ETERNAFOLD_PATH=/opt/conda/bin/eternafold-bin\n%env ETERNAFOLD_PARAMETERS=/opt/conda/lib/eternafold-lib/parameters/EternaFoldParams.v1","metadata":{"execution":{"iopub.status.busy":"2023-10-24T15:50:21.568929Z","iopub.execute_input":"2023-10-24T15:50:21.569855Z","iopub.status.idle":"2023-10-24T15:53:30.572842Z","shell.execute_reply.started":"2023-10-24T15:50:21.569812Z","shell.execute_reply":"2023-10-24T15:53:30.571233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from arnie.mfe import mfe\nfrom arnie.bpps import bpps\nfrom draw_rna.ipynb_draw import draw_struct","metadata":{"execution":{"iopub.status.busy":"2023-10-24T15:53:30.575514Z","iopub.execute_input":"2023-10-24T15:53:30.575927Z","iopub.status.idle":"2023-10-24T15:53:30.597585Z","shell.execute_reply.started":"2023-10-24T15:53:30.575893Z","shell.execute_reply":"2023-10-24T15:53:30.596122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_bpps_pps(sequences):\n    bpps_ = [ bpps(sequence, package=\"eternafold\") for sequence in sequences ]\n    pps_ = [ np.sum(bpps_[i], axis=0) for i in range(len(bpps_)) ]\n    return bpps_, pps_","metadata":{"execution":{"iopub.status.busy":"2023-10-24T15:53:30.598916Z","iopub.execute_input":"2023-10-24T15:53:30.599273Z","iopub.status.idle":"2023-10-24T15:53:30.646178Z","shell.execute_reply.started":"2023-10-24T15:53:30.599232Z","shell.execute_reply":"2023-10-24T15:53:30.644620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sequences_ = train.sequence.values[2000:2010]\nbpps_, pps_ = get_bpps_pps(sequences_)\nplot_pairing_proba(bpps_, pps_, figsize=(20, 4))","metadata":{"execution":{"iopub.status.busy":"2023-10-24T15:55:35.016629Z","iopub.execute_input":"2023-10-24T15:55:35.016996Z","iopub.status.idle":"2023-10-24T15:55:40.976571Z","shell.execute_reply.started":"2023-10-24T15:55:35.016964Z","shell.execute_reply":"2023-10-24T15:55:40.975042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predict Structure\nfor sequence in train.sequence.values[2000:2010]:\n    bpp = bpps(sequence, package=\"eternafold\")\n    pp = np.sum(bpp, axis=0)\n    structure = mfe(sequence, package=\"eternafold\")\n#     reactivity = [val for val in train.iloc[0, 7: 7 + len(sequence)].values]\n    print(f\"Sequence len: {len(sequence)}\")\n    print(sequence)\n    print(structure)\n    draw_struct(sequence, structure, c = pp, cmap='plasma')","metadata":{"execution":{"iopub.status.busy":"2023-10-24T15:59:10.663340Z","iopub.execute_input":"2023-10-24T15:59:10.663818Z","iopub.status.idle":"2023-10-24T15:59:42.587575Z","shell.execute_reply.started":"2023-10-24T15:59:10.663778Z","shell.execute_reply":"2023-10-24T15:59:42.586271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bpp_eternafold_df = pd.DataFrame(bpps(sequence, package=\"eternafold\"))\nbpp_eternafold_df","metadata":{"execution":{"iopub.status.busy":"2023-10-24T16:03:38.780714Z","iopub.execute_input":"2023-10-24T16:03:38.781173Z","iopub.status.idle":"2023-10-24T16:03:38.941786Z","shell.execute_reply.started":"2023-10-24T16:03:38.781138Z","shell.execute_reply":"2023-10-24T16:03:38.940627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.array([val for val in train.iloc[0, 7: 7 + len(sequence)]])","metadata":{"execution":{"iopub.status.busy":"2023-10-24T16:04:15.880936Z","iopub.execute_input":"2023-10-24T16:04:15.881399Z","iopub.status.idle":"2023-10-24T16:04:15.896372Z","shell.execute_reply.started":"2023-10-24T16:04:15.881366Z","shell.execute_reply":"2023-10-24T16:04:15.894827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Retrieve Base Pairs Proba from data sources\nseq_id = train[train.sequence == sequence].sequence_id.values[0]\npath = get_bpp_path(seq_id)\nwc_proba_one_seq = pd.read_csv(path, header=None, sep=' ', names=['base_1', 'base_2', 'proba'])\nwc_proba_one_seq.sort_values(by='proba', ascending=False, inplace=True)\nwc_proba_one_seq","metadata":{"execution":{"iopub.status.busy":"2023-10-22T12:40:46.333487Z","iopub.execute_input":"2023-10-22T12:40:46.334204Z","iopub.status.idle":"2023-10-22T12:45:11.122363Z","shell.execute_reply.started":"2023-10-22T12:40:46.334157Z","shell.execute_reply":"2023-10-22T12:45:11.121146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bpp_given = pd.DataFrame(columns = range(len(sequence)), index = range(len(sequence)))\nfor i in range(len(sequence)):\n    for j in range(len(sequence)):\n        try: bpp_given.iloc[i, j] = wc_proba_one_seq[ (wc_proba_one_seq.base_1 == i + 1) & (wc_proba_one_seq.base_2 == j + 1) ].proba.values[0]\n        except: bpp_given.iloc[i, j] = 0.0","metadata":{"execution":{"iopub.status.busy":"2023-10-24T16:03:44.730317Z","iopub.execute_input":"2023-10-24T16:03:44.730754Z","iopub.status.idle":"2023-10-24T16:03:46.394196Z","shell.execute_reply.started":"2023-10-24T16:03:44.730720Z","shell.execute_reply":"2023-10-24T16:03:46.392720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <font color='blue'>Good Luck for the competition and keep me updated!</font>","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}