{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11553390,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom IPython.display import display\nimport itertools\nimport Bio\nprint(\"Biopython v\" + Bio.__version__)\nfrom Bio import Align","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T10:12:46.677276Z","iopub.execute_input":"2025-04-10T10:12:46.677652Z","iopub.status.idle":"2025-04-10T10:12:46.683125Z","shell.execute_reply.started":"2025-04-10T10:12:46.677625Z","shell.execute_reply":"2025-04-10T10:12:46.682156Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Understanding Data","metadata":{}},{"cell_type":"code","source":"# Load datasets:\ntrain_seq = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_sequences.csv\")\ntrain_lab = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_labels.csv\")\nval_seq = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/validation_sequences.csv\")\nval_lab = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/validation_labels.csv\")\ntest_seq =pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/test_sequences.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T10:12:46.684304Z","iopub.execute_input":"2025-04-10T10:12:46.684590Z","iopub.status.idle":"2025-04-10T10:12:46.975100Z","shell.execute_reply.started":"2025-04-10T10:12:46.684559Z","shell.execute_reply":"2025-04-10T10:12:46.974190Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Quick look at the data:\nfor data in [train_seq, train_lab, val_seq, val_lab, test_seq]:\n    display(data[:3])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T10:12:46.976741Z","iopub.execute_input":"2025-04-10T10:12:46.977097Z","iopub.status.idle":"2025-04-10T10:12:47.023016Z","shell.execute_reply.started":"2025-04-10T10:12:46.977064Z","shell.execute_reply":"2025-04-10T10:12:47.021899Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The `temporal_cutoff`, `description`, and `all_sequences` columns in the **train_**, **val_**, and **test_seq** datasets contain general information, but nothing about RNA structure, therefore, they can be safely removed.","metadata":{"execution":{"iopub.status.busy":"2025-04-09T09:37:16.688204Z","iopub.execute_input":"2025-04-09T09:37:16.688635Z","iopub.status.idle":"2025-04-09T09:37:16.697145Z","shell.execute_reply.started":"2025-04-09T09:37:16.688605Z","shell.execute_reply":"2025-04-09T09:37:16.695438Z"}}},{"cell_type":"code","source":"train_seq = train_seq.drop([\"temporal_cutoff\", \"description\", \"all_sequences\"], axis = 1)\nval_seq = val_seq.drop([\"temporal_cutoff\", \"description\", \"all_sequences\"], axis = 1)\ntest_seq = test_seq.drop([\"temporal_cutoff\", \"description\", \"all_sequences\"], axis = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T10:12:47.024478Z","iopub.execute_input":"2025-04-10T10:12:47.024906Z","iopub.status.idle":"2025-04-10T10:12:47.031502Z","shell.execute_reply.started":"2025-04-10T10:12:47.024871Z","shell.execute_reply":"2025-04-10T10:12:47.030611Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"In **train_**, **val_**, and **test_seq** datasets rename `target_id` column to `ID`, and from `ID` just leave `pdb_id`.","metadata":{}},{"cell_type":"code","source":"train_seq.rename({\"target_id\": \"ID\"}, axis=1, inplace=True)\nval_seq.rename({\"target_id\": \"ID\"}, axis=1, inplace=True)\ntest_seq.rename({\"target_id\": \"ID\"}, axis=1, inplace=True)\n\nfor df in [train_lab, val_lab]:\n    df['ID'] = df['ID'].str.extract('(.*)_[\\d]+$')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T10:12:47.032629Z","iopub.execute_input":"2025-04-10T10:12:47.032990Z","iopub.status.idle":"2025-04-10T10:12:47.346923Z","shell.execute_reply.started":"2025-04-10T10:12:47.032958Z","shell.execute_reply":"2025-04-10T10:12:47.346160Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for data in [train_seq, train_lab, val_seq, val_lab, test_seq]:\n    display(data[:3])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T10:12:47.347618Z","iopub.execute_input":"2025-04-10T10:12:47.347905Z","iopub.status.idle":"2025-04-10T10:12:47.389704Z","shell.execute_reply.started":"2025-04-10T10:12:47.347882Z","shell.execute_reply":"2025-04-10T10:12:47.388720Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Missing data","metadata":{}},{"cell_type":"code","source":"#cheching for 0 values in train_lab, val_lab datasets\nfor data in [train_lab, val_lab]:\n    display(data.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T10:12:47.392562Z","iopub.execute_input":"2025-04-10T10:12:47.392925Z","iopub.status.idle":"2025-04-10T10:12:47.423226Z","shell.execute_reply.started":"2025-04-10T10:12:47.392891Z","shell.execute_reply":"2025-04-10T10:12:47.422456Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The **train_lab** (training labels) dataset has 6145 rows with empty x_1, y_1, z_1 coordinates. Let's take a closer look on it:","metadata":{}},{"cell_type":"code","source":"#Checking for sequences in which the x_1, y_1, z_1 coordinates are completely missing: \nempty_ids = []\nfor ID in train_lab.ID.unique():\n    x_1 = train_lab[train_lab.ID == ID].x_1\n    if x_1.isnull().all():\n        empty_ids.append(ID)\n\nprint(f\"There are {len(empty_ids)} sequences without x_1, y_1, z_1 coordinates.\\n\")\nprint(empty_ids)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T10:12:47.424565Z","iopub.execute_input":"2025-04-10T10:12:47.424864Z","iopub.status.idle":"2025-04-10T10:12:57.661665Z","shell.execute_reply.started":"2025-04-10T10:12:47.424840Z","shell.execute_reply":"2025-04-10T10:12:57.660526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#empty sequences contain no information about x_1, y_1, z_1 coordinates => deleting them\ntrain_lab = train_lab[~train_lab.ID.isin(empty_ids)]\ntrain_seq = train_seq[~train_seq.ID.isin(empty_ids)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T10:12:57.662713Z","iopub.execute_input":"2025-04-10T10:12:57.663075Z","iopub.status.idle":"2025-04-10T10:12:57.694847Z","shell.execute_reply.started":"2025-04-10T10:12:57.663040Z","shell.execute_reply":"2025-04-10T10:12:57.693887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#The description mentions that train_sequences.csv contains other than other 'G', 'U', 'C', 'A' nucleotides.\ntrain_lab.resname.unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T10:12:57.695726Z","iopub.execute_input":"2025-04-10T10:12:57.696170Z","iopub.status.idle":"2025-04-10T10:12:57.708211Z","shell.execute_reply.started":"2025-04-10T10:12:57.696136Z","shell.execute_reply":"2025-04-10T10:12:57.707118Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Sequences with 'X' nucleotide:\ndisplay(train_seq[train_seq.sequence.str.contains('X')])\n# x_1, y_1, z_1 coordinates for 'X' nucleotide:\ndisplay(train_lab[train_lab.resname == 'X'])\n# ignore the error, there are NaN values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T10:12:57.709076Z","iopub.execute_input":"2025-04-10T10:12:57.709353Z","iopub.status.idle":"2025-04-10T10:12:57.739122Z","shell.execute_reply.started":"2025-04-10T10:12:57.709331Z","shell.execute_reply":"2025-04-10T10:12:57.738202Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# \"X\" is an unknown nucleotide and its x_1, y_1, z_1 coordinates are also unknown. \n# Moreover, \"X\" nucleotide is at the beginning (1st nucleotide) of the sequences => it's safe to delete and forget it :)\ntrain_lab = train_lab[train_lab.resname != 'X']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T10:12:57.740104Z","iopub.execute_input":"2025-04-10T10:12:57.740324Z","iopub.status.idle":"2025-04-10T10:12:57.766011Z","shell.execute_reply.started":"2025-04-10T10:12:57.740305Z","shell.execute_reply":"2025-04-10T10:12:57.765215Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Sequences with '-' nucleotide:\ndisplay(train_seq[train_seq.sequence.str.contains('-')])\n#x_1, y_1, z_1 coordinates for '-' nucleotide:\ndisplay(train_lab[train_lab.resname == '-'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T10:12:57.766593Z","iopub.execute_input":"2025-04-10T10:12:57.766851Z","iopub.status.idle":"2025-04-10T10:12:57.794789Z","shell.execute_reply.started":"2025-04-10T10:12:57.766829Z","shell.execute_reply":"2025-04-10T10:12:57.793873Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Let's go to NCBI and find those missing (-) nucleotides: <br />\n**6WB1_C** FASTA sequence: `GACTCTGGTAACTAGAGATCCCTCAGACCCTTTTAGTCAGTGTGGAAAATCTCTAGCAGTGGCGCCCGAACAGGGACTTGAAAGCGAAAGTAAAGCCAGAG` <br /> \nhttps://www.ncbi.nlm.nih.gov/nuccore/6WB1_C?report=fasta <br />\n**6Y0C_IN1** FASTA sequence: `AGTAGAAACAAGGGTATTTTTCTTTACTAGTCTACCCTGCTTTTGCT` <br />\nhttps://www.ncbi.nlm.nih.gov/nuccore/6Y0C_IN1?report=fasta <br />\n...and compare the sequences:","metadata":{}},{"cell_type":"code","source":"# old sequences from the train_sequences.csv\nold_6WB1_C = train_seq[train_seq.ID == \"6WB1_C\"].sequence.item()\nold_6Y0C_IN1 = train_seq[train_seq.ID == \"6Y0C_IN1\"].sequence.item()\n\n# FASTA sequences (NOTE: FASTA is a DNA file):\nncbi_6WB1_C = \"GACTCTGGTAACTAGAGATCCCTCAGACCCTTTTAGTCAGTGTGGAAAATCTCTAGCAGTGGCGCCCGAACAGGGACTTGAAAGCGAAAGTAAAGCCAGAG\"\nncbi_6Y0C_IN1 = \"AGTAGAAACAAGGGTATTTTTCTTTACTAGTCTACCCTGCTTTTGCT\"\n\n# \"Converting\" DNA to RNA (replacing 'T' to 'U') \nnew_6WB1_C = ncbi_6WB1_C.replace('T', 'U')\nnew_6Y0C_IN1 = ncbi_6Y0C_IN1.replace('T', 'U')\n\nprint(f\" old_6WB1_C: {old_6WB1_C}\\n new_6WB1_C: {new_6WB1_C}\\n \\n old_6Y0C_IN1: {old_6Y0C_IN1}\\n new_6Y0C_IN1: {new_6Y0C_IN1}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T10:12:57.795722Z","iopub.execute_input":"2025-04-10T10:12:57.796060Z","iopub.status.idle":"2025-04-10T10:12:57.803965Z","shell.execute_reply.started":"2025-04-10T10:12:57.796027Z","shell.execute_reply":"2025-04-10T10:12:57.802895Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 6WB1_C and 6Y0C_IN1 sequences aligment (comparing sequences from the train_sequences.csv and FASTA files)\naligner = Align.PairwiseAligner()\n\nalignments_6WB1_C = aligner.align(old_6WB1_C, new_6WB1_C)\nalignments_6Y0C_IN1 = aligner.align(old_6Y0C_IN1, new_6Y0C_IN1)\n\nfor alignment in alignments_6WB1_C:\n    print(\"6WB1_C:\\n\")\n    print(alignment)\n\nfor alignment in alignments_6Y0C_IN1:\n    print(\"6Y0C_IN1:\\n\")\n    print(alignment)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T10:12:57.804889Z","iopub.execute_input":"2025-04-10T10:12:57.805126Z","iopub.status.idle":"2025-04-10T10:12:57.819861Z","shell.execute_reply.started":"2025-04-10T10:12:57.805104Z","shell.execute_reply":"2025-04-10T10:12:57.818934Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"I guess there is an Easter egg, Easter is comming after all :)  <br /> \n'-' unknown nucleotides are not nucleotides at all! <br /> \nLet's see if there are some more Easter eggs!","metadata":{}},{"cell_type":"code","source":"#deleting '-' \"nucleotides\"\ntrain_lab = train_lab[train_lab.resname != '-']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T10:12:57.820841Z","iopub.execute_input":"2025-04-10T10:12:57.821171Z","iopub.status.idle":"2025-04-10T10:12:57.859097Z","shell.execute_reply.started":"2025-04-10T10:12:57.821138Z","shell.execute_reply":"2025-04-10T10:12:57.858092Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Working with the test dataset","metadata":{}},{"cell_type":"code","source":"#Data view:\ntest_seq","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T10:12:57.859996Z","iopub.execute_input":"2025-04-10T10:12:57.860253Z","iopub.status.idle":"2025-04-10T10:12:57.868662Z","shell.execute_reply.started":"2025-04-10T10:12:57.860231Z","shell.execute_reply":"2025-04-10T10:12:57.867503Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sequences = train_seq.sequence\nval_sequences = val_seq.sequence\ntest_sequences = test_seq.sequence\n\nprint (\"Comparing test sequences with train sequences:\")\nfor _, test_row in test_seq.iterrows():\n    for _, train_row in train_seq.iterrows():\n        if test_row.sequence == train_row.sequence:\n            print(f\"Test seq {test_row.ID} is train seq {train_row.ID}\")\n            \nprint() #adding an extra line to make it easier to read\n\nprint (\"Comparing test sequences with validation sequences:\")\nfor _, test_row in test_seq.iterrows():\n    for _, val_row in val_seq.iterrows():\n        if test_row.sequence == val_row.sequence:\n            print(f\"Test seq {test_row.ID} is val seq {val_row.ID}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T10:12:57.869624Z","iopub.execute_input":"2025-04-10T10:12:57.869972Z","iopub.status.idle":"2025-04-10T10:12:58.470795Z","shell.execute_reply.started":"2025-04-10T10:12:57.869939Z","shell.execute_reply":"2025-04-10T10:12:58.469888Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Besides the fact that all **test** sequences are present in the **validation** dataset, their IDs are also the same... <br />\nWell, then I can ~~be lazy~~  use less computer memory and power and create test labels using validation labels. <br />\n**R1189** is also **R1190** (dublicate sequence in validation and test datasets).","metadata":{}},{"cell_type":"code","source":"test_lab = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/validation_labels.csv\")\n#slicing till x_6 column\ntest_lab = test_lab.iloc[:,:18]\ntest_lab[:2]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T10:12:58.473318Z","iopub.execute_input":"2025-04-10T10:12:58.473575Z","iopub.status.idle":"2025-04-10T10:12:58.539212Z","shell.execute_reply.started":"2025-04-10T10:12:58.473553Z","shell.execute_reply":"2025-04-10T10:12:58.538189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_sub = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/sample_submission.csv\")\nsample_s = sample_sub.iloc[:,:3]\nsample_s[:2]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T10:12:58.540511Z","iopub.execute_input":"2025-04-10T10:12:58.540876Z","iopub.status.idle":"2025-04-10T10:12:58.557237Z","shell.execute_reply.started":"2025-04-10T10:12:58.540843Z","shell.execute_reply":"2025-04-10T10:12:58.556357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.merge(sample_s, test_lab, how=\"inner\", on = [\"ID\", \"resname\", \"resid\"])\nsubmission[:2]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T10:12:58.557985Z","iopub.execute_input":"2025-04-10T10:12:58.558255Z","iopub.status.idle":"2025-04-10T10:12:58.582584Z","shell.execute_reply.started":"2025-04-10T10:12:58.558234Z","shell.execute_reply":"2025-04-10T10:12:58.581665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#randomly adding some data \nsize = submission.ID.count()\nfor coord, i in itertools.product(\"xyz\", \"2345\"):\n    submission[f\"{coord}_{i}\"] = submission[f\"{coord}_1\"]\n\nfor coord, i in itertools.product(\"xyz\", \"12345\"):\n    submission[f\"{coord}_{1}\"] += np.random.normal(scale=0.000001, size=size)\n    \nsubmission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T10:35:13.254442Z","iopub.execute_input":"2025-04-10T10:35:13.254933Z","iopub.status.idle":"2025-04-10T10:35:13.284465Z","shell.execute_reply.started":"2025-04-10T10:35:13.254889Z","shell.execute_reply":"2025-04-10T10:35:13.283471Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index=False)\nprint(\"submission.csv file is created\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T10:35:20.374175Z","iopub.execute_input":"2025-04-10T10:35:20.374509Z","iopub.status.idle":"2025-04-10T10:35:20.443894Z","shell.execute_reply.started":"2025-04-10T10:35:20.374483Z","shell.execute_reply":"2025-04-10T10:35:20.442725Z"}},"outputs":[],"execution_count":null}]}