{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11228175,"sourceType":"competition"}],"dockerImageVersionId":30749,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\n# <div align=\"center\"> Stanford RNA 3D Folding \n    \n\n<div align=\"center\"> <img src=\"https://www.kaggle.com/competitions/87793/images/header\" width=\"300\"></div><br>\n\n**From the competition material**: For each sequence in the test set, you can predict <ins>five structures</ins>. Your notebook should look for a file test_sequences.csv and output submission.csv. This file should contain x, y, z coordinates of the C1' atom in each residue across your predicted structures 1 to 5:\n\n\n**ID**,**resname**,**resid**,**x_1**,**y_1**,**z_1**,... **x_5**,**y_5**,**z_5** <br>\nR1107_1,G,1,-7.561,9.392,9.361,... -7.301,9.023,8.932<br>\nR1107_2,G,1,-8.02,11.014,14.606,... -7.953,10.02,12.127<br>\netc.<br>\n\n\n**Evaluation**: Submissions are scored using <ins>TM-score</ins> (\"template modeling\" score), which goes from 0.0 to 1.0 (higher is better):\n","metadata":{}},{"cell_type":"markdown","source":"# 📚 Libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n# 导入必要的库\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n","metadata":{"_uuid":"051d70d956493feee0c6d64651c6a088724dca2a","_execution_state":"idle","trusted":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2025-03-06T09:39:58.643946Z","iopub.execute_input":"2025-03-06T09:39:58.644315Z","iopub.status.idle":"2025-03-06T09:40:01.345708Z","shell.execute_reply.started":"2025-03-06T09:39:58.644285Z","shell.execute_reply":"2025-03-06T09:40:01.344545Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🔍 Exploratory Data Analysis (EDA)","metadata":{}},{"cell_type":"code","source":"# 读取训练序列数据\ntrain_sequences = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/train_sequences.csv')\n# 读取训练标签数据\ntrain_labels = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/train_labels.csv')\n# 读取验证序列数据\nvalidation_sequences = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/validation_sequences.csv')\n# 读取验证标签数据\nvalidation_labels = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/validation_labels.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:40:01.347506Z","iopub.execute_input":"2025-03-06T09:40:01.347971Z","iopub.status.idle":"2025-03-06T09:40:01.781066Z","shell.execute_reply.started":"2025-03-06T09:40:01.347942Z","shell.execute_reply":"2025-03-06T09:40:01.779779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sequences.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:40:01.782286Z","iopub.execute_input":"2025-03-06T09:40:01.782618Z","iopub.status.idle":"2025-03-06T09:40:01.790677Z","shell.execute_reply.started":"2025-03-06T09:40:01.782590Z","shell.execute_reply":"2025-03-06T09:40:01.789343Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### train_sequences.csv","metadata":{}},{"cell_type":"code","source":"# 打印训练序列的行数和列数\nprint(f'train_sequences has {train_sequences.shape[0]} rows and {train_sequences.shape[1]} columns')\n# 打印训练序列中缺失值的总数\nprint(f'train_sequences has {train_sequences.isna().sum().sum()} NAs.')\n# 快速查看数据的前3行\ntrain_sequences.head(12)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:40:01.793101Z","iopub.execute_input":"2025-03-06T09:40:01.793423Z","iopub.status.idle":"2025-03-06T09:40:01.822992Z","shell.execute_reply.started":"2025-03-06T09:40:01.793396Z","shell.execute_reply":"2025-03-06T09:40:01.821979Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# sequence column...specifically the length of the sequences.","metadata":{}},{"cell_type":"code","source":"# 设置画布的大小\nplt.figure(figsize=(15, 8))\n# 计算序列长度并创建新的 DataFrame\ntrain_sequences['length'] = train_sequences['sequence'].str.len()\n# 绘制箱线图\nsns.boxplot(x='length', data=train_sequences)\nplt.xlabel(\"Sequence length\")\nplt.ylabel(\"Sequence\")\nplt.title(\"Boxplot of sequence length values (with outliers)\")\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:40:01.824319Z","iopub.execute_input":"2025-03-06T09:40:01.824643Z","iopub.status.idle":"2025-03-06T09:40:02.032961Z","shell.execute_reply.started":"2025-03-06T09:40:01.824617Z","shell.execute_reply":"2025-03-06T09:40:02.031859Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 设置画布的大小\nplt.figure(figsize=(15, 8))\n\n# 计算序列长度并创建新的 DataFrame\ntrain_sequences['length'] = train_sequences['sequence'].str.len()\n\n# 绘制箱线图，不显示异常值\nsns.boxplot(x='length', data=train_sequences, showfliers=False)\nplt.xlabel(\"Sequence length\")\nplt.ylabel(\"Sequence\")\nplt.title(\"Boxplot of sequence length values (without outliers)\")\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:40:02.034352Z","iopub.execute_input":"2025-03-06T09:40:02.034682Z","iopub.status.idle":"2025-03-06T09:40:02.212125Z","shell.execute_reply.started":"2025-03-06T09:40:02.034656Z","shell.execute_reply":"2025-03-06T09:40:02.210686Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 看 temporary_cutoff 列，看看已发布序列的长度是如何随着时间的推移而演变的","metadata":{}},{"cell_type":"code","source":"# 创建新的列 'cutoff_year' 和 'sequence_length'\ntrain_sequences['cutoff_year'] = pd.to_datetime(train_sequences['temporal_cutoff']).dt.year\ntrain_sequences['sequence_length'] = train_sequences['sequence'].str.len()\n\n# 计算每年平均序列长度\navg_sequence_length = (\n    train_sequences\n    .groupby('cutoff_year')\n    .agg(avg_sequence_length=('sequence_length', 'mean'))\n    .reset_index()\n)\n\n# 设置画布的大小\nplt.figure(figsize=(15, 8))\n\n# 绘制实际序列长度和多项式平滑线\nsns.regplot(x='cutoff_year', y='avg_sequence_length', data=avg_sequence_length, \n            order=3, ci=None, label='Smoothed Line', color='blue')\nplt.plot(avg_sequence_length['cutoff_year'], avg_sequence_length['avg_sequence_length'], \n         color='black', label='Actual Sequence Length')\n\n# 添加标签和标题\nplt.xlabel(\"Temporal Cutoff Year\")\nplt.ylabel(\"Average Sequence Length discovered\")\nplt.title(\"Average length of published sequence over time\")\nplt.legend()\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:40:02.213602Z","iopub.execute_input":"2025-03-06T09:40:02.213981Z","iopub.status.idle":"2025-03-06T09:40:02.579389Z","shell.execute_reply.started":"2025-03-06T09:40:02.213954Z","shell.execute_reply":"2025-03-06T09:40:02.577970Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Focussing on the target_id...from the competition material:\n>\"In train_sequences.csv, this is formatted as pdb_id_chain_id, where pdb_id is the id of the entry in the Protein Data Bank and chain_id is the chain id of the monomer in the pdb file\n>### 关注来自竞赛材料的 target_id...\n>“在 train_sequences.csv 中，其格式为 pdb_id_chain_id，其中 pdb_id 是蛋白质数据库中条目的 id，chain_id 是 pdb 文件中单体的链 id","metadata":{}},{"cell_type":"markdown","source":"#### Let's first look at how many unique pdb_id's and chain_id's there are in the train_sequences","metadata":{"execution":{"iopub.status.busy":"2025-03-02T22:06:18.770555Z","iopub.execute_input":"2025-03-02T22:06:18.772152Z","iopub.status.idle":"2025-03-02T22:06:18.78293Z","shell.execute_reply":"2025-03-02T22:06:18.781167Z"}}},{"cell_type":"code","source":"# 导入必要的库\nimport pandas as pd\n\n# 假设 train_sequences 数据框中包含 'target_id' 列\n# 提取 pdb_id 和 chain_id\ntrain_sequences[['pdb_id', 'chain_id']] = train_sequences['target_id'].str.split('_', expand=True)\n\n# 计算唯一的 pdb_id 数量\npdb_id_count = train_sequences['pdb_id'].nunique()\n\n# 计算唯一的 chain_id 数量\nchain_id_count = train_sequences['chain_id'].nunique()\n\n# 打印结果\nprint(f\"There are {pdb_id_count} unique pdb_id's in the dataset.\")\nprint(f\"There are {chain_id_count} unique chain_id's in the dataset.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:40:02.580766Z","iopub.execute_input":"2025-03-06T09:40:02.581083Z","iopub.status.idle":"2025-03-06T09:40:02.591687Z","shell.execute_reply.started":"2025-03-06T09:40:02.581056Z","shell.execute_reply":"2025-03-06T09:40:02.590660Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Now let's look at which pdb_id's are repeated in the train_sequences","metadata":{}},{"cell_type":"code","source":"# 导入必要的库\nimport pandas as pd\n\n# 假设 train_sequences 数据框中包含 'target_id' 列\n# 提取 pdb_id 和 chain_id\ntrain_sequences[['pdb_id', 'chain_id']] = train_sequences['target_id'].str.split('_', expand=True)\n\n# 计算每个 pdb_id 的总数，并筛选出重复的 pdb_id\nrepeated_pdb_ids = (\n    train_sequences\n    .groupby('pdb_id')\n    .size()\n    .reset_index(name='total')\n    .query('total > 1')\n    .sort_values(by='total', ascending=False)\n)\n\n# 获取前 10 个重复的 pdb_id\ntop_repeated_pdb_ids = repeated_pdb_ids.head(10)\n\n# 打印结果\nprint(\"In total, there are\", repeated_pdb_ids.shape[0], \"pdb_id's that are repeated in train_sequences.\")\nprint(\"Top 10 most repeated pdb_id's:\")\ntop_repeated_pdb_ids# In total, there are 60 pdb_id's that are repeated in train_sequences, with the top 10 most repeated shown below","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:40:02.593343Z","iopub.execute_input":"2025-03-06T09:40:02.593794Z","iopub.status.idle":"2025-03-06T09:40:02.623779Z","shell.execute_reply.started":"2025-03-06T09:40:02.593757Z","shell.execute_reply":"2025-03-06T09:40:02.622385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sequences[train_sequences['pdb_id']=='4V5Z']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:40:02.629468Z","iopub.execute_input":"2025-03-06T09:40:02.629888Z","iopub.status.idle":"2025-03-06T09:40:02.650859Z","shell.execute_reply.started":"2025-03-06T09:40:02.629858Z","shell.execute_reply":"2025-03-06T09:40:02.649390Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### train_labels.csv","metadata":{}},{"cell_type":"code","source":"# 假设 train_labels 是一个 Pandas DataFrame\n# 打印行数和列数\nprint(f'train_labels has {train_labels.shape[0]} rows and {train_labels.shape[1]} columns.')\n# 打印缺失值的总数\nprint(f'train_labels has {train_labels.isna().sum().sum()} NAs.')\n# 快速查看数据\ntrain_labels.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:40:02.652476Z","iopub.execute_input":"2025-03-06T09:40:02.652842Z","iopub.status.idle":"2025-03-06T09:40:02.684486Z","shell.execute_reply.started":"2025-03-06T09:40:02.652815Z","shell.execute_reply":"2025-03-06T09:40:02.683029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_labels.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:40:12.304668Z","iopub.execute_input":"2025-03-06T09:40:12.305059Z","iopub.status.idle":"2025-03-06T09:40:12.312089Z","shell.execute_reply.started":"2025-03-06T09:40:12.305029Z","shell.execute_reply":"2025-03-06T09:40:12.310909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 查找包含缺失值的行并显示前 3 行\nna_rows = train_labels[train_labels.isna().any(axis=1)]\n# 打印前 3 行\nna_rows.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:40:02.686401Z","iopub.execute_input":"2025-03-06T09:40:02.686892Z","iopub.status.idle":"2025-03-06T09:40:02.719011Z","shell.execute_reply.started":"2025-03-06T09:40:02.686851Z","shell.execute_reply":"2025-03-06T09:40:02.717937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"na_rows.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:40:02.720280Z","iopub.execute_input":"2025-03-06T09:40:02.720605Z","iopub.status.idle":"2025-03-06T09:40:02.726836Z","shell.execute_reply.started":"2025-03-06T09:40:02.720578Z","shell.execute_reply":"2025-03-06T09:40:02.725688Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Let's create a boxplot of the coordinate values for x, y, & z ","metadata":{}},{"cell_type":"code","source":"# 选择 x_1, y_1, z_1 列并转换为长格式\ntrain_labels_long = train_labels[['x_1', 'y_1', 'z_1']].melt(var_name='coordinate', value_name='value')\n# 设置画布的大小\nplt.figure(figsize=(15, 8))\n# 绘制箱线图\nsns.boxplot(x='value', y='coordinate', data=train_labels_long, hue='coordinate', palette='YlGnBu', dodge=False)\n# 添加标签和标题\nplt.xlabel(\"Coordinate Values\")\nplt.ylabel(\"Coordinates\")\nplt.title(\"Boxplot of coordinate values in train_labels\")\nplt.legend(title='Coordinate', loc='upper right')\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:41:19.018956Z","iopub.execute_input":"2025-03-06T09:41:19.019346Z","iopub.status.idle":"2025-03-06T09:41:19.678690Z","shell.execute_reply.started":"2025-03-06T09:41:19.019315Z","shell.execute_reply":"2025-03-06T09:41:19.677578Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Resname occurrences in train_labels","metadata":{}},{"cell_type":"code","source":"# 计算每种 resname 的出现次数\nresname_counts = train_labels.groupby('resname').size().reset_index(name='total')\n# 设置画布的大小\nplt.figure(figsize=(15, 8))\n# 绘制柱状图\nsns.barplot(x='resname', y='total', data=resname_counts, palette='Blues')\n# 添加柱子上的计数标签\nfor index, row in resname_counts.iterrows():\n    plt.text(index, row['total'], row['total'], ha='center', va='bottom')\n# 设置标签和标题\nplt.xlabel(\"Resname\")\nplt.ylabel(\"Count\")\nplt.title(\"Barplot of resname occurrences in train_labels\")\n# 隐藏图例\nplt.legend().set_visible(False)\n# 使用简约主题\nsns.set_theme(style=\"whitegrid\")\n# 显示图形\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:49:04.396199Z","iopub.execute_input":"2025-03-06T09:49:04.396630Z","iopub.status.idle":"2025-03-06T09:49:04.703145Z","shell.execute_reply.started":"2025-03-06T09:49:04.396599Z","shell.execute_reply":"2025-03-06T09:49:04.701964Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### validation_sequences.csv","metadata":{}},{"cell_type":"code","source":"# 导入必要的库\nimport pandas as pd\n\n# 假设 validation_sequences 是一个 Pandas DataFrame\n# 打印行数和列数\nprint(f'validation_sequences has {validation_sequences.shape[0]} rows and {validation_sequences.shape[1]} columns.')\n\n# 打印缺失值的总数\nprint(f'validation_sequences has {validation_sequences.isna().sum().sum()} NAs.')\n\n# 快速查看数据\nvalidation_sequences.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:52:30.198564Z","iopub.execute_input":"2025-03-06T09:52:30.198937Z","iopub.status.idle":"2025-03-06T09:52:30.211998Z","shell.execute_reply.started":"2025-03-06T09:52:30.198909Z","shell.execute_reply":"2025-03-06T09:52:30.210858Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Let's recreate the boxplot of the sequence lengths we used in train_sequences, using outliers = FALSE","metadata":{}},{"cell_type":"code","source":"# 导入必要的库\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# 假设 validation_sequences 是一个 Pandas DataFrame\n# 计算序列长度并创建新列\nvalidation_sequences['length'] = validation_sequences['sequence'].str.len()\n\n# 设置画布的大小\nplt.figure(figsize=(15, 8))\n\n# 绘制箱线图\nsns.boxplot(x='length', data=validation_sequences, showfliers=False)\n\n# 设置标签和标题\nplt.xlabel(\"Sequence length\")\nplt.ylabel(\"Sequence\")\nplt.title(\"Boxplot of sequence length values (without outliers)\")\n\n# 使用简约主题\nsns.set_theme(style=\"whitegrid\")\n\n# 显示图形\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:53:13.572916Z","iopub.execute_input":"2025-03-06T09:53:13.573285Z","iopub.status.idle":"2025-03-06T09:53:13.865883Z","shell.execute_reply.started":"2025-03-06T09:53:13.573257Z","shell.execute_reply":"2025-03-06T09:53:13.864865Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 导入必要的库\nimport pandas as pd\n\n# 假设 train_sequences 和 validation_sequences 是 Pandas DataFrame\n# 计算 train_sequences 的中位数序列长度\ntrain_sequence_med = train_sequences['sequence'].str.len().median()\n\n# 计算 validation_sequences 的中位数序列长度\nvalid_sequence_med = validation_sequences['sequence'].str.len().median()\n\n# 打印结果\nprint(f'The train_sequences median sequence length is {train_sequence_med}.')\nprint(f'The validation_sequences median sequence length is {valid_sequence_med}.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:54:19.393992Z","iopub.execute_input":"2025-03-06T09:54:19.394336Z","iopub.status.idle":"2025-03-06T09:54:19.403608Z","shell.execute_reply.started":"2025-03-06T09:54:19.394311Z","shell.execute_reply":"2025-03-06T09:54:19.402388Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### validation_labels.csv","metadata":{}},{"cell_type":"code","source":"# 导入必要的库\nimport pandas as pd\n\n# 假设 validation_labels 是一个 Pandas DataFrame\n# 打印行数和列数\nprint(f'validation_labels has {validation_labels.shape[0]} rows and {validation_labels.shape[1]} columns.')\n\n# 打印缺失值的总数\nprint(f'validation_labels has {validation_labels.isna().sum().sum()} NAs.')\n\nprint(\"Note validation_labels has 123 columns; train_labels has 6.\")\n# 快速查看数据\nvalidation_labels.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:54:54.845309Z","iopub.execute_input":"2025-03-06T09:54:54.845741Z","iopub.status.idle":"2025-03-06T09:54:54.873625Z","shell.execute_reply.started":"2025-03-06T09:54:54.845708Z","shell.execute_reply":"2025-03-06T09:54:54.872491Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### The number of NAs is misleading; from a discussion post [here](https://www.kaggle.com/competitions/stanford-rna-3d-folding/discussion/565746), the -1e+18 should be interpreted as NA/NaN values.\n\n### We will recreate the resname boxplot from train_labels from earlier:","metadata":{}},{"cell_type":"code","source":"# 导入必要的库\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# 假设 validation_labels 是一个 Pandas DataFrame\n# 计算每种 resname 的出现次数\nresname_counts = validation_labels.groupby('resname').size().reset_index(name='total')\n\n# 设置画布的大小\nplt.figure(figsize=(15, 8))\n\n# 绘制柱状图\nsns.barplot(x='resname', y='total', data=resname_counts, palette='Blues', order=resname_counts.sort_values('total')['resname'])\n\n# 添加柱子上的计数标签\nfor index, row in resname_counts.iterrows():\n    plt.text(index, row['total'], row['total'], ha='center', va='bottom')\n\n# 设置标签和标题\nplt.xlabel(\"Resname\")\nplt.ylabel(\"Count\")\nplt.title(\"Barplot of resname occurrences in validation_labels\")\n\n# 隐藏图例\nplt.legend().set_visible(False)\n\n# 使用简约主题\nsns.set_theme(style=\"whitegrid\")\n\n# 显示图形\nplt.show()\n\n# 说明\nprint(\"Note there are no missing or 'X' resnames.\")\nprint(\"Also of note is in this dataset, 'U' occurs more frequently than 'A'. 'C' and 'G' maintain their ordering.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T10:07:26.227778Z","iopub.execute_input":"2025-03-06T10:07:26.228140Z","iopub.status.idle":"2025-03-06T10:07:26.534317Z","shell.execute_reply.started":"2025-03-06T10:07:26.228116Z","shell.execute_reply":"2025-03-06T10:07:26.533205Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Submission format","metadata":{}},{"cell_type":"code","source":"# 导入必要的库\nimport pandas as pd\n\n# 读取样本提交文件\nsample_submission = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/sample_submission.csv')\n\n# 打印行数和列数\nprint(f'There are {sample_submission.shape[0]} rows and {sample_submission.shape[1]} columns.')\n\n# 快速查看前 6 行数据\nprint(\"Note: It is required to submit 5 sets of coordinate plots (x,y,z), per ID, resname & resid from the target_id, which is 18 columns.\")\nprint(\"For rows in the submission, we need one row for each resname/resid for each target.\")\nprint(\"From the sample submission, we need 2515 rows.\")\nsample_submission.head(6)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T10:08:15.201531Z","iopub.execute_input":"2025-03-06T10:08:15.201899Z","iopub.status.idle":"2025-03-06T10:08:15.235889Z","shell.execute_reply.started":"2025-03-06T10:08:15.201866Z","shell.execute_reply":"2025-03-06T10:08:15.234899Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 模型训练\n","metadata":{}},{"cell_type":"markdown","source":"### 从 test_sequence 生成输出文件","metadata":{}},{"cell_type":"code","source":"# 导入必要的库\nimport pandas as pd\n\n# 读取测试序列文件\ntest_sequences = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/test_sequences.csv')\n\n# 打印行数和列数\nprint(f'There are {test_sequences.shape[0]} rows and {test_sequences.shape[1]} columns.')\n\n# 说明\nprint(\"Reading in the test sequences, we can see we have 12 rows and the same 5 columns found in train & validation sequences.\")\ntest_sequences.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T10:12:36.558270Z","iopub.execute_input":"2025-03-06T10:12:36.558707Z","iopub.status.idle":"2025-03-06T10:12:36.575349Z","shell.execute_reply.started":"2025-03-06T10:12:36.558616Z","shell.execute_reply":"2025-03-06T10:12:36.574119Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 导入必要的库\nimport pandas as pd\n\n# 读取测试序列文件\ntest_sequences = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/test_sequences.csv')\n\n# 计算所有序列长度的总和\ntotal_length = test_sequences['sequence'].str.len().sum()\n\nprint(f'Total length of sequences in test_sequences is {total_length}.')\n\n# 创建一个 DataFrame 来存储提交数据\nsubmission_df = pd.DataFrame(columns=['target_id', 'resname', 'resid', 'x', 'y', 'z'])  # 根据需要定义列\n\n# 说明\nprint(\"Now, we need to create the df to house our submission.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T10:14:53.151690Z","iopub.execute_input":"2025-03-06T10:14:53.152205Z","iopub.status.idle":"2025-03-06T10:14:53.166850Z","shell.execute_reply.started":"2025-03-06T10:14:53.152168Z","shell.execute_reply":"2025-03-06T10:14:53.165827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# This function will accept a test_sequence ID & sequence and return df in format of submission (one resid/resname per row in sequence)\nparse_target <- function(tmp_ID, tmp_sequence){\n    seq_length <- nchar(tmp_sequence)\n    tmp_df=data.frame(matrix(ncol = 3, nrow = seq_length)) \n    colnames(tmp_df) <- c('ID','resname','resid')\n    tmp_df$resname <- unlist(strsplit(tmp_sequence,split=''))\n    tmp_df$ID <- tmp_ID\n    tmp_df$resid <- seq(1:seq_length)\n\n    return(tmp_df)\n}\n# Create the test_clean df in the format of the submission\ntest_id_seq <- test_sequences %>% select(target_id, sequence)\ntest_clean <- data.frame(matrix(ncol = 3, nrow = 0)) \ncolnames(test_clean) <- c('ID','resname','resid')\n\n# For each target_id / sequence, apply the function and rbind to previous results\nfor(i in 1:nrow(test_id_seq)){\n    tmp_df <- as.data.frame(mapply(parse_target,test_id_seq$target_id[i],test_id_seq$sequence[i], SIMPLIFY=FALSE))\n    colnames(tmp_df) <- c('ID','resname','resid')\n    test_clean <- rbind(test_clean,tmp_df)\n}\n\nprint(paste('There are',nrow(test_clean),'rows in the test_clean df.'))\nprint(head(test_clean))\n\n# That works...now let's get back to the training data, add some features, & train a model.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:40:02.765143Z","iopub.status.idle":"2025-03-06T09:40:02.765651Z","shell.execute_reply.started":"2025-03-06T09:40:02.765385Z","shell.execute_reply":"2025-03-06T09:40:02.765406Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🎯 Model Training","metadata":{}},{"cell_type":"code","source":"# Function to create some features based off resname combinations & sequence length\nfeat_eng <- function(df){\n    df <- df %>% mutate(seq_length = nchar(sequence), \n               A_cnt = str_count(sequence,\"A\"), \n               C_cnt = str_count(sequence,\"C\"), \n               U_cnt = str_count(sequence,\"U\"), \n               G_cnt = str_count(sequence,\"G\"),                  \n               AC_cnt = str_count(sequence,\"AC\"), \n               AU_cnt = str_count(sequence,\"AU\"), \n               AG_cnt = str_count(sequence,\"AG\"),\n               CA_cnt = str_count(sequence,\"CA\"), \n               CU_cnt = str_count(sequence,\"CU\"), \n               CG_cnt = str_count(sequence,\"CG\"),\n               UA_cnt = str_count(sequence,\"UA\"), \n               UC_cnt = str_count(sequence,\"UC\"), \n               UG_cnt = str_count(sequence,\"UG\"),\n               GA_cnt = str_count(sequence,\"GA\"), \n               GC_cnt = str_count(sequence,\"GC\"), \n               GU_cnt = str_count(sequence,\"GU\"), \n               AA_cnt = str_count(sequence,\"AA\"), \n               CC_cnt = str_count(sequence,\"CC\"), \n               UU_cnt = str_count(sequence,\"UU\"), \n               GG_cnt = str_count(sequence,\"GG\"),\n               begin_seq = substr(sequence,1,1),\n               end_seq = substr(sequence,nchar(sequence),nchar(sequence))) %>% select(-c(sequence, temporal_cutoff, description, all_sequences))\n    return(df)\n}\ntrain_sequences <- feat_eng(train_sequences)\n\n# Create the target_id from ID to facilitate joining with the meta_train data we just created\ntrain_labels <- train_labels %>% group_by(ID) %>% mutate(target_id = paste(unlist(strsplit(ID,'_'))[1],unlist(strsplit(ID,'_'))[2],sep='_'))\n\n# Create the final df to train\ntrain_data_clean <- merge(train_labels, train_sequences, by=\"target_id\", all.x = TRUE)\n\n# For now, we will impute the missing x_1,y_1,z_1 values with the group average for that target_id, resname\ntrain_data_clean <- train_data_clean %>% group_by(target_id, resname) %>% mutate(x_1 = ifelse(is.na(x_1),mean(x_1, na.rm=TRUE),x_1), y_1 = ifelse(is.na(y_1),mean(y_1, na.rm=TRUE),y_1), z_1 = ifelse(is.na(z_1),mean(z_1, na.rm=TRUE),z_1)) %>% ungroup() %>% select(-target_id)\n\n# The above imputation for train_data_clean doesn't take care of all NAs...so for now we will remove them(~ 1979 rows)\ntrain_data_clean <- train_data_clean %>% na.exclude() \nprint(paste('There are ',sum(is.na(train_data_clean)),\"NA's in the dataset!\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:40:02.768687Z","iopub.status.idle":"2025-03-06T09:40:02.769060Z","shell.execute_reply.started":"2025-03-06T09:40:02.768892Z","shell.execute_reply":"2025-03-06T09:40:02.768908Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features <- train_data_clean %>% select(-c(x_1,y_1,z_1, ID)) %>% colnames()  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:40:02.770188Z","iopub.status.idle":"2025-03-06T09:40:02.770560Z","shell.execute_reply.started":"2025-03-06T09:40:02.770369Z","shell.execute_reply":"2025-03-06T09:40:02.770383Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### For now, our model approach will consist of 3 separate models for x,y,z values","metadata":{}},{"cell_type":"code","source":"# Train model for X\n\ntrain_hframe = train_data_clean %>%  select(-c(ID,y_1,z_1)) %>% as.h2o() # will need to separate out x_1,y_1,z_1 & ID for each iteration\ntrain_hframe$resname <- as.factor(train_hframe$resname)\ntrain_hframe$begin_seq <- as.factor(train_hframe$begin_seq)\ntrain_hframe$end_seq <- as.factor(train_hframe$end_seq)\n# split the data into training and validation sets:\nsplits <- h2o.splitFrame(data = train_hframe, ratio = 0.8, seed = 1)\ntrain <- splits[[1]]\nvalid <- splits[[2]]\nxgb_x <- h2o.xgboost(x = features,\n        y = 'x_1',\n        nfolds = 5,\n        seed = 1,\n        booster = 'gbtree',\n        ntrees=200,\n        max_depth=15,\n        keep_cross_validation_predictions = TRUE,\n        training_frame = train_hframe,\n        validation_frame = valid)\n\nprint(h2o.mse(xgb_x,train=TRUE,valid=TRUE))\n\n# Train model for Y\ntrain_hframe = train_data_clean %>% select(-c(ID,x_1,z_1)) %>% as.h2o() # will need to separate out x_1,y_1,z_1 & ID for each iteration\ntrain_hframe$resname <- as.factor(train_hframe$resname)\ntrain_hframe$begin_seq <- as.factor(train_hframe$begin_seq)\ntrain_hframe$end_seq <- as.factor(train_hframe$end_seq)\n# split the data into training and validation sets:\nsplits <- h2o.splitFrame(data = train_hframe, ratio = 0.8, seed = 1)\ntrain <- splits[[1]]\nvalid <- splits[[2]]\nxgb_y <- h2o.xgboost(x = features,\n        y = 'y_1',\n        nfolds = 5,\n        seed = 1,\n        booster = 'gbtree',\n        ntrees=200,\n        max_depth=15,\n        keep_cross_validation_predictions = TRUE,\n        training_frame = train_hframe,\n        validation_frame = valid)\n\nprint(h2o.mse(xgb_y,train=TRUE,valid=TRUE))\n\n# Train model for Z\ntrain_hframe = train_data_clean %>% select(-c(ID,y_1,x_1)) %>% as.h2o() # will need to separate out x_1,y_1,z_1 & ID for each iteration\ntrain_hframe$resname <- as.factor(train_hframe$resname)\ntrain_hframe$begin_seq <- as.factor(train_hframe$begin_seq)\ntrain_hframe$end_seq <- as.factor(train_hframe$end_seq)\n# split the data into training and validation sets:\nsplits <- h2o.splitFrame(data = train_hframe, ratio = 0.8, seed = 1)\ntrain <- splits[[1]]\nvalid <- splits[[2]]\nxgb_z <- h2o.xgboost(x = features,\n        y = 'z_1',\n        nfolds = 5,\n        seed = 1,\n        booster = 'gbtree',\n        ntrees=200,\n        max_depth=15,\n        keep_cross_validation_predictions = TRUE,\n        training_frame = train_hframe,\n        validation_frame = valid)\n\nprint(h2o.mse(xgb_z,train=TRUE,valid=TRUE))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:40:02.772233Z","iopub.status.idle":"2025-03-06T09:40:02.772594Z","shell.execute_reply.started":"2025-03-06T09:40:02.772397Z","shell.execute_reply":"2025-03-06T09:40:02.772409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create features on test sequence\ntest_sequences <- feat_eng(test_sequences)\n\n# Merge test_clean we created earlier with test_sequences\ntest_data_clean <- merge(test_clean, test_sequences, by.x=\"ID\",by.y=\"target_id\", all.x = TRUE)\ntest_data_clean$resname <- as.factor(test_data_clean$resname)\ntest_data_clean$begin_seq <- as.factor(test_data_clean$begin_seq)\ntest_data_clean$end_seq <- as.factor(test_data_clean$end_seq)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:40:02.773936Z","iopub.status.idle":"2025-03-06T09:40:02.774309Z","shell.execute_reply.started":"2025-03-06T09:40:02.774120Z","shell.execute_reply":"2025-03-06T09:40:02.774139Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preds_x = h2o.predict(xgb_x,newdata = as.h2o(test_data_clean %>% select(-ID)))\npreds_y = h2o.predict(xgb_y,newdata = as.h2o(test_data_clean %>% select(-ID)))\npreds_z = h2o.predict(xgb_z,newdata = as.h2o(test_data_clean %>% select(-ID)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:40:02.775723Z","iopub.status.idle":"2025-03-06T09:40:02.776053Z","shell.execute_reply.started":"2025-03-06T09:40:02.775900Z","shell.execute_reply":"2025-03-06T09:40:02.775914Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🤞 Submission","metadata":{}},{"cell_type":"markdown","source":"### For now, we will use the same predictions for all x,y,z combinations.  Future version will incorporate different methods for these predictions.","metadata":{}},{"cell_type":"code","source":"submission <- data.frame(test_data_clean$ID, \n                         test_data_clean$resname, \n                         test_data_clean$resid) \n\nsubmission <- cbind(submission, \n                         as.data.frame(preds_x),\n                         as.data.frame(preds_y),\n                         as.data.frame(preds_z),\n                    \n                         as.data.frame(preds_x),\n                         as.data.frame(preds_y),\n                         as.data.frame(preds_z),\n                    \n                         as.data.frame(preds_x),\n                         as.data.frame(preds_y),\n                         as.data.frame(preds_z),\n                    \n                         as.data.frame(preds_x),\n                         as.data.frame(preds_y),\n                         as.data.frame(preds_z),\n                    \n                         as.data.frame(preds_x),\n                         as.data.frame(preds_y),\n                         as.data.frame(preds_z))\ncolnames(submission) <- sample_submission %>% colnames()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:40:02.777382Z","iopub.status.idle":"2025-03-06T09:40:02.777871Z","shell.execute_reply.started":"2025-03-06T09:40:02.777681Z","shell.execute_reply":"2025-03-06T09:40:02.777699Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### My first two submissions in R errored at the scoring phase.  This was due to the ID columns not being in the same order as the sample_submission, so we will create a simple sort column based off the sample_submission, and apply that to our submission df.","metadata":{}},{"cell_type":"code","source":"# Update ID to match submission ID column\nsubmission <- submission %>% mutate(ID = paste(ID,resid,sep=\"_\"))\n\n# Create a sort_order column within the sample_submission so we can apply it to our submission\nsample_submission$sort_order <- seq(1:nrow(sample_submission))\nsubmission <- merge(submission,sample_submission %>% select(ID,sort_order), by='ID',all.x=TRUE) %>% arrange(sort_order) %>% select(-sort_order)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:40:02.779181Z","iopub.status.idle":"2025-03-06T09:40:02.779715Z","shell.execute_reply.started":"2025-03-06T09:40:02.779429Z","shell.execute_reply":"2025-03-06T09:40:02.779468Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission %>% head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:40:02.781443Z","iopub.status.idle":"2025-03-06T09:40:02.781854Z","shell.execute_reply.started":"2025-03-06T09:40:02.781686Z","shell.execute_reply":"2025-03-06T09:40:02.781701Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"write_csv(submission, 'submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T09:40:02.783176Z","iopub.status.idle":"2025-03-06T09:40:02.783590Z","shell.execute_reply.started":"2025-03-06T09:40:02.783369Z","shell.execute_reply":"2025-03-06T09:40:02.783384Z"}},"outputs":[],"execution_count":null}]}