{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11228175,"sourceType":"competition"},{"sourceId":10857821,"sourceType":"datasetVersion","datasetId":6734160},{"sourceId":10984389,"sourceType":"datasetVersion","datasetId":6836394},{"sourceId":11016970,"sourceType":"datasetVersion","datasetId":6859619},{"sourceId":11024457,"sourceType":"datasetVersion","datasetId":6865232}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"ext_label_df_I converted [UW Synthetic Dataset](https://www.kaggle.com/datasets/andrewfavor/uw-synthetic-rna-final) into the same format of this competition.\n\nYou can download the dataset from [here](https://www.kaggle.com/datasets/tomooinubushi/converted-uw-synthetic-rna-final)\n\nI am not sure it works for this competition. I appreciate if you share us your results.\n\nSee [related discussion](https://www.kaggle.com/competitions/stanford-rna-3d-folding/discussion/567959)","metadata":{}},{"cell_type":"code","source":"!pip install biopandas","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-13T23:37:51.152840Z","iopub.execute_input":"2025-03-13T23:37:51.153186Z","iopub.status.idle":"2025-03-13T23:37:58.515682Z","shell.execute_reply.started":"2025-03-13T23:37:51.153157Z","shell.execute_reply":"2025-03-13T23:37:58.514422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from biopandas.pdb import PandasPdb\nimport pandas as pd\nfrom typing import Optional\nfrom glob import glob\nfrom tqdm.notebook import tqdm\nimport os\nimport numpy as np\nimport re","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-13T23:37:58.517163Z","iopub.execute_input":"2025-03-13T23:37:58.517586Z","iopub.status.idle":"2025-03-13T23:37:59.108902Z","shell.execute_reply.started":"2025-03-13T23:37:58.517548Z","shell.execute_reply":"2025-03-13T23:37:59.107566Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# https://medium.com/@jgbrasier/working-with-pdb-files-in-python-7b538ee1b5e4\n\ndef read_pdb_to_dataframe(\n    pdb_path: Optional[str] = None,\n    model_index: int = 1,\n    ) -> pd.DataFrame:\n    \"\"\"\n    Read a PDB file, and return a Pandas DataFrame containing the atomic coordinates and metadata.\n\n    Args:\n        pdb_path (str, optional): Path to a local PDB file to read. Defaults to None.\n        model_index (int, optional): Index of the model to extract from the PDB file, in case\n            it contains multiple models. Defaults to 1.\n        parse_header (bool, optional): Whether to parse the PDB header and extract metadata.\n            Defaults to True.\n\n    Returns:\n        pd.DataFrame: A DataFrame containing the atomic coordinates and metadata, with one row\n            per atom\n    \"\"\"\n    atomic_df = PandasPdb().read_pdb(pdb_path)\n    header = None\n    atomic_df = atomic_df.get_model(model_index)\n    if len(atomic_df.df[\"ATOM\"]) == 0:\n        raise ValueError(f\"No model found for index: {model_index}\")\n\n    return pd.concat([atomic_df.df[\"ATOM\"], atomic_df.df[\"HETATM\"]])\n\ndef preprocess_pdb(pdb_path):\n    df = read_pdb_to_dataframe(pdb_path)\n    df = df[df.atom_name==\"C1'\"].sort_values('residue_number').reset_index(drop = True)\n    df = df[['residue_name', 'residue_number', 'x_coord', 'y_coord', 'z_coord']].copy()\n    df.columns = ['resname', 'resid', 'x_1', 'y_1', 'z_1']\n    df[['x_1', 'y_1', 'z_1']] =  df[['x_1', 'y_1', 'z_1']].astype(np.float32) \n    df['resid'] =  df['resid'].astype(np.int32)\n    assert len(df['resid'].unique())  == (df['resid'].max()-df['resid'].min())+1, 'resid is discontinuous.'\n    return df\n\n\ndef convert_pdb(pdb_paths):\n    ext_label_df = []\n    for pdb_path in tqdm(pdb_paths):\n        try:\n            pdb_id = os.path.basename(pdb_path).split('.', 1)[0]\n            ext_label_df_ = preprocess_pdb(pdb_path)\n            ext_label_df_['ID'] =  [f'{pdb_id}_{resid}' for resid in ext_label_df_['resid']]\n            ext_label_df_['pdb_id'] =  pdb_id\n            ext_label_df.append(ext_label_df_[['ID', 'resname', 'resid', 'x_1', 'y_1', 'z_1', 'pdb_id']])\n        except Exception as e:\n            print(f'Failed to read{pdb_path}')\n            print('Error:' + str(e))\n    ext_label_df = pd.concat(ext_label_df).reset_index(drop = True)\n    ext_sequence_df = ext_label_df[[\"pdb_id\", 'resname']].groupby(\"pdb_id\", as_index = False).apply(lambda x: ''.join(x['resname']), include_groups=False).reset_index(drop = True)\n    ext_sequence_df.columns = [\"target_id\", 'sequence']\n    ext_label_df = ext_label_df.drop(\"pdb_id\", axis = 1)\n    return ext_label_df, ext_sequence_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-13T23:50:25.681041Z","iopub.execute_input":"2025-03-13T23:50:25.681542Z","iopub.status.idle":"2025-03-13T23:50:25.693492Z","shell.execute_reply.started":"2025-03-13T23:50:25.681504Z","shell.execute_reply":"2025-03-13T23:50:25.692479Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_label_df = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_labels.csv\")\ndisplay(train_label_df.head())\ntest_sequence_df = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/test_sequences.csv\")\ndisplay(test_sequence_df.head())","metadata":{"trusted":true,"editable":false,"execution":{"iopub.status.busy":"2025-03-13T23:37:59.124468Z","iopub.execute_input":"2025-03-13T23:37:59.125194Z","iopub.status.idle":"2025-03-13T23:37:59.467568Z","shell.execute_reply.started":"2025-03-13T23:37:59.125149Z","shell.execute_reply":"2025-03-13T23:37:59.466560Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Check my conversion script\nI downloaded pdb file of 1rnk.pdb from [here](https://www.rcsb.org/structure/1rnk).\nI confirmed that my conversion script is correct","metadata":{}},{"cell_type":"code","source":"train_label_df[\"pdb_id\"] = train_label_df[\"ID\"].apply(lambda x: x.split(\"_\")[0]+'_'+x.split(\"_\")[1])\ncheck_df_train=train_label_df[train_label_df.pdb_id=='1RNK_A'].reset_index(drop=True)\ndisplay(check_df_train.head())\n\npdb_path='/kaggle/input/some-pdb-files-for-rna-competition/1rnk.pdb'\ncheck_df_mine=preprocess_pdb(pdb_path)\ndisplay(check_df_mine.head())\n\n#Check sequences are the same\nprint(all(check_df_train.resname == check_df_mine.resname))\n#Check coordinates are the same\nprint(np.abs(check_df_train[['x_1','y_1','z_1']].values-check_df_mine[['x_1','y_1','z_1']].values).sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-13T23:44:27.892989Z","iopub.execute_input":"2025-03-13T23:44:27.893370Z","iopub.status.idle":"2025-03-13T23:44:28.069042Z","shell.execute_reply.started":"2025-03-13T23:44:27.893339Z","shell.execute_reply":"2025-03-13T23:44:28.068036Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nalpha = 0.3\ncoord=[check_df_train[['x_1','y_1','z_1']].values, check_df_mine[['x_1','y_1','z_1']].values]\nCOLOR = ['red', 'blue', 'green', 'black', 'yellow', 'cyan', 'magenta']\nfig = plt.figure(figsize=(10, 10))\nax = fig.add_subplot(111, projection='3d')\n# ax.clear()\n\nfor j in range(len(coord)):\n    x, y, z = coord[j][:, 0], coord[j][:, 1], coord[j][:, 2]\n    ax.scatter(x, y, z, c=COLOR[j], s=30, alpha=alpha)\n    ax.plot(x, y, z, color=COLOR[j], linewidth=1, alpha=alpha, label=f'{j}')\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-13T23:42:41.738635Z","iopub.execute_input":"2025-03-13T23:42:41.739012Z","iopub.status.idle":"2025-03-13T23:42:41.985240Z","shell.execute_reply.started":"2025-03-13T23:42:41.738979Z","shell.execute_reply":"2025-03-13T23:42:41.984105Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Convert all data","metadata":{}},{"cell_type":"code","source":"ext_dir = '/kaggle/input/uw-synthetic-rna-final/compile_all'\npdb_paths = glob(f'{ext_dir}/*/*.pdb')\nprint(len(pdb_paths))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-13T23:38:00.106072Z","iopub.execute_input":"2025-03-13T23:38:00.106411Z","iopub.status.idle":"2025-03-13T23:38:07.895461Z","shell.execute_reply.started":"2025-03-13T23:38:00.106381Z","shell.execute_reply":"2025-03-13T23:38:07.893772Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ext_label_df, ext_sequence_df = convert_pdb(pdb_paths)\ndisplay(ext_label_df.head())\ndisplay(ext_sequence_df.head())\n\next_label_df.to_parquet(f'ext_labels.parquet')\next_sequence_df.to_parquet(f'ext_sequences.parquet')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-13T23:38:07.895988Z","iopub.status.idle":"2025-03-13T23:38:07.896320Z","shell.execute_reply":"2025-03-13T23:38:07.896174Z"}},"outputs":[],"execution_count":null}]}