{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":51294,"databundleVersionId":7331882,"sourceType":"competition"},{"sourceId":87793,"databundleVersionId":11228175,"sourceType":"competition"},{"sourceId":10857821,"sourceType":"datasetVersion","datasetId":6734160},{"sourceId":10984389,"sourceType":"datasetVersion","datasetId":6836394}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"I converted [Stanford Ribonanza RNA Folding Dataset](https://www.kaggle.com/competitions/stanford-ribonanza-rna-folding/data) to the same format of this competition.\n\nI am not sure it is useful for this competition. I appreciate if you share us your results.","metadata":{}},{"cell_type":"code","source":"!pip install biopandas","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"editable":false,"execution":{"iopub.status.busy":"2025-03-13T11:39:14.496020Z","iopub.execute_input":"2025-03-13T11:39:14.496517Z","iopub.status.idle":"2025-03-13T11:39:21.569025Z","shell.execute_reply.started":"2025-03-13T11:39:14.496474Z","shell.execute_reply":"2025-03-13T11:39:21.567731Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nimport pandas as pd\nfrom biopandas.pdb import PandasPdb\nfrom typing import Optional\nfrom glob import glob\nfrom tqdm.notebook import tqdm\nimport os\nimport numpy as np\nimport re","metadata":{"trusted":true,"editable":false,"execution":{"iopub.status.busy":"2025-03-13T11:39:21.570337Z","iopub.execute_input":"2025-03-13T11:39:21.570652Z","iopub.status.idle":"2025-03-13T11:39:22.155031Z","shell.execute_reply.started":"2025-03-13T11:39:21.570621Z","shell.execute_reply":"2025-03-13T11:39:22.153993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# https://medium.com/@jgbrasier/working-with-pdb-files-in-python-7b538ee1b5e4\n\ndef read_pdb_to_dataframe(\n    pdb_path: Optional[str] = None,\n    model_index: int = 1,\n    ) -> pd.DataFrame:\n    \"\"\"\n    Read a PDB file, and return a Pandas DataFrame containing the atomic coordinates and metadata.\n\n    Args:\n        pdb_path (str, optional): Path to a local PDB file to read. Defaults to None.\n        model_index (int, optional): Index of the model to extract from the PDB file, in case\n            it contains multiple models. Defaults to 1.\n        parse_header (bool, optional): Whether to parse the PDB header and extract metadata.\n            Defaults to True.\n\n    Returns:\n        pd.DataFrame: A DataFrame containing the atomic coordinates and metadata, with one row\n            per atom\n    \"\"\"\n    atomic_df = PandasPdb().read_pdb(pdb_path)\n    header = None\n    atomic_df = atomic_df.get_model(model_index)\n    if len(atomic_df.df[\"ATOM\"]) == 0:\n        raise ValueError(f\"No model found for index: {model_index}\")\n\n    return pd.concat([atomic_df.df[\"ATOM\"], atomic_df.df[\"HETATM\"]])\n\ndef preprocess_pdb(pdb_path):\n    df = read_pdb_to_dataframe(pdb_path)\n    df = df[df.atom_name==\"C1'\"].sort_values('residue_number').reset_index(drop = True)\n    df = df[['residue_name', 'residue_number', 'x_coord', 'y_coord', 'z_coord']].copy()\n    df.columns = ['resname', 'resid', 'x_1', 'y_1', 'z_1']\n    df[['x_1', 'y_1', 'z_1']] =  df[['x_1', 'y_1', 'z_1']].astype(np.float32) \n    df['resid'] =  df['resid'].astype(np.int32)\n    assert len(df['resid'].unique())  == (df['resid'].max()-df['resid'].min())+1\n    return df\n\n\ndef convert_pdb(pdb_paths):\n    ext_label_df = []\n    for pdb_path in tqdm(pdb_paths):\n        try:\n            # pdb_id = os.path.basename(pdb_path).split('.', 1)[0]\n            pdb_id = pdb_path.split('/')[-2]\n            ext_label_df_ = preprocess_pdb(pdb_path)\n            ext_label_df_['ID'] =  [f'{pdb_id}_{resid}' for resid in ext_label_df_['resid']]\n            ext_label_df_['pdb_id'] =  pdb_id\n            ext_label_df.append(ext_label_df_[['ID', 'resname', 'resid', 'x_1', 'y_1', 'z_1', 'pdb_id']])\n        except Exception as e:\n            print(f'Failed to read{pdb_path}')\n            print('Error:' + str(e))\n    ext_label_df = pd.concat(ext_label_df).reset_index(drop = True)\n    ext_sequence_df = ext_label_df[[\"pdb_id\", 'resname']].groupby(\"pdb_id\", as_index = False).apply(lambda x: ''.join(x['resname']), include_groups=False).reset_index(drop = True)\n    ext_sequence_df.columns = [\"target_id\", 'sequence']\n    ext_label_df = ext_label_df.drop(\"pdb_id\", axis = 1)\n    return ext_label_df, ext_sequence_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-13T11:39:22.156170Z","iopub.execute_input":"2025-03-13T11:39:22.156771Z","iopub.status.idle":"2025-03-13T11:39:22.168436Z","shell.execute_reply.started":"2025-03-13T11:39:22.156731Z","shell.execute_reply":"2025-03-13T11:39:22.167127Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ext_dir = '/kaggle/input/stanford-ribonanza-rna-folding/rhofold_pdbs/rhofold_pdbs'\npdb_paths = glob(f'{ext_dir}/*/*/*/*/*.pdb')\nprint(len(pdb_paths))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-13T11:39:22.170519Z","iopub.execute_input":"2025-03-13T11:39:22.170854Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_label_df = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_labels.csv\")\ndisplay(train_label_df.head())\ntest_sequence_df = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/test_sequences.csv\")\ndisplay(test_sequence_df.head())","metadata":{"trusted":true,"editable":false},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ext_label_df, ext_sequence_df = convert_pdb(pdb_paths[:5])\ndisplay(ext_label_df.head())\ndisplay(ext_sequence_df.head())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\next_label_df, ext_sequence_df = convert_pdb(pdb_paths)\next_label_df.to_parquet(f'ext_ribonanza_labels.parquet')\next_sequence_df.to_parquet(f'ext_ribonanza_sequences.parquet')","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}