{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":87793,"databundleVersionId":12276181,"sourceType":"competition"},{"sourceId":11118830,"sourceType":"datasetVersion","datasetId":6933267},{"sourceId":11413730,"sourceType":"datasetVersion","datasetId":7148547},{"sourceId":11421128,"sourceType":"datasetVersion","datasetId":7152794},{"sourceId":11428421,"sourceType":"datasetVersion","datasetId":7157768},{"sourceId":11593585,"sourceType":"datasetVersion","datasetId":7270084},{"sourceId":11989607,"sourceType":"datasetVersion","datasetId":7541175},{"sourceId":11990395,"sourceType":"datasetVersion","datasetId":7541717},{"sourceId":11997691,"sourceType":"datasetVersion","datasetId":7546928},{"sourceId":11999509,"sourceType":"datasetVersion","datasetId":7548267}],"dockerImageVersionId":30919,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import sys\nimport os\nimport shutil\nfrom pathlib import Path\nimport subprocess\nimport glob\nimport gc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T08:51:10.786494Z","iopub.execute_input":"2025-05-28T08:51:10.786693Z","iopub.status.idle":"2025-05-28T08:51:10.790456Z","shell.execute_reply.started":"2025-05-28T08:51:10.786675Z","shell.execute_reply":"2025-05-28T08:51:10.789604Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val = False\nif val:\n    !pip install --no-deps protenix\n    !pip install biopython\n    !pip install ml-collections\n    !pip install biotite==1.0.1\n    !pip install rdkit\n\nsrc = \"/kaggle/input/arena-with-permission/Arena\"\ndst = \"/kaggle/working/Arena\"\n\nshutil.copytree(src, dst)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T08:51:10.791266Z","iopub.execute_input":"2025-05-28T08:51:10.791554Z","iopub.status.idle":"2025-05-28T08:51:32.217773Z","shell.execute_reply.started":"2025-05-28T08:51:10.791527Z","shell.execute_reply":"2025-05-28T08:51:32.216923Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !pip install --no-index /kaggle/input/boltz-dependencies/*whl --no-deps\n# !pip install --no-index /kaggle/input/fairscale-0413/*whl --no-deps","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T14:04:21.77767Z","iopub.execute_input":"2025-05-22T14:04:21.778042Z","iopub.status.idle":"2025-05-22T14:04:25.185102Z","shell.execute_reply.started":"2025-05-22T14:04:21.777995Z","shell.execute_reply":"2025-05-22T14:04:25.184174Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"os.chmod('/kaggle/working/Arena/Arena', 0o755)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T08:51:58.955496Z","iopub.execute_input":"2025-05-28T08:51:58.955777Z","iopub.status.idle":"2025-05-28T08:51:58.959832Z","shell.execute_reply.started":"2025-05-28T08:51:58.955755Z","shell.execute_reply":"2025-05-28T08:51:58.959003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport os, sys\nimport pandas as pd\nimport numpy as np\nfrom Bio import SeqIO, AlignIO\nfrom Bio.Seq import Seq\nfrom Bio.SeqRecord import SeqRecord\nfrom Bio.PDB import Atom, Model, Chain, Residue, Structure, PDBParser\nimport json\nimport time\n\nprint('torch',torch.__version__)\nprint('torch.cuda',torch.version.cuda)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T08:52:00.417889Z","iopub.execute_input":"2025-05-28T08:52:00.418249Z","iopub.status.idle":"2025-05-28T08:52:03.861653Z","shell.execute_reply.started":"2025-05-28T08:52:00.418221Z","shell.execute_reply":"2025-05-28T08:52:03.860942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"local = False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T08:52:03.862554Z","iopub.execute_input":"2025-05-28T08:52:03.862989Z","iopub.status.idle":"2025-05-28T08:52:03.866266Z","shell.execute_reply.started":"2025-05-28T08:52:03.862964Z","shell.execute_reply":"2025-05-28T08:52:03.865497Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"PROTENIX_DATA_ROOT_DIR = '/kaggle/input/protenix-checkpoints'\n# LONG_MODEL_PATH = f'{PROTENIX_DATA_ROOT_DIR}/model_v0.2.0.pt'\nSHORT_MODEL_PATH = f'{PROTENIX_DATA_ROOT_DIR}/model_v0.2.0.pt'\n# LONG_MODEL_PATH = '/kaggle/input/rna16999/16999_ema_0.995.pt'\n# SHORT_MODEL_PATH = '/kaggle/input/rna16999/16999_ema_0.995.pt'\n# LONG_MODEL_PATH = '/kaggle/input/3798-all-msa/3798_ema_0.995_all.pt'\n# SHORT_MODEL_PATH = '/kaggle/input/3798-all-msa/3798_ema_0.995_all.pt'\nSEQUENCE_LENGTH_THRESHOLD = 1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T08:52:03.867851Z","iopub.execute_input":"2025-05-28T08:52:03.868133Z","iopub.status.idle":"2025-05-28T08:52:03.884157Z","shell.execute_reply.started":"2025-05-28T08:52:03.868085Z","shell.execute_reply":"2025-05-28T08:52:03.883442Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"! mkdir /af3-dev \n! ln -s /kaggle/input/protenix-checkpoints /af3-dev/release_data\n! ls /af3-dev/release_data/","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T08:52:03.885053Z","iopub.execute_input":"2025-05-28T08:52:03.885369Z","iopub.status.idle":"2025-05-28T08:52:04.263664Z","shell.execute_reply.started":"2025-05-28T08:52:03.885341Z","shell.execute_reply":"2025-05-28T08:52:04.262857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if not local:\n    KAGGLE_INPUT = '/kaggle/input/drfold2-12/DRfold2_optimized'\n    WORKING_DIR = '/kaggle/working'\n    DRFOLD_DIR = WORKING_DIR\n    \n\n    # if not os.path.exists(DRFOLD_DIR):\n    #     print(f\"Copying DRfold2 from {KAGGLE_INPUT} to {DRFOLD_DIR}\")\n    #     shutil.copytree(KAGGLE_INPUT, DRFOLD_DIR)\n\n\n    os.chdir(DRFOLD_DIR)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T08:52:04.264663Z","iopub.execute_input":"2025-05-28T08:52:04.264902Z","iopub.status.idle":"2025-05-28T08:52:04.268929Z","shell.execute_reply.started":"2025-05-28T08:52:04.264873Z","shell.execute_reply":"2025-05-28T08:52:04.268143Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from os import listdir\nfrom os.path import isfile, join\nfrom copy import deepcopy\n\n# sto generation\nmsa_dir = \"/kaggle/input/stanford-rna-3d-folding/MSA/\"\nsto_dir = \"/kaggle/working/MSA-STO/\"\n# os.makedirs(sto_dir, exist_ok=True)\n\nsrc = \"/kaggle/input/sto-files/MSA-STO\"\ndst = \"/kaggle/working/MSA-STO\"\n\nshutil.copytree(src, dst)\n\nmsa_files = [f for f in listdir(msa_dir) if isfile(join(msa_dir, f))]\nfor file_name in msa_files:\n    target_name = file_name[:-6]\n\n    if isfile(f\"{sto_dir}/{target_name}.sto\"):\n        continue\n\n    # Read the FASTA MSA\n    alignment = AlignIO.read(f\"{msa_dir}{file_name}\", \"fasta\")\n    # Write to Stockholm format\n    AlignIO.write(alignment, f\"{sto_dir}/{target_name}.sto\", \"stockholm\")\n\nsto_cut_bound = 0\nfor file_name in msa_files:\n    target_name = file_name[:-6]\n    sto_file_path = f\"{sto_dir}/{target_name}.sto\"\n\n    query_count = 0\n    with open(sto_file_path, \"r\") as f:\n        lines = f.readlines()\n        query_count = min((len(lines) - 3) // 3, 0)\n\n    if query_count < sto_cut_bound:\n        # delete file\n        print(f\"delete {target_name}.sto under sto_cut_bound\")\n        os.remove(sto_file_path)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T08:52:09.288085Z","iopub.execute_input":"2025-05-28T08:52:09.288409Z","iopub.status.idle":"2025-05-28T08:52:27.036525Z","shell.execute_reply.started":"2025-05-28T08:52:09.288384Z","shell.execute_reply":"2025-05-28T08:52:27.035835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_c1_coords(pdb_file, target_id, model_idx=0):\n    \"\"\"Extract C1' atom coordinates from PDB file for a specific model index.\n    \n    Args:\n        pdb_file: Path to the PDB file\n        target_id: Target ID for the RNA sequence\n        model_idx: Model index (0-4) corresponding to which model this is (will be mapped to model_idx+1 in output)\n    \"\"\"\n    coords = []\n    residue_ids = []\n    residue_names = []\n    \n    with open(pdb_file, 'r') as f:\n        for line in f:\n            if line.startswith('ATOM'):\n                atom_name = line[12:16].strip()\n                residue_name = line[17:20].strip()\n                residue_id = int(line[22:26].strip())\n                \n                if atom_name == \"C1'\":\n                    x = float(line[30:38].strip())\n                    y = float(line[38:46].strip())\n                    z = float(line[46:54].strip())\n                    \n                    residue_ids.append(residue_id)\n                    residue_names.append(residue_name)\n                    coords.append([x, y, z])\n    \n    # Process results\n    results = {}\n    for i, (res_id, res_name, coord) in enumerate(zip(residue_ids, residue_names, coords)):\n        # Use the first letter of the residue name\n        res_code = res_name[0]\n        \n        # Create key for each residue\n        key = f\"{target_id}_{res_id}\"\n        \n        # Initialize entry or update existing entry\n        if key not in results:\n            results[key] = {\n                'ID': key,\n                'resname': res_code,\n                'resid': res_id\n            }\n            \n        # Add coordinates for this specific model (model_idx + 1 because models are 1-indexed in output)\n        results[key][f'x_{model_idx+1}'] = coord[0]\n        results[key][f'y_{model_idx+1}'] = coord[1]\n        results[key][f'z_{model_idx+1}'] = coord[2]\n    \n    return results\n\ndef create_fasta_from_sequence(target_id, sequence, fasta_path, max_length=400):\n    \"\"\"Create a FASTA file from sequence, with sequence on a single line\"\"\"\n    # Truncate sequence if it exceeds max_length\n    truncated = False\n    if len(sequence) > max_length:\n        truncated_sequence = sequence[:max_length]\n        truncated = True\n    else:\n        truncated_sequence = sequence\n        \n    record = SeqRecord(Seq(truncated_sequence), id=target_id, description=\"\")\n    with open(fasta_path, \"w\") as f:\n        # Write header and sequence manually\n        f.write(f\">{record.id} {record.description}\\n\")\n        f.write(f\"{str(record.seq)}\\n\")\n    \n    return fasta_path, truncated, len(sequence)\n\ndef clean_output_dirs(outdir, ret_dir, folddir, refdir):\n    \"\"\"Clean output directories between runs\"\"\"\n    for dir_path in [ret_dir, folddir, refdir]:\n        if os.path.exists(dir_path):\n            for file in os.listdir(dir_path):\n                file_path = os.path.join(dir_path, file)\n                if os.path.isfile(file_path):\n                    os.remove(file_path)\n                    \ndef parse_output_to_df(output, seq, target_id):\n    \"\"\"Convert Protenix output to DataFrame format\"\"\"\n    df = []\n    chain_data = []\n    for i, res in enumerate(seq):\n        d = dict(ID = f\"{target_id}_{i+1}\",\n                 resname=res,\n                 resid=i+1)\n        for n in range(len(output)):\n            d = {**d, \n                 f'x_{n+1}': round(output[n,i,0].item(),3),\n                 f'y_{n+1}': round(output[n,i,1].item(),3),\n                 f'z_{n+1}': round(output[n,i,2].item(),3)}\n        chain_data.append(d)\n\n    if len(chain_data) != 0:\n        chain_df = pd.DataFrame(chain_data)\n        df.append(chain_df)\n    return df\n\ndef run_with_time(cmd, description):\n    \"\"\"Run a command and measure its execution time\"\"\"\n    print(f\"Running {description}...\")\n    print(f\"Command: {cmd}\")\n    start_time = time.time()\n    ret_code = os.system(cmd)\n    end_time = time.time()\n    elapsed = end_time - start_time\n    print(f\"✓ {description} completed in {elapsed:.2f} seconds (return code: {ret_code})\")\n    return ret_code, elapsed","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T08:52:27.037676Z","iopub.execute_input":"2025-05-28T08:52:27.037933Z","iopub.status.idle":"2025-05-28T08:52:27.049909Z","shell.execute_reply.started":"2025-05-28T08:52:27.037901Z","shell.execute_reply":"2025-05-28T08:52:27.049173Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"exp_dir = KAGGLE_INPUT\ndevice = \"cuda\" if torch.cuda.is_available() else \"cpu\"\nprint(device)\n#dlexps = ['cfg_97']\ndlexps = ['cfg_95', 'cfg_96', 'cfg_97']\ndirs = [os.path.join(exp_dir, 'model_hub', one_exp) for one_exp in dlexps]\n\n# Setting up the fasta path\nfasta_path = './test/seq.fasta'\nos.makedirs(os.path.dirname(fasta_path), exist_ok=True)\n\n# Setup directories (use the original structure)\noutdir = './outputs'\nif not os.path.isdir(outdir):\n    os.makedirs(outdir)\n    \nret_dir = os.path.join(outdir, 'rets_dir')\nif not os.path.isdir(ret_dir):\n    os.makedirs(ret_dir)\n\nfolddir = os.path.join(outdir, 'folds')\nif not os.path.isdir(folddir):\n    os.makedirs(folddir)\n\nrefdir = os.path.join(outdir, 'relax')\nif not os.path.isdir(refdir):\n    os.makedirs(refdir)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T08:52:27.051495Z","iopub.execute_input":"2025-05-28T08:52:27.051701Z","iopub.status.idle":"2025-05-28T08:52:27.117355Z","shell.execute_reply.started":"2025-05-28T08:52:27.051683Z","shell.execute_reply":"2025-05-28T08:52:27.116528Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if local:\n    csv_path = '../data/rna_fold_kaggle_data/test_sequences.csv'\nelse:\n    csv_path = '/kaggle/input/stanford-rna-3d-folding/test_sequences.csv'\n\nsequences_df = pd.read_csv(csv_path)\nprint(f\"Loaded {len(sequences_df)} sequences from {csv_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T08:52:27.118294Z","iopub.execute_input":"2025-05-28T08:52:27.118610Z","iopub.status.idle":"2025-05-28T08:52:27.146288Z","shell.execute_reply.started":"2025-05-28T08:52:27.118588Z","shell.execute_reply":"2025-05-28T08:52:27.145521Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"exp_dir","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T08:52:27.147124Z","iopub.execute_input":"2025-05-28T08:52:27.147413Z","iopub.status.idle":"2025-05-28T08:52:27.152151Z","shell.execute_reply.started":"2025-05-28T08:52:27.147383Z","shell.execute_reply":"2025-05-28T08:52:27.151530Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_results = {}\ndlmains = [os.path.join(exp_dir, one_exp, 'test_modeldir.py') for one_exp in dlexps]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T08:52:28.158186Z","iopub.execute_input":"2025-05-28T08:52:28.158471Z","iopub.status.idle":"2025-05-28T08:52:28.162339Z","shell.execute_reply.started":"2025-05-28T08:52:28.158451Z","shell.execute_reply":"2025-05-28T08:52:28.161431Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\nimport os\nimport shutil\nfrom abc import ABC, abstractmethod\nfrom collections import defaultdict\nfrom copy import deepcopy\nfrom os.path import exists as opexists\nfrom os.path import join as opjoin\nfrom typing import Any, Mapping, Optional, Sequence, Union\n\nimport numpy as np\nimport torch\nfrom biotite.structure import AtomArray\n\nfrom protenix.data.constants import STD_RESIDUES, rna_order_with_x\nfrom protenix.data.msa_utils import (\n    PROT_TYPE_NAME,\n    FeatureDict,\n    add_assembly_features,\n    clip_msa,\n    convert_monomer_features,\n    get_identifier_func,\n    load_and_process_msa,\n    make_sequence_features,\n    merge_features_from_prot_rna,\n    msa_parallel,\n    pair_and_merge,\n    rna_merge,\n)\nfrom typing import Any, Mapping\nfrom biotite.structure import AtomArray\nfrom protenix.data.json_to_feature import SampleDictToFeatures\nfrom protenix.data.data_pipeline import DataPipeline\nfrom protenix.data.msa_featurizer import InferenceMSAFeaturizer, process_single_sequence, SEQ_LIMITS, tokenize_msa, convert_monomer_features\nfrom protenix.data.utils import data_type_transform, make_dummy_feature\nfrom protenix.utils.distributed import DIST_WRAPPER\nfrom protenix.utils.torch_utils import dict_to_tensor\nfrom protenix.data.tokenizer import TokenArray\nfrom protenix.utils.logger import get_logger\n\n# Common function for train and inference\ndef merge_all_chain_features(\n    pdb_id: str,\n    all_chain_features: dict[str, FeatureDict],\n    asym_to_entity_id: dict,\n    is_homomer_or_monomer: bool = False,\n    merge_method: str = \"dense_max\",\n    max_size: int = 16384,\n    msa_entity_type: str = \"prot\",\n) -> dict[str, np.ndarray]:\n    \"\"\"\n    Merges features from all chains in the bioassembly.\n\n    Args:\n        pdb_id (str): The PDB ID of the bioassembly.\n        all_chain_features (dict[str, FeatureDict]): Features for each chain in the bioassembly.\n        asym_to_entity_id (dict): Mapping from asym ID to entity ID.\n        is_homomer_or_monomer (bool): Indicates if the bioassembly is a homomer or monomer. Defaults to False.\n        merge_method (str): Method used for merging features. Defaults to \"dense_max\".\n        max_size (int): Maximum size of the MSA. Defaults to 16384.\n        msa_entity_type (str): Type of MSA entity, either \"prot\" or \"rna\". Defaults to \"prot\".\n\n    Returns:\n        dict[str, np.ndarray]: Merged features for the bioassembly.\n    \"\"\"\n    all_chain_features = add_assembly_features(\n        pdb_id,\n        all_chain_features,\n        asym_to_entity_id=asym_to_entity_id,\n    )\n    if msa_entity_type == \"rna\":\n        np_example = rna_merge(\n            all_chain_features=all_chain_features,\n            merge_method=merge_method,\n            msa_crop_size=max_size,\n        )\n    elif msa_entity_type == \"prot\":\n        np_example = pair_and_merge(\n            is_homomer_or_monomer=is_homomer_or_monomer,\n            all_chain_features=all_chain_features,\n            merge_method=merge_method,\n            msa_crop_size=max_size,\n        )\n    np_example = clip_msa(np_example, max_num_msa=max_size)\n    return np_example","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T08:52:31.036341Z","iopub.execute_input":"2025-05-28T08:52:31.036613Z","iopub.status.idle":"2025-05-28T08:52:32.397410Z","shell.execute_reply.started":"2025-05-28T08:52:31.036592Z","shell.execute_reply":"2025-05-28T08:52:32.396486Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sequence_info = {}\n    \nprotenix_initialized = False\n#has_long_sequences = any(len(seq) >= 400 for seq in sequences_df['sequence'])\nhas_long_sequences = True\nif has_long_sequences:\n    try:\n        # Import Protenix modules\n        from runner.batch_inference import get_default_runner\n        from runner.inference import update_inference_configs, InferenceRunner\n        from protenix.data.infer_data_pipeline import InferenceDataset\n        from configs.configs_base import configs as configs_base\n        from configs.configs_data import data_configs\n        from configs.configs_inference import inference_configs\n        from protenix.config.config import parse_configs\n        \n        # Set random seeds for reproducibility\n        np.random.seed(0)\n        torch.random.manual_seed(0)\n        torch.cuda.manual_seed_all(0)\n        \n        # Define DictDataset for Protenix\n        class DictDataset(InferenceDataset):\n            def __init__(\n                self,\n                seq_list: list,\n                dump_dir: str,\n                id_list: list = None,\n                use_msa: bool = True,\n            ) -> None:\n                self.dump_dir = dump_dir\n                self.use_msa = use_msa\n                if isinstance(id_list, type(None)):\n                    self.inputs = [{\"sequences\": \n                                    [{\"rnaSequence\": \n                                      {\"sequence\": seq, \n                                       \"count\": 1}}],\n                                    \"name\": \"query\"} for seq in seq_list]\n                else:\n                    self.inputs = [\n                        {\n                            \"sequences\": [\n                                {\n                                    \"rnaSequence\": {\n                                        \"sequence\": seq,\n                                        \"count\": 1,\n                                        \"msa\": {\n                                            \"precomputed_msa_dir\": f\"{sto_dir}/\",\n                                            \"pairing_db\": \"no_pairing_db\"\n                                        }\n                                    }\n                                }\n                            ],\n                            \"name\": i\n                        } for i, seq in zip(id_list,seq_list)\n                    ]\n\n            # override for RNA MSA functionality\n            def process_one(\n                self,\n                single_sample_dict: Mapping[str, Any],\n            ) -> tuple[dict[str, torch.Tensor], AtomArray, dict[str, float]]:\n                \"\"\"\n                Processes a single sample from the input JSON to generate features and statistics.\n    \n                Args:\n                    single_sample_dict: A dictionary containing the sample data.\n    \n                Returns:\n                    A tuple containing:\n                        - A dictionary of features.\n                        - An AtomArray object.\n                        - A dictionary of time tracking statistics.\n                \"\"\"\n                # general features\n                t0 = time.time()\n                sample2feat = SampleDictToFeatures(\n                    single_sample_dict,\n                )\n                features_dict, atom_array, token_array = sample2feat.get_feature_dict()\n                features_dict[\"distogram_rep_atom_mask\"] = torch.Tensor(\n                    atom_array.distogram_rep_atom_mask\n                ).long()\n                entity_poly_type = sample2feat.entity_poly_type\n                t1 = time.time()\n    \n                # Msa features\n                entity_to_asym_id = DataPipeline.get_label_entity_id_to_asym_id_int(atom_array)\n    \n                try:\n                    if not self.use_msa:\n                        msa_features = {}\n                    else:\n                        sequence_to_features: dict[str, dict[str, Any]] = {}\n                        name = single_sample_dict[\"name\"]\n                        sequence = single_sample_dict[\"sequences\"][0][\"rnaSequence\"][\"sequence\"]\n                        msa_dir = single_sample_dict[\"sequences\"][0][\"rnaSequence\"][\"msa\"][\"precomputed_msa_dir\"]\n                        sequence_feat = process_single_sequence(\n                            pdb_name=name,\n                            sequence=sequence,\n                            raw_msa_paths=[f\"{msa_dir}{name}.MSA.sto\"], \n                            seq_limits=SEQ_LIMITS, # Optional[list[str]],\n                            msa_entity_type =\"rna\",\n                            msa_type = \"non_pairing\",\n                        )\n                        sequence_feat = convert_monomer_features(sequence_feat)\n                        sequence_to_features[sequence] = sequence_feat\n                        all_chain_features = {\n                            0: deepcopy(sequence_to_features[sequence])\n                        }\n                        del sequence_to_features\n    \n                        asym_to_entity_id = {\n                            0: '1'\n                        }\n    \n                        msa_feats = merge_all_chain_features(\n                            pdb_id=\"test_assembly\",\n                            all_chain_features=all_chain_features,\n                            asym_to_entity_id=asym_to_entity_id,\n                            is_homomer_or_monomer=False,\n                            msa_entity_type=\"rna\",\n                        )\n                        if msa_feats is None:\n                            return {}\n    \n                        msa_feats = tokenize_msa(\n                            msa_feats=msa_feats,\n                            token_array=token_array,\n                            atom_array=atom_array,\n                        )\n    \n                        msa_features = {\n                            k: v\n                            for (k, v) in msa_feats.items()\n                            if k\n                            in [\"msa\", \"has_deletion\", \"deletion_value\", \"deletion_mean\", \"profile\"]\n                        }\n                except Exception as e:\n                    print(\"msa feature skipped as error occur, run without msa\")\n                    msa_features = {}\n    \n                # Make dummy features for not implemented features\n                dummy_feats = [\"template\"]\n                if len(msa_features) == 0:\n                    dummy_feats.append(\"msa\")\n                else:\n                    msa_features = dict_to_tensor(msa_features)\n                    features_dict.update(msa_features)\n                features_dict = make_dummy_feature(\n                    features_dict=features_dict,\n                    dummy_feats=dummy_feats,\n                )\n    \n                # Transform to right data type\n                feat = data_type_transform(feat_or_label_dict=features_dict)\n    \n                t2 = time.time()\n    \n                data = {}\n                data[\"input_feature_dict\"] = feat\n    \n                # Add dimension related items\n                N_token = feat[\"token_index\"].shape[0]\n                N_atom = feat[\"atom_to_token_idx\"].shape[0]\n                N_msa = feat[\"msa\"].shape[0]\n    \n                stats = {}\n                for mol_type in [\"ligand\", \"protein\", \"dna\", \"rna\"]:\n                    mol_type_mask = feat[f\"is_{mol_type}\"].bool()\n                    stats[f\"{mol_type}/atom\"] = int(mol_type_mask.sum(dim=-1).item())\n                    stats[f\"{mol_type}/token\"] = len(\n                        torch.unique(feat[\"atom_to_token_idx\"][mol_type_mask])\n                    )\n    \n                N_asym = len(torch.unique(data[\"input_feature_dict\"][\"asym_id\"]))\n                data.update(\n                    {\n                        \"N_asym\": torch.tensor([N_asym]),\n                        \"N_token\": torch.tensor([N_token]),\n                        \"N_atom\": torch.tensor([N_atom]),\n                        \"N_msa\": torch.tensor([N_msa]),\n                    }\n                )\n    \n                def formatted_key(key):\n                    type_, unit = key.split(\"/\")\n                    if type_ == \"protein\":\n                        type_ = \"prot\"\n                    elif type_ == \"ligand\":\n                        type_ = \"lig\"\n                    else:\n                        pass\n                    return f\"N_{type_}_{unit}\"\n    \n                data.update(\n                    {\n                        formatted_key(k): torch.tensor([stats[k]])\n                        for k in [\n                            \"protein/atom\",\n                            \"ligand/atom\",\n                            \"dna/atom\",\n                            \"rna/atom\",\n                            \"protein/token\",\n                            \"ligand/token\",\n                            \"dna/token\",\n                            \"rna/token\",\n                        ]\n                    }\n                )\n                data.update({\"entity_poly_type\": entity_poly_type})\n                t3 = time.time()\n                time_tracker = {\n                    \"crop\": t1 - t0,\n                    \"featurizer\": t2 - t1,\n                    \"added_feature\": t3 - t2,\n                }\n    \n                return data, atom_array, time_tracker\n        \n        # Setup Protenix model configuration\n        configs_base[\"use_deepspeed_evo_attention\"] = (\n            os.environ.get(\"USE_DEEPSPEED_EVO_ATTENTION\", False) == \"true\")\n        configs_base[\"model\"][\"N_cycle\"] = 10\n        configs_base[\"sample_diffusion\"][\"N_sample\"] = 5  # Generate 5 models\n        configs_base[\"sample_diffusion\"][\"N_step\"] = 200\n        inference_configs['load_checkpoint_path'] = SHORT_MODEL_PATH\n        configs = {**configs_base, **{\"data\": data_configs}, **inference_configs}\n        \n        configs = parse_configs(\n            configs=configs,\n            fill_required_with_null=True,\n        )\n        \n        protenix_runner = InferenceRunner(configs)\n        protenix_initialized = True\n        print(\"Protenix model initialized successfully!\")\n        \n    except Exception as e:\n        print(f\"Error initializing Protenix model: {e}\")\n        print(\"Will attempt to use DRfold for all sequences.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T08:52:35.225692Z","iopub.execute_input":"2025-05-28T08:52:35.226269Z","iopub.status.idle":"2025-05-28T08:52:53.050776Z","shell.execute_reply.started":"2025-05-28T08:52:35.226239Z","shell.execute_reply":"2025-05-28T08:52:53.050050Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"csv_df = pd.read_csv('/kaggle/input/testingcsv/run_test.csv')\n\n# Model paths for three different Protenix models\nPROTENIX1_MODEL_PATH = \"/kaggle/input/rnaonly19799/19799_ema_0.995.pt\"  #2\nPROTENIX2_MODEL_PATH = \"/kaggle/input/37989all/37989_ema_0.995.pt\"   #2\nPROTENIX3_MODEL_PATH = \"/kaggle/input/protenix-checkpoints/model_v0.2.0.pt\"  #1\n\nMAX_SEQUENCE_LENGTH = 1000\n\n\nfor idx, row in sequences_df.iterrows():\n    \n    sequence_results = {}\n    \n    clean_output_dirs(outdir, ret_dir, folddir, refdir)\n    start_time = time.time()\n    target_id = row['target_id']\n    sequence = row['sequence']\n    optimize = False\n\n    target_rows = csv_df[csv_df['ID'].str.startswith(target_id + '_')]\n    \n    if not target_rows.empty:\n        for _, csv_row in target_rows.iterrows():\n            key = csv_row['ID']  # e.g., R1107_1\n            resid = int(csv_row['ID'].split('_')[-1])  # Extract resid from ID\n            resname = csv_row['resname']\n            \n            # Create dictionary for this residue\n            result_entry = {\n                'ID': key,\n                'resname': resname,\n                'resid': resid,\n                'x_1': csv_row['x_1'],\n                'y_1': csv_row['y_1'],\n                'z_1': csv_row['z_1'],\n                'x_2': csv_row['x_2'],\n                'y_2': csv_row['y_2'],\n                'z_2': csv_row['z_2'],\n                'x_3': csv_row['x_3'],\n                'y_3': csv_row['y_3'],\n                'z_3': csv_row['z_3'],\n                'x_4': csv_row['x_4'],\n                'y_4': csv_row['y_4'],\n                'z_4': csv_row['z_4'],\n                'x_5': csv_row['x_5'],\n                'y_5': csv_row['y_5'],\n                'z_5': csv_row['z_5']\n            }\n            \n            sequence_results[key] = result_entry\n        \n        # Update all_results with CSV data\n        all_results.update(sequence_results)\n        print(f\"Added {len(sequence_results)} residue positions from CSV for {target_id}\")\n        print(f\"Time taken for {target_id}: {time.time() - start_time:.2f} seconds\")\n        continue\n        \n    \n    seq_len = len(sequence)\n    \n    # Handle sequence truncation for sequences > 1000\n    original_sequence = sequence\n    was_truncated = len(sequence) > MAX_SEQUENCE_LENGTH\n    if was_truncated:\n        sequence = sequence[:MAX_SEQUENCE_LENGTH]\n        print(f\"Truncated sequence from {len(original_sequence)} to {len(sequence)} residues\")\n\n    sequence_info[target_id] = {\n        'was_truncated': was_truncated,\n        'original_length': len(original_sequence)\n    }\n    \n    # Update sequence length after potential truncation\n    seq_len = len(sequence)\n    \n    use_drfold = seq_len <= 400\n    use_protenix = True\n\n    print(f\"Strategy for {target_id} (length: {seq_len}):\")\n    if seq_len <= 400:\n        print(f\"- Using 4 DRfold models + 1 Protenix model\")\n    else:\n        print(f\"- Using 3 Protenix models (2 baseline + 2 MSA-trained + 1 RNA-only)\")\n\n    protenix_results = {}\n    drfold_results = {}\n    \n    \n    if use_protenix:\n        print(f\"Using Protenix pipeline for {target_id} (length: {len(sequence)})\")\n        protenix_start = time.time()\n        try:\n            if seq_len <= 400:\n                # For short sequences: use only 1 model (MSA-trained)\n                PROTENIX_MODEL_PATHS = [PROTENIX1_MODEL_PATH]\n                protenix_models_needed = [1]  # 1 sample from MSA-trained model\n                print(\"Using 1 Protenix model (MSA-trained) for short sequence\")\n            else:\n                # For long sequences: use 3 different models\n                PROTENIX_MODEL_PATHS = [\n                    PROTENIX1_MODEL_PATH,  \n                    PROTENIX2_MODEL_PATH,  \n                    PROTENIX3_MODEL_PATH   \n                ]\n                protenix_models_needed = [2, 2, 1]  # 2+2+1 = 5 total samples\n                print(\"Using 3 Protenix models for long sequence\")\n            \n            # Run Protenix predictions with different models\n            all_protenix_predictions = []\n            \n            for model_run, (model_path, n_samples) in enumerate(zip(PROTENIX_MODEL_PATHS, protenix_models_needed)):\n                print(f\"Running Protenix model {model_run + 1} ({model_path}) with {n_samples} samples...\")\n                \n                # Update model checkpoint path\n                inference_configs['load_checkpoint_path'] = model_path\n                \n                configs_base[\"sample_diffusion\"][\"N_sample\"] = n_samples\n                \n                configs = {**configs_base, **{\"data\": data_configs}, **inference_configs}\n                configs = parse_configs(\n                    configs=configs,\n                    fill_required_with_null=True,\n                )\n                \n                # Reinitialize the model with the appropriate checkpoint\n                protenix_runner = InferenceRunner(configs)\n                \n                # Create dataset with single sequence\n                dataset = DictDataset([sequence], dump_dir='output', id_list=[target_id], use_msa=True)\n                \n                # Process with Protenix\n                data, atom_array, data_error_message = dataset[0]\n                \n                # Check for errors\n                if data_error_message:\n                    raise ValueError(f\"Protenix data error: {data_error_message}\")\n                \n                # Update model configs for this sequence\n                new_configs = update_inference_configs(configs, data[\"N_token\"].item())\n                protenix_runner.update_model_configs(new_configs)\n                \n                # Run prediction\n                prediction = protenix_runner.predict(data)\n                print(prediction['summary_confidence'])\n                ptm_scores = [summary['gpde'].item() for summary in prediction['summary_confidence']]\n                \n                protenix_indices_by_score = torch.tensor(ptm_scores).argsort(descending=True)\n                best_protenix_indices = protenix_indices_by_score.tolist()\n                print(f\"Protenix model {model_run + 1} ranked by GPDE: {best_protenix_indices}\")\n                \n                prediction['coordinate'] = prediction['coordinate'][protenix_indices_by_score]\n                prediction = prediction['coordinate'][:, data['input_feature_dict']['atom_to_tokatom_idx']==12]\n                \n                # Store predictions from this run\n                all_protenix_predictions.extend(prediction)\n            \n            # Process all predictions\n            total_predictions = len(all_protenix_predictions)\n            print(f\"Total Protenix predictions: {total_predictions}\")\n            \n            sequence_results_list = parse_output_to_df(torch.stack(all_protenix_predictions), sequence, target_id)\n\n            if sequence_results_list:\n                protenix_results_df = sequence_results_list[0]\n                \n                # Convert to dictionary format\n                protenix_results_dict = {}\n                for _, row in protenix_results_df.iterrows():\n                    key = row['ID']\n                    protenix_results_dict[key] = row.to_dict()\n                \n                protenix_results = protenix_results_dict\n                \n                print(f\"Successfully processed {target_id} with Protenix in {time.time() - protenix_start:.2f} seconds\")\n            \n        except Exception as e:\n            print(f\"Error processing {target_id} with Protenix: {e}\")\n            use_protenix = False\n            \n    if use_drfold:\n        print(f\"Running DRfold pipeline for {target_id}\")       \n        drfold_start = time.time()\n        max_length = 400\n        print(f\"Processing sequence {idx+1}/{len(sequences_df)}: {target_id}\")\n        fasta_path, was_truncated_drfold, orig_length_drfold = create_fasta_from_sequence(target_id, sequence, fasta_path, max_length=max_length)\n        \n        # Update sequence info if DRfold truncation is different\n        if was_truncated_drfold:\n            sequence_info[target_id]['drfold_was_truncated'] = was_truncated_drfold\n            sequence_info[target_id]['drfold_original_length'] = orig_length_drfold\n    \n        for dlmain, one_exp, mdir in zip(dlmains, dlexps, dirs):\n            cmd = f'python {dlmain} {device} {fasta_path} {ret_dir}/{one_exp}_ {mdir}'\n            print(cmd)\n            os.system(cmd)\n            with open(ret_dir+'/done', 'w') as wfile:\n                wfile.write('1')\n           \n        config_sel = os.path.join(exp_dir, 'cfg_for_selection.json')\n        foldconfig = os.path.join(exp_dir, 'cfg_for_folding.json')\n        selpython = os.path.join(exp_dir, 'PotentialFold', 'Selection.py')\n        optpython = os.path.join(exp_dir, 'PotentialFold', 'Optimization.py')\n        \n        optsaveprefix = os.path.join(folddir, f'opt_0')\n        save_prefix = os.path.join(folddir, f'sel_0')\n       \n        rets = os.listdir(ret_dir)\n        rets = [afile for afile in rets if afile.endswith('.ret')]\n        rets = [os.path.join(ret_dir, aret) for aret in rets]\n        ret_str = ' '.join(rets)\n        \n        cmd = f'python {selpython} {fasta_path} {config_sel} {save_prefix} {ret_str} --optimize' if optimize else f'python {selpython} {fasta_path} {config_sel} {save_prefix} {ret_str}'\n        print(cmd)\n        os.system(cmd)\n        print('done ret')\n        if optimize:\n            cmd = f'python {optpython} {fasta_path} {optsaveprefix} {ret_dir} {save_prefix} {foldconfig}'\n            os.system(cmd)\n        print('done c')\n    \n        drfold_results = {}\n\n        arena = '/kaggle/working/Arena/Arena'\n        if not os.path.exists(refdir):\n            os.makedirs(refdir)\n\n        # Process 4 DRfold models instead of 3\n        for model_idx in range(4):\n            pdb_file = f'opt_{model_idx}.pdb'\n            if pdb_file:\n                cgpdb = os.path.join(folddir, pdb_file)\n\n                savepdb = os.path.join(refdir, f'model_{model_idx+1}.pdb')\n                cmd = f'{arena} {cgpdb} {savepdb} 7'\n                ret_code, elapsed = run_with_time(cmd, f\"Arena for model {model_idx+1}/4\")\n                \n                # Extract coordinates for this model\n                model_results = extract_c1_coords(cgpdb, target_id, model_idx)\n\n                print(f\"Model {model_idx+1} results contain keys: {list(model_results.keys())[:5]}...\")\n                \n                # Merge with existing results\n                for key, data in model_results.items():\n                    model_pos = model_idx + 1\n                    \n                    if key not in drfold_results:\n                        drfold_results[key] = {\n                            'ID': data['ID'],\n                            'resname': data['resname'],\n                            'resid': data['resid']\n                        }\n                    \n                    # Add coordinates for this specific model\n                    drfold_results[key][f'x_{model_pos}'] = data[f'x_{model_pos}']\n                    drfold_results[key][f'y_{model_pos}'] = data[f'y_{model_pos}']\n                    drfold_results[key][f'z_{model_pos}'] = data[f'z_{model_pos}']\n                    \n                print(f\"Processed model {model_idx+1}/4 for {target_id}\")\n            else:\n                print(f\"Model {model_idx+1}/4 not found for {target_id}\")\n        \n        print(f\"DRfold processing completed in {time.time() - drfold_start:.2f} seconds\")\n\n    # Combine results from different methods\n    if seq_len > 400:\n        # For sequences > 400: Use 3 Protenix models (5 total predictions)\n        all_keys = set()\n        if protenix_results:\n            all_keys.update(protenix_results.keys())\n            \n        for key in all_keys:\n            if key in protenix_results:\n                result_entry = {\n                    'ID': protenix_results[key]['ID'],\n                    'resname': protenix_results[key]['resname'],\n                    'resid': protenix_results[key]['resid']\n                }\n                \n                # Fill all 5 positions from Protenix models\n                for out_pos in range(1, 6):\n                    model_idx = out_pos\n                    if f'x_{model_idx}' in protenix_results[key]:\n                        result_entry[f'x_{out_pos}'] = protenix_results[key][f'x_{model_idx}']\n                        result_entry[f'y_{out_pos}'] = protenix_results[key][f'y_{model_idx}']\n                        result_entry[f'z_{out_pos}'] = protenix_results[key][f'z_{model_idx}']\n                    else:\n                        result_entry[f'x_{out_pos}'] = 0.0\n                        result_entry[f'y_{out_pos}'] = 0.0\n                        result_entry[f'z_{out_pos}'] = 0.0\n                \n                sequence_results[key] = result_entry\n\n    else:\n        # For sequences <= 400: Use 4 DRfold + 1 Protenix\n        all_keys = set()\n        if drfold_results:\n            all_keys.update(drfold_results.keys())\n        if protenix_results:\n            all_keys.update(protenix_results.keys())\n                    \n        for key in all_keys:\n            # Initialize with base info from any available source\n            if key in drfold_results:\n                result_entry = {\n                    'ID': drfold_results[key]['ID'],\n                    'resname': drfold_results[key]['resname'],\n                    'resid': drfold_results[key]['resid']\n                }\n            elif key in protenix_results:\n                result_entry = {\n                    'ID': protenix_results[key]['ID'],\n                    'resname': protenix_results[key]['resname'],\n                    'resid': protenix_results[key]['resid']\n                }\n            else:\n                continue\n            \n            # Fill positions 1-4 from DRfold (4 models)\n            for model_idx in range(1, 5):\n                if key in drfold_results and f'x_{model_idx}' in drfold_results[key]:\n                    result_entry[f'x_{model_idx}'] = drfold_results[key][f'x_{model_idx}']\n                    result_entry[f'y_{model_idx}'] = drfold_results[key][f'y_{model_idx}']\n                    result_entry[f'z_{model_idx}'] = drfold_results[key][f'z_{model_idx}']\n                elif key in protenix_results and f'x_1' in protenix_results[key]:\n                    # Fallback to Protenix if DRfold is missing\n                    result_entry[f'x_{model_idx}'] = protenix_results[key]['x_1']\n                    result_entry[f'y_{model_idx}'] = protenix_results[key]['y_1']\n                    result_entry[f'z_{model_idx}'] = protenix_results[key]['z_1']\n                else:\n                    # Use zeros as fallback\n                    result_entry[f'x_{model_idx}'] = 0.0\n                    result_entry[f'y_{model_idx}'] = 0.0\n                    result_entry[f'z_{model_idx}'] = 0.0\n            \n            # Fill position 5 from Protenix model\n            if key in protenix_results and f'x_1' in protenix_results[key]:\n                result_entry[f'x_5'] = protenix_results[key]['x_1']\n                result_entry[f'y_5'] = protenix_results[key]['y_1']\n                result_entry[f'z_5'] = protenix_results[key]['z_1']\n            elif key in drfold_results and 'x_1' in drfold_results[key]:\n                # Fallback to DRfold if Protenix is missing\n                result_entry[f'x_5'] = drfold_results[key]['x_1']\n                result_entry[f'y_5'] = drfold_results[key]['y_1']\n                result_entry[f'z_5'] = drfold_results[key]['z_1']\n            else:\n                # Use zeros as fallback\n                result_entry[f'x_5'] = 0.0\n                result_entry[f'y_5'] = 0.0\n                result_entry[f'z_5'] = 0.0\n                    \n            sequence_results[key] = result_entry\n        \n        print(f\"Combined 4 DRfold and 1 Protenix models for {target_id} (length ≤ 400)\")\n\n    # Handle truncated sequences by adding dummy entries with zero coordinates\n    if sequence_info[target_id]['was_truncated']:\n        orig_length = sequence_info[target_id]['original_length']\n        for res_id in range(MAX_SEQUENCE_LENGTH + 1, orig_length + 1):\n            key = f\"{target_id}_{res_id}\"\n            # Determine residue name from original sequence\n            res_code = original_sequence[res_id - 1]\n            \n            # Create dummy entry with zero coordinates\n            sequence_results[key] = {\n                'ID': key,\n                'resname': res_code,\n                'resid': res_id\n            }\n            \n            # Add zero coordinates for all 5 models\n            for model_idx in range(1, 6):\n                sequence_results[key][f'x_{model_idx}'] = 0.0\n                sequence_results[key][f'y_{model_idx}'] = 0.0\n                sequence_results[key][f'z_{model_idx}'] = 0.0\n        \n        print(f\"Added {orig_length - MAX_SEQUENCE_LENGTH} dummy entries with zero coordinates for {target_id}\")\n    \n    # ADD RESULTS TO MAIN DICTIONARY\n    if sequence_results:\n        all_results.update(sequence_results)\n        print(f\"Successfully processed {target_id} with multi-model approach\")\n        print(f\"Added {len(sequence_results)} residue entries to all_results\")\n    else:\n        print(f\"ERROR: No results available for {target_id}\")        \n    \n    # Clean up\n    torch.cuda.empty_cache()\n    if torch.cuda.is_available():\n        torch.cuda.synchronize()\n    \n    print(f\"Total time taken for {target_id}: {time.time() - start_time:.2f} seconds\")\n    print(f\"Current all_results size: {len(all_results)}\")\n    print(\"-\" * 50)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T08:53:11.559187Z","iopub.execute_input":"2025-05-28T08:53:11.559781Z","iopub.status.idle":"2025-05-28T08:53:11.844626Z","shell.execute_reply.started":"2025-05-28T08:53:11.559751Z","shell.execute_reply":"2025-05-28T08:53:11.843920Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# results_list = []\n# for key, result in all_results.items():\n#     results_list.append(result)\n\n# submit_df = pd.DataFrame(results_list)\nresults_df = pd.DataFrame(all_results.values())\n# Sort by target ID first, then by residue ID\nresults_df['target_id'] = results_df['ID'].str.split('_').str[0]\nresults_df['residue_num'] = results_df['ID'].str.split('_').str[1].astype(int)\nresults_df = results_df.sort_values(['target_id', 'residue_num'])\n# Remove helper columns if needed\nresults_df = results_df.drop(['target_id', 'residue_num'], axis=1)\nsubmit_df = results_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T07:41:41.821295Z","iopub.execute_input":"2025-05-27T07:41:41.821673Z","iopub.status.idle":"2025-05-27T07:41:41.833650Z","shell.execute_reply.started":"2025-05-27T07:41:41.821640Z","shell.execute_reply":"2025-05-27T07:41:41.832892Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submit_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T07:41:42.986769Z","iopub.execute_input":"2025-05-27T07:41:42.987063Z","iopub.status.idle":"2025-05-27T07:41:43.008456Z","shell.execute_reply.started":"2025-05-27T07:41:42.987031Z","shell.execute_reply":"2025-05-27T07:41:43.007714Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submit_df = submit_df.fillna(0)\n# Save submission file\nif local:\n    submit_df.to_csv('submission.csv', index=False)\nelse:\n    submit_df.to_csv('/kaggle/working/submission.csv', index=False)\n\nprint(f\"Saved submission file with {len(submit_df)} predictions\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T07:41:49.313584Z","iopub.execute_input":"2025-05-27T07:41:49.313998Z","iopub.status.idle":"2025-05-27T07:41:49.327578Z","shell.execute_reply.started":"2025-05-27T07:41:49.313966Z","shell.execute_reply":"2025-05-27T07:41:49.326621Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"directory = '/kaggle/working'\n\n# Loop through all files and directories\nfor filename in os.listdir(directory):\n    file_path = os.path.join(directory, filename)\n    if filename != 'submission.csv':\n        try:\n            if os.path.isfile(file_path) or os.path.islink(file_path):\n                os.remove(file_path)  # remove file or symbolic link\n            elif os.path.isdir(file_path):\n                import shutil\n                shutil.rmtree(file_path)  # remove directory and its contents\n        except Exception as e:\n            print(f'Failed to delete {file_path}. Reason: {e}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T07:41:51.011598Z","iopub.execute_input":"2025-05-27T07:41:51.011894Z","iopub.status.idle":"2025-05-27T07:41:51.575356Z","shell.execute_reply.started":"2025-05-27T07:41:51.011870Z","shell.execute_reply":"2025-05-27T07:41:51.574418Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}