{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":87793,"databundleVersionId":12255112,"sourceType":"competition"},{"sourceId":11411364,"sourceType":"datasetVersion","datasetId":6896780},{"sourceId":11641448,"sourceType":"datasetVersion","datasetId":7304841}],"dockerImageVersionId":30919,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import sys\nimport os\nfrom pathlib import Path\n\n# Assuming you've mounted your dataset containing the wheels and Protenix code\nDATASET_PATH = '/kaggle/input/required-presets'\n\n# Install dependencies from wheels\nwheel_path = os.path.join(DATASET_PATH, 'wheels')\n!pip install -q --no-index --find-links {wheel_path} torch torchvision torchaudio \\\nml-collections dm-tree deepspeed protobuf scipy biopython numpy matplotlib biotite","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T00:54:16.520155Z","iopub.execute_input":"2025-05-12T00:54:16.520522Z","iopub.status.idle":"2025-05-12T00:54:20.027417Z","shell.execute_reply.started":"2025-05-12T00:54:16.520490Z","shell.execute_reply":"2025-05-12T00:54:20.026172Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from biotite.structure.io.pdb import PDBFile\nfrom biotite.structure import AtomArray\nimport numpy as np\nimport os\nimport pandas as pd \nfrom tqdm import tqdm\nimport subprocess\nimport glob\nimport shutil","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T00:54:20.028962Z","iopub.execute_input":"2025-05-12T00:54:20.029235Z","iopub.status.idle":"2025-05-12T00:54:20.033768Z","shell.execute_reply.started":"2025-05-12T00:54:20.029212Z","shell.execute_reply":"2025-05-12T00:54:20.032825Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_input(sequence , target_id, input_dir) :\n\n    # 실행 가능한 바이너리로 복사\n    ipknot_src = \"/kaggle/input/nufold-kaggle-dataset/ipknot-1.1.0-x86_64-linux/ipknot\"\n    ipknot_dst = \"/kaggle/working/ipknot\"\n    \n    shutil.copy(ipknot_src, ipknot_dst)\n    os.chmod(ipknot_dst, 0o755)  # 실행 권한 부여\n\n    ipknot_path = \"/kaggle/working/ipknot\" # ipknot 경로 \n    a3m_path = '/kaggle/input/stanford-rna-3d-folding/MSA' # MSA 경로 \n\n    input_target_dir = os.path.join(input_dir, target_id)\n    os.makedirs(input_target_dir, exist_ok=True)\n\n    fasta_path = os.path.join(input_target_dir, f\"{target_id}.fasta\")\n    with open(fasta_path, \"w\") as f:\n        f.write(f\">{target_id}\\n\")\n        f.write(f\"{sequence}\\n\")\n\n    ss_path = os.path.join(input_target_dir, f\"{target_id}.ipknot.ss\")\n    os.system(f\"{ipknot_path} {fasta_path} > {ss_path}\")\n\n\n    original_file = f\"{target_id}.MSA.fasta\"\n    source_path = os.path.join(a3m_path, original_file)  \n    \n    new_filename = f\"{target_id}.a3m\"\n    input_a3m_path = os.path.join(input_target_dir, new_filename)  \n    \n    if os.path.exists(source_path):\n        shutil.copy(source_path, input_a3m_path)\n    else:\n        print(f\"파일이 존재하지 않습니다: {source_path}\")\n\n    return None ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T00:54:20.035296Z","iopub.execute_input":"2025-05-12T00:54:20.035559Z","iopub.status.idle":"2025-05-12T00:54:20.050646Z","shell.execute_reply.started":"2025-05-12T00:54:20.035537Z","shell.execute_reply":"2025-05-12T00:54:20.049815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# NuFold 실행 \ndef run_inference(input_dir , output_dir , target_id) : \n\t\n    target_output_dir = f\"{output_dir}/{target_id}\"\n    os.makedirs(target_output_dir, exist_ok=True) \n    \n    cmd = [\n    \"python3\", \"/kaggle/input/nufold-kaggle-dataset/NuFold/run_nufold.py\",\n    \"--ckpt_path\", \"/kaggle/input/nufold-kaggle-dataset/checkpoints/global_step145245.pt\", \n    \"--input_fasta\", f\"{input_dir}/{target_id}/{target_id}.fasta\", \n    \"--input_dir\",  f\"{input_dir}/\", \n    \"--output_dir\", f\"{output_dir}\", \n    \"--config_preset\", \"initial_training\",\n    ]\n    \n    print(f\"Running inference for {target_id}\")\n    \n    try:\n    \t# cmd  명령어 실행 \n        result = subprocess.run(cmd, capture_output=True, text=True)\n        \n        # 실행 처리 \n        if result.returncode != 0:\n            print(f\"Error running inference for {target_id}: {result.stderr}\")\n            return False\n        return True\n    \n    # 예외처리 \n    except Exception as e:\n        print(f\"Exception running inference for {target_id}: {e}\")\n        return False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T00:54:20.051881Z","iopub.execute_input":"2025-05-12T00:54:20.052118Z","iopub.status.idle":"2025-05-12T00:54:20.069885Z","shell.execute_reply.started":"2025-05-12T00:54:20.052099Z","shell.execute_reply":"2025-05-12T00:54:20.069106Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# PDB에서 C1 좌표 반환 \ndef extract_c1_coordinates(pdb_file_path):\n    \n    try:\n        pdb_file = PDBFile.read(pdb_file_path)\n        \n        # Get structure from PDB dat \n        atom_array = pdb_file.get_structure(model=1)\n        \n        # Clean atom names and find C1' atoms (C1' 원자만 골라내기)\n        atom_names_clean = np.char.strip(atom_array.atom_name.astype(str))\t\t\n        # True/False로 구성된 마스크 배열 \n        mask_c1 = atom_names_clean == \"C1'\"\n        # True인 원자들만 추출 \n        c1_atoms = atom_array[mask_c1]\n        \n        # 구조파일에 C1' 원자가 없다면 경고 출력 \n        if len(c1_atoms) == 0:\n            print(f\"Warning: No C1' atoms found in {pdb_file_path}\")\n            return None\n        \n        # Sort by residue ID and return coordinates (순서대로 정렬된 C1') \n        # res_id = 염기들의 순서 번호 \n        sort_indices = np.argsort(c1_atoms.res_id)\n        # 인덱스를 활용해 정렬 \n        c1_atoms_sorted = c1_atoms[sort_indices]\n        # C1' 원자들의 (x,y,z) 좌표 추출 \n        c1_coords = c1_atoms_sorted.coord\n        \n        return c1_coords\n    \n    # 예외 처리 \n    except Exception as e:\n        print(f\"Error extracting C1' coordinates from {cif_file_path}: {e}\")\n        return None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T00:54:20.070800Z","iopub.execute_input":"2025-05-12T00:54:20.071044Z","iopub.status.idle":"2025-05-12T00:54:20.086110Z","shell.execute_reply.started":"2025-05-12T00:54:20.071013Z","shell.execute_reply":"2025-05-12T00:54:20.085268Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_sequence(sequence, target_id, input_dir, output_dir):\n    \"\"\"\n    Process a single RNA sequence and return C1' coordinates\n    \"\"\"\n    \n    # 어떤 sequence를 처리하고 있는지 print \n    print(f\"Processing {target_id}: {sequence}\")\n    # input data 구조 갖추기\n    create_input(sequence , target_id, input_dir)\n    \n    # Run inference (추론) \n    success = run_inference(input_dir , output_dir , target_id)\n    \n    if not success:\n        print(f\"Inference failed for {target_id}\")\n        return None\n    \n    target_prediction_dir = os.path.join(output_dir , target_id)\n    print(target_prediction_dir)\n    \n    if not os.path.exists(target_prediction_dir) :\n        print(\"Prediction directory not found for {target_id}\") \n    \n    pdb_files = sorted(glob.glob(os.path.join(target_prediction_dir , f\"{target_id}_*.pdb\")))\n    \n    if not pdb_files:\n        print(f\"No PDB files found for {target_id}\")\n    \n    # PDB 개수 확인 출력 \n    print(f\"Found {len(pdb_files)} PDB files for {target_id}\")\n    \n    all_coords = []\n    \n    for pdb_file in pdb_files : \n        \n        coords = extract_c1_coordinates(pdb_file)\n        if coords is not None:\n            all_coords.append(coords)\n    \n    if not all_coords :\n        print(f\"No valid C1' coordinates found for {target_id}\")\n        \n     # Ensure we have 5 models (if we have fewer, duplicate the last one)\n    while len(all_coords) < 5:\n        print(f\"Only {len(all_coords)} models found for {target_id}, duplicating last model\")\n        all_coords.append(all_coords[-1])\n    \n    return all_coords[:5]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T00:54:20.086966Z","iopub.execute_input":"2025-05-12T00:54:20.087260Z","iopub.status.idle":"2025-05-12T00:54:20.101788Z","shell.execute_reply.started":"2025-05-12T00:54:20.087232Z","shell.execute_reply":"2025-05-12T00:54:20.101159Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission file 생성 \ndef create_submission(test_sequences_df, c1_coords_dict, output_file):\n    \"\"\"\n    Create the submission CSV file with C1' coordinates\n    \"\"\"\n    rows = []\n    \n    # Process each sequence\n    for _, row in test_sequences_df.iterrows():\n        target_id = row['target_id']\n        sequence = row['sequence']\n        \n        if target_id not in c1_coords_dict or c1_coords_dict[target_id] is None:\n            print(f\"No prediction found for {target_id}, using zeros\")\n            # Create empty predictions (all zeros)\n            for i, residue in enumerate(sequence):\n                row_data = {\n                    'ID': f\"{target_id}_{i+1}\",\n                    'resname': residue,\n                    'resid': i+1\n                }\n                for model in range(1, 6):\n                    row_data[f'x_{model}'] = 0.0\n                    row_data[f'y_{model}'] = 0.0\n                    row_data[f'z_{model}'] = 0.0\n                rows.append(row_data)\n        else:\n            # Get the 5 models for this target\n            models = c1_coords_dict[target_id]\n            \n            # Create a row for each residue\n            for i, residue in enumerate(sequence):\n                row_data = {\n                    'ID': f\"{target_id}_{i+1}\",\n                    'resname': residue,\n                    'resid': i+1\n                }\n                \n                # Add coordinates for each model\n                for model_idx in range(5):\n                    if model_idx < len(models) and i < len(models[model_idx]):\n                        row_data[f'x_{model_idx+1}'] = models[model_idx][i][0]\n                        row_data[f'y_{model_idx+1}'] = models[model_idx][i][1]\n                        row_data[f'z_{model_idx+1}'] = models[model_idx][i][2]\n                    else:\n                        # If coordinates are not available, use zeros\n                        row_data[f'x_{model_idx+1}'] = 0.0\n                        row_data[f'y_{model_idx+1}'] = 0.0\n                        row_data[f'z_{model_idx+1}'] = 0.0\n                \n                rows.append(row_data)\n    \n    # Create DataFrame and save to CSV\n    df = pd.DataFrame(rows)\n    df.to_csv(output_file, index=False)\n    print(f\"Created submission file: {output_file}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T00:54:20.102628Z","iopub.execute_input":"2025-05-12T00:54:20.102846Z","iopub.status.idle":"2025-05-12T00:54:20.118614Z","shell.execute_reply.started":"2025-05-12T00:54:20.102828Z","shell.execute_reply":"2025-05-12T00:54:20.117947Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def main():\n    \"\"\"\n    Main function\n    \"\"\"\n    input_dir = \"/kaggle/working/input_dir\"\n    output_dir = \"/kaggle/output_dir\"\n    \n    os.makedirs(input_dir, exist_ok=True)\n    os.makedirs(output_dir, exist_ok=True)\n    \n    # Read test sequences\n    test_sequences_df = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/test_sequences.csv\")\n    print(f\"Loaded {len(test_sequences_df)} test sequences\")\n\n    # Process each sequence\n    c1_coords_dict = {}\n    for _, row in tqdm(test_sequences_df.iterrows(), total=len(test_sequences_df)):\n        \n        target_id = row['target_id']\n        sequence = row['sequence']\n        \n        # Check if we already have predictions for this target\n        target_prediction_dir = os.path.join(output_dir, target_id)\n        \n        if os.path.exists(target_prediction_dir):\n            print(f\"Found existing prediction for {target_id}, loading coordinates\")\n            \n            # Extract coordinates from existing predictions\n            pdb_files = sorted(glob.glob(os.path.join(target_prediction_dir , f\"{target_id}_*.pdb\")))\n            \n            all_coords = []\n            \n            for pdb_file in pdb_files:\n                coords = extract_c1_coordinates(pdb_file)\n                if coords is not None:\n                    all_coords.append(coords)\n            \n            if all_coords:\n                # Ensure we have 5 models\n                while len(all_coords) < 5:\n                    all_coords.append(all_coords[-1])\n                c1_coords_dict[target_id] = all_coords[:5]\n                continue\n        \n        # Process the sequence if no existing prediction was found or was invalid\n        c1_coords = process_sequence(sequence, target_id, input_dir, output_dir)\n        c1_coords_dict[target_id] = c1_coords\n    \n    # Create submission file\n    create_submission(test_sequences_df, c1_coords_dict, \"submission.csv\")\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T00:54:20.119439Z","iopub.execute_input":"2025-05-12T00:54:20.119711Z","iopub.status.idle":"2025-05-12T00:54:21.816842Z","shell.execute_reply.started":"2025-05-12T00:54:20.119683Z","shell.execute_reply":"2025-05-12T00:54:21.816000Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.read_csv(\"submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T00:54:21.819226Z","iopub.execute_input":"2025-05-12T00:54:21.819521Z","iopub.status.idle":"2025-05-12T00:54:21.848909Z","shell.execute_reply.started":"2025-05-12T00:54:21.819490Z","shell.execute_reply":"2025-05-12T00:54:21.848186Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.read_csv(\"submission.csv\")\nsubmission.to_csv('submission.csv', index=False)\nsubmission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T00:54:21.849772Z","iopub.execute_input":"2025-05-12T00:54:21.850063Z","iopub.status.idle":"2025-05-12T00:54:21.904934Z","shell.execute_reply.started":"2025-05-12T00:54:21.850029Z","shell.execute_reply":"2025-05-12T00:54:21.903998Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generate minimal CSV with predicted structure file paths\noutput_dir = \"/kaggle/output_dir\"\ntest_sequences_df = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/test_sequences.csv\")\n\noutput_data = []\n\nfor _, row in test_sequences_df.iterrows():\n    target_id = row[\"target_id\"]\n    sequence = row[\"sequence\"]\n    pdb_path = f\"{output_dir}/{target_id}/{target_id}_1.pdb\"  # change to _1.pdb, since that's what's generated\n    if os.path.exists(pdb_path):\n        output_data.append({\n            \"target_id\": target_id,\n            \"sequence\": sequence,\n            \"predicted_model_path\": pdb_path\n        })\n    else:\n        print(f\"Missing prediction for {target_id}\")\n\ndf_proteinx = pd.DataFrame(output_data)\ndf_proteinx.to_csv(\"submission_proteinx.csv\", index=False)\ndf_proteinx.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T00:54:21.905771Z","iopub.execute_input":"2025-05-12T00:54:21.906041Z","iopub.status.idle":"2025-05-12T00:54:21.920929Z","shell.execute_reply.started":"2025-05-12T00:54:21.906007Z","shell.execute_reply":"2025-05-12T00:54:21.919955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ndf = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_sequences.csv\")\ndf.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T00:54:21.921941Z","iopub.execute_input":"2025-05-12T00:54:21.922231Z","iopub.status.idle":"2025-05-12T00:54:21.959594Z","shell.execute_reply.started":"2025-05-12T00:54:21.922202Z","shell.execute_reply":"2025-05-12T00:54:21.958793Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport subprocess\nfrom tqdm import tqdm\n\n# Paths\ninput_csv = \"/kaggle/input/stanford-rna-3d-folding/train_sequences.csv\"\ninput_dir = \"/kaggle/working/input_dir\"\noutput_base = \"/kaggle/working/nufold_preds\"\nos.makedirs(input_dir, exist_ok=True)\nos.makedirs(output_base, exist_ok=True)\n\n# Load first 5 sequences\ndf = pd.read_csv(input_csv)\nsequences = df[[\"target_id\", \"sequence\"]].values.tolist()[:5]\n\n# Output list\noutput_rows = []\n\n# Main loop: 3 seeds per target\nfor target_id, seq in tqdm(sequences):\n    for seed in [1, 2, 3]:\n        run_id = f\"{target_id}_seed{seed}\"\n        run_input_dir = os.path.join(input_dir, run_id)\n        run_output_dir = os.path.join(output_base, run_id)\n        os.makedirs(run_input_dir, exist_ok=True)\n        os.makedirs(run_output_dir, exist_ok=True)\n\n        fasta_path = os.path.join(run_input_dir, f\"{run_id}.fasta\")\n        with open(fasta_path, \"w\") as f:\n            f.write(f\">{run_id}\\n{seq}\\n\")\n\n        pdb_out_path = os.path.join(run_output_dir, f\"{run_id}.pdb\")\n\n        cmd = [\n            \"python3\", \"/kaggle/input/nufold-kaggle-dataset/NuFold/run_nufold.py\",\n            \"--ckpt_path\", \"/kaggle/input/nufold-kaggle-dataset/checkpoints/global_step145245.pt\",\n            \"--input_fasta\", fasta_path,\n            \"--input_dir\", run_input_dir,\n            \"--output_dir\", run_output_dir,\n            \"--config_preset\", \"initial_training\"\n        ]\n\n        # Set seed via environment variable\n        env = os.environ.copy()\n        env[\"NUFOLD_RANDOM_SEED\"] = str(seed)\n\n        print(f\"Running {run_id}...\")\n        try:\n            result = subprocess.run(cmd, capture_output=True, text=True, env=env)\n            if result.returncode != 0:\n                print(f\"Error for {run_id}: {result.stderr}\")\n                continue\n        except Exception as e:\n            print(f\"Exception for {run_id}: {e}\")\n            continue\n\n        # Record output path\n        output_rows.append({\n            \"sequence\": seq,\n            \"predicted_model_path\": pdb_out_path,\n            \"target_id\": run_id\n        })\n\n# Save test output CSV\ndf_out = pd.DataFrame(output_rows)\ndf_out.to_csv(\"nufold_training_output_TEST.csv\", index=False)\nprint(\"✅ Test run complete! Saved nufold_training_output_TEST.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T00:54:21.960412Z","iopub.execute_input":"2025-05-12T00:54:21.960719Z","iopub.status.idle":"2025-05-12T00:59:16.735903Z","shell.execute_reply.started":"2025-05-12T00:54:21.960686Z","shell.execute_reply":"2025-05-12T00:59:16.734955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport subprocess\nfrom tqdm import tqdm\n\n# Paths\ninput_csv = \"/kaggle/input/stanford-rna-3d-folding/train_sequences.csv\"\ninput_dir = \"/kaggle/working/input_dir\"\noutput_base = \"/kaggle/working/nufold_preds\"\nos.makedirs(input_dir, exist_ok=True)\nos.makedirs(output_base, exist_ok=True)\n\n# Load full dataset and slice chunk 5–154 (150 entries)\ndf = pd.read_csv(input_csv)\nsequences = df[[\"target_id\", \"sequence\"]].values.tolist()[5:80]\n\n# Output list\noutput_rows = []\n\n# Loop over sequences and seeds\nfor target_id, seq in tqdm(sequences):\n    for seed in [1, 2, 3]:\n        run_id = f\"{target_id}_seed{seed}\"\n        run_input_dir = os.path.join(input_dir, run_id)\n        run_output_dir = os.path.join(output_base, run_id)\n        os.makedirs(run_input_dir, exist_ok=True)\n        os.makedirs(run_output_dir, exist_ok=True)\n\n        fasta_path = os.path.join(run_input_dir, f\"{run_id}.fasta\")\n        with open(fasta_path, \"w\") as f:\n            f.write(f\">{run_id}\\n{seq}\\n\")\n\n        pdb_out_path = os.path.join(run_output_dir, f\"{run_id}.pdb\")\n\n        cmd = [\n            \"python3\", \"/kaggle/input/nufold-kaggle-dataset/NuFold/run_nufold.py\",\n            \"--ckpt_path\", \"/kaggle/input/nufold-kaggle-dataset/checkpoints/global_step145245.pt\",\n            \"--input_fasta\", fasta_path,\n            \"--input_dir\", run_input_dir,\n            \"--output_dir\", run_output_dir,\n            \"--config_preset\", \"initial_training\"\n        ]\n\n        env = os.environ.copy()\n        env[\"NUFOLD_RANDOM_SEED\"] = str(seed)\n\n        print(f\"Running {run_id}...\")\n        try:\n            result = subprocess.run(cmd, capture_output=True, text=True, env=env)\n            if result.returncode != 0:\n                print(f\"Error for {run_id}: {result.stderr}\")\n                continue\n        except Exception as e:\n            print(f\"Exception for {run_id}: {e}\")\n            continue\n\n        output_rows.append({\n            \"sequence\": seq,\n            \"predicted_model_path\": pdb_out_path,\n            \"target_id\": run_id\n        })\n\n# Save chunk to CSV\ndf_out = pd.DataFrame(output_rows)\ndf_out.to_csv(\"nufold_training_output_chunk2.csv\", index=False)\nprint(\"✅ Chunk 2 complete! Saved nufold_training_output_chunk2.csv\")\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-05-12T02:25:27.086Z"}},"outputs":[],"execution_count":null}]}