{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":12276181,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":11938779,"sourceType":"datasetVersion","datasetId":7505879}],"dockerImageVersionId":31040,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# heyy","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q /kaggle/input/rna-wheels/wheels/biopython-1.85-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T01:00:02.813321Z","iopub.execute_input":"2025-05-27T01:00:02.813689Z","iopub.status.idle":"2025-05-27T01:00:11.079611Z","shell.execute_reply.started":"2025-05-27T01:00:02.813663Z","shell.execute_reply":"2025-05-27T01:00:11.077836Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls \"/kaggle/input/stanford-rna-3d-folding\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T01:00:11.082181Z","iopub.execute_input":"2025-05-27T01:00:11.083252Z","iopub.status.idle":"2025-05-27T01:00:11.224637Z","shell.execute_reply.started":"2025-05-27T01:00:11.083193Z","shell.execute_reply":"2025-05-27T01:00:11.222873Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom tqdm import tqdm\nfrom collections import Counter\n\nfiles = os.listdir(\"/kaggle/input/stanford-rna-3d-folding/PDB_RNA\")\nprint(f\"Total number of files: {len(files)}\")\n\n# Get file extensions\nextensions = [os.path.splitext(file)[1] for file in files if os.path.splitext(file)[1]]\n\n# Count occurrences of each extension\nextension_counts = Counter(extensions)\n\nprint(f\"File types and their counts:\")\nfor ext, count in sorted(extension_counts.items()):\n    print(f\"  {ext}: {count}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T01:00:25.042981Z","iopub.execute_input":"2025-05-27T01:00:25.043462Z","iopub.status.idle":"2025-05-27T01:00:25.165674Z","shell.execute_reply.started":"2025-05-27T01:00:25.043423Z","shell.execute_reply":"2025-05-27T01:00:25.164607Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\n\n# Find the CSV file name\nfiles = os.listdir(\"/kaggle/input/stanford-rna-3d-folding/PDB_RNA\")\ncsv_files = [f for f in files if f.endswith('.csv')]\nprint(f\"CSV file(s): {csv_files}\")\n\n# Read the CSV file\ncsv_file_path = f\"/kaggle/input/stanford-rna-3d-folding/PDB_RNA/{csv_files[0]}\"\n\n# Skip the problematic header lines\ndf = pd.read_csv(csv_file_path, on_bad_lines='skip')\nprint(f\"CSV file shape: {df.shape}\")\n\nprint(f\"Column names: {list(df.columns)}\")\nprint(df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T01:00:25.406949Z","iopub.execute_input":"2025-05-27T01:00:25.407484Z","iopub.status.idle":"2025-05-27T01:00:25.900611Z","shell.execute_reply.started":"2025-05-27T01:00:25.407451Z","shell.execute_reply":"2025-05-27T01:00:25.898544Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check unique counts for Entry ID\nprint(f\"Total rows: {len(df)}\")\nprint(f\"Unique Entry IDs: {df['Entry ID'].nunique()}\")\nprint(f\"Duplicate Entry IDs: {len(df) - df['Entry ID'].nunique()}\")\n\n# Convert Release Date to datetime for better analysis\ndf['Release Date'] = pd.to_datetime(df['Release Date'])\n\n# Get date range\nprint(f\"\\nRelease Date range:\")\nprint(f\"Earliest date: {df['Release Date'].min()}\")\nprint(f\"Latest date: {df['Release Date'].max()}\")\nprint(f\"Date span: {(df['Release Date'].max() - df['Release Date'].min()).days} days\")\n\n# Check unique release dates\nprint(f\"\\nUnique Release Dates: {df['Release Date'].nunique()}\")\n\n# Show some examples of duplicates if any exist\nif len(df) > df['Entry ID'].nunique():\n    print(f\"\\nExamples of duplicate Entry IDs:\")\n    duplicates = df[df['Entry ID'].duplicated(keep=False)].sort_values('Entry ID')\n    print(duplicates.head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T01:00:27.557780Z","iopub.execute_input":"2025-05-27T01:00:27.558850Z","iopub.status.idle":"2025-05-27T01:00:27.599694Z","shell.execute_reply.started":"2025-05-27T01:00:27.558808Z","shell.execute_reply":"2025-05-27T01:00:27.598482Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## FASTA:  \n---","metadata":{}},{"cell_type":"code","source":"import os\n\n# Find the FASTA file\nfiles = os.listdir(\"/kaggle/input/stanford-rna-3d-folding/PDB_RNA\")\nfasta_files = [f for f in files if f.endswith('.fasta')]\nprint(f\"FASTA file(s): {fasta_files}\")\n\n# Read and examine the FASTA file\nfasta_file_path = f\"/kaggle/input/stanford-rna-3d-folding/PDB_RNA/{fasta_files[0]}\"\n\n# Read the file and look at its structure\nwith open(fasta_file_path, 'r') as f:\n    lines = f.readlines()\n\nprint(f\"Total lines in FASTA file: {len(lines)}\")\nprint(f\"\\nFirst 20 lines:\")\nfor i, line in enumerate(lines[:20]):\n    print(f\"Line {i+1}: {repr(line)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T01:00:29.977386Z","iopub.execute_input":"2025-05-27T01:00:29.977839Z","iopub.status.idle":"2025-05-27T01:00:30.224202Z","shell.execute_reply.started":"2025-05-27T01:00:29.977812Z","shell.execute_reply":"2025-05-27T01:00:30.222940Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Parse FASTA file to count sequences\nsequences = []\ncurrent_seq = \"\"\nheaders = []\n\nwith open(fasta_file_path, 'r') as f:\n    for line in f:\n        line = line.strip()\n        if line.startswith('>'):  # Header line\n            if current_seq:  # Save previous sequence\n                sequences.append(current_seq)\n                current_seq = \"\"\n            headers.append(line)\n        else:  # Sequence line\n            current_seq += line\n    \n    # Don't forget the last sequence\n    if current_seq:\n        sequences.append(current_seq)\n\nprint(f\"\\nFASTA file summary:\")\nprint(f\"Number of sequences: {len(sequences)}\")\nprint(f\"Number of headers: {len(headers)}\")\n\nif sequences:\n    seq_lengths = [len(seq) for seq in sequences]\n    print(f\"Sequence lengths - Min: {min(seq_lengths)}, Max: {max(seq_lengths)}, Average: {sum(seq_lengths)/len(seq_lengths):.1f}\")\n    \n    print(f\"\\nFirst few headers:\")\n    for i, header in enumerate(headers[:5]):\n        print(f\"  {header}\")\n    \n    print(f\"\\nFirst sequence (first 100 chars):\")\n    print(f\"  {sequences[0][:100]}...\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T01:00:30.227861Z","iopub.execute_input":"2025-05-27T01:00:30.228481Z","iopub.status.idle":"2025-05-27T01:00:30.364485Z","shell.execute_reply.started":"2025-05-27T01:00:30.228443Z","shell.execute_reply":"2025-05-27T01:00:30.363339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Continue with the parsing code from before to see the complete statistics\nprint(f\"\\nFASTA file summary:\")\nprint(f\"Number of sequences: {len(sequences)}\")\nprint(f\"Sequence lengths - Min: {min(seq_lengths)}, Max: {max(seq_lengths)}\")\n\n# Count DNA vs RNA vs mixed\ndna_count = sum(1 for h in headers if 'DNA' in h and 'RNA' not in h)\nrna_count = sum(1 for h in headers if 'RNA' in h and 'DNA' not in h)\nmixed_count = sum(1 for h in headers if 'DNA' in h and 'RNA' in h)\nprint(f\"DNA sequences: {dna_count}\")\nprint(f\"RNA sequences: {rna_count}\")  \nprint(f\"Mixed DNA/RNA: {mixed_count}\")\n\n# Count unique PDB IDs\npdb_ids = [h.split('_')[0][1:] for h in headers]  # Remove '>' and chain part\nunique_pdbs = len(set(pdb_ids))\nprint(f\"Unique PDB structures: {unique_pdbs}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T01:00:32.582616Z","iopub.execute_input":"2025-05-27T01:00:32.583034Z","iopub.status.idle":"2025-05-27T01:00:32.644255Z","shell.execute_reply.started":"2025-05-27T01:00:32.583006Z","shell.execute_reply":"2025-05-27T01:00:32.642933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count RNA sequences with less than 1000 nucleotides\nrna_short_count = 0\nrna_lengths = []\n\nfor i, header in enumerate(headers):\n   if 'RNA' in header and 'DNA' not in header:  # Pure RNA sequences only\n       seq_length = len(sequences[i])\n       rna_lengths.append(seq_length)\n       if seq_length < 1000:\n           rna_short_count += 1\n\nprint(f\"RNA sequences with less than 1000 nucleotides: {rna_short_count}\")\nprint(f\"Total RNA sequences: {len(rna_lengths)}\")\nprint(f\"Percentage of RNA sequences < 1000 nt: {rna_short_count/len(rna_lengths)*100:.1f}%\")\n\n# Additional statistics for RNA sequences\nif rna_lengths:\n   print(f\"\\nRNA sequence length statistics:\")\n   print(f\"Min length: {min(rna_lengths)}\")\n   print(f\"Max length: {max(rna_lengths)}\")\n   print(f\"Average length: {sum(rna_lengths)/len(rna_lengths):.1f}\")\n   \n   # Length distribution\n   length_ranges = [\n       (0, 100, \"1-100\"),\n       (100, 500, \"100-500\"), \n       (500, 1000, \"500-1000\"),\n       (1000, 2000, \"1000-2000\"),\n       (2000, float('inf'), \"2000+\")\n   ]\n   \n   print(f\"\\nRNA sequence length distribution:\")\n   for min_len, max_len, label in length_ranges:\n       count = sum(1 for length in rna_lengths if min_len < length <= max_len)\n       print(f\"  {label} nt: {count}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T01:00:35.023446Z","iopub.execute_input":"2025-05-27T01:00:35.023845Z","iopub.status.idle":"2025-05-27T01:00:35.063265Z","shell.execute_reply.started":"2025-05-27T01:00:35.023818Z","shell.execute_reply":"2025-05-27T01:00:35.061929Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train Sequences:  \n---","metadata":{}},{"cell_type":"code","source":"train_sequences = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_sequences.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T01:00:38.331965Z","iopub.execute_input":"2025-05-27T01:00:38.332399Z","iopub.status.idle":"2025-05-27T01:00:38.446182Z","shell.execute_reply.started":"2025-05-27T01:00:38.332375Z","shell.execute_reply":"2025-05-27T01:00:38.445162Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sequences.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T01:00:38.578197Z","iopub.execute_input":"2025-05-27T01:00:38.578590Z","iopub.status.idle":"2025-05-27T01:00:38.604212Z","shell.execute_reply.started":"2025-05-27T01:00:38.578563Z","shell.execute_reply":"2025-05-27T01:00:38.602828Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sequences_v2 = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_sequences.v2.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T01:00:38.806927Z","iopub.execute_input":"2025-05-27T01:00:38.807556Z","iopub.status.idle":"2025-05-27T01:00:39.961878Z","shell.execute_reply.started":"2025-05-27T01:00:38.807522Z","shell.execute_reply":"2025-05-27T01:00:39.960541Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sequences_v2.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T01:00:40.532953Z","iopub.execute_input":"2025-05-27T01:00:40.533425Z","iopub.status.idle":"2025-05-27T01:00:40.560692Z","shell.execute_reply.started":"2025-05-27T01:00:40.533396Z","shell.execute_reply":"2025-05-27T01:00:40.559311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check for target_ids in v1 that are not in v2\nv1_only_targets = set(train_sequences['target_id']) - set(train_sequences_v2['target_id'])\nprint(f\"Target IDs in v1 but not in v2: {len(v1_only_targets)}\")\nprint(f\"First 10: {list(v1_only_targets)[:10]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T01:00:40.798828Z","iopub.execute_input":"2025-05-27T01:00:40.799265Z","iopub.status.idle":"2025-05-27T01:00:40.808372Z","shell.execute_reply.started":"2025-05-27T01:00:40.799225Z","shell.execute_reply":"2025-05-27T01:00:40.807174Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_labels = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_labels.csv\")\ntrain_labels_v2 = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_labels.v2.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T01:00:40.989426Z","iopub.execute_input":"2025-05-27T01:00:40.989827Z","iopub.status.idle":"2025-05-27T01:00:49.574302Z","shell.execute_reply.started":"2025-05-27T01:00:40.989799Z","shell.execute_reply":"2025-05-27T01:00:49.572663Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_labels.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T01:00:49.576394Z","iopub.execute_input":"2025-05-27T01:00:49.576815Z","iopub.status.idle":"2025-05-27T01:00:49.591492Z","shell.execute_reply.started":"2025-05-27T01:00:49.576789Z","shell.execute_reply":"2025-05-27T01:00:49.589935Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_labels_v2.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T01:00:49.592628Z","iopub.execute_input":"2025-05-27T01:00:49.593010Z","iopub.status.idle":"2025-05-27T01:00:49.621240Z","shell.execute_reply.started":"2025-05-27T01:00:49.592977Z","shell.execute_reply":"2025-05-27T01:00:49.620201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extract target_ids from labels by removing the residue suffix\nv1_label_targets = set(train_labels['ID'].str.rsplit('_', n=1).str[0])\nv2_label_targets = set(train_labels_v2['ID'].str.rsplit('_', n=1).str[0])\n\n# Check for target_ids in v1 labels that are not in v2 labels\nv1_only_label_targets = v1_label_targets - v2_label_targets\nprint(f\"Target IDs in v1 labels but not in v2 labels: {len(v1_only_label_targets)}\")\nprint(f\"First 10: {list(v1_only_label_targets)[:10]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T01:00:49.623208Z","iopub.execute_input":"2025-05-27T01:00:49.623670Z","iopub.status.idle":"2025-05-27T01:00:58.005392Z","shell.execute_reply.started":"2025-05-27T01:00:49.623638Z","shell.execute_reply":"2025-05-27T01:00:58.004191Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Concatenate both sequence dataframes\ncombined_sequences = pd.concat([train_sequences, train_sequences_v2], ignore_index=True)\n\n# Concatenate both label dataframes  \ncombined_labels = pd.concat([train_labels, train_labels_v2], ignore_index=True)\n\nprint(f\"Combined sequences shape: {combined_sequences.shape}\")\nprint(f\"Combined labels shape: {combined_labels.shape}\")\nprint(f\"Unique target_ids in combined sequences: {combined_sequences['target_id'].nunique()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T01:00:58.006350Z","iopub.execute_input":"2025-05-27T01:00:58.006597Z","iopub.status.idle":"2025-05-27T01:00:58.426248Z","shell.execute_reply.started":"2025-05-27T01:00:58.006577Z","shell.execute_reply":"2025-05-27T01:00:58.425298Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"combined_sequences.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T01:00:58.427252Z","iopub.execute_input":"2025-05-27T01:00:58.427506Z","iopub.status.idle":"2025-05-27T01:00:58.438877Z","shell.execute_reply.started":"2025-05-27T01:00:58.427485Z","shell.execute_reply":"2025-05-27T01:00:58.437590Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extract target_ids from FASTA headers for RNA sequences only\nrna_headers = [h for h in headers if 'RNA' in h and 'DNA' not in h]\nfasta_rna_targets = set()\n\nfor header in rna_headers:\n    # Extract PDB_ID and chain from header like \">1SCL_A mol:na...\"\n    target_part = header.split()[0][1:]  # Remove '>' and take first part\n    fasta_rna_targets.add(target_part)\n\nprint(f\"RNA targets in FASTA: {len(fasta_rna_targets)}\")\n\n# Check overlap with combined sequences\ncombined_targets = set(combined_sequences['target_id'])\noverlap = combined_targets.intersection(fasta_rna_targets)\nmissing_in_fasta = combined_targets - fasta_rna_targets\n\nprint(f\"Combined sequence targets: {len(combined_targets)}\")\nprint(f\"Overlap (targets in both): {len(overlap)}\")\nprint(f\"Missing in FASTA: {len(missing_in_fasta)}\")\nprint(f\"Coverage: {len(overlap)/len(combined_targets)*100:.1f}%\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T01:00:58.439918Z","iopub.execute_input":"2025-05-27T01:00:58.440393Z","iopub.status.idle":"2025-05-27T01:00:58.484987Z","shell.execute_reply.started":"2025-05-27T01:00:58.440368Z","shell.execute_reply":"2025-05-27T01:00:58.483547Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check unique target_id counts in combined sequences\ntarget_counts = combined_sequences['target_id'].value_counts()\nprint(f\"Total unique target_ids: {len(target_counts)}\")\nprint(f\"Target_ids appearing more than once: {(target_counts > 1).sum()}\")\nprint(f\"\\nFirst 10 target_ids and their counts:\")\nprint(target_counts.head(10))\n\nprint(f\"\\nSample target_ids from combined_sequences:\")\nprint(combined_sequences['target_id'].head(10).tolist())\n\nprint(f\"\\nSample target_ids from FASTA RNA headers:\")\nsample_fasta_targets = list(fasta_rna_targets)[:10]\nprint(sample_fasta_targets)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T01:00:58.486250Z","iopub.execute_input":"2025-05-27T01:00:58.486606Z","iopub.status.idle":"2025-05-27T01:00:58.503260Z","shell.execute_reply.started":"2025-05-27T01:00:58.486575Z","shell.execute_reply":"2025-05-27T01:00:58.502142Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Compare the naming patterns more clearly\nprint(\"Combined sequences target_id pattern:\")\nprint(\"Format appears to be: [4-char PDB]_[1-2 char chain]\")\nfor target in combined_sequences['target_id'].head(5):\n    print(f\"  {target}\")\n\nprint(f\"\\nFASTA RNA target_id pattern:\")\nprint(\"Format appears to be: [4-char PDB]_[1-2 char chain]\")\nfor target in list(fasta_rna_targets)[:5]:\n    print(f\"  {target}\")\n\n# Check if there's any case sensitivity or format difference\ncombined_lower = set(t.lower() for t in combined_sequences['target_id'])\nfasta_lower = set(t.lower() for t in fasta_rna_targets)\noverlap_lower = combined_lower.intersection(fasta_lower)\n\nprint(f\"\\nCase-insensitive check:\")\nprint(f\"Overlap when ignoring case: {len(overlap_lower)}\")\n\n# Check PDB codes only (without chain)\ncombined_pdbs = set(t.split('_')[0] for t in combined_sequences['target_id'])\nfasta_pdbs = set(t.split('_')[0] for t in fasta_rna_targets)\npdb_overlap = combined_pdbs.intersection(fasta_pdbs)\n\nprint(f\"\\nPDB code overlap (ignoring chains):\")\nprint(f\"Combined PDB codes: {len(combined_pdbs)}\")\nprint(f\"FASTA PDB codes: {len(fasta_pdbs)}\")\nprint(f\"PDB overlap: {len(pdb_overlap)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T01:00:58.504497Z","iopub.execute_input":"2025-05-27T01:00:58.504802Z","iopub.status.idle":"2025-05-27T01:00:58.563013Z","shell.execute_reply.started":"2025-05-27T01:00:58.504774Z","shell.execute_reply":"2025-05-27T01:00:58.561355Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# The case-insensitive overlap suggests the issue is case sensitivity\n# Let's examine this more closely\n\nprint(\"Case comparison examples:\")\ncombined_sample = list(combined_sequences['target_id'])[:5]\nfasta_sample = list(fasta_rna_targets)[:5]\n\nfor target in combined_sample:\n    print(f\"Combined: {target} -> lowercase: {target.lower()}\")\n\nfor target in fasta_sample:\n    print(f\"FASTA: {target} -> lowercase: {target.lower()}\")\n\n# Check if FASTA uses lowercase PDB codes\nprint(f\"\\nPDB code case analysis:\")\ncombined_pdb_sample = [t.split('_')[0] for t in combined_sample]\nfasta_pdb_sample = [t.split('_')[0] for t in fasta_sample]\n\nprint(\"Combined PDB codes:\", combined_pdb_sample)\nprint(\"FASTA PDB codes:\", fasta_pdb_sample)\n\n# Check PDB overlap with case-insensitive comparison\ncombined_pdbs_lower = set(t.split('_')[0].lower() for t in combined_sequences['target_id'])\nfasta_pdbs_lower = set(t.split('_')[0].lower() for t in fasta_rna_targets)\npdb_overlap_lower = combined_pdbs_lower.intersection(fasta_pdbs_lower)\n\nprint(f\"\\nCase-insensitive PDB overlap: {len(pdb_overlap_lower)}\")\nprint(f\"This explains the discrepancy - FASTA uses lowercase PDB codes!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T01:00:58.566464Z","iopub.execute_input":"2025-05-27T01:00:58.566874Z","iopub.status.idle":"2025-05-27T01:00:58.604453Z","shell.execute_reply.started":"2025-05-27T01:00:58.566848Z","shell.execute_reply":"2025-05-27T01:00:58.603269Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## `cif` files  \n---","metadata":{}},{"cell_type":"code","source":"# !ls /kaggle/input/stanford-rna-3d-folding/PDB_RNA","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T01:01:06.055760Z","iopub.execute_input":"2025-05-27T01:01:06.056177Z","iopub.status.idle":"2025-05-27T01:01:06.264705Z","shell.execute_reply.started":"2025-05-27T01:01:06.056149Z","shell.execute_reply":"2025-05-27T01:01:06.262909Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extract RNA sequences and coordinates with deduplication\nfrom Bio.PDB import MMCIFParser\nimport pandas as pd\nfrom pathlib import Path\n\ndef extract_rna_data_from_cif(cif_file_path):\n    \"\"\"Extract unique RNA sequences and C1' coordinates from a CIF file\"\"\"\n    parser = MMCIFParser(QUIET=True)\n    \n    try:\n        structure = parser.get_structure('structure', cif_file_path)\n        pdb_id = Path(cif_file_path).stem.upper()\n        \n        sequences_data = []\n        coordinates_data = []\n        seen_sequences = set()  # Track unique sequences\n        \n        for model in structure:\n            for chain in model:\n                chain_id = chain.id\n                target_id = f\"{pdb_id}_{chain_id}\"\n                \n                # Check if chain contains RNA residues\n                rna_residues = []\n                for residue in chain:\n                    if residue.get_resname() in ['A', 'U', 'G', 'C']:  # RNA nucleotides\n                        rna_residues.append(residue)\n                \n                if rna_residues:  # Only process if RNA residues found\n                    # Build sequence\n                    sequence = ''.join([res.get_resname() for res in rna_residues])\n                    \n                    # Only add if sequence is unique\n                    if sequence not in seen_sequences:\n                        seen_sequences.add(sequence)\n                        sequences_data.append({\n                            'target_id': target_id,\n                            'sequence': sequence\n                        })\n                        \n                        # Extract C1' coordinates for this unique sequence\n                        for i, residue in enumerate(rna_residues, 1):\n                            if \"C1'\" in residue:\n                                atom = residue[\"C1'\"]\n                                coordinates_data.append({\n                                    'ID': f\"{target_id}_{i}\",\n                                    'resname': residue.get_resname(),\n                                    'resid': i,\n                                    'x_1': atom.coord[0],\n                                    'y_1': atom.coord[1], \n                                    'z_1': atom.coord[2]\n                                })\n        \n        return sequences_data, coordinates_data\n        \n    except Exception as e:\n        print(f\"Error processing {cif_file_path}: {e}\")\n        return [], []","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T01:08:02.620679Z","iopub.execute_input":"2025-05-27T01:08:02.622474Z","iopub.status.idle":"2025-05-27T01:08:49.018000Z","shell.execute_reply.started":"2025-05-27T01:08:02.622422Z","shell.execute_reply":"2025-05-27T01:08:49.016846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Process all CIF files and save to CSV\nimport os\nfrom tqdm import tqdm\n\ndef process_all_cif_files():\n    \"\"\"Process all CIF files in the directory and extract RNA data\"\"\"\n    cif_dir = \"/kaggle/input/stanford-rna-3d-folding/PDB_RNA\"\n    cif_files = [f for f in os.listdir(cif_dir) if f.endswith('.cif')]\n    \n    all_sequences = []\n    all_coordinates = []\n    \n    print(f\"Processing {len(cif_files)} CIF files...\")\n    \n    for cif_file in tqdm(cif_files):\n        cif_path = os.path.join(cif_dir, cif_file)\n        sequences, coordinates = extract_rna_data_from_cif(cif_path)\n        \n        all_sequences.extend(sequences)\n        all_coordinates.extend(coordinates)\n    \n    return all_sequences, all_coordinates\n\n# Process all files\nprint(\"Starting full extraction...\")\nall_sequences, all_coordinates = process_all_cif_files()\n\nprint(f\"\\nFull extraction summary:\")\nprint(f\"Total unique RNA sequences: {len(all_sequences)}\")\nprint(f\"Total coordinate entries: {len(all_coordinates)}\")\n\n# Create DataFrames\nsequences_df = pd.DataFrame(all_sequences)\ncoordinates_df = pd.DataFrame(all_coordinates)\n\n# Save to CSV files\nsequences_df.to_csv('rna_sequences.csv', index=False)\ncoordinates_df.to_csv('rna_coordinates.csv', index=False)\n\nprint(f\"\\nSaved files:\")\nprint(f\"rna_sequences.csv: {sequences_df.shape}\")\nprint(f\"rna_coordinates.csv: {coordinates_df.shape}\")\n\nprint(f\"\\nFirst few entries:\")\nprint(sequences_df.head())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# MSA     - 856\n# MSA_v2  - 2534\n# PDB_RNA - 8672","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}