{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11553390,"sourceType":"competition"},{"sourceId":10933126,"sourceType":"datasetVersion","datasetId":6798383}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Setup\n\nThis notebook performs an exploratory data analysis (EDA) on the Stanford RNA 3D Folding dataset. The datasets include training, validation, and test RNA sequences along with corresponding per-residue label data and a sample submission template.\n\n**Key Points:**\n- **Data Overview:**  \n  - The `train_sequences.csv` file contains 844 entries with basic information about each RNA target.  \n  - The `train_labels.csv` file provides per-residue coordinate data (with some missing values) for 137,095 rows corresponding to 735 unique targets.  \n  - The validation and test sequence files each have 12 entries.  \n  - The sample submission file is provided as a template with coordinate columns set to zero.\n\n- **Sequence Analysis:**  \n  - RNA sequence lengths vary widely (from as few as 3 nucleotides up to 4298 nucleotides).  \n  - Base composition analysis reveals that the sequences are predominantly composed of G, C, A, and U. Note that extra characters such as '-' and 'X' appear in the training sequences.\n\n- **Label Data Analysis:**  \n  - Coordinate columns in the label files have meaningful ranges and summary statistics.  \n  - Correlation heatmaps of coordinate columns indicate expected relationships.\n\n- **Data Integration & Quality:**  \n  - Merging of sequence and label data shows discrepancies in the training set (i.e. target IDs do not match perfectly).  \n  - Missing values exist, particularly in the training label coordinates.\n\n**Conclusion Preview:**  \nThe EDA reveals essential characteristics of the dataset, including wide variation in RNA sequence lengths, predominant base composition, and areas for data cleaning (missing values and merge key adjustments). These insights provide a strong foundation for further analysis and modeling.\n","metadata":{}},{"cell_type":"code","source":"!pip install -q --no-deps /kaggle/input/biopython-1-85/biopython-1.85-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:30.012072Z","iopub.execute_input":"2025-04-01T00:09:30.012398Z","iopub.status.idle":"2025-04-01T00:09:33.176273Z","shell.execute_reply.started":"2025-04-01T00:09:30.012371Z","shell.execute_reply":"2025-04-01T00:09:33.174862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEED = 42\nimport os\nimport numpy as np\nfrom numpy import random as np_rnd\nimport random as rnd\nimport pandas as pd\nimport pickle\nfrom Bio import SeqIO\n\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.preprocessing import MinMaxScaler\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:33.177608Z","iopub.execute_input":"2025-04-01T00:09:33.178013Z","iopub.status.idle":"2025-04-01T00:09:34.778986Z","shell.execute_reply.started":"2025-04-01T00:09:33.177976Z","shell.execute_reply":"2025-04-01T00:09:34.777878Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def seed_everything(seed=42):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    # python random\n    rnd.seed(seed)\n    # numpy random\n    np_rnd.seed(seed)\n    # RAPIDS random\n    try:\n        cupy.random.seed(seed)\n    except:\n        pass\n    # tf random\n    try:\n        tf_rnd.set_seed(seed)\n    except:\n        pass\n    # pytorch random\n    try:\n        torch.backends.cudnn.benchmark = False\n        torch.backends.cudnn.deterministic = True\n        torch.manual_seed(seed)\n        torch.cuda.manual_seed(seed)\n        torch.cuda.manual_seed_all(seed)\n    except:\n        pass\n\ndef pickleIO(obj, src, op=\"r\"):\n    if op==\"w\":\n        with open(src, op + \"b\") as f:\n            pickle.dump(obj, f)\n    elif op==\"r\":\n        with open(src, op + \"b\") as f:\n            tmp = pickle.load(f)\n        return tmp\n    else:\n        print(\"unknown operation\")\n        return obj\n    \ndef createFolder(directory):\n    try:\n        if not os.path.exists(directory):\n            os.makedirs(directory)\n    except OSError:\n        print('Error: Creating directory. ' + directory)\n\ndef findIdx(data_x, col_names):\n    return [int(i) for i, j in enumerate(data_x) if j in col_names]\n\ndef diff(first, second):\n    second = set(second)\n    return [item for item in first if item not in second]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:34.779984Z","iopub.execute_input":"2025-04-01T00:09:34.780495Z","iopub.status.idle":"2025-04-01T00:09:34.790066Z","shell.execute_reply.started":"2025-04-01T00:09:34.780455Z","shell.execute_reply":"2025-04-01T00:09:34.788943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CFG:\n    debug = False\n    n_folds = 5\n    max_seq = 1024","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:34.792859Z","iopub.execute_input":"2025-04-01T00:09:34.793344Z","iopub.status.idle":"2025-04-01T00:09:34.813688Z","shell.execute_reply.started":"2025-04-01T00:09:34.793301Z","shell.execute_reply":"2025-04-01T00:09:34.812609Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Loading data","metadata":{}},{"cell_type":"code","source":"df_full = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_sequences.csv\")\nlabel_full = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_labels.csv\")\nlabel_full[\"target_id\"] = label_full[\"ID\"].apply(lambda x: \"_\".join(x.split(\"_\")[:-1]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:34.815693Z","iopub.execute_input":"2025-04-01T00:09:34.816100Z","iopub.status.idle":"2025-04-01T00:09:35.316245Z","shell.execute_reply.started":"2025-04-01T00:09:34.816062Z","shell.execute_reply":"2025-04-01T00:09:35.315188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_full.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:35.317340Z","iopub.execute_input":"2025-04-01T00:09:35.317699Z","iopub.status.idle":"2025-04-01T00:09:35.346335Z","shell.execute_reply.started":"2025-04-01T00:09:35.317659Z","shell.execute_reply":"2025-04-01T00:09:35.345101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_full.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:35.347549Z","iopub.execute_input":"2025-04-01T00:09:35.347972Z","iopub.status.idle":"2025-04-01T00:09:35.385302Z","shell.execute_reply.started":"2025-04-01T00:09:35.347943Z","shell.execute_reply":"2025-04-01T00:09:35.384282Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/validation_sequences.csv\")\nlabel_test = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/validation_labels.csv\").iloc[:, :6]\nlabel_test[\"target_id\"] = label_test[\"ID\"].apply(lambda x: \"_\".join(x.split(\"_\")[:-1]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:35.386288Z","iopub.execute_input":"2025-04-01T00:09:35.386698Z","iopub.status.idle":"2025-04-01T00:09:35.496411Z","shell.execute_reply.started":"2025-04-01T00:09:35.386659Z","shell.execute_reply":"2025-04-01T00:09:35.495326Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:35.497409Z","iopub.execute_input":"2025-04-01T00:09:35.497795Z","iopub.status.idle":"2025-04-01T00:09:35.507785Z","shell.execute_reply.started":"2025-04-01T00:09:35.497753Z","shell.execute_reply":"2025-04-01T00:09:35.506480Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_test.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:35.508779Z","iopub.execute_input":"2025-04-01T00:09:35.509165Z","iopub.status.idle":"2025-04-01T00:09:35.534702Z","shell.execute_reply.started":"2025-04-01T00:09:35.509135Z","shell.execute_reply":"2025-04-01T00:09:35.533399Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Simple EDA","metadata":{}},{"cell_type":"code","source":"residues = set()\nfor i in df_full[\"sequence\"]:\n    residues = residues.union(set(i))\nresidues","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:35.535883Z","iopub.execute_input":"2025-04-01T00:09:35.536250Z","iopub.status.idle":"2025-04-01T00:09:35.560329Z","shell.execute_reply.started":"2025-04-01T00:09:35.536212Z","shell.execute_reply":"2025-04-01T00:09:35.559270Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_full[\"sequence\"].apply(len).describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:35.561297Z","iopub.execute_input":"2025-04-01T00:09:35.561685Z","iopub.status.idle":"2025-04-01T00:09:35.592155Z","shell.execute_reply.started":"2025-04-01T00:09:35.561659Z","shell.execute_reply":"2025-04-01T00:09:35.591078Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_full[\"sequence\"].apply(len).hist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:35.596689Z","iopub.execute_input":"2025-04-01T00:09:35.597016Z","iopub.status.idle":"2025-04-01T00:09:35.952231Z","shell.execute_reply.started":"2025-04-01T00:09:35.596990Z","shell.execute_reply":"2025-04-01T00:09:35.951131Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Filtering","metadata":{}},{"cell_type":"code","source":"# filtering label is not exists\nlabel_nans = label_full.groupby(\"target_id\")[[\"x_1\", \"y_1\", \"z_1\"]].apply(lambda x: x.isna().any().any())\ntarget_nans = label_nans.index[label_nans]\ndf_full = df_full[~df_full[\"target_id\"].isin(target_nans)].reset_index(drop=True)\nlabel_full = label_full[~label_full[\"target_id\"].isin(target_nans)].reset_index(drop=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:35.954315Z","iopub.execute_input":"2025-04-01T00:09:35.954708Z","iopub.status.idle":"2025-04-01T00:09:36.292246Z","shell.execute_reply.started":"2025-04-01T00:09:35.954678Z","shell.execute_reply":"2025-04-01T00:09:36.291436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_full.shape, label_full.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:36.293062Z","iopub.execute_input":"2025-04-01T00:09:36.293300Z","iopub.status.idle":"2025-04-01T00:09:36.298998Z","shell.execute_reply.started":"2025-04-01T00:09:36.293280Z","shell.execute_reply":"2025-04-01T00:09:36.298115Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"residues = set()\nfor i in df_full[\"sequence\"]:\n    residues = residues.union(set(i))\nresidues","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:36.299945Z","iopub.execute_input":"2025-04-01T00:09:36.300239Z","iopub.status.idle":"2025-04-01T00:09:36.317200Z","shell.execute_reply.started":"2025-04-01T00:09:36.300201Z","shell.execute_reply":"2025-04-01T00:09:36.316272Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_full[\"sequence\"].apply(len).describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:36.318207Z","iopub.execute_input":"2025-04-01T00:09:36.318564Z","iopub.status.idle":"2025-04-01T00:09:36.341441Z","shell.execute_reply.started":"2025-04-01T00:09:36.318530Z","shell.execute_reply":"2025-04-01T00:09:36.340335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_full[\"sequence\"].apply(len).hist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:36.342482Z","iopub.execute_input":"2025-04-01T00:09:36.342887Z","iopub.status.idle":"2025-04-01T00:09:36.629151Z","shell.execute_reply.started":"2025-04-01T00:09:36.342850Z","shell.execute_reply":"2025-04-01T00:09:36.628175Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Sequence Chunking","metadata":{}},{"cell_type":"code","source":"def seq_chunking(seq, chunk_size, overlap_size, min_chunk_size):\n    \"\"\"\n    긴 서열을 일정 크기의 청크로 나누는 함수\n    \n    :param seq: 원본 시퀀스 (문자열)\n    :param chunk_size: 각 청크의 기본 크기\n    :param overlap_size: 청크 간의 오버랩 크기\n    :param min_chunk_size: 최소 청크 크기 (너무 작은 청크 방지)\n    :return: 나뉜 청크 리스트\n    \"\"\"\n    chunks = []\n    seq_locs = []\n    start = 0\n    \n    while start < len(seq):\n        end = start + chunk_size\n        chunk = seq[start:end]\n        \n        # 최소 청크 크기보다 작은 경우 제외\n        if len(chunk) >= min_chunk_size:\n            chunks.append(chunk)\n            seq_locs.append((start, end))\n        \n        # 다음 청크의 시작 위치 설정 (오버랩 적용)\n        start += chunk_size - overlap_size\n        \n        # 마지막 청크가 너무 작으면 제거\n        if start + min_chunk_size > len(seq):\n            break\n    \n    return chunks, seq_locs","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:36.630317Z","iopub.execute_input":"2025-04-01T00:09:36.630706Z","iopub.status.idle":"2025-04-01T00:09:36.637171Z","shell.execute_reply.started":"2025-04-01T00:09:36.630669Z","shell.execute_reply":"2025-04-01T00:09:36.636072Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_tmp = []\nfor idx, row in df_full.iterrows():\n    if len(row[\"sequence\"]) >= CFG.max_seq:\n        for chunk_id, data in enumerate(zip(*seq_chunking(row[\"sequence\"], chunk_size=CFG.max_seq, overlap_size=CFG.max_seq // 4, min_chunk_size=3)) ): \n            chunk, seq_loc = data\n            df_tmp.append({\n                **row,\n                \"chunk_id\": chunk_id,\n                \"sequence\": chunk,\n                \"seq_loc\": seq_loc,\n            })\n    else:\n        df_tmp.append({\n            **row,\n            \"chunk_id\": 0,\n            \"seq_loc\": (0, len(row[\"sequence\"])),\n        })\ndf_full = pd.DataFrame(df_tmp)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:36.638076Z","iopub.execute_input":"2025-04-01T00:09:36.638327Z","iopub.status.idle":"2025-04-01T00:09:36.710690Z","shell.execute_reply.started":"2025-04-01T00:09:36.638306Z","shell.execute_reply":"2025-04-01T00:09:36.709866Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_full[\"sequence\"].apply(len).describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:36.711700Z","iopub.execute_input":"2025-04-01T00:09:36.712081Z","iopub.status.idle":"2025-04-01T00:09:36.722003Z","shell.execute_reply.started":"2025-04-01T00:09:36.712042Z","shell.execute_reply":"2025-04-01T00:09:36.721008Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_tmp = []\nfor idx, row in df_test.iterrows():\n    if len(row[\"sequence\"]) >= CFG.max_seq:\n        for chunk_id, data in enumerate(zip(*seq_chunking(row[\"sequence\"], chunk_size=CFG.max_seq, overlap_size=CFG.max_seq // 4, min_chunk_size=3)) ): \n            chunk, seq_loc = data\n            df_tmp.append({\n                **row,\n                \"chunk_id\": chunk_id,\n                \"sequence\": chunk,\n                \"seq_loc\": seq_loc,\n            })\n    else:\n        df_tmp.append({\n            **row,\n            \"chunk_id\": 0,\n            \"seq_loc\": (0, len(row[\"sequence\"])),\n        })\ndf_test = pd.DataFrame(df_tmp)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:36.723009Z","iopub.execute_input":"2025-04-01T00:09:36.723314Z","iopub.status.idle":"2025-04-01T00:09:36.746706Z","shell.execute_reply.started":"2025-04-01T00:09:36.723288Z","shell.execute_reply":"2025-04-01T00:09:36.745574Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test[\"sequence\"].apply(len).describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:36.747696Z","iopub.execute_input":"2025-04-01T00:09:36.747994Z","iopub.status.idle":"2025-04-01T00:09:36.773461Z","shell.execute_reply.started":"2025-04-01T00:09:36.747969Z","shell.execute_reply":"2025-04-01T00:09:36.772475Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Coordinate scaling","metadata":{}},{"cell_type":"code","source":"target_scaler = MinMaxScaler((-1, 1))\nlabel_full = label_full.set_index(\"target_id\")\nlabel_full[[\"x_1\", \"y_1\", \"z_1\"]] = target_scaler.fit_transform(label_full[[\"x_1\", \"y_1\", \"z_1\"]].astype(\"float32\"))\nlabel_full","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:36.774390Z","iopub.execute_input":"2025-04-01T00:09:36.774714Z","iopub.status.idle":"2025-04-01T00:09:36.818170Z","shell.execute_reply.started":"2025-04-01T00:09:36.774689Z","shell.execute_reply":"2025-04-01T00:09:36.817302Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_test = label_test.set_index(\"target_id\")\nlabel_test[[\"x_1\", \"y_1\", \"z_1\"]] = target_scaler.transform(label_test[[\"x_1\", \"y_1\", \"z_1\"]].astype(\"float32\"))\nlabel_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:36.819218Z","iopub.execute_input":"2025-04-01T00:09:36.819506Z","iopub.status.idle":"2025-04-01T00:09:36.837912Z","shell.execute_reply.started":"2025-04-01T00:09:36.819465Z","shell.execute_reply":"2025-04-01T00:09:36.836993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"residues = set()\nfor i in df_full[\"sequence\"]:\n    residues = residues.union(set(i))\nresidues = pd.Series(range(len(residues)), index=residues, dtype=\"int64\") + 1\nresidues","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:36.838957Z","iopub.execute_input":"2025-04-01T00:09:36.839322Z","iopub.status.idle":"2025-04-01T00:09:36.866448Z","shell.execute_reply.started":"2025-04-01T00:09:36.839287Z","shell.execute_reply":"2025-04-01T00:09:36.865187Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## CV split","metadata":{}},{"cell_type":"code","source":"skf = StratifiedGroupKFold(CFG.n_folds, shuffle=True, random_state=SEED)\nstartVec = pd.qcut(df_full[\"sequence\"].apply(len), CFG.n_folds, labels=range(CFG.n_folds)).astype(\"int64\")\ndf_full[\"fold_target\"] = -1\nfor fold, (train_idx, valid_idx) in enumerate(skf.split(df_full, startVec, groups=df_full[\"target_id\"])):\n    df_full[\"fold_target\"].iloc[valid_idx] = fold","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:36.867654Z","iopub.execute_input":"2025-04-01T00:09:36.868002Z","iopub.status.idle":"2025-04-01T00:09:37.195273Z","shell.execute_reply.started":"2025-04-01T00:09:36.867969Z","shell.execute_reply":"2025-04-01T00:09:37.194180Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Save data","metadata":{}},{"cell_type":"code","source":"pickleIO(df_full, \"df_full.pkl\", \"w\")\npickleIO(label_full, \"label_full.pkl\", \"w\")\npickleIO(df_full, \"df_test.pkl\", \"w\")\npickleIO(label_full, \"label_test.pkl\", \"w\")\npickleIO(target_scaler, \"target_scaler.pkl\", \"w\")\npickleIO(residues, \"residues.pkl\", \"w\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T00:09:37.196258Z","iopub.execute_input":"2025-04-01T00:09:37.196506Z","iopub.status.idle":"2025-04-01T00:09:37.320160Z","shell.execute_reply.started":"2025-04-01T00:09:37.196485Z","shell.execute_reply":"2025-04-01T00:09:37.319193Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}