{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11553390,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Imports ","metadata":{}},{"cell_type":"code","source":"import pandas as pd \nimport numpy as np \nimport matplotlib.pyplot as plt\nimport time\nfrom tqdm import tqdm\n\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import train_test_split\n\nfrom joblib import Parallel, delayed\n\nfrom xgboost import XGBRegressor\nfrom sklearn.multioutput import MultiOutputRegressor, RegressorChain\nfrom sklearn.neighbors import KNeighborsRegressor, RadiusNeighborsRegressor\nfrom sklearn.ensemble import GradientBoostingRegressor, ExtraTreesRegressor, RandomForestRegressor\nimport lightgbm as lgb\n\nimport torch\nimport gc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T09:56:33.687441Z","iopub.execute_input":"2025-04-11T09:56:33.687713Z","iopub.status.idle":"2025-04-11T09:56:43.858535Z","shell.execute_reply.started":"2025-04-11T09:56:33.687689Z","shell.execute_reply":"2025-04-11T09:56:43.857438Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Functions","metadata":{}},{"cell_type":"code","source":"def report_gpu():\n    gc.collect()\n    torch.cuda.empty_cache()\nreport_gpu()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T09:56:43.859741Z","iopub.execute_input":"2025-04-11T09:56:43.860562Z","iopub.status.idle":"2025-04-11T09:56:44.026839Z","shell.execute_reply.started":"2025-04-11T09:56:43.860525Z","shell.execute_reply":"2025-04-11T09:56:44.025806Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def getKmers(sequence, size):\n    return [sequence[x:x+size].lower() for x in range(len(sequence) - size + 1)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T09:56:44.028035Z","iopub.execute_input":"2025-04-11T09:56:44.028479Z","iopub.status.idle":"2025-04-11T09:56:44.042921Z","shell.execute_reply.started":"2025-04-11T09:56:44.028439Z","shell.execute_reply":"2025-04-11T09:56:44.041804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def kmers_rna(seq_data, size, label_data=None):\n    kmers, sequence_lengths = [], []\n    seq_data = seq_data.copy()\n\n    for _, row in seq_data.iterrows(): \n        target_id, seq = row.iloc[0], row.iloc[1]\n        seq_length = len(seq)\n        \n        if label_data is not None:\n            label_matches = label_data[label_data['ID'].str.startswith(f\"{target_id}_\")]\n            if seq_length != len(label_matches):\n                continue\n                \n        kmers.append(getKmers(sequence=seq, size=size))\n        sequence_lengths.append(seq_length)\n\n    seq_data['seq_length'] = sequence_lengths\n    seq_data['seq_kmers'] = kmers\n\n    return seq_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T09:56:44.044027Z","iopub.execute_input":"2025-04-11T09:56:44.044515Z","iopub.status.idle":"2025-04-11T09:56:44.053837Z","shell.execute_reply.started":"2025-04-11T09:56:44.044484Z","shell.execute_reply":"2025-04-11T09:56:44.052552Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def one_hot_encode_rna(seq):\n    mapping = {'A': [1, 0, 0, 0],\n               'U': [0, 1, 0, 0],\n               'G': [0, 0, 1, 0],\n               'C': [0, 0, 0, 1]}\n    return [mapping.get(nt, [0, 0, 0, 0]) for nt in seq]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T09:56:44.055105Z","iopub.execute_input":"2025-04-11T09:56:44.055561Z","iopub.status.idle":"2025-04-11T09:56:44.068741Z","shell.execute_reply.started":"2025-04-11T09:56:44.055473Z","shell.execute_reply":"2025-04-11T09:56:44.067633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def fix_dataset(seq_df, labels_df=None):\n    data = {}\n    start_time = time.time()  # Start timing\n    \n    for indx, row in tqdm(seq_df.iterrows(), total=len(seq_df), desc=\"Processing Sequences\"):\n        all_encoded = []\n        seq_id, seq, seq_len = row.target_id, row.sequence, row.sequence_legnth\n\n        # Encode RNA Sequence\n        encode_seq = one_hot_encode_rna(seq)\n        for pos, one_hot in enumerate(encode_seq):\n            all_encoded.append([seq_id, pos] + one_hot)\n        # Df of encoder\n        encoded_df = pd.DataFrame(all_encoded, columns=['ID', 'seq_len', 'A', 'U', 'G', 'C'])\n        \n        # Convert sequence to numerical values\n        numerical_seq = [seq_map.get(nuc, 4) for nuc in seq]\n\n        # Count nucleotide occurrences using pandas' value_counts\n        counts = pd.Series(list(seq)).value_counts()\n        A_count = counts.get('A', 0)\n        C_count = counts.get('C', 0)\n        G_count = counts.get('G', 0)\n        U_count = counts.get('U', 0)\n\n        # Create DataFrame efficiently\n        encoded_df['RNA_seq'] = numerical_seq\n        encoded_df['seq_id'] = [f\"{seq_id}_{i}\" for i in range(1, seq_len+1)]\n        encoded_df['A_count'] = A_count\n        encoded_df['C_count'] = C_count\n        encoded_df['G_count'] = G_count\n        encoded_df['U_count'] = U_count\n\n               # Check for labels if they exist\n        if labels_df is not None:\n            seq_id_df = labels_df[labels_df['ID'].str.startswith(f\"{seq_id}_\")]\n\n            if not seq_id_df.empty:\n                coords = seq_id_df[['x_1', 'y_1', 'z_1']].to_numpy()\n                if coords.shape[0] == seq_len:\n                    # Normalize coordinates efficiently\n                    mean = coords.mean(axis=0)\n                    std = coords.std(axis=0) + 1e-8  # Avoid division by zero\n                    coords_norm = (coords - mean) / std\n                    encoded_df[['x_1', 'y_1', 'z_1']] = coords_norm\n                else:\n                    print(f\"Warning: Mismatch for {seq_id} - coords: {coords.shape[0]}, seq_len: {seq_len}\")\n                    continue  \n    \n        data[seq_id] = encoded_df\n        \n    # Merge all data after loop\n    merge_data = pd.concat(data.values(), ignore_index=True)\n    end_time = time.time()\n    print(f\"Total time taken: {end_time - start_time:.2f} seconds\")\n    \n    return merge_data\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T09:56:44.071835Z","iopub.execute_input":"2025-04-11T09:56:44.072108Z","iopub.status.idle":"2025-04-11T09:56:44.084947Z","shell.execute_reply.started":"2025-04-11T09:56:44.072084Z","shell.execute_reply":"2025-04-11T09:56:44.083986Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def clean_data(data, pop_col, remove_col, std_cols): \n    \n    print(f'BEFORE Data:{data.shape}')\n    # Remove missing values\n    data = data.dropna().copy()  \n\n    # Remove and store the ID column efficiently\n    id_col = data.pop(pop_col) \n    data = data.drop(columns=remove_col)\n     \n    # StandardScaler\n    data[std_cols] = StandardScaler().fit_transform(data[std_cols])\n    print(f'AFTER Data:{data.shape}')\n\n    return data, id_col","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T09:56:44.087020Z","iopub.execute_input":"2025-04-11T09:56:44.087337Z","iopub.status.idle":"2025-04-11T09:56:44.102720Z","shell.execute_reply.started":"2025-04-11T09:56:44.087313Z","shell.execute_reply":"2025-04-11T09:56:44.101685Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_models(df_train, df_val, target_data, models):\n    start_time = time.time()\n\n    \n    df_merged = pd.concat([df_train, df_val], ignore_index=True)\n    print(f'merge Train Val:{df_merged.shape}')\n    \n    X_train = df_merged.drop(columns=target_data)\n    y_train = df_merged[target_data]\n    print(f\"Train shapes: X={X_train.shape}, y={y_train.shape}\")\n    \n    def train_model(name, model):\n        model.fit(X_train, y_train)\n        return (name, model)\n    \n    trained_models = Parallel(n_jobs=-1)( delayed(train_model)(name, model) for name, model in models.items())\n    \n    for name, model in trained_models:\n        print(f\"{name} model trained\")\n    print(f\"Total training time: {time.time() - start_time:.2f} seconds\")\n    \n    return trained_models","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T09:56:44.103871Z","iopub.execute_input":"2025-04-11T09:56:44.104214Z","iopub.status.idle":"2025-04-11T09:56:44.122057Z","shell.execute_reply.started":"2025-04-11T09:56:44.104182Z","shell.execute_reply":"2025-04-11T09:56:44.120931Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predictions(models, unseen_target): \n    # Submission dataframe\n    sample_sub = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/sample_submission.csv')\n    id_s, resname_seq, resid_seq = sample_sub.ID, sample_sub.resname, sample_sub.resid\n    submission = pd.DataFrame({'ID':id_s,\n                              'resname':resname_seq, \n                              'resid':resid_seq})\n\n    #make the predictions \n    predictions = {}\n    for (name, model), trgt_lst in zip(models, unseen_target): \n        y_pred = model.predict(clean_test_data) # lst [x,y,z]\n        predictions[name] = pd.DataFrame(y_pred, columns=trgt_lst)\n        print(f'{name} model predict')\n    \n    # Final submision Df\n    final_df = submission.copy()\n    for df in predictions.values():\n        final_df = final_df.merge(df, left_index=True, right_index=True, how='left') \n    print('... Submision ready')\n    \n    return  final_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T09:56:44.123359Z","iopub.execute_input":"2025-04-11T09:56:44.123759Z","iopub.status.idle":"2025-04-11T09:56:44.137737Z","shell.execute_reply.started":"2025-04-11T09:56:44.123724Z","shell.execute_reply":"2025-04-11T09:56:44.136766Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Variables","metadata":{}},{"cell_type":"code","source":"train_deq_path  = '/kaggle/input/stanford-rna-3d-folding/train_sequences.csv'\ntrain_label_path ='/kaggle/input/stanford-rna-3d-folding/train_labels.csv' \n\nval_seq_path ='/kaggle/input/stanford-rna-3d-folding/validation_sequences.csv'\nval_label_path ='/kaggle/input/stanford-rna-3d-folding/validation_labels.csv' \n\n\ntest_seq_path ='/kaggle/input/stanford-rna-3d-folding/test_sequences.csv' ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T09:56:44.138682Z","iopub.execute_input":"2025-04-11T09:56:44.138960Z","iopub.status.idle":"2025-04-11T09:56:44.155954Z","shell.execute_reply.started":"2025-04-11T09:56:44.138938Z","shell.execute_reply":"2025-04-11T09:56:44.154860Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"used_columns = ['target_id', 'sequence']\nseq_map = {'A': 0, 'C': 1, 'G': 2, 'U': 3}\nstd_cols = ['seq_len', 'A_count', 'C_count', 'G_count', 'U_count']\nremove_col, pop_col = ['ID'], 'seq_id'\ntarget_coord = ['x_1', 'y_1', 'z_1']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T09:56:44.156943Z","iopub.execute_input":"2025-04-11T09:56:44.157200Z","iopub.status.idle":"2025-04-11T09:56:44.173087Z","shell.execute_reply.started":"2025-04-11T09:56:44.157177Z","shell.execute_reply":"2025-04-11T09:56:44.171967Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"unseen_target_name =[['x_1', 'y_1', 'z_1'], ['x_2', 'y_2', 'z_2'], ['x_3', 'y_3', 'z_3'],\n             ['x_4', 'y_4', 'z_4'], ['x_5', 'y_5', 'z_5'] ]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T09:56:44.174295Z","iopub.execute_input":"2025-04-11T09:56:44.174675Z","iopub.status.idle":"2025-04-11T09:56:44.187349Z","shell.execute_reply.started":"2025-04-11T09:56:44.174639Z","shell.execute_reply":"2025-04-11T09:56:44.186345Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# LGBMRegressor (x1, y1, z1)\nlgbm = lgb.LGBMRegressor(n_estimators=200,\n                              learning_rate=1,\n                              max_depth=-1,\n                              random_state=42)\nlgbm_model = RegressorChain(lgbm)\n\n# GBDTRegression (x2, y2, z2)\ngbdtr = GradientBoostingRegressor(n_estimators=20, \n                                learning_rate=0.1, \n                                max_depth=50, \n                                random_state=42)\ngbdtr_model = RegressorChain(gbdtr)\n\n# ExtraTreeRegression (x3, y3, z3)\nextree = ExtraTreesRegressor(n_estimators=10,\n                                 max_depth=50, \n                                 criterion='friedman_mse',\n                                 random_state=42)\nextree_model = RegressorChain(extree)\n\n# RandomForest Regression (x4, y4, z4)\nrf_r = RandomForestRegressor(criterion='friedman_mse',\n                                n_estimators=10, \n                                max_depth=80, \n                                bootstrap=True,\n                                random_state=42)\nrf_r_model = RegressorChain(rf_r)\n\n# # RadiusNeighborsRegressor (x5, y5, z5)\n# knn_r = RadiusNeighborsRegressor(radius= 1.0,\n#                          weights=  'distance', #'uniform', 'distance'\n#                          metric= 'minkowski', #'euclidean', manhattan, minkowski\n#                          algorithm = 'auto',  #kd_tree, ball_tree\n#                               p=1)\n# knn_r_model = RegressorChain(knn_r)\n\n# XGB regressor (x5, y5, z5)\nxgb_r = XGBRegressor(n_estimators=80, \n                     learning_rate=1, \n                     max_depth=30, \n                     random_state=42)\n\nxgb_r_model = RegressorChain(xgb_r)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T09:56:44.188517Z","iopub.execute_input":"2025-04-11T09:56:44.188802Z","iopub.status.idle":"2025-04-11T09:56:44.202692Z","shell.execute_reply.started":"2025-04-11T09:56:44.188778Z","shell.execute_reply":"2025-04-11T09:56:44.201634Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"models = {\n    'LGBMRegressor': lgbm_model,\n    'GradientBoostingRegressor': gbdtr_model,\n    'ExtraTreesRegressor': extree_model,\n    'RandomForestRegressor': rf_r_model,\n    'XGBRegressor': xgb_r_model\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T09:56:44.203950Z","iopub.execute_input":"2025-04-11T09:56:44.204296Z","iopub.status.idle":"2025-04-11T09:56:44.222248Z","shell.execute_reply.started":"2025-04-11T09:56:44.204265Z","shell.execute_reply":"2025-04-11T09:56:44.220936Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Main","metadata":{}},{"cell_type":"code","source":"#train data\ntrain_sequences = pd.read_csv(train_deq_path, usecols=used_columns)\ntrain_labels = pd.read_csv(train_label_path)\n\n# Val Data\nvalidation_sequences = pd.read_csv(val_seq_path, usecols=used_columns)\nvalidation_labels = pd.read_csv(val_label_path)\n\n# Test data\ntest_sequences = pd.read_csv(test_seq_path ,usecols=used_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T09:56:44.223376Z","iopub.execute_input":"2025-04-11T09:56:44.223773Z","iopub.status.idle":"2025-04-11T09:56:44.686209Z","shell.execute_reply.started":"2025-04-11T09:56:44.223738Z","shell.execute_reply":"2025-04-11T09:56:44.685114Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# rna_train_seq = kmers_rna(seq_data=train_sequences, size=3, label_data=train_labels)\n# rna_val_seq = kmers_rna(seq_data=validation_sequences, size=3, label_data=validation_labels)\n# rna_test_seq = kmers_rna(seq_data=test_sequences, size=3, label_data=None)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T09:56:44.687393Z","iopub.execute_input":"2025-04-11T09:56:44.687734Z","iopub.status.idle":"2025-04-11T09:56:44.691639Z","shell.execute_reply.started":"2025-04-11T09:56:44.687706Z","shell.execute_reply":"2025-04-11T09:56:44.690518Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sequences['sequence_legnth'] = train_sequences['sequence'].str.len()\nvalidation_sequences['sequence_legnth'] = validation_sequences['sequence'].str.len()\ntest_sequences['sequence_legnth'] = test_sequences['sequence'].str.len()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T09:56:44.692492Z","iopub.execute_input":"2025-04-11T09:56:44.692773Z","iopub.status.idle":"2025-04-11T09:56:44.714967Z","shell.execute_reply.started":"2025-04-11T09:56:44.692750Z","shell.execute_reply":"2025-04-11T09:56:44.713840Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('.... Train Data Procesed')\ntrain_data  = fix_dataset(seq_df=train_sequences, labels_df=train_labels)\n\nprint('.... Validation Data Procesed')\nval_data  = fix_dataset(seq_df=validation_sequences, labels_df=validation_labels)\n\nprint('.... Test Data Procesed')\ntest_data  = fix_dataset(seq_df=test_sequences, labels_df=None)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T09:56:44.715942Z","iopub.execute_input":"2025-04-11T09:56:44.716314Z","iopub.status.idle":"2025-04-11T09:57:19.069683Z","shell.execute_reply.started":"2025-04-11T09:56:44.716276Z","shell.execute_reply":"2025-04-11T09:57:19.068374Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" print('... Clean Train Data')\nclean_train_data, _ = clean_data(data = train_data, pop_col = pop_col,\n                              remove_col = remove_col, std_cols =  std_cols)\nprint('... Clean Validation Data')\nclean_val_data, _ = clean_data(data = val_data, pop_col = pop_col,\n                              remove_col = remove_col, std_cols =  std_cols)\n\nprint('... Clean Test Data')\nclean_test_data, unseen_id = clean_data(data = test_data, pop_col = pop_col,\n                              remove_col = remove_col, std_cols =  std_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T09:57:19.070751Z","iopub.execute_input":"2025-04-11T09:57:19.071108Z","iopub.status.idle":"2025-04-11T09:57:19.216770Z","shell.execute_reply.started":"2025-04-11T09:57:19.071064Z","shell.execute_reply":"2025-04-11T09:57:19.215550Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"trained_models = train_models(df_train=clean_train_data, \n                              df_val=clean_val_data,\n                              target_data =target_coord, \n                              models=models) #  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T09:57:19.220171Z","iopub.execute_input":"2025-04-11T09:57:19.220518Z","iopub.status.idle":"2025-04-11T09:58:33.787069Z","shell.execute_reply.started":"2025-04-11T09:57:19.220489Z","shell.execute_reply":"2025-04-11T09:58:33.785984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = predictions(models= trained_models, \n                         unseen_target=unseen_target_name)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T09:58:33.788564Z","iopub.execute_input":"2025-04-11T09:58:33.788937Z","iopub.status.idle":"2025-04-11T09:58:33.972277Z","shell.execute_reply.started":"2025-04-11T09:58:33.788907Z","shell.execute_reply":"2025-04-11T09:58:33.971219Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T09:58:33.973113Z","iopub.execute_input":"2025-04-11T09:58:33.973378Z","iopub.status.idle":"2025-04-11T09:58:34.057495Z","shell.execute_reply.started":"2025-04-11T09:58:33.973355Z","shell.execute_reply":"2025-04-11T09:58:34.056651Z"}},"outputs":[],"execution_count":null}]}