{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11403143,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# This notebook mimics the approach of my other notebook (except for the EDA) that is written in R. That notebook can be found [here](https://www.kaggle.com/code/chrisk321/rna-folding-r-eda-submission).","metadata":{}},{"cell_type":"markdown","source":"\n# <div align=\"center\"> Stanford RNA 3D Folding \n    \n\n<div align=\"center\"> <img src=\"https://www.kaggle.com/competitions/87793/images/header\" width=\"300\"></div><br>\n\n**From the competition material**: For each sequence in the test set, you can predict <ins>five structures</ins>. Your notebook should look for a file test_sequences.csv and output submission.csv. This file should contain x, y, z coordinates of the C1' atom in each residue across your predicted structures 1 to 5:\n\n\n**ID**,**resname**,**resid**,**x_1**,**y_1**,**z_1**,... **x_5**,**y_5**,**z_5** <br>\nR1107_1,G,1,-7.561,9.392,9.361,... -7.301,9.023,8.932<br>\nR1107_2,G,1,-8.02,11.014,14.606,... -7.953,10.02,12.127<br>\netc.<br>\n\n\n**Evaluation**: Submissions are scored using <ins>TM-score</ins> (\"template modeling\" score), which goes from 0.0 to 1.0 (higher is better):\n","metadata":{}},{"cell_type":"markdown","source":"# 📚 Libraries","metadata":{}},{"cell_type":"code","source":"import h2o\nh2o.init()\nfrom h2o.estimators import H2OXGBoostEstimator\nimport numpy as np \nimport pandas as pd \nimport xgboost as xgb\nfrom sklearn.model_selection import KFold,StratifiedKFold\nfrom sklearn.metrics import mean_squared_error","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-15T14:40:06.393774Z","iopub.execute_input":"2025-03-15T14:40:06.394243Z","iopub.status.idle":"2025-03-15T14:40:17.336616Z","shell.execute_reply.started":"2025-03-15T14:40:06.394199Z","shell.execute_reply":"2025-03-15T14:40:17.335275Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🔍 Reading competition material","metadata":{}},{"cell_type":"code","source":"# Reading in the competition material\ntrain_sequences =  pd.read_csv('/kaggle/input/stanford-rna-3d-folding/train_sequences.csv') \ntrain_labels =  pd.read_csv('/kaggle/input/stanford-rna-3d-folding/train_labels.csv') \ntest_sequences = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/test_sequences.csv') \nsample_submission = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T14:40:17.338056Z","iopub.execute_input":"2025-03-15T14:40:17.338591Z","iopub.status.idle":"2025-03-15T14:40:17.720256Z","shell.execute_reply.started":"2025-03-15T14:40:17.338557Z","shell.execute_reply":"2025-03-15T14:40:17.719174Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🔢 Feature Engineering & Pre-Processing","metadata":{}},{"cell_type":"markdown","source":"### Basic feature engineering using occurrences of A, C, U, G, sequence length, and begin / end of sequence","metadata":{}},{"cell_type":"code","source":"def feat_eng(df):\n    df['seq_length'] = df['sequence'].str.len()\n    df['a_cnt'] = df['sequence'].str.count(\"A\")\n    df['c_cnt'] = df['sequence'].str.count(\"C\")\n    df['u_cnt'] = df['sequence'].str.count(\"U\")\n    df['g_cnt'] = df['sequence'].str.count(\"G\")\n    df['ac_cnt'] = df['sequence'].str.count(\"AC\")\n    df['au_cnt'] = df['sequence'].str.count(\"AU\")\n    df['ag_cnt'] = df['sequence'].str.count(\"AG\")\n    df['ca_cnt'] = df['sequence'].str.count(\"CA\")\n    df['cu_cnt'] = df['sequence'].str.count(\"CU\")\n    df['cg_cnt'] = df['sequence'].str.count(\"CG\")\n    df['ua_cnt'] = df['sequence'].str.count(\"UA\")\n    df['uc_cnt'] = df['sequence'].str.count(\"UC\")\n    df['ug_cnt'] = df['sequence'].str.count(\"UG\")\n    df['ga_cnt'] = df['sequence'].str.count(\"GA\")\n    df['gc_cnt'] = df['sequence'].str.count(\"GC\")\n    df['gu_cnt'] = df['sequence'].str.count(\"GU\")\n    df['aa_cnt'] = df['sequence'].str.count(\"AA\")\n    df['cc_cnt'] = df['sequence'].str.count(\"CC\")\n    df['uu_cnt'] = df['sequence'].str.count(\"UU\")\n    df['gg_cnt'] = df['sequence'].str.count(\"GG\")\n    df['begin_sequence'] = df['sequence'].str[0]\n    df['end_sequence'] = df['sequence'].str[-1]\n    df = df.drop(['sequence', 'temporal_cutoff', 'description', 'all_sequences'],axis=1)\n\n    return df\n\ntrain_sequences_fe = feat_eng(train_sequences)\ntest_sequences_fe = feat_eng(test_sequences)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T14:40:17.721948Z","iopub.execute_input":"2025-03-15T14:40:17.722258Z","iopub.status.idle":"2025-03-15T14:40:17.818116Z","shell.execute_reply.started":"2025-03-15T14:40:17.722231Z","shell.execute_reply":"2025-03-15T14:40:17.817018Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Let's check the shape of our df\nprint(train_sequences_fe.shape)\nprint(test_sequences_fe.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T14:40:17.819605Z","iopub.execute_input":"2025-03-15T14:40:17.819944Z","iopub.status.idle":"2025-03-15T14:40:17.838164Z","shell.execute_reply.started":"2025-03-15T14:40:17.819915Z","shell.execute_reply":"2025-03-15T14:40:17.837039Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### The following pre-processing steps join the train_labels & train_sequences datasets, and performs several steps to impute missing data.","metadata":{}},{"cell_type":"code","source":"# create target_id to join with train_sequences\ntrain_labels['target_id'] = train_labels['ID'].str.split('_',expand=True)[0] + '_' + train_labels['ID'].str.split('_',expand=True)[1]\n\n# Combined features with train_labels\ntrain_combined = train_labels.merge(train_sequences_fe,how='left',on='target_id')\n\n# Handling missing values\nprint(f\"There are {train_combined.isna().sum().sum()} missing values before imputation.\")\n\n# Impute missing values with the target_id, resname group average\ntrain_combined['x_1'] = train_combined.groupby(['target_id','resname'])['x_1'].transform(lambda x: x.fillna(x.mean()))\ntrain_combined['y_1'] = train_combined.groupby(['target_id','resname'])['y_1'].transform(lambda x: x.fillna(x.mean()))\ntrain_combined['z_1'] = train_combined.groupby(['target_id','resname'])['z_1'].transform(lambda x: x.fillna(x.mean()))\n\n# Handling missing values\nprint(f\"There are {train_combined.isna().sum().sum()} missing values after imputation.\")\n\n# Some targets only have NA for x,y,z coordinates...we're going to remove those\ntrain_combined = train_combined.dropna()\n\nprint(f\"The final NA count is {train_combined.isna().sum().sum()}.\")\n\n# Drop target_id column...no longer needed\ntrain_combined = train_combined.drop('target_id',axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T14:40:17.839263Z","iopub.execute_input":"2025-03-15T14:40:17.839565Z","iopub.status.idle":"2025-03-15T14:40:21.099255Z","shell.execute_reply.started":"2025-03-15T14:40:17.839540Z","shell.execute_reply":"2025-03-15T14:40:21.097982Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### The following section generates the test_label format for submission and combines that with the test_sequences","metadata":{}},{"cell_type":"code","source":"def generate_test_label(tmp_id, tmp_seq):\n\n    # Take a target_id & sequence, for each sequence, generate a ID resname resid format, append to df for all sequences\n \n    tmp_seq_len = len(tmp_seq)\n    tmp_df = pd.DataFrame()\n    \n    tmp_df['resname'] = [x for x in tmp_seq]\n    tmp_df.insert(0, 'ID', tmp_id) \n    tmp_df['target_id'] = tmp_df['ID']\n    tmp_df['resid'] = list(range(1,tmp_seq_len+1))\n    tmp_df['ID'] = [x+\"_\" + str(y) for x,y in zip(tmp_df['ID'],tmp_df['resid'])]\n    return tmp_df\n    \ntest_labels = pd.DataFrame()\nfor x,y in zip(test_sequences['target_id'],test_sequences['sequence']):\n    test_labels =  pd.concat([test_labels, generate_test_label(x,y)])\n\n# Combined test features with test_labels\ntest_combined = test_labels.merge(test_sequences_fe,how='left',on='target_id')\n# Drop target_id column...no longer needed\ntest_combined = test_combined.drop('target_id',axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T14:40:21.100259Z","iopub.execute_input":"2025-03-15T14:40:21.100617Z","iopub.status.idle":"2025-03-15T14:40:21.158403Z","shell.execute_reply.started":"2025-03-15T14:40:21.100568Z","shell.execute_reply":"2025-03-15T14:40:21.157220Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🎯 Model Training","metadata":{}},{"cell_type":"markdown","source":"### The model approach, for now, is to train three separate models for x, y, & z predictions","metadata":{}},{"cell_type":"code","source":"features = train_combined.drop(['ID','x_1','y_1','z_1'],axis=1).columns.to_list()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T14:40:21.159475Z","iopub.execute_input":"2025-03-15T14:40:21.159889Z","iopub.status.idle":"2025-03-15T14:40:21.174222Z","shell.execute_reply.started":"2025-03-15T14:40:21.159849Z","shell.execute_reply":"2025-03-15T14:40:21.172818Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Predict X ###\ntrain_hframe = h2o.H2OFrame(train_combined.drop(['ID','y_1','z_1'],axis=1)) # will need to separate out x_1,y_1,z_1 & ID for each iteration\n\n# Split the dataset into a train and valid set:\ntrain, valid = train_hframe.split_frame(ratios=[.8], seed=1)\n\nxgb_x = H2OXGBoostEstimator(booster='gbtree', ntrees=10,max_depth=13,learn_rate=.3,tree_method='exact',grow_policy='depthwise',nfolds=5,keep_cross_validation_predictions = True,seed=1)\nxgb_x.train(x = features, y = 'x_1', training_frame = train, validation_frame = valid)\n\n# Model Performance\nprint(xgb_x.mse(valid=True))\n\n\n### Predict Y ###\ntrain_hframe = h2o.H2OFrame(train_combined.drop(['ID','x_1','z_1'],axis=1)) # will need to separate out x_1,y_1,z_1 & ID for each iteration\n\n# Split the dataset into a train and valid set:\ntrain, valid = train_hframe.split_frame(ratios=[.8], seed=1)\n\nxgb_y = H2OXGBoostEstimator(booster='gbtree', ntrees=10,max_depth=13,learn_rate=.3,tree_method='exact',grow_policy='depthwise',nfolds=5,keep_cross_validation_predictions = True,seed=1)\nxgb_y.train(x = features, y = 'y_1', training_frame = train, validation_frame = valid)\n\n# Model Performance\nprint(xgb_y.mse(valid=True))\n\n\n### Predict Z ###\ntrain_hframe = h2o.H2OFrame(train_combined.drop(['ID','y_1','x_1'],axis=1)) # will need to separate out x_1,y_1,z_1 & ID for each iteration\n\n# Split the dataset into a train and valid set:\ntrain, valid = train_hframe.split_frame(ratios=[.8], seed=1)\n\nxgb_z = H2OXGBoostEstimator(booster='gbtree', ntrees=10,max_depth=13,learn_rate=.3,tree_method='exact',grow_policy='depthwise',nfolds=5,keep_cross_validation_predictions = True,seed=1)\nxgb_z.train(x = features, y = 'z_1', training_frame = train, validation_frame = valid)\n\n# Model Performance\nprint(xgb_z.mse(valid=True))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T14:40:21.177087Z","iopub.execute_input":"2025-03-15T14:40:21.177408Z","iopub.status.idle":"2025-03-15T14:41:39.593707Z","shell.execute_reply.started":"2025-03-15T14:40:21.177374Z","shell.execute_reply":"2025-03-15T14:41:39.592235Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🤞 Submission","metadata":{}},{"cell_type":"code","source":"test_hframe = h2o.H2OFrame(test_combined.drop(['ID'],axis=1)) \npreds_x = xgb_x.predict(test_hframe)\npreds_y = xgb_y.predict(test_hframe)\npreds_z = xgb_z.predict(test_hframe)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T14:41:39.595649Z","iopub.execute_input":"2025-03-15T14:41:39.596102Z","iopub.status.idle":"2025-03-15T14:41:40.588720Z","shell.execute_reply.started":"2025-03-15T14:41:39.596047Z","shell.execute_reply":"2025-03-15T14:41:40.587635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop the target_id from test_labels as it is not in the submission format\nsubmission = test_labels.drop('target_id',axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T14:41:40.589737Z","iopub.execute_input":"2025-03-15T14:41:40.590054Z","iopub.status.idle":"2025-03-15T14:41:40.595171Z","shell.execute_reply.started":"2025-03-15T14:41:40.590028Z","shell.execute_reply":"2025-03-15T14:41:40.594202Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### For now, we will use the same coordinates for x_1,y_1,z_1 .... x_5,y_5,z_5.\n\n### Future versions will incorporate different methodologies.","metadata":{}},{"cell_type":"code","source":"submission['x_1'] =  preds_x.as_data_frame(use_pandas=True, header=True,use_multi_thread=True)\nsubmission['y_1'] =preds_y.as_data_frame(use_pandas=True, header=True,use_multi_thread=True)\nsubmission['z_1'] = preds_z.as_data_frame(use_pandas=True, header=True,use_multi_thread=True)\n\nsubmission['x_2'] = preds_x.as_data_frame(use_pandas=True, header=True,use_multi_thread=True)\nsubmission['y_2'] =preds_y.as_data_frame(use_pandas=True, header=True,use_multi_thread=True)\nsubmission['z_2'] = preds_z.as_data_frame(use_pandas=True, header=True,use_multi_thread=True)\n\nsubmission['x_3'] = preds_x.as_data_frame(use_pandas=True, header=True,use_multi_thread=True)\nsubmission['y_3'] =preds_y.as_data_frame(use_pandas=True, header=True,use_multi_thread=True)\nsubmission['z_3'] = preds_z.as_data_frame(use_pandas=True, header=True,use_multi_thread=True)\n\nsubmission['x_4'] = preds_x.as_data_frame(use_pandas=True, header=True,use_multi_thread=True)\nsubmission['y_4'] =preds_y.as_data_frame(use_pandas=True, header=True,use_multi_thread=True)\nsubmission['z_4'] = preds_z.as_data_frame(use_pandas=True, header=True,use_multi_thread=True)\n\nsubmission['x_5'] = preds_x.as_data_frame(use_pandas=True, header=True,use_multi_thread=True)\nsubmission['y_5'] =preds_y.as_data_frame(use_pandas=True, header=True,use_multi_thread=True)\nsubmission['z_5'] = preds_z.as_data_frame(use_pandas=True, header=True,use_multi_thread=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T14:41:40.596323Z","iopub.execute_input":"2025-03-15T14:41:40.596733Z","iopub.status.idle":"2025-03-15T14:41:42.612055Z","shell.execute_reply.started":"2025-03-15T14:41:40.596668Z","shell.execute_reply":"2025-03-15T14:41:42.610717Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# This is what our submission looks like\nsubmission.tail(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T14:45:10.295845Z","iopub.execute_input":"2025-03-15T14:45:10.296282Z","iopub.status.idle":"2025-03-15T14:45:10.318383Z","shell.execute_reply.started":"2025-03-15T14:45:10.296245Z","shell.execute_reply":"2025-03-15T14:45:10.317138Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.index","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T14:44:23.067392Z","iopub.execute_input":"2025-03-15T14:44:23.067859Z","iopub.status.idle":"2025-03-15T14:44:23.075105Z","shell.execute_reply.started":"2025-03-15T14:44:23.067825Z","shell.execute_reply":"2025-03-15T14:44:23.073831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('submission.csv',index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T14:41:42.647795Z","iopub.execute_input":"2025-03-15T14:41:42.648150Z","iopub.status.idle":"2025-03-15T14:41:42.746145Z","shell.execute_reply.started":"2025-03-15T14:41:42.648060Z","shell.execute_reply":"2025-03-15T14:41:42.745038Z"}},"outputs":[],"execution_count":null}]}