{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":12024591,"sourceType":"competition"},{"sourceId":11588653,"sourceType":"datasetVersion","datasetId":7266502},{"sourceId":11589946,"sourceType":"datasetVersion","datasetId":7267481}],"dockerImageVersionId":31012,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport os\nimport time\nimport gc\nimport traceback\nfrom collections import Counter\n\n# Data Manipulation Libraries\nimport numpy as np\nimport pandas as pd\n\n# Machine Learning Libraries\nimport tensorflow as tf\nfrom sklearn.model_selection import train_test_split\n\n# Visualization Libraries\nimport matplotlib.pyplot as plt\nimport matplotlib.colors as mcolors\n\n# Configuration\nwarnings = __import__('warnings')  # Import warnings\nwarnings.filterwarnings('ignore')  # Suppress warnings\nnp.random.seed(0)\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:02.064662Z","iopub.execute_input":"2025-05-05T22:01:02.064978Z","iopub.status.idle":"2025-05-05T22:01:02.072158Z","shell.execute_reply.started":"2025-05-05T22:01:02.064956Z","shell.execute_reply":"2025-05-05T22:01:02.070737Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Import the sequence and labels dataframes\n\ntrain_labels = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/train_labels.csv')\ntrain_sequences = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/train_sequences.csv')\n\n#Merge the training dataframe with the labels to generate a single dataframe\ntrain_labels['ID'] = train_labels['ID'].str.rsplit('_', n=1).str[0]\ntrain_df  = train_labels.merge(how = 'left' , left_on = 'ID' , right_on = 'target_id' , right = train_sequences )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:02.073701Z","iopub.execute_input":"2025-05-05T22:01:02.074042Z","iopub.status.idle":"2025-05-05T22:01:02.750728Z","shell.execute_reply.started":"2025-05-05T22:01:02.074018Z","shell.execute_reply":"2025-05-05T22:01:02.749562Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Fill the Null value of coordinates with the mean coordinates of the same residue in the same structure\n\nfor col in ['x_1', 'y_1', 'z_1']:\n        train_df[col] = train_df.groupby(['target_id', 'resname'])[col].transform(lambda x: x.fillna(x.mean()))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:02.751644Z","iopub.execute_input":"2025-05-05T22:01:02.751897Z","iopub.status.idle":"2025-05-05T22:01:04.689242Z","shell.execute_reply.started":"2025-05-05T22:01:02.751878Z","shell.execute_reply":"2025-05-05T22:01:04.688499Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:04.692012Z","iopub.execute_input":"2025-05-05T22:01:04.692292Z","iopub.status.idle":"2025-05-05T22:01:04.707701Z","shell.execute_reply.started":"2025-05-05T22:01:04.692270Z","shell.execute_reply":"2025-05-05T22:01:04.706858Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:04.708607Z","iopub.execute_input":"2025-05-05T22:01:04.708959Z","iopub.status.idle":"2025-05-05T22:01:04.734992Z","shell.execute_reply.started":"2025-05-05T22:01:04.708918Z","shell.execute_reply":"2025-05-05T22:01:04.734157Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Relative Resid calculates the relative position of the residue in the sequence by dividing the \n#Residue number with the sequence length. This helps us obtain the normalized relative position of the\n#reside between 0-1. Values closer to 1 denote residues closer to end of sequence and those near 0s are at the start\n\ntrain_df['relative_resid'] = (train_df['resid']/train_df['sequence'].apply(len)).astype('float')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:04.735957Z","iopub.execute_input":"2025-05-05T22:01:04.736203Z","iopub.status.idle":"2025-05-05T22:01:04.800204Z","shell.execute_reply.started":"2025-05-05T22:01:04.736185Z","shell.execute_reply":"2025-05-05T22:01:04.799186Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Keep only the important columns that will be used\ntrain_df = train_df[['sequence','x_1','y_1','z_1','resname','resid','relative_resid']] #,'chains'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:04.801177Z","iopub.execute_input":"2025-05-05T22:01:04.801425Z","iopub.status.idle":"2025-05-05T22:01:04.811222Z","shell.execute_reply.started":"2025-05-05T22:01:04.801406Z","shell.execute_reply":"2025-05-05T22:01:04.810331Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = train_df.dropna()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:04.812218Z","iopub.execute_input":"2025-05-05T22:01:04.812596Z","iopub.status.idle":"2025-05-05T22:01:04.852299Z","shell.execute_reply.started":"2025-05-05T22:01:04.812563Z","shell.execute_reply":"2025-05-05T22:01:04.851293Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:04.853359Z","iopub.execute_input":"2025-05-05T22:01:04.853682Z","iopub.status.idle":"2025-05-05T22:01:04.868160Z","shell.execute_reply.started":"2025-05-05T22:01:04.853660Z","shell.execute_reply":"2025-05-05T22:01:04.867115Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/sample_submission.csv')\nsubmission['ID']  = submission['ID'].str.rsplit('_' ,n =1  ).str[0]\ntest_sequences = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/test_sequences.csv')\ntest  = test_sequences.merge(how = 'left' , left_on = 'target_id' , right_on = 'ID' , right = submission)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:04.871575Z","iopub.execute_input":"2025-05-05T22:01:04.871917Z","iopub.status.idle":"2025-05-05T22:01:04.903143Z","shell.execute_reply.started":"2025-05-05T22:01:04.871893Z","shell.execute_reply":"2025-05-05T22:01:04.902342Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Returns the relative residue id of each nucleotide, which represents the normalized position of the\n#nucleotide in the sequence. Since the resid usually contains numbers 1,2,3,4... denoting positions of each\n#nucleotide the relative_resid will return its relative position in the range [0-1] providing a sense of how closer \n#to beginning (near 0) or end of the sequence (near 1) the nucleotide is.\n\ntest['relative_resid'] = (test['resid']/test['sequence'].apply(len)).astype('float')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:04.904243Z","iopub.execute_input":"2025-05-05T22:01:04.904554Z","iopub.status.idle":"2025-05-05T22:01:04.911902Z","shell.execute_reply.started":"2025-05-05T22:01:04.904524Z","shell.execute_reply":"2025-05-05T22:01:04.910915Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected_cols = ['sequence','resname','ID','resid','relative_resid'] #,'chains'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:04.912968Z","iopub.execute_input":"2025-05-05T22:01:04.913848Z","iopub.status.idle":"2025-05-05T22:01:04.933864Z","shell.execute_reply.started":"2025-05-05T22:01:04.913812Z","shell.execute_reply":"2025-05-05T22:01:04.932738Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = test[selected_cols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:04.934837Z","iopub.execute_input":"2025-05-05T22:01:04.935111Z","iopub.status.idle":"2025-05-05T22:01:04.952802Z","shell.execute_reply.started":"2025-05-05T22:01:04.935091Z","shell.execute_reply":"2025-05-05T22:01:04.951592Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in selected_cols:\n    # Handle potential NaN values before encoding, substitute them with NE\n    test[col] = test[col].astype(str).fillna('NE')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:04.954054Z","iopub.execute_input":"2025-05-05T22:01:04.954372Z","iopub.status.idle":"2025-05-05T22:01:04.980171Z","shell.execute_reply.started":"2025-05-05T22:01:04.954348Z","shell.execute_reply":"2025-05-05T22:01:04.979056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:04.981356Z","iopub.execute_input":"2025-05-05T22:01:04.981788Z","iopub.status.idle":"2025-05-05T22:01:05.011199Z","shell.execute_reply.started":"2025-05-05T22:01:04.981759Z","shell.execute_reply":"2025-05-05T22:01:05.009908Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# File paths\nDATA_DIR = \"/kaggle/input/stanford-rna-3d-folding/\"\nOUTPUT_DIR = \"/kaggle/working/\"\nos.makedirs(OUTPUT_DIR, exist_ok=True)\n\n#Function to one hot encode a single character to a vector of 5 dimensions. Also handles unknown letters.\ndef one_hot_enc(char):\n    end_stream = []\n    if char == 'A':\n        end_stream = [1, 0, 0, 0, 0]\n    elif char == 'C':\n        end_stream = [0, 1, 0, 0, 0]\n    elif char == 'G':\n        end_stream = [0, 0, 1, 0, 0]\n    elif char == 'U':\n        end_stream = [0, 0, 0, 1, 0]\n    else:\n        end_stream = [0, 0, 0, 0, 1]\n    return end_stream\n\n### Function to process the features and create processed data for training and testing\ndef create_test_data(dataframe, train=False):\n    X_data = []; y_data = [];\n    previous_nucleotide='-'; previous_seq = '-'\n\n    '''The following loop iterates through the nuceotides of each sequence and uses both the local and global sequence\n    information to create new features'''\n    for index, row in dataframe.iterrows():\n        \n        seq = row['sequence']                     #Gets the whole sequence as a string\n        nucleotide = row['resname']               #Gets the current nucleotide (A,U,G,C)\n        resid = int(row['resid']);                #Gets the residue id/index.\n        rel_resid = float(row['relative_resid'])  #Gets the relative residue id (given by resid/sequence_length) \n        \n        '''The following condition gets the current and previous nucleotides in the sequence, \n        one hot enocode them and concates the result.Since neighboring nucleotides can be base pairs, \n        this gives an indication of potential base-pairs'''\n        if seq==previous_seq:\n            current_base_pair = one_hot_enc(nucleotide) + one_hot_enc(previous_nucleotide) #dataframe.iloc[index-1,dataframe.columns.get_loc('resname')]#[]\n        else:\n            current_base_pair = one_hot_enc(nucleotide) + one_hot_enc('-')\n        previous_nucleotide = nucleotide; previous_seq = seq\n\n        '''Gets the pairs of nucleotide in the start and end of the sequence and one_hot encodes them. The result\n        are stored in variables: starting_basepair and end_basepair'''\n        if(len(seq)<2):\n            nucleotide_beg = seq[0]+'-'; nucleotide_end = '-'+seq[0]\n        else:\n            nucleotide_beg = seq[0:2]; nucleotide_end = seq[-2:]\n        starting_basepair =  one_hot_enc(nucleotide_beg[0]) + one_hot_enc(nucleotide_beg[1])\n        end_basepair = one_hot_enc(nucleotide_end[-2]) + one_hot_enc(nucleotide_end[-1])\n        \n        \n        features = []\n        length = len(seq)\n\n        #Get the GC content which is essentially the sum total of the G and C nucleotides in the sequence and indicates the potentially available pairs for forming G-C base-pairs which are useful for prediction.\n        gc_content = seq.count('G') + seq.count('C')\n        \n        #Similarly AU content, which is sum of A + U nucleotides essentially denotes the potentially available pairs that can form A-U bond/base-pair.\n        au_content = seq.count('A') + seq.count('U')\n\n        #We now add the total count of all possible two lettered pairs of numbers that are possible. \n        au_cnt = seq.count('AU'); aur_cnt = seq.count('UA')\n        gc_cnt = seq.count('GC'); gcr_cnt = seq.count('CG')\n        rem_counts = [length, seq.count('AC'),seq.count('CA'),seq.count('CU'),seq.count('UC'),seq.count('AG'),seq.count('GA'),seq.count('AA'),seq.count('GG'),seq.count('GU'),seq.count('UG'),seq.count('UC'),seq.count('CU'),seq.count('CC'),seq.count('UU'),seq.count('GGGA'), seq.count('GAAA'),seq.count('GAGA'),seq.count('GUGA')] #,chains,seq.count('GUGA') ,seq.count('UUCG'),seq.count('GGGA') #\n\n        #Append all the features to X_data\n        X_data.append(current_base_pair+starting_basepair+[gc_content, au_content, au_cnt, aur_cnt, gc_cnt, gcr_cnt]+rem_counts+end_basepair+[resid,rel_resid])\n        \n        if train:\n            y_data.append([row['x_1'],row['y_1'],row['z_1']])\n            \n    if not train:\n        return X_data\n    else:\n        return X_data, y_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:05.012245Z","iopub.execute_input":"2025-05-05T22:01:05.012622Z","iopub.status.idle":"2025-05-05T22:01:05.039404Z","shell.execute_reply.started":"2025-05-05T22:01:05.012592Z","shell.execute_reply":"2025-05-05T22:01:05.038440Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error, make_scorer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:05.040593Z","iopub.execute_input":"2025-05-05T22:01:05.040876Z","iopub.status.idle":"2025-05-05T22:01:05.063415Z","shell.execute_reply.started":"2025-05-05T22:01:05.040855Z","shell.execute_reply":"2025-05-05T22:01:05.062172Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import optuna\n## Additional code for \n#### HyperparameterOptimization\ndef objective2(trial):     \n     params = {\n         \"n_estimators\": trial.suggest_int(\"n_estimators\", 100, 900),\n         \"learning_rate\": trial.suggest_float(\"learning_rate\", 1e-4, 0.2, log=True),\n         \"num_leaves\": trial.suggest_int(\"num_leaves\", 20, 300),\n         \"max_depth\": trial.suggest_int(\"max_depth\", 3, 25),\n         \"min_child_samples\": trial.suggest_int(\"min_child_samples\", 5, 150),\n         \"subsample\": trial.suggest_float(\"subsample\", 0.5, 1.0),\n         \"colsample_bytree\": trial.suggest_float(\"colsample_bytree\", 0.5, 1.0),\n         \"reg_alpha\": trial.suggest_float(\"reg_alpha\", 1e-4, 10.0, log=True),\n         \"reg_lambda\": trial.suggest_float(\"reg_lambda\", 1e-4, 10.0, log=True),\n         \"random_state\": 42,\n         \"device\" : 'cpu',\n         \"n_jobs\": -1,\n     }\n\n     # Train one regressor per coordinate\n     losses = []\n     #print(i)\n     model = MultiOutputRegressor(xgb.XGBRegressor(verbose=-1,**params))\n    \n     kf = KFold(n_splits=3, shuffle=True, random_state=42)\n     mse_scorer = make_scorer(mean_squared_error, greater_is_better=False)\n     scores = cross_val_score(model, X_train, y_train, cv=kf, scoring=mse_scorer)\n     #mse_scorer = make_scorer(mean_squared_error, greater_is_better=False)\n     print(-scores)\n    \n     return -scores.mean()#sum(losses) / len(losses)\n\n#study2 = optuna.create_study(direction=\"minimize\")\n#study2.optimize(objective2, n_trials=20 )\n\n#print(\"Best params:\", study.best_params)\n\n'''\nBest params: {'n_estimators': 430, 'max_depth': 16, 'reg_alpha': 3.143887428307033}\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:05.064247Z","iopub.execute_input":"2025-05-05T22:01:05.064526Z","iopub.status.idle":"2025-05-05T22:01:05.080215Z","shell.execute_reply.started":"2025-05-05T22:01:05.064476Z","shell.execute_reply":"2025-05-05T22:01:05.078954Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nfrom sklearn.model_selection import cross_val_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:05.081343Z","iopub.execute_input":"2025-05-05T22:01:05.081692Z","iopub.status.idle":"2025-05-05T22:01:05.102389Z","shell.execute_reply.started":"2025-05-05T22:01:05.081666Z","shell.execute_reply":"2025-05-05T22:01:05.101366Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data = create_test_data(test,train=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:05.103610Z","iopub.execute_input":"2025-05-05T22:01:05.103944Z","iopub.status.idle":"2025-05-05T22:01:05.354951Z","shell.execute_reply.started":"2025-05-05T22:01:05.103916Z","shell.execute_reply":"2025-05-05T22:01:05.354087Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:05.356166Z","iopub.execute_input":"2025-05-05T22:01:05.356681Z","iopub.status.idle":"2025-05-05T22:01:05.371978Z","shell.execute_reply.started":"2025-05-05T22:01:05.356645Z","shell.execute_reply":"2025-05-05T22:01:05.371015Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, y_train = create_test_data(train_df,train=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:05.373151Z","iopub.execute_input":"2025-05-05T22:01:05.373527Z","iopub.status.idle":"2025-05-05T22:01:41.391464Z","shell.execute_reply.started":"2025-05-05T22:01:05.373461Z","shell.execute_reply":"2025-05-05T22:01:41.390358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(X_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:41.392616Z","iopub.execute_input":"2025-05-05T22:01:41.393838Z","iopub.status.idle":"2025-05-05T22:01:41.399166Z","shell.execute_reply.started":"2025-05-05T22:01:41.393807Z","shell.execute_reply":"2025-05-05T22:01:41.398323Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features=len(X_train[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:41.400317Z","iopub.execute_input":"2025-05-05T22:01:41.400595Z","iopub.status.idle":"2025-05-05T22:01:41.418029Z","shell.execute_reply.started":"2025-05-05T22:01:41.400576Z","shell.execute_reply":"2025-05-05T22:01:41.416928Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#valx.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:41.419395Z","iopub.execute_input":"2025-05-05T22:01:41.419731Z","iopub.status.idle":"2025-05-05T22:01:41.439096Z","shell.execute_reply.started":"2025-05-05T22:01:41.419709Z","shell.execute_reply":"2025-05-05T22:01:41.438218Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train = np.reshape(np.array(X_train),[-1,features])\ny_train = np.reshape(np.array(y_train),[-1,3])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:41.440659Z","iopub.execute_input":"2025-05-05T22:01:41.441035Z","iopub.status.idle":"2025-05-05T22:01:42.375348Z","shell.execute_reply.started":"2025-05-05T22:01:41.441006Z","shell.execute_reply":"2025-05-05T22:01:42.374374Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#trainx , valx , trainy , valy = train_test_split(X_train , y_train , random_state = 0 , test_size = 0.25)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:42.376288Z","iopub.execute_input":"2025-05-05T22:01:42.376636Z","iopub.status.idle":"2025-05-05T22:01:42.382002Z","shell.execute_reply.started":"2025-05-05T22:01:42.376608Z","shell.execute_reply":"2025-05-05T22:01:42.380755Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"nan_indices = np.isnan(y_train).any(axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:42.386538Z","iopub.execute_input":"2025-05-05T22:01:42.386840Z","iopub.status.idle":"2025-05-05T22:01:42.413742Z","shell.execute_reply.started":"2025-05-05T22:01:42.386819Z","shell.execute_reply":"2025-05-05T22:01:42.412602Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train = X_train[~nan_indices]\ny_train = y_train[~nan_indices]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:42.414725Z","iopub.execute_input":"2025-05-05T22:01:42.415002Z","iopub.status.idle":"2025-05-05T22:01:42.461799Z","shell.execute_reply.started":"2025-05-05T22:01:42.414980Z","shell.execute_reply":"2025-05-05T22:01:42.460836Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = np.reshape(np.array(test_data),[-1,features]) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:42.462832Z","iopub.execute_input":"2025-05-05T22:01:42.463149Z","iopub.status.idle":"2025-05-05T22:01:42.483301Z","shell.execute_reply.started":"2025-05-05T22:01:42.463123Z","shell.execute_reply":"2025-05-05T22:01:42.482361Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Found after hyperparameter optimization\nparams={'n_estimators': 430, 'max_depth': 16, 'reg_alpha': 3.143887428307033}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:42.484569Z","iopub.execute_input":"2025-05-05T22:01:42.484926Z","iopub.status.idle":"2025-05-05T22:01:42.502461Z","shell.execute_reply.started":"2025-05-05T22:01:42.484897Z","shell.execute_reply":"2025-05-05T22:01:42.501430Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import xgboost as xgb\nfrom sklearn.multioutput import MultiOutputRegressor\n\n#Multioutput regressor predicts multiple target variables together (here x,y, and z coordinates)\nmultioutputregressor = MultiOutputRegressor(xgb.XGBRegressor(objective='reg:linear',**params))#.fit(trainx, trainy)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:42.503703Z","iopub.execute_input":"2025-05-05T22:01:42.503965Z","iopub.status.idle":"2025-05-05T22:01:42.526909Z","shell.execute_reply.started":"2025-05-05T22:01:42.503947Z","shell.execute_reply":"2025-05-05T22:01:42.525873Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"multioutputregressor.fit(X_train,y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:01:42.527915Z","iopub.execute_input":"2025-05-05T22:01:42.528237Z","iopub.status.idle":"2025-05-05T22:03:38.143450Z","shell.execute_reply.started":"2025-05-05T22:01:42.528210Z","shell.execute_reply":"2025-05-05T22:03:38.142539Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#multioutputregressor.estimators_[0].feature_importances_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:03:38.144409Z","iopub.execute_input":"2025-05-05T22:03:38.144701Z","iopub.status.idle":"2025-05-05T22:03:38.148894Z","shell.execute_reply.started":"2025-05-05T22:03:38.144681Z","shell.execute_reply":"2025-05-05T22:03:38.147934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:03:38.149802Z","iopub.execute_input":"2025-05-05T22:03:38.150107Z","iopub.status.idle":"2025-05-05T22:03:38.170644Z","shell.execute_reply.started":"2025-05-05T22:03:38.150079Z","shell.execute_reply":"2025-05-05T22:03:38.169403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#mse = mean_squared_error(trainy, multioutputregressor.predict(trainx))\nmse = mean_squared_error(y_train, multioutputregressor.predict(X_train))\n#mse_val = mean_squared_error(valy, multioutputregressor.predict(valx))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:03:38.172061Z","iopub.execute_input":"2025-05-05T22:03:38.172376Z","iopub.status.idle":"2025-05-05T22:03:46.648923Z","shell.execute_reply.started":"2025-05-05T22:03:38.172350Z","shell.execute_reply":"2025-05-05T22:03:46.647792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mse","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:03:46.649600Z","iopub.execute_input":"2025-05-05T22:03:46.649836Z","iopub.status.idle":"2025-05-05T22:03:46.655592Z","shell.execute_reply.started":"2025-05-05T22:03:46.649818Z","shell.execute_reply":"2025-05-05T22:03:46.654699Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions = multioutputregressor.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:03:46.656711Z","iopub.execute_input":"2025-05-05T22:03:46.657031Z","iopub.status.idle":"2025-05-05T22:03:46.840535Z","shell.execute_reply.started":"2025-05-05T22:03:46.656998Z","shell.execute_reply":"2025-05-05T22:03:46.839891Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:03:46.841071Z","iopub.execute_input":"2025-05-05T22:03:46.841273Z","iopub.status.idle":"2025-05-05T22:03:46.848630Z","shell.execute_reply.started":"2025-05-05T22:03:46.841255Z","shell.execute_reply":"2025-05-05T22:03:46.845878Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_submission_template_final(test_df, sample_submission_df,preds):\n    \"\"\"\n    Creates a submission template based on test data.\n    \"\"\"\n    # Check if sample_submission.csv is available\n    if sample_submission_df is None:\n        print(\"Sample submission file not found. Creating a new template.\")\n        \n        # Create a new DataFrame for submission\n        submission_df = pd.DataFrame()\n        \n        # Example code to fill the template (adjust as needed)\n        ids = []\n        resnames = []\n        resids = []\n        \n        for _, row in test_df.iterrows():\n            sequence = row['sequence']\n            target_id = row['target_id']\n        #    \n            for i, nucleotide in enumerate(sequence, 1):\n                ids.append(f\"{target_id}_{i}\")\n                resnames.append(nucleotide)\n                resids.append(i)\n        \n        submission_df['ID'] = ids\n        submission_df['resname'] = resnames\n        submission_df['resid'] = resids\n        \n        # Add coordinate columns (5 structures)\n        for i in range(1, 6):\n            submission_df[f'x_{i}'] = preds[:,0]\n            submission_df[f'y_{i}'] = preds[:,1]\n            submission_df[f'z_{i}'] = preds[:,2]\n    else:\n        submission_df = sample_submission_df.copy()\n        print(\"Submission template created based on the provided example.\")\n    \n    return submission_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:03:46.851044Z","iopub.execute_input":"2025-05-05T22:03:46.851398Z","iopub.status.idle":"2025-05-05T22:03:46.870992Z","shell.execute_reply.started":"2025-05-05T22:03:46.851348Z","shell.execute_reply":"2025-05-05T22:03:46.869830Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:03:46.871885Z","iopub.execute_input":"2025-05-05T22:03:46.872129Z","iopub.status.idle":"2025-05-05T22:03:46.905827Z","shell.execute_reply.started":"2025-05-05T22:03:46.872111Z","shell.execute_reply":"2025-05-05T22:03:46.904674Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:03:46.906816Z","iopub.execute_input":"2025-05-05T22:03:46.907080Z","iopub.status.idle":"2025-05-05T22:03:46.913522Z","shell.execute_reply.started":"2025-05-05T22:03:46.907060Z","shell.execute_reply":"2025-05-05T22:03:46.912040Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df = create_submission_template_final(test_sequences,None,predictions)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:03:46.914696Z","iopub.execute_input":"2025-05-05T22:03:46.915076Z","iopub.status.idle":"2025-05-05T22:03:46.942048Z","shell.execute_reply.started":"2025-05-05T22:03:46.915045Z","shell.execute_reply":"2025-05-05T22:03:46.940987Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:03:46.943313Z","iopub.execute_input":"2025-05-05T22:03:46.943629Z","iopub.status.idle":"2025-05-05T22:03:46.966821Z","shell.execute_reply.started":"2025-05-05T22:03:46.943606Z","shell.execute_reply":"2025-05-05T22:03:46.965792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reverse_mapping(encoded):\n    seq=[]\n    for c in encoded:\n        pos=-1\n        if 1 in c:\n            pos = list(c).index(1)\n        if pos == 0:\n            seq.append('A')\n        elif pos==1:\n            seq.append('C')\n        elif pos==2:\n            seq.append('G')\n        elif pos==3:\n            seq.append('U')\n        elif pos==4:\n            seq.append('N')\n        else:\n            seq.append('K')\n    return seq","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:03:46.967908Z","iopub.execute_input":"2025-05-05T22:03:46.968218Z","iopub.status.idle":"2025-05-05T22:03:46.985806Z","shell.execute_reply.started":"2025-05-05T22:03:46.968192Z","shell.execute_reply":"2025-05-05T22:03:46.984671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df.to_csv('submission.csv',index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:03:46.986734Z","iopub.execute_input":"2025-05-05T22:03:46.987011Z","iopub.status.idle":"2025-05-05T22:03:47.061463Z","shell.execute_reply.started":"2025-05-05T22:03:46.986985Z","shell.execute_reply":"2025-05-05T22:03:47.060350Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}