{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":87793,"databundleVersionId":11553390,"sourceType":"competition"}],"dockerImageVersionId":31011,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-13T08:50:15.643955Z","iopub.execute_input":"2025-04-13T08:50:15.644656Z","iopub.status.idle":"2025-04-13T08:50:15.936284Z","shell.execute_reply.started":"2025-04-13T08:50:15.644611Z","shell.execute_reply":"2025-04-13T08:50:15.935644Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_labels = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/train_labels.csv')\ntrain_sequences = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/train_sequences.csv')\nsubmission = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/sample_submission.csv')\ntest_sequences = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/test_sequences.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T08:50:16.008873Z","iopub.execute_input":"2025-04-13T08:50:16.009436Z","iopub.status.idle":"2025-04-13T08:50:16.400002Z","shell.execute_reply.started":"2025-04-13T08:50:16.009413Z","shell.execute_reply":"2025-04-13T08:50:16.399412Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## left join ##\ntrain_labels['ID'] = train_labels['ID'].str.rsplit('_', n=1).str[0]\ntrain_df  = train_labels.merge(how = 'left' , left_on = 'ID' , right_on = 'target_id' , right = train_sequences )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T08:50:16.834373Z","iopub.execute_input":"2025-04-13T08:50:16.834711Z","iopub.status.idle":"2025-04-13T08:50:17.126426Z","shell.execute_reply.started":"2025-04-13T08:50:16.834672Z","shell.execute_reply":"2025-04-13T08:50:17.125670Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('No of rows and cols ' , train_df.shape)\nprint()\nprint('Missing values' , train_df.isna().sum())\nprint()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T08:50:17.142419Z","iopub.execute_input":"2025-04-13T08:50:17.142636Z","iopub.status.idle":"2025-04-13T08:50:17.199919Z","shell.execute_reply.started":"2025-04-13T08:50:17.142619Z","shell.execute_reply":"2025-04-13T08:50:17.199270Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"##datetime conversion ###\ntrain_df['temporal_cutoff'] = pd.to_datetime(train_df['temporal_cutoff']).astype('int64') // 10**9\ntest_sequences['temporal_cutoff'] = pd.to_datetime(test_sequences['temporal_cutoff']).astype('int64') // 10**9\n## using len of sequence \ntrain_df['seq_length'] = train_df['sequence'].str.len()\ntest_sequences['seq_length'] =  test_sequences['sequence'].str.len()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T08:50:17.325202Z","iopub.execute_input":"2025-04-13T08:50:17.325864Z","iopub.status.idle":"2025-04-13T08:50:17.414539Z","shell.execute_reply.started":"2025-04-13T08:50:17.325832Z","shell.execute_reply":"2025-04-13T08:50:17.413822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission['ID']  = submission['ID'].str.rsplit('_' ,n =1  ).str[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T08:50:19.030265Z","iopub.execute_input":"2025-04-13T08:50:19.030825Z","iopub.status.idle":"2025-04-13T08:50:19.037288Z","shell.execute_reply.started":"2025-04-13T08:50:19.030797Z","shell.execute_reply":"2025-04-13T08:50:19.036680Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test  = test_sequences.merge(how = 'left' , left_on = 'target_id' , right_on = 'ID' , right = submission)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T08:50:19.305727Z","iopub.execute_input":"2025-04-13T08:50:19.306269Z","iopub.status.idle":"2025-04-13T08:50:19.314147Z","shell.execute_reply.started":"2025-04-13T08:50:19.306245Z","shell.execute_reply":"2025-04-13T08:50:19.313431Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define columns\ncategorical_cols = ['resname', 'target_id', 'description', 'all_sequences']\nnumerical_cols = ['temporal_cutoff', 'resid', 'seq_length']\ntarget_cols = ['x_1', 'y_1', 'z_1']\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T08:50:20.449771Z","iopub.execute_input":"2025-04-13T08:50:20.450035Z","iopub.status.idle":"2025-04-13T08:50:20.454113Z","shell.execute_reply.started":"2025-04-13T08:50:20.450014Z","shell.execute_reply":"2025-04-13T08:50:20.453499Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Encode categoricals\nfrom sklearn.preprocessing import LabelEncoder\n\n# Encode categoricals\ndef encode_categoricals(df, categorical_cols):\n    df_encoded = df.copy()\n    encoders = {}\n    for col in categorical_cols:\n        le = LabelEncoder()\n        df_encoded[col] = le.fit_transform(df_encoded[col].astype(str))\n        encoders[col] = le\n    return df_encoded, encoders","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T08:50:21.049873Z","iopub.execute_input":"2025-04-13T08:50:21.050128Z","iopub.status.idle":"2025-04-13T08:50:21.478986Z","shell.execute_reply.started":"2025-04-13T08:50:21.050109Z","shell.execute_reply":"2025-04-13T08:50:21.478245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## dropna values \ntrain_df.dropna(inplace = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T08:50:22.103094Z","iopub.execute_input":"2025-04-13T08:50:22.103948Z","iopub.status.idle":"2025-04-13T08:50:22.164085Z","shell.execute_reply.started":"2025-04-13T08:50:22.103924Z","shell.execute_reply":"2025-04-13T08:50:22.163354Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prepare data\ndf_encoded, encoders = encode_categoricals(train_df, categorical_cols)\nX = df_encoded[categorical_cols + numerical_cols]\ny = df_encoded[target_cols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T08:50:22.693882Z","iopub.execute_input":"2025-04-13T08:50:22.694192Z","iopub.status.idle":"2025-04-13T08:50:22.815241Z","shell.execute_reply.started":"2025-04-13T08:50:22.694163Z","shell.execute_reply":"2025-04-13T08:50:22.814510Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrainx , testx , trainy , testy = train_test_split(X , y , random_state = 0 , test_size = 0.25)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T08:50:23.930689Z","iopub.execute_input":"2025-04-13T08:50:23.931239Z","iopub.status.idle":"2025-04-13T08:50:24.038100Z","shell.execute_reply.started":"2025-04-13T08:50:23.931215Z","shell.execute_reply":"2025-04-13T08:50:24.037352Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # # Objective function for Optuna\n\n# import optuna\n# from lightgbm import LGBMRegressor\n# from sklearn.metrics import mean_squared_error\n\n# import warnings\n\n# warnings.filterwarnings(\"ignore\", category=UserWarning)\n# warnings.filterwarnings(\"ignore\", category=FutureWarning)\n\n\n# # Prepare training data (from your earlier split)\n# X_train = trainx[categorical_cols + numerical_cols]\n# y_train = trainy[target_cols]\n\n# X_valid = testx[categorical_cols + numerical_cols]\n# y_valid = testy[target_cols]\n\n# def objective(trial):\n#     params = {\n#         \"n_estimators\": trial.suggest_int(\"n_estimators\", 100, 2000),\n#         \"learning_rate\": trial.suggest_float(\"learning_rate\", 1e-4, 0.1, log=True),\n#         \"num_leaves\": trial.suggest_int(\"num_leaves\", 20, 300),\n#         \"max_depth\": trial.suggest_int(\"max_depth\", 3, 30),\n#         \"min_child_samples\": trial.suggest_int(\"min_child_samples\", 5, 150),\n#         \"subsample\": trial.suggest_float(\"subsample\", 0.5, 1.0),\n#         \"colsample_bytree\": trial.suggest_float(\"colsample_bytree\", 0.5, 1.0),\n#         \"reg_alpha\": trial.suggest_float(\"reg_alpha\", 1e-4, 10.0, log=True),\n#         \"reg_lambda\": trial.suggest_float(\"reg_lambda\", 1e-4, 10.0, log=True),\n#         \"random_state\": 42,\n#         \"n_jobs\": -1,\n#         \"device\" : 'gpu' ,\n#         \"verbose\" : -1 , \n#     }\n\n#     # Train one regressor per coordinate\n#     losses = []\n#     for target in target_cols:\n#         model = LGBMRegressor(**params)\n#         model.fit(X_train, y_train[target])\n#         preds = model.predict(X_valid)\n#         mse = mean_squared_error(y_valid[target], preds)\n#         losses.append(mse)\n    \n#     return sum(losses) / len(losses)  # average loss\n\n# # Run Optuna optimization\n# study = optuna.create_study(direction=\"minimize\")\n# study.optimize(objective, n_trials=100 )\n\n# print(\"Best params:\", study.best_params)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T08:52:04.576160Z","iopub.execute_input":"2025-04-13T08:52:04.576873Z","iopub.status.idle":"2025-04-13T10:46:09.851921Z","shell.execute_reply.started":"2025-04-13T08:52:04.576850Z","shell.execute_reply":"2025-04-13T10:46:09.851396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(\"Best params:\", study.best_params)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T08:27:10.577678Z","iopub.execute_input":"2025-04-13T08:27:10.577937Z","iopub.status.idle":"2025-04-13T08:27:10.601955Z","shell.execute_reply.started":"2025-04-13T08:27:10.577916Z","shell.execute_reply":"2025-04-13T08:27:10.600972Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Best parameters obtained from optuna and model training \n\nfrom lightgbm import LGBMRegressor\nfrom sklearn.metrics import mean_squared_error\nimport warnings\n\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\n\nX_train = trainx[categorical_cols + numerical_cols]\ny_train = trainy[target_cols]\n\nX_valid = testx[categorical_cols + numerical_cols]\ny_valid = testy[target_cols]\n\nparams =  {'n_estimators': 1886, 'learning_rate': 0.028848642094635626, 'num_leaves': 161, 'max_depth': 29, 'min_child_samples': 26, \n                'subsample': 0.7474245117029201, 'colsample_bytree': 0.9882183006199234, \n                'reg_alpha': 0.0001174276705873561, 'reg_lambda': 0.0039170175656948825}\nmodels = {}\nfor target in target_cols:\n    lgb_reg = LGBMRegressor(**params)\n    lgb_reg.fit(X_train, y_train[target])\n    models[target] = lgb_reg\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T10:50:55.777119Z","iopub.execute_input":"2025-04-13T10:50:55.777643Z","iopub.status.idle":"2025-04-13T10:51:27.982243Z","shell.execute_reply.started":"2025-04-13T10:50:55.777620Z","shell.execute_reply":"2025-04-13T10:51:27.981681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T10:51:27.983330Z","iopub.execute_input":"2025-04-13T10:51:27.983614Z","iopub.status.idle":"2025-04-13T10:51:28.000643Z","shell.execute_reply.started":"2025-04-13T10:51:27.983589Z","shell.execute_reply":"2025-04-13T10:51:28.000116Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_test = test[categorical_cols + numerical_cols]\nx_test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T10:51:28.001206Z","iopub.execute_input":"2025-04-13T10:51:28.001478Z","iopub.status.idle":"2025-04-13T10:51:28.010123Z","shell.execute_reply.started":"2025-04-13T10:51:28.001461Z","shell.execute_reply":"2025-04-13T10:51:28.009604Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"##  Data Preprocessing for inference\nfrom sklearn.preprocessing import LabelEncoder\n\n\n\nfor col in categorical_cols:\n    # Handle potential NaN values before encoding\n    test[col] = test[col].astype(str).fillna('missing')\n    \n    # Create and fit label encoder\n    le = LabelEncoder()\n    test[col] = le.fit_transform(test[col])\n    encoders[col] = le","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T10:51:28.011703Z","iopub.execute_input":"2025-04-13T10:51:28.011883Z","iopub.status.idle":"2025-04-13T10:51:28.028085Z","shell.execute_reply.started":"2025-04-13T10:51:28.011870Z","shell.execute_reply":"2025-04-13T10:51:28.027452Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_test = test[categorical_cols + numerical_cols].copy()\nx_test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T10:51:28.028674Z","iopub.execute_input":"2025-04-13T10:51:28.028828Z","iopub.status.idle":"2025-04-13T10:51:28.047112Z","shell.execute_reply.started":"2025-04-13T10:51:28.028816Z","shell.execute_reply":"2025-04-13T10:51:28.046450Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Prediction on test data ##\n\n# Predict on test set (or new data)\n\npredictions = pd.DataFrame()\n\nfor target in target_cols:\n    predictions[target] = models[target].predict(x_test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T10:51:28.047906Z","iopub.execute_input":"2025-04-13T10:51:28.048146Z","iopub.status.idle":"2025-04-13T10:51:29.245211Z","shell.execute_reply.started":"2025-04-13T10:51:28.048124Z","shell.execute_reply":"2025-04-13T10:51:29.244645Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission[['x_1', 'y_1', 'z_1']] = predictions\nsubmission[['ID', 'resname', 'resid', 'x_1', 'y_1', 'z_1']].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T10:51:29.245923Z","iopub.execute_input":"2025-04-13T10:51:29.246140Z","iopub.status.idle":"2025-04-13T10:51:29.257527Z","shell.execute_reply.started":"2025-04-13T10:51:29.246125Z","shell.execute_reply":"2025-04-13T10:51:29.256945Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create x_2-x_5, y_2-y_5, z_2-z_5 columns with same values as x_1/y_1/z_1\nfor i in range(2, 6):\n    submission[f'x_{i}'] = submission['x_1']\n    submission[f'y_{i}'] = submission['y_1']\n    submission[f'z_{i}'] = submission['z_1']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T10:51:29.258159Z","iopub.execute_input":"2025-04-13T10:51:29.258483Z","iopub.status.idle":"2025-04-13T10:51:29.272926Z","shell.execute_reply.started":"2025-04-13T10:51:29.258461Z","shell.execute_reply":"2025-04-13T10:51:29.272144Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission['ID'] = submission['ID'].astype(str) + '_' + submission['resid'].astype(str)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T10:51:29.273673Z","iopub.execute_input":"2025-04-13T10:51:29.273846Z","iopub.status.idle":"2025-04-13T10:51:29.282121Z","shell.execute_reply.started":"2025-04-13T10:51:29.273832Z","shell.execute_reply":"2025-04-13T10:51:29.281299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nsubmission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T10:51:29.283773Z","iopub.execute_input":"2025-04-13T10:51:29.284170Z","iopub.status.idle":"2025-04-13T10:51:29.310563Z","shell.execute_reply.started":"2025-04-13T10:51:29.284154Z","shell.execute_reply":"2025-04-13T10:51:29.309946Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('submission.csv' , index = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-13T10:51:29.311204Z","iopub.execute_input":"2025-04-13T10:51:29.311441Z","iopub.status.idle":"2025-04-13T10:51:29.382778Z","shell.execute_reply.started":"2025-04-13T10:51:29.311416Z","shell.execute_reply":"2025-04-13T10:51:29.382277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}