{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11553390,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-16T08:10:03.321487Z","iopub.execute_input":"2025-04-16T08:10:03.321862Z","iopub.status.idle":"2025-04-16T08:10:03.747492Z","shell.execute_reply.started":"2025-04-16T08:10:03.321831Z","shell.execute_reply":"2025-04-16T08:10:03.746490Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_labels = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/train_labels.csv')\ntrain_sequences = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/train_sequences.csv')\nsubmission = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/sample_submission.csv')\ntest_sequences = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/test_sequences.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T08:10:03.749029Z","iopub.execute_input":"2025-04-16T08:10:03.749400Z","iopub.status.idle":"2025-04-16T08:10:04.037671Z","shell.execute_reply.started":"2025-04-16T08:10:03.749353Z","shell.execute_reply":"2025-04-16T08:10:04.031429Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## left join ##\ntrain_labels['ID'] = train_labels['ID'].str.rsplit('_', n=1).str[0]\ntrain_df  = train_labels.merge(how = 'left' , left_on = 'ID' , right_on = 'target_id' , right = train_sequences )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T08:10:04.038869Z","iopub.execute_input":"2025-04-16T08:10:04.039245Z","iopub.status.idle":"2025-04-16T08:10:04.342618Z","shell.execute_reply.started":"2025-04-16T08:10:04.039221Z","shell.execute_reply":"2025-04-16T08:10:04.341723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('No of rows and cols ' , train_df.shape)\nprint()\nprint('Missing values' , train_df.isna().sum())\nprint()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T08:10:04.344502Z","iopub.execute_input":"2025-04-16T08:10:04.344823Z","iopub.status.idle":"2025-04-16T08:10:04.414088Z","shell.execute_reply.started":"2025-04-16T08:10:04.344793Z","shell.execute_reply":"2025-04-16T08:10:04.413162Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"##datetime conversion ###\ntrain_df['temporal_cutoff'] = pd.to_datetime(train_df['temporal_cutoff']).astype('int64') // 10**9\ntest_sequences['temporal_cutoff'] = pd.to_datetime(test_sequences['temporal_cutoff']).astype('int64') // 10**9\n## using len of sequence \ntrain_df['seq_length'] = train_df['sequence'].str.len()\ntest_sequences['seq_length'] =  test_sequences['sequence'].str.len()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T08:10:04.415033Z","iopub.execute_input":"2025-04-16T08:10:04.415325Z","iopub.status.idle":"2025-04-16T08:10:04.495482Z","shell.execute_reply.started":"2025-04-16T08:10:04.415299Z","shell.execute_reply":"2025-04-16T08:10:04.494505Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission['ID']  = submission['ID'].str.rsplit('_' ,n =1  ).str[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T08:10:05.238327Z","iopub.execute_input":"2025-04-16T08:10:05.238685Z","iopub.status.idle":"2025-04-16T08:10:05.246355Z","shell.execute_reply.started":"2025-04-16T08:10:05.238663Z","shell.execute_reply":"2025-04-16T08:10:05.245421Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test  = test_sequences.merge(how = 'left' , left_on = 'target_id' , right_on = 'ID' , right = submission)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T08:10:06.616043Z","iopub.execute_input":"2025-04-16T08:10:06.616678Z","iopub.status.idle":"2025-04-16T08:10:06.625783Z","shell.execute_reply.started":"2025-04-16T08:10:06.616653Z","shell.execute_reply":"2025-04-16T08:10:06.624940Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define columns\ncategorical_cols = ['resname', 'target_id', 'description', 'all_sequences']\nnumerical_cols = ['temporal_cutoff', 'resid', 'seq_length']\ntarget_cols = ['x_1', 'y_1', 'z_1']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T08:10:07.322698Z","iopub.execute_input":"2025-04-16T08:10:07.323025Z","iopub.status.idle":"2025-04-16T08:10:07.328368Z","shell.execute_reply.started":"2025-04-16T08:10:07.323001Z","shell.execute_reply":"2025-04-16T08:10:07.327152Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Encode categoricals\nfrom sklearn.preprocessing import LabelEncoder\n\n# Encode categoricals\ndef encode_categoricals(df, categorical_cols):\n    df_encoded = df.copy()\n    encoders = {}\n    for col in categorical_cols:\n        le = LabelEncoder()\n        df_encoded[col] = le.fit_transform(df_encoded[col].astype(str))\n        encoders[col] = le\n    return df_encoded, encoders","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T08:10:08.440721Z","iopub.execute_input":"2025-04-16T08:10:08.441057Z","iopub.status.idle":"2025-04-16T08:10:09.096359Z","shell.execute_reply.started":"2025-04-16T08:10:08.441033Z","shell.execute_reply":"2025-04-16T08:10:09.095455Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## dropna values \ntrain_df.dropna(inplace = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T08:10:09.097773Z","iopub.execute_input":"2025-04-16T08:10:09.098214Z","iopub.status.idle":"2025-04-16T08:10:09.172613Z","shell.execute_reply.started":"2025-04-16T08:10:09.098170Z","shell.execute_reply":"2025-04-16T08:10:09.171470Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prepare data\ndf_encoded, encoders = encode_categoricals(train_df, categorical_cols)\nX = df_encoded[categorical_cols + numerical_cols]\ny = df_encoded[target_cols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T08:10:09.173573Z","iopub.execute_input":"2025-04-16T08:10:09.173798Z","iopub.status.idle":"2025-04-16T08:10:09.322812Z","shell.execute_reply.started":"2025-04-16T08:10:09.173781Z","shell.execute_reply":"2025-04-16T08:10:09.321840Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrainx , testx , trainy , testy = train_test_split(X , y , random_state = 0 , test_size = 0.25)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T08:10:10.154801Z","iopub.execute_input":"2025-04-16T08:10:10.155114Z","iopub.status.idle":"2025-04-16T08:10:10.208354Z","shell.execute_reply.started":"2025-04-16T08:10:10.155092Z","shell.execute_reply":"2025-04-16T08:10:10.207558Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # # # Objective function for Optuna\n\n# import optuna\n# from sklearn.ensemble import RandomForestRegressor\n# from sklearn.metrics import mean_squared_error\n# import warnings\n\n# warnings.filterwarnings(\"ignore\", category=UserWarning)\n# warnings.filterwarnings(\"ignore\", category=FutureWarning)\n\n# # Prepare training and validation data\n# X_train = trainx[categorical_cols + numerical_cols]\n# y_train = trainy[target_cols]\n# X_valid = testx[categorical_cols + numerical_cols]\n# y_valid = testy[target_cols]\n\n# def objective(trial):\n#     params = {\n#         \"n_estimators\": trial.suggest_int(\"n_estimators\", 25, 1000),\n#         \"max_depth\": trial.suggest_int(\"max_depth\", 5, 35),\n#         \"min_samples_split\": trial.suggest_int(\"min_samples_split\", 2, 20),\n#         \"min_samples_leaf\": trial.suggest_int(\"min_samples_leaf\", 1, 20),\n#         \"max_features\": trial.suggest_categorical(\"max_features\", [\"auto\", \"sqrt\", \"log2\"]),\n#         \"bootstrap\": trial.suggest_categorical(\"bootstrap\", [True, False]),\n#         \"random_state\": 42,\n#         \"n_jobs\": -1 ,\n#         \"verbose\" : 0  \n#     }\n\n#     # Train one model per target column\n#     losses = []\n#     for target in target_cols:\n#         model = RandomForestRegressor(**params)\n#         model.fit(X_train, y_train[target])\n#         preds = model.predict(X_valid)\n#         mse = mean_squared_error(y_valid[target], preds)\n#         losses.append(mse)\n\n#     return sum(losses) / len(losses)\n\n# # Run Optuna optimization\n# study = optuna.create_study(direction=\"minimize\")\n# study.optimize(objective, n_trials=100)\n\n# print(\"Best params:\", study.best_params)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T08:10:40.878020Z","iopub.execute_input":"2025-04-16T08:10:40.878596Z","iopub.status.idle":"2025-04-16T10:03:53.110250Z","shell.execute_reply.started":"2025-04-16T08:10:40.878572Z","shell.execute_reply":"2025-04-16T10:03:53.109542Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## training models on hyperparameters obtained from optuna \n\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_error\n\n\n\n# Prepare training and validation data\nX_train = trainx[categorical_cols + numerical_cols]\ny_train = trainy[target_cols]\nX_valid = testx[categorical_cols + numerical_cols]\ny_valid = testy[target_cols]\n\n\nparams = {'n_estimators': 520, 'max_depth': 34, 'min_samples_split': 8, 'min_samples_leaf': 1, 'max_features': 'auto', \n              'bootstrap': True , 'n_jobs' : -1 }\n\nmodels = {}\nfor target in target_cols:\n    lgb_reg = RandomForestRegressor(**params)\n    lgb_reg.fit(X_train, y_train[target])\n    models[target] = lgb_reg","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T10:06:44.530788Z","iopub.execute_input":"2025-04-16T10:06:44.531086Z","iopub.status.idle":"2025-04-16T10:09:07.196786Z","shell.execute_reply.started":"2025-04-16T10:06:44.531065Z","shell.execute_reply":"2025-04-16T10:09:07.196056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Feature importance \nimport pandas as pd\nimport matplotlib.pyplot as plt\n\n# Collect feature importances for each target\nfor target in target_cols:\n    model = models[target]\n    importances = model.feature_importances_\n    \n    # Create a DataFrame for better visualization\n    importance_df = pd.DataFrame({\n        'feature': X_train.columns,\n        'importance': importances\n    }).sort_values(by='importance', ascending=False)\n    \n    print(f\"\\nTop Features for Target: {target}\")\n    print(importance_df.head(10))\n\n    # Optional: Plot\n    plt.figure(figsize=(10, 5))\n    plt.title(f'Feature Importance for {target}')\n    plt.bar(importance_df['feature'], importance_df['importance'])\n    plt.xticks(rotation=45)\n    plt.tight_layout()\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T10:09:07.198202Z","iopub.execute_input":"2025-04-16T10:09:07.198532Z","iopub.status.idle":"2025-04-16T10:09:08.403282Z","shell.execute_reply.started":"2025-04-16T10:09:07.198504Z","shell.execute_reply":"2025-04-16T10:09:08.402413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_test = test[categorical_cols + numerical_cols]\nx_test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T10:09:08.404244Z","iopub.execute_input":"2025-04-16T10:09:08.404595Z","iopub.status.idle":"2025-04-16T10:09:08.414765Z","shell.execute_reply.started":"2025-04-16T10:09:08.404567Z","shell.execute_reply":"2025-04-16T10:09:08.413832Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"##  Data Preprocessing for inference\nfrom sklearn.preprocessing import LabelEncoder\n\n\n\nfor col in categorical_cols:\n    # Handle potential NaN values before encoding\n    test[col] = test[col].astype(str).fillna('missing')\n    \n    # Create and fit label encoder\n    le = LabelEncoder()\n    test[col] = le.fit_transform(test[col])\n    encoders[col] = le\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T10:09:08.416482Z","iopub.execute_input":"2025-04-16T10:09:08.416738Z","iopub.status.idle":"2025-04-16T10:09:08.440577Z","shell.execute_reply.started":"2025-04-16T10:09:08.416720Z","shell.execute_reply":"2025-04-16T10:09:08.439757Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_test = test[categorical_cols + numerical_cols].copy()\nx_test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T10:09:08.441367Z","iopub.execute_input":"2025-04-16T10:09:08.441682Z","iopub.status.idle":"2025-04-16T10:09:08.466923Z","shell.execute_reply.started":"2025-04-16T10:09:08.441656Z","shell.execute_reply":"2025-04-16T10:09:08.465947Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Prediction on test data ##\n\n# Predict on test set (or new data)\n\npredictions = pd.DataFrame()\n\nfor target in target_cols:\n    predictions[target] = models[target].predict(x_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T10:09:08.467923Z","iopub.execute_input":"2025-04-16T10:09:08.468209Z","iopub.status.idle":"2025-04-16T10:09:08.953516Z","shell.execute_reply.started":"2025-04-16T10:09:08.468179Z","shell.execute_reply":"2025-04-16T10:09:08.952807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission[['x_1', 'y_1', 'z_1']] = predictions\nsubmission[['ID', 'resname', 'resid', 'x_1', 'y_1', 'z_1']].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T10:09:08.954316Z","iopub.execute_input":"2025-04-16T10:09:08.954591Z","iopub.status.idle":"2025-04-16T10:09:08.966559Z","shell.execute_reply.started":"2025-04-16T10:09:08.954573Z","shell.execute_reply":"2025-04-16T10:09:08.965831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create x_2-x_5, y_2-y_5, z_2-z_5 columns with same values as x_1/y_1/z_1\nfor i in range(2, 6):\n    submission[f'x_{i}'] = submission['x_1']\n    submission[f'y_{i}'] = submission['y_1']\n    submission[f'z_{i}'] = submission['z_1']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T10:09:08.967508Z","iopub.execute_input":"2025-04-16T10:09:08.967754Z","iopub.status.idle":"2025-04-16T10:09:08.984485Z","shell.execute_reply.started":"2025-04-16T10:09:08.967736Z","shell.execute_reply":"2025-04-16T10:09:08.983480Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission['ID'] = submission['ID'].astype(str) + '_' + submission['resid'].astype(str)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T10:09:08.985408Z","iopub.execute_input":"2025-04-16T10:09:08.985645Z","iopub.status.idle":"2025-04-16T10:09:09.005083Z","shell.execute_reply.started":"2025-04-16T10:09:08.985628Z","shell.execute_reply":"2025-04-16T10:09:09.004016Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('submission.csv' , index = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T10:09:09.007476Z","iopub.execute_input":"2025-04-16T10:09:09.007742Z","iopub.status.idle":"2025-04-16T10:09:09.091732Z","shell.execute_reply.started":"2025-04-16T10:09:09.007723Z","shell.execute_reply":"2025-04-16T10:09:09.090730Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T10:09:09.092666Z","iopub.execute_input":"2025-04-16T10:09:09.092905Z","iopub.status.idle":"2025-04-16T10:09:09.108696Z","shell.execute_reply.started":"2025-04-16T10:09:09.092886Z","shell.execute_reply":"2025-04-16T10:09:09.107865Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}