{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-23T13:39:31.560671Z","iopub.execute_input":"2023-09-23T13:39:31.561688Z","iopub.status.idle":"2023-09-23T13:39:31.590338Z","shell.execute_reply.started":"2023-09-23T13:39:31.561643Z","shell.execute_reply":"2023-09-23T13:39:31.589149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import necessary libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestRegressor  # You can choose a different model\nfrom sklearn.metrics import mean_squared_error\n","metadata":{"execution":{"iopub.status.busy":"2023-09-23T13:45:42.501154Z","iopub.execute_input":"2023-09-23T13:45:42.502725Z","iopub.status.idle":"2023-09-23T13:45:44.011843Z","shell.execute_reply.started":"2023-09-23T13:45:42.502672Z","shell.execute_reply":"2023-09-23T13:45:44.010643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Step 1: Data Preprocessing","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv('/kaggle/input/stanford-ribonanza-rna-folding/train_data.csv')\ntest_sequences = pd.read_csv('/kaggle/input/stanford-ribonanza-rna-folding/test_sequences.csv')\n","metadata":{"execution":{"iopub.status.busy":"2023-09-23T13:45:48.742360Z","iopub.execute_input":"2023-09-23T13:45:48.742911Z","iopub.status.idle":"2023-09-23T13:47:35.396544Z","shell.execute_reply.started":"2023-09-23T13:45:48.742879Z","shell.execute_reply":"2023-09-23T13:47:35.395407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data","metadata":{"execution":{"iopub.status.busy":"2023-09-23T13:48:05.824175Z","iopub.execute_input":"2023-09-23T13:48:05.825327Z","iopub.status.idle":"2023-09-23T13:48:06.456703Z","shell.execute_reply.started":"2023-09-23T13:48:05.825277Z","shell.execute_reply":"2023-09-23T13:48:06.455606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_sequences","metadata":{"execution":{"iopub.status.busy":"2023-09-23T13:48:09.803579Z","iopub.execute_input":"2023-09-23T13:48:09.803964Z","iopub.status.idle":"2023-09-23T13:48:09.819228Z","shell.execute_reply.started":"2023-09-23T13:48:09.803937Z","shell.execute_reply":"2023-09-23T13:48:09.818071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Handle missing values if any","metadata":{}},{"cell_type":"code","source":"train_data.fillna(0, inplace=True)  # Replace NaN with 0 for simplicity","metadata":{"execution":{"iopub.status.busy":"2023-09-23T13:48:12.951123Z","iopub.execute_input":"2023-09-23T13:48:12.951504Z","iopub.status.idle":"2023-09-23T13:48:25.789316Z","shell.execute_reply.started":"2023-09-23T13:48:12.951473Z","shell.execute_reply":"2023-09-23T13:48:25.788295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data","metadata":{"execution":{"iopub.status.busy":"2023-09-23T13:48:28.805669Z","iopub.execute_input":"2023-09-23T13:48:28.806079Z","iopub.status.idle":"2023-09-23T13:48:29.417328Z","shell.execute_reply.started":"2023-09-23T13:48:28.806051Z","shell.execute_reply":"2023-09-23T13:48:29.416232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Split the data into features and target variables","metadata":{}},{"cell_type":"code","source":"X = train_data.drop(columns=['sequence_id', 'sequence', 'experiment_type', 'dataset_name', 'SN_filter'])\ny = train_data[['reactivity_0001', 'reactivity_0002', 'reactivity_0003']]  # Add all target columns\n","metadata":{"execution":{"iopub.status.busy":"2023-09-23T13:48:34.393412Z","iopub.execute_input":"2023-09-23T13:48:34.394357Z","iopub.status.idle":"2023-09-23T13:48:36.550472Z","shell.execute_reply.started":"2023-09-23T13:48:34.394321Z","shell.execute_reply":"2023-09-23T13:48:36.549292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X","metadata":{"execution":{"iopub.status.busy":"2023-09-23T13:48:40.338474Z","iopub.execute_input":"2023-09-23T13:48:40.339228Z","iopub.status.idle":"2023-09-23T13:48:40.589088Z","shell.execute_reply.started":"2023-09-23T13:48:40.339192Z","shell.execute_reply":"2023-09-23T13:48:40.588059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y","metadata":{"execution":{"iopub.status.busy":"2023-09-23T13:48:46.227418Z","iopub.execute_input":"2023-09-23T13:48:46.228309Z","iopub.status.idle":"2023-09-23T13:48:46.243712Z","shell.execute_reply.started":"2023-09-23T13:48:46.228274Z","shell.execute_reply":"2023-09-23T13:48:46.242702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Step 2: Feature Engineering (Not implemented here, you can add your own features)","metadata":{}},{"cell_type":"markdown","source":"# Step 3: Data Splitting","metadata":{}},{"cell_type":"code","source":"X_train, X_valid, y_train, y_valid = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-09-23T13:48:55.214666Z","iopub.execute_input":"2023-09-23T13:48:55.215021Z","iopub.status.idle":"2023-09-23T13:49:03.204322Z","shell.execute_reply.started":"2023-09-23T13:48:55.214995Z","shell.execute_reply":"2023-09-23T13:49:03.203209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Step 4: Model Building","metadata":{}},{"cell_type":"code","source":"model = RandomForestRegressor(n_estimators=100, random_state=42)  # Random Forest as an example","metadata":{"execution":{"iopub.status.busy":"2023-09-23T13:49:12.967477Z","iopub.execute_input":"2023-09-23T13:49:12.967970Z","iopub.status.idle":"2023-09-23T13:49:12.973663Z","shell.execute_reply.started":"2023-09-23T13:49:12.967925Z","shell.execute_reply":"2023-09-23T13:49:12.972584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model","metadata":{"execution":{"iopub.status.busy":"2023-09-23T13:49:19.603946Z","iopub.execute_input":"2023-09-23T13:49:19.604337Z","iopub.status.idle":"2023-09-23T13:49:19.616822Z","shell.execute_reply.started":"2023-09-23T13:49:19.604309Z","shell.execute_reply":"2023-09-23T13:49:19.615618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Step 5: Model Training","metadata":{}},{"cell_type":"code","source":"model.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-09-23T13:49:23.033577Z","iopub.execute_input":"2023-09-23T13:49:23.034002Z","iopub.status.idle":"2023-09-23T13:49:34.316228Z","shell.execute_reply.started":"2023-09-23T13:49:23.033970Z","shell.execute_reply":"2023-09-23T13:49:34.315103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model","metadata":{"execution":{"iopub.status.busy":"2023-09-23T13:49:34.317823Z","iopub.execute_input":"2023-09-23T13:49:34.318143Z","iopub.status.idle":"2023-09-23T13:49:34.325562Z","shell.execute_reply.started":"2023-09-23T13:49:34.318115Z","shell.execute_reply":"2023-09-23T13:49:34.324611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Step 6: Model Evaluation","metadata":{}},{"cell_type":"code","source":"y_pred = model.predict(X_valid)\nmse = mean_squared_error(y_valid, y_pred)\nprint(f'Mean Squared Error: {mse}')","metadata":{"execution":{"iopub.status.busy":"2023-09-23T13:49:41.012220Z","iopub.execute_input":"2023-09-23T13:49:41.012633Z","iopub.status.idle":"2023-09-23T13:49:42.078212Z","shell.execute_reply.started":"2023-09-23T13:49:41.012600Z","shell.execute_reply":"2023-09-23T13:49:42.076948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Step 7: Predictions on Test Data","metadata":{}},{"cell_type":"code","source":"# Load and preprocess the test sequences\ntest_data = pd.read_csv('/kaggle/input/stanford-ribonanza-rna-folding/test_sequences.csv')\n","metadata":{"execution":{"iopub.status.busy":"2023-09-23T13:52:06.497314Z","iopub.execute_input":"2023-09-23T13:52:06.497767Z","iopub.status.idle":"2023-09-23T13:52:11.373282Z","shell.execute_reply.started":"2023-09-23T13:52:06.497737Z","shell.execute_reply":"2023-09-23T13:52:11.372355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data","metadata":{"execution":{"iopub.status.busy":"2023-09-23T13:52:11.374881Z","iopub.execute_input":"2023-09-23T13:52:11.375214Z","iopub.status.idle":"2023-09-23T13:52:11.390271Z","shell.execute_reply.started":"2023-09-23T13:52:11.375184Z","shell.execute_reply":"2023-09-23T13:52:11.389276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_predictions = model.predict(test_data)  # Use the entire test_data DataFrame","metadata":{"execution":{"iopub.status.busy":"2023-09-23T13:57:07.301540Z","iopub.execute_input":"2023-09-23T13:57:07.302307Z","iopub.status.idle":"2023-09-23T13:57:07.374804Z","shell.execute_reply.started":"2023-09-23T13:57:07.302271Z","shell.execute_reply":"2023-09-23T13:57:07.373075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"common_columns","metadata":{"execution":{"iopub.status.busy":"2023-09-23T13:54:22.281619Z","iopub.execute_input":"2023-09-23T13:54:22.282464Z","iopub.status.idle":"2023-09-23T13:54:22.290004Z","shell.execute_reply.started":"2023-09-23T13:54:22.282424Z","shell.execute_reply":"2023-09-23T13:54:22.288965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make predictions\ntest_predictions = model.predict(test_data.drop(columns=['sequence_id', 'sequence']))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Continue to work","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}