{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":51294,"databundleVersionId":6923401,"sourceType":"competition"}],"dockerImageVersionId":30589,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"pip install xgboost","metadata":{"execution":{"iopub.status.busy":"2023-11-29T17:51:36.961789Z","iopub.execute_input":"2023-11-29T17:51:36.962546Z","iopub.status.idle":"2023-11-29T17:51:49.600081Z","shell.execute_reply.started":"2023-11-29T17:51:36.962492Z","shell.execute_reply":"2023-11-29T17:51:49.598974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nimport xgboost as xgb\nfrom sklearn.metrics import mean_absolute_error\n\n# Load data\ntrain_data = pd.read_csv('/kaggle/input/stanford-ribonanza-rna-folding/train_data_QUICK_START.csv')\n\n# Basic preprocessing\n# Encode 'experiment_type' as it is categorical\nencoder = LabelEncoder()\ntrain_data['experiment_type'] = encoder.fit_transform(train_data['experiment_type'])\n\n# Feature Engineering for RNA sequences\n# Replace this placeholder with the actual encoding for RNA sequences\n# For example, you can use one-hot encoding or other advanced techniques\n# Here, we'll use the length of the sequence as a simple placeholder\ntrain_data['sequence_encoded'] = train_data['sequence'].apply(lambda x: len(x))\n\n# Select features and target\nfeatures = train_data[['sequence_encoded', 'experiment_type']]\ntargets = train_data.filter(regex='^reactivity_')\n\n# Split data into training and validation sets\nX_train, X_val, y_train, y_val = train_test_split(features, targets, test_size=0.2, random_state=42)\n\nif y_train.isnull().values.any():\n    y_train.fillna(0, inplace=True)  # You can replace NaN values with 0 or use another appropriate imputation strategy\n\nif y_val.isnull().values.any():\n    y_val.fillna(0, inplace=True)  # You can replace NaN values with 0 or use another appropriate imputation strategy\n\n# Initialize the XGBoost Regressor\nxgb_regressor = xgb.XGBRegressor(objective='reg:squarederror', random_state=42)\n\n# Train the model on the training data\nxgb_regressor.fit(X_train, y_train)\n\n# Predict on the validation set\ny_val_pred_xgb = xgb_regressor.predict(X_val)\n\n# Calculate Mean Absolute Error (MAE) for XGBoost\nmae_xgb = mean_absolute_error(y_val, y_val_pred_xgb)\nprint(f\"MAE (XGBoost): {mae_xgb}\")\n\n# Now you can proceed with testing on the test data, similar to the previous example\n","metadata":{"execution":{"iopub.status.busy":"2023-11-29T17:51:49.601977Z","iopub.execute_input":"2023-11-29T17:51:49.602279Z","iopub.status.idle":"2023-11-29T17:51:56.996928Z","shell.execute_reply.started":"2023-11-29T17:51:49.602250Z","shell.execute_reply":"2023-11-29T17:51:56.995143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np","metadata":{"execution":{"iopub.status.busy":"2023-11-29T17:51:56.997852Z","iopub.status.idle":"2023-11-29T17:51:56.998309Z","shell.execute_reply.started":"2023-11-29T17:51:56.998075Z","shell.execute_reply":"2023-11-29T17:51:56.998096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate mean/median predictions from validation set\nmean_val_predictions = y_val_pred_xgb.mean(axis=0)\nmedian_val_predictions = np.median(y_val_pred_xgb, axis=0)\n\n# Use these mean/median values to fill in your test data\nsubmission_data = pd.read_csv('/kaggle/input/stanford-ribonanza-rna-folding/sample_submission.csv')\nfor i, column in enumerate(['reactivity_DMS_MaP', 'reactivity_2A3_MaP']):\n    submission_data[column] = mean_val_predictions[i]  # or median_val_predictions[i]\n\n# Save the filled submission file\nsubmission_data.to_csv('submission_with_mean_predictions.csv', index=False)\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-11-29T17:51:56.999843Z","iopub.status.idle":"2023-11-29T17:51:57.000179Z","shell.execute_reply.started":"2023-11-29T17:51:57.000006Z","shell.execute_reply":"2023-11-29T17:51:57.000021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming 'train_data' is your DataFrame containing the training data\nprint(f\"Number of rows in training data: {train_data.shape[0]}\")\nprint(f\"Number of columns in training data: {train_data.shape[1]}\")\n\ntrain_data1 = pd.read_csv('/kaggle/input/stanford-ribonanza-rna-folding/sample_submission.csv')\n# Assuming 'train_data' is your DataFrame containing the training data\nprint(f\"Number of rows in training data: {train_data1.shape[0]}\")\nprint(f\"Number of columns in training data: {train_data1.shape[1]}\")\n\n","metadata":{"execution":{"iopub.status.busy":"2023-11-29T17:51:57.001592Z","iopub.status.idle":"2023-11-29T17:51:57.001911Z","shell.execute_reply.started":"2023-11-29T17:51:57.001752Z","shell.execute_reply":"2023-11-29T17:51:57.001767Z"},"trusted":true},"execution_count":null,"outputs":[]}]}