{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11512973,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport xgboost as xgb\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-23T11:43:10.808071Z","iopub.execute_input":"2025-03-23T11:43:10.808438Z","iopub.status.idle":"2025-03-23T11:43:11.216381Z","shell.execute_reply.started":"2025-03-23T11:43:10.808406Z","shell.execute_reply":"2025-03-23T11:43:11.215258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_labels = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/train_labels.csv')\ntrain_sequences = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/train_sequences.csv')\nsubmission = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/sample_submission.csv')\ntest_sequences = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/test_sequences.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T11:41:20.042591Z","iopub.execute_input":"2025-03-23T11:41:20.042991Z","iopub.status.idle":"2025-03-23T11:41:20.445264Z","shell.execute_reply.started":"2025-03-23T11:41:20.042960Z","shell.execute_reply":"2025-03-23T11:41:20.444279Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extract target from ID by removing the last segment after final underscore\ntrain_labels['target'] = train_labels['ID'].str.rsplit('_', n=1).str[0]\n# Merge train_sequences with train_labels using left join on target_id and target\ntrain = train_sequences.merge(train_labels, how='left', left_on='target_id', right_on='target')\n# Preprocess data for modeling\n# Convert temporal cutoff to numerical feature\ntrain['temporal_cutoff'] = pd.to_datetime(train['temporal_cutoff']).astype('int64') // 10**9\ntest_sequences['temporal_cutoff'] = pd.to_datetime(test_sequences['temporal_cutoff']).astype('int64') // 10**9\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T11:41:38.821446Z","iopub.execute_input":"2025-03-23T11:41:38.821820Z","iopub.status.idle":"2025-03-23T11:41:39.237894Z","shell.execute_reply.started":"2025-03-23T11:41:38.821788Z","shell.execute_reply":"2025-03-23T11:41:39.236782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nsubmission['target'] = submission['ID'].str.rsplit('_', n=1).str[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T11:42:04.536963Z","iopub.execute_input":"2025-03-23T11:42:04.537411Z","iopub.status.idle":"2025-03-23T11:42:04.548717Z","shell.execute_reply.started":"2025-03-23T11:42:04.537374Z","shell.execute_reply":"2025-03-23T11:42:04.547379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = test_sequences.merge(submission, how='left', left_on='target_id', right_on='target')\ntest['temporal_cutoff'] = pd.to_datetime(test['temporal_cutoff']).astype('int64') // 10**9\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T11:42:10.938506Z","iopub.execute_input":"2025-03-23T11:42:10.938900Z","iopub.status.idle":"2025-03-23T11:42:10.953246Z","shell.execute_reply.started":"2025-03-23T11:42:10.938869Z","shell.execute_reply":"2025-03-23T11:42:10.952138Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create sequence length feature\ntrain['seq_length'] = train['sequence'].str.len()\ntest['seq_length'] = test['sequence'].str.len()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T11:42:15.783368Z","iopub.execute_input":"2025-03-23T11:42:15.783759Z","iopub.status.idle":"2025-03-23T11:42:15.837706Z","shell.execute_reply.started":"2025-03-23T11:42:15.783728Z","shell.execute_reply":"2025-03-23T11:42:15.836474Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_columns = ['ID', 'target_id', 'description','all_sequences','sequence','target','resname']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T11:42:24.164374Z","iopub.execute_input":"2025-03-23T11:42:24.164787Z","iopub.status.idle":"2025-03-23T11:42:24.169354Z","shell.execute_reply.started":"2025-03-23T11:42:24.164754Z","shell.execute_reply":"2025-03-23T11:42:24.168150Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Label encode categorical columns\nfrom sklearn.preprocessing import LabelEncoder\n\n# Create dictionary to store label encoders\nlabel_encoders = {}\n\nfor col in categorical_columns:\n    # Handle potential NaN values before encoding\n    train[col] = train[col].astype(str).fillna('missing')\n    \n    # Create and fit label encoder\n    le = LabelEncoder()\n    train[col] = le.fit_transform(train[col])\n    label_encoders[col] = le\n\nfor col in categorical_columns:\n    # Handle potential NaN values before encoding\n    test[col] = test[col].astype(str).fillna('missing')\n    \n    # Create and fit label encoder\n    le = LabelEncoder()\n    test[col] = le.fit_transform(test[col])\n    label_encoders[col] = le\n\n# Optional: Add embedding dimensions for high-cardinality features\n# (This would require neural network approach and more complex handling)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T11:42:29.918229Z","iopub.execute_input":"2025-03-23T11:42:29.918539Z","iopub.status.idle":"2025-03-23T11:42:31.033859Z","shell.execute_reply.started":"2025-03-23T11:42:29.918516Z","shell.execute_reply":"2025-03-23T11:42:31.032833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Features and targets for training\nfeatures = ['temporal_cutoff', 'resname', 'resid', 'seq_length', 'target_id','ID','description','all_sequences']\ncoord_targets = ['x_1','y_1','z_1']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T11:42:45.498790Z","iopub.execute_input":"2025-03-23T11:42:45.499320Z","iopub.status.idle":"2025-03-23T11:42:45.504233Z","shell.execute_reply.started":"2025-03-23T11:42:45.499287Z","shell.execute_reply":"2025-03-23T11:42:45.503109Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"models = {}\nfor coord in coord_targets:\n    print(f'Training model for {coord}...')\n    model = xgb.XGBRegressor(\n        objective='reg:squarederror',\n        n_estimators=1000,\n        max_depth=7,\n        learning_rate=0.1,\n        subsample=0.8,\n        colsample_bytree=0.8\n    )\n    \n    # Filter out rows with missing coordinates\n    mean_val = train[coord].mean()\n    train[coord].fillna(mean_val, inplace=True)\n    valid_idx = train[coord].notna()  # Now just indicates all rows are valid\n    model.fit(train[valid_idx][features], train[valid_idx][coord])\n    models[coord] = model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T11:43:21.301560Z","iopub.execute_input":"2025-03-23T11:43:21.301991Z","iopub.status.idle":"2025-03-23T11:43:37.568840Z","shell.execute_reply.started":"2025-03-23T11:43:21.301956Z","shell.execute_reply":"2025-03-23T11:43:37.567602Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = test[['ID','resname','resid','sequence','target_id','target','description','all_sequences','temporal_cutoff','seq_length']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T11:46:39.110140Z","iopub.execute_input":"2025-03-23T11:46:39.110519Z","iopub.status.idle":"2025-03-23T11:46:39.117230Z","shell.execute_reply.started":"2025-03-23T11:46:39.110487Z","shell.execute_reply":"2025-03-23T11:46:39.116329Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Predict coordinates for test set\ntest_predictions = test.copy()\nfor coord in coord_targets:\n    # Use trained model to make predictions\n    test_predictions[coord] = models[coord].predict(test[features])\n    \n# Display sample predictions\nprint(\"Test predictions sample:\")\ntest_predictions[['ID', 'resname', 'resid', 'x_1', 'y_1', 'z_1']].head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T11:46:47.268045Z","iopub.execute_input":"2025-03-23T11:46:47.268393Z","iopub.status.idle":"2025-03-23T11:46:47.408739Z","shell.execute_reply.started":"2025-03-23T11:46:47.268367Z","shell.execute_reply":"2025-03-23T11:46:47.407746Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create x_2-x_5, y_2-y_5, z_2-z_5 columns with same values as x_1/y_1/z_1\nfor i in range(2, 6):\n    test_predictions[f'x_{i}'] = test_predictions['x_1']\n    test_predictions[f'y_{i}'] = test_predictions['y_1']\n    test_predictions[f'z_{i}'] = test_predictions['z_1']\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T11:46:57.468974Z","iopub.execute_input":"2025-03-23T11:46:57.469333Z","iopub.status.idle":"2025-03-23T11:46:57.479968Z","shell.execute_reply.started":"2025-03-23T11:46:57.469307Z","shell.execute_reply":"2025-03-23T11:46:57.478885Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission[['x_1','y_1','z_1','x_2','y_2','z_2','x_3','y_3','z_3','x_4','y_4','z_4','x_5','y_5','z_5']] = test_predictions[['x_1','y_1','z_1','x_2','y_2','z_2','x_3','y_3','z_3','x_4','y_4','z_4','x_5','y_5','z_5']] \n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T11:47:09.733194Z","iopub.execute_input":"2025-03-23T11:47:09.733612Z","iopub.status.idle":"2025-03-23T11:47:09.745605Z","shell.execute_reply.started":"2025-03-23T11:47:09.733575Z","shell.execute_reply":"2025-03-23T11:47:09.744383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T11:47:17.878257Z","iopub.execute_input":"2025-03-23T11:47:17.878638Z","iopub.status.idle":"2025-03-23T11:47:17.938351Z","shell.execute_reply.started":"2025-03-23T11:47:17.878589Z","shell.execute_reply":"2025-03-23T11:47:17.936855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}