{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":12276181,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\ncsv_files = []\nfor dirname, _, filenames in os.walk('/kaggle/input/stanford-rna-3d-folding'): \n    for filename in filenames:\n        if filename.endswith('.csv'):\n            csv_files.append(filename)\n\nprint(csv_files)\n\n\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-17T00:28:08.901119Z","iopub.execute_input":"2025-05-17T00:28:08.901426Z","iopub.status.idle":"2025-05-17T00:28:08.954761Z","shell.execute_reply.started":"2025-05-17T00:28:08.901403Z","shell.execute_reply":"2025-05-17T00:28:08.953757Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Let's inspect the files provided by the user to understand their structure.\n# Adjust the paths if needed for local or Kaggle environments.\n\n# Attempt to read the test_sequences file to inspect its structure\ntry:\n    test_sequences = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/test_sequences.csv\")\nexcept Exception as e:\n    test_sequences = str(e)\n\n# Attempt to read the sample_submission file to inspect its structure\ntry:\n    sample_submission = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/sample_submission.csv\")\nexcept Exception as e:\n    sample_submission = str(e)\n\ntest_sequences, sample_submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-17T00:28:21.759460Z","iopub.execute_input":"2025-05-17T00:28:21.760691Z","iopub.status.idle":"2025-05-17T00:28:21.790468Z","shell.execute_reply.started":"2025-05-17T00:28:21.760649Z","shell.execute_reply":"2025-05-17T00:28:21.789639Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Let's inspect the files provided by the user to understand their structure.\n# Adjust the paths if needed for local or Kaggle environments.\n\n# Attempt to read the test_sequences file to inspect its structure\ntry:\n    test_sequences = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/test_sequences.csv\")\nexcept Exception as e:\n    test_sequences = str(e)\n\n# Attempt to read the sample_submission file to inspect its structure\ntry:\n    sample_submission = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/sample_submission.csv\")\nexcept Exception as e:\n    sample_submission = str(e)\n\ntest_sequences, sample_submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-17T00:28:27.856194Z","iopub.execute_input":"2025-05-17T00:28:27.856533Z","iopub.status.idle":"2025-05-17T00:28:27.886607Z","shell.execute_reply.started":"2025-05-17T00:28:27.856506Z","shell.execute_reply":"2025-05-17T00:28:27.885713Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ndf_sample = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/sample_submission.csv\")\ndf_sample","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-17T00:28:38.472309Z","iopub.execute_input":"2025-05-17T00:28:38.472702Z","iopub.status.idle":"2025-05-17T00:28:38.512769Z","shell.execute_reply.started":"2025-05-17T00:28:38.472679Z","shell.execute_reply":"2025-05-17T00:28:38.511801Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/test_sequences.csv\")  # or use `read_excel` for Excel files\ndf_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-17T00:28:46.313012Z","iopub.execute_input":"2025-05-17T00:28:46.313318Z","iopub.status.idle":"2025-05-17T00:28:46.326807Z","shell.execute_reply.started":"2025-05-17T00:28:46.313296Z","shell.execute_reply":"2025-05-17T00:28:46.325844Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df_sample.head())\nprint(df_test.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-17T00:28:57.525743Z","iopub.execute_input":"2025-05-17T00:28:57.526035Z","iopub.status.idle":"2025-05-17T00:28:57.540982Z","shell.execute_reply.started":"2025-05-17T00:28:57.526015Z","shell.execute_reply":"2025-05-17T00:28:57.539665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Assuming df_test and df_sample are your two dataframes\n\n# Extract base target ID from sample\ndf_sample['target_id'] = df_sample['ID'].str.extract(r'^(R\\d+)', expand=False)\n\n# Now join with test dataframe on 'target_id'\ndf_merged = df_sample.merge(df_test, on='target_id', how='left')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-17T00:29:03.980330Z","iopub.execute_input":"2025-05-17T00:29:03.980768Z","iopub.status.idle":"2025-05-17T00:29:04.011009Z","shell.execute_reply.started":"2025-05-17T00:29:03.980744Z","shell.execute_reply":"2025-05-17T00:29:04.010102Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_merged","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-17T00:29:10.224260Z","iopub.execute_input":"2025-05-17T00:29:10.224539Z","iopub.status.idle":"2025-05-17T00:29:10.255079Z","shell.execute_reply.started":"2025-05-17T00:29:10.224519Z","shell.execute_reply":"2025-05-17T00:29:10.254145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sequence length\ndf_test['seq_length'] = df_test['sequence'].str.len()\n\n# Number of residues per target in sample\nresidue_counts = df_sample['target_id'].value_counts().reset_index()\nresidue_counts.columns = ['target_id', 'residue_count']\n\n# Merge back\ndf_test = df_test.merge(residue_counts, on='target_id', how='left')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-17T00:29:22.977952Z","iopub.execute_input":"2025-05-17T00:29:22.978235Z","iopub.status.idle":"2025-05-17T00:29:22.990483Z","shell.execute_reply.started":"2025-05-17T00:29:22.978215Z","shell.execute_reply":"2025-05-17T00:29:22.989410Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Example: Suppose df_sample already has encoded features\nX = df_sample.drop(columns=[\"target_id\"])  # all columns except target\nX","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-17T00:29:29.855858Z","iopub.execute_input":"2025-05-17T00:29:29.856166Z","iopub.status.idle":"2025-05-17T00:29:29.882266Z","shell.execute_reply.started":"2025-05-17T00:29:29.856143Z","shell.execute_reply":"2025-05-17T00:29:29.881285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y = df_sample[\"target_id\"]  # Adjust this based on your actual target column name\ny","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-17T00:29:34.613230Z","iopub.execute_input":"2025-05-17T00:29:34.613554Z","iopub.status.idle":"2025-05-17T00:29:34.621071Z","shell.execute_reply.started":"2025-05-17T00:29:34.613530Z","shell.execute_reply":"2025-05-17T00:29:34.620137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_test, y_train, y_test = train_test_split(\n    X, y, test_size=0.2, random_state=42, stratify=y if len(set(y)) > 1 else None\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-17T00:29:42.784970Z","iopub.execute_input":"2025-05-17T00:29:42.785580Z","iopub.status.idle":"2025-05-17T00:29:42.797117Z","shell.execute_reply.started":"2025-05-17T00:29:42.785552Z","shell.execute_reply":"2025-05-17T00:29:42.796426Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check for non-numeric columns\nnon_numeric_cols = X.select_dtypes(include=['object']).columns\nprint(non_numeric_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-17T00:29:49.028292Z","iopub.execute_input":"2025-05-17T00:29:49.028703Z","iopub.status.idle":"2025-05-17T00:29:49.035318Z","shell.execute_reply.started":"2025-05-17T00:29:49.028673Z","shell.execute_reply":"2025-05-17T00:29:49.034151Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\nlabel_encoder = LabelEncoder()\n\nfor col in non_numeric_cols:\n    X[col] = label_encoder.fit_transform(X[col])\n\n# Verify that the encoding worked\nprint(X.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-17T00:29:53.377175Z","iopub.execute_input":"2025-05-17T00:29:53.377465Z","iopub.status.idle":"2025-05-17T00:29:53.395295Z","shell.execute_reply.started":"2025-05-17T00:29:53.377439Z","shell.execute_reply":"2025-05-17T00:29:53.394305Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"X_train shape:\", X_train.shape)\nprint(\"y_train shape:\", y_train.shape)\nprint(\"X_test shape:\", X_test.shape)\nprint(\"y_test shape:\", y_test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-17T00:29:58.811355Z","iopub.execute_input":"2025-05-17T00:29:58.811708Z","iopub.status.idle":"2025-05-17T00:29:58.817003Z","shell.execute_reply.started":"2025-05-17T00:29:58.811679Z","shell.execute_reply":"2025-05-17T00:29:58.815935Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-17T00:30:05.528209Z","iopub.execute_input":"2025-05-17T00:30:05.528540Z","iopub.status.idle":"2025-05-17T00:30:05.585217Z","shell.execute_reply.started":"2025-05-17T00:30:05.528507Z","shell.execute_reply":"2025-05-17T00:30:05.584447Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# For example, using Logistic Regression again\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\n\n# Split data into train and test\nX_train, X_test, y_train, y_test = train_test_split(X_scaled, y, test_size=0.2, random_state=42)\n\n# Initialize and train the model\nclassifier = LogisticRegression(max_iter=10000)\nclassifier.fit(X_train, y_train)\n\n# Predict on the test set\ny_pred = classifier.predict(X_test)\n\n# Evaluate the model\nprint(\"Accuracy:\", accuracy_score(y_test, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-17T00:30:10.747283Z","iopub.execute_input":"2025-05-17T00:30:10.747604Z","iopub.status.idle":"2025-05-17T00:30:11.256239Z","shell.execute_reply.started":"2025-05-17T00:30:10.747582Z","shell.execute_reply":"2025-05-17T00:30:11.253456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Assuming you have test data (X_test) and a trained model (classifier)\ny_pred_test = classifier.predict(X_test)\n\n# Check the length of predictions and ensure it matches the number of rows in the test set\nprint(len(y_pred_test))  # Should match the number of rows in the test data (2515)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-17T00:30:22.819221Z","iopub.execute_input":"2025-05-17T00:30:22.819525Z","iopub.status.idle":"2025-05-17T00:30:22.825085Z","shell.execute_reply.started":"2025-05-17T00:30:22.819503Z","shell.execute_reply":"2025-05-17T00:30:22.824205Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df_test.columns)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-17T00:30:29.175790Z","iopub.execute_input":"2025-05-17T00:30:29.176716Z","iopub.status.idle":"2025-05-17T00:30:29.181300Z","shell.execute_reply.started":"2025-05-17T00:30:29.176687Z","shell.execute_reply":"2025-05-17T00:30:29.180256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# Use this if you already have df_test loaded\n# You'll need a list of all the test IDs that require predictions\n\n# If you have a sample_submission.csv, read from there\nsample_submission = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/sample_submission.csv\")\n\n# Extract the 'ID' column from it\ntest_ids = sample_submission[\"ID\"]\n\n# Generate dummy predictions (shape must be [n_samples, 15])\n# Replace this with your model predictions\ny_pred = np.zeros((len(test_ids), 15))  # 5 atoms × (x, y, z) = 15 values per ID\n\n# Create a DataFrame for submission\nsubmission_df = pd.DataFrame(y_pred, columns=[\n    'x_1', 'y_1', 'z_1',\n    'x_2', 'y_2', 'z_2',\n    'x_3', 'y_3', 'z_3',\n    'x_4', 'y_4', 'z_4',\n    'x_5', 'y_5', 'z_5'\n])\n\n# Add ID column\nsubmission_df.insert(0, 'ID', test_ids)\n\n# Save to CSV\nsubmission_df.to_csv(\"sample_submission.csv\", index=False)\nprint(\"✅ Submission file created successfully with shape:\", submission_df.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-17T00:30:43.400700Z","iopub.execute_input":"2025-05-17T00:30:43.401361Z","iopub.status.idle":"2025-05-17T00:30:43.446759Z","shell.execute_reply.started":"2025-05-17T00:30:43.401333Z","shell.execute_reply":"2025-05-17T00:30:43.445420Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## CHECK SUBMISSON","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\n# Load your submission file\nsubmission = pd.read_csv(\"sample_submission.csv\")\n\n# Show the first few rows\nprint(submission.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-17T00:30:50.272208Z","iopub.execute_input":"2025-05-17T00:30:50.272521Z","iopub.status.idle":"2025-05-17T00:30:50.292881Z","shell.execute_reply.started":"2025-05-17T00:30:50.272497Z","shell.execute_reply":"2025-05-17T00:30:50.291651Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"expected_columns = ['ID',\n    'x_1', 'y_1', 'z_1',\n    'x_2', 'y_2', 'z_2',\n    'x_3', 'y_3', 'z_3',\n    'x_4', 'y_4', 'z_4',\n    'x_5', 'y_5', 'z_5'\n]\n\n# Compare to your submission file\nmissing = [col for col in expected_columns if col not in submission.columns]\nextra = [col for col in submission.columns if col not in expected_columns]\n\nprint(\"Missing columns:\", missing)\nprint(\"Extra columns:\", extra)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-17T00:30:56.746420Z","iopub.execute_input":"2025-05-17T00:30:56.746792Z","iopub.status.idle":"2025-05-17T00:30:56.753484Z","shell.execute_reply.started":"2025-05-17T00:30:56.746766Z","shell.execute_reply":"2025-05-17T00:30:56.752472Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check if there are any NaN values\nprint(\"Any NaN values?\", submission.isnull().values.any())\n\n# Show how many in each column\nprint(submission.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-17T00:31:00.929433Z","iopub.execute_input":"2025-05-17T00:31:00.929768Z","iopub.status.idle":"2025-05-17T00:31:00.938100Z","shell.execute_reply.started":"2025-05-17T00:31:00.929745Z","shell.execute_reply":"2025-05-17T00:31:00.936718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Submission shape:\", submission.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-17T00:31:09.396981Z","iopub.execute_input":"2025-05-17T00:31:09.397278Z","iopub.status.idle":"2025-05-17T00:31:09.401857Z","shell.execute_reply.started":"2025-05-17T00:31:09.397255Z","shell.execute_reply":"2025-05-17T00:31:09.400961Z"}},"outputs":[],"execution_count":null}]}