{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":87793,"databundleVersionId":11403143,"sourceType":"competition"}],"dockerImageVersionId":30919,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-18T06:52:23.190101Z","iopub.execute_input":"2025-03-18T06:52:23.190389Z","iopub.status.idle":"2025-03-18T06:52:25.730297Z","shell.execute_reply.started":"2025-03-18T06:52:23.190365Z","shell.execute_reply":"2025-03-18T06:52:25.729488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\ntrain_sequences = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_sequences.csv\")\ntrain_labels = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_labels.csv\")\ntest_sequences = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/test_sequences.csv\")\nsample_submission = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/sample_submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T06:52:41.453321Z","iopub.execute_input":"2025-03-18T06:52:41.453673Z","iopub.status.idle":"2025-03-18T06:52:42.207083Z","shell.execute_reply.started":"2025-03-18T06:52:41.453643Z","shell.execute_reply":"2025-03-18T06:52:42.206343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"unique_chars = set(\"\".join(train_sequences[\"sequence\"].dropna().values))\nprint(\"Unique RNA bases:\", unique_chars)\n\nbase_encoder = LabelEncoder()\nbase_encoder.fit(['A', 'U', 'G', 'C', 'N'])\n\ndef clean_and_encode_sequence(seq):\n    clean_seq = seq.replace(\"-\", \"\").replace(\"X\", \"N\") \n    return np.array(base_encoder.transform(list(clean_seq)))  \n\ntrain_sequences[\"encoded_seq\"] = train_sequences[\"sequence\"].apply(clean_and_encode_sequence)\n\ntrain_sequences.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T06:52:44.925103Z","iopub.execute_input":"2025-03-18T06:52:44.925382Z","iopub.status.idle":"2025-03-18T06:52:45.069314Z","shell.execute_reply.started":"2025-03-18T06:52:44.925359Z","shell.execute_reply":"2025-03-18T06:52:45.068310Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Sample target_id values (train_sequences):\", train_sequences[\"target_id\"].unique()[:5])\nprint(\"Sample ID values (train_labels):\", train_labels[\"ID\"].unique()[:5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T06:52:48.966309Z","iopub.execute_input":"2025-03-18T06:52:48.966645Z","iopub.status.idle":"2025-03-18T06:52:48.990644Z","shell.execute_reply.started":"2025-03-18T06:52:48.966616Z","shell.execute_reply":"2025-03-18T06:52:48.989966Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_labels[\"target_base\"] = train_labels[\"ID\"].apply(lambda x: \"_\".join(x.split(\"_\")[:2]))\n\nmerged_data = train_sequences.merge(\n    train_labels, left_on=\"target_id\", right_on=\"target_base\"\n)\n\nmerged_data.drop(columns=[\"target_base\"], inplace=True)\n\nmerged_data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T06:52:51.247338Z","iopub.execute_input":"2025-03-18T06:52:51.247664Z","iopub.status.idle":"2025-03-18T06:52:51.430504Z","shell.execute_reply.started":"2025-03-18T06:52:51.247635Z","shell.execute_reply":"2025-03-18T06:52:51.429777Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_counts = merged_data.groupby(\"target_id\").size()\n\nprint(label_counts.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T06:54:04.421245Z","iopub.execute_input":"2025-03-18T06:54:04.421655Z","iopub.status.idle":"2025-03-18T06:54:04.440389Z","shell.execute_reply.started":"2025-03-18T06:54:04.421620Z","shell.execute_reply":"2025-03-18T06:54:04.439533Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Y_grouped = merged_data.groupby(\"target_id\")[[\"x_1\", \"y_1\", \"z_1\"]].apply(lambda x: x.values)\n\nY = np.array(Y_grouped.tolist(), dtype=object)\n\nprint(f\"Before padding: {Y.shape}\")  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T06:54:06.614372Z","iopub.execute_input":"2025-03-18T06:54:06.614736Z","iopub.status.idle":"2025-03-18T06:54:06.654879Z","shell.execute_reply.started":"2025-03-18T06:54:06.614709Z","shell.execute_reply":"2025-03-18T06:54:06.654001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_grouped = merged_data.groupby(\"target_id\")[\"encoded_seq\"].apply(lambda x: list(x)).tolist()\n\nX_grouped = [np.concatenate(seq).tolist() for seq in X_grouped]\nY_grouped = merged_data.groupby(\"target_id\")[[\"x_1\", \"y_1\", \"z_1\"]].apply(lambda x: x.values).tolist()\n\nY_grouped = [np.array(seq).tolist() for seq in Y_grouped]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T06:54:08.988074Z","iopub.execute_input":"2025-03-18T06:54:08.988348Z","iopub.status.idle":"2025-03-18T06:54:13.862093Z","shell.execute_reply.started":"2025-03-18T06:54:08.988326Z","shell.execute_reply":"2025-03-18T06:54:13.861413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"max_seq_length = train_sequences.groupby(\"target_id\")[\"sequence\"].apply(len).max()\nprint(f\"Max sequence length: {max_seq_length}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T06:54:42.089932Z","iopub.execute_input":"2025-03-18T06:54:42.090212Z","iopub.status.idle":"2025-03-18T06:54:42.109910Z","shell.execute_reply.started":"2025-03-18T06:54:42.090187Z","shell.execute_reply":"2025-03-18T06:54:42.109147Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.sequence import pad_sequences\n\nX_padded = pad_sequences(X_grouped, maxlen=max_seq_length, padding=\"post\", dtype=\"float32\")\n\nY_padded = pad_sequences(Y_grouped, maxlen=max_seq_length, padding=\"post\", dtype=\"float32\")\n\nY_padded = np.array(Y_padded)\n\nprint(f\"X shape: {X_padded.shape}, Y shape: {Y_padded.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T06:54:44.548594Z","iopub.execute_input":"2025-03-18T06:54:44.548896Z","iopub.status.idle":"2025-03-18T06:54:44.559018Z","shell.execute_reply.started":"2025-03-18T06:54:44.548873Z","shell.execute_reply":"2025-03-18T06:54:44.558328Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_val, Y_train, Y_val = train_test_split(X_padded, Y_padded, test_size=0.2, random_state=42)\n\nprint(f\"Train Shape: X={X_train.shape}, Y={Y_train.shape}\")\nprint(f\"Validation Shape: X={X_val.shape}, Y={Y_val.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T06:54:49.359045Z","iopub.execute_input":"2025-03-18T06:54:49.359321Z","iopub.status.idle":"2025-03-18T06:54:49.479768Z","shell.execute_reply.started":"2025-03-18T06:54:49.359298Z","shell.execute_reply":"2025-03-18T06:54:49.478916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"NaNs in X_padded:\", np.isnan(X_padded).sum())\nprint(\"NaNs in Y_padded:\", np.isnan(Y_padded).sum())\n\nprint(\"Infs in X_padded:\", np.isinf(X_padded).sum())\nprint(\"Infs in Y_padded:\", np.isinf(Y_padded).sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T06:21:29.656955Z","iopub.execute_input":"2025-03-18T06:21:29.657461Z","iopub.status.idle":"2025-03-18T06:21:29.772923Z","shell.execute_reply.started":"2025-03-18T06:21:29.657421Z","shell.execute_reply":"2025-03-18T06:21:29.771672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"nan_rows = np.isnan(Y_padded).sum(axis=1) > 0\n\nprint(f\"Number of sequences with NaNs: {nan_rows.sum()} out of {len(Y_padded)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T06:54:54.274861Z","iopub.execute_input":"2025-03-18T06:54:54.275154Z","iopub.status.idle":"2025-03-18T06:54:54.279992Z","shell.execute_reply.started":"2025-03-18T06:54:54.275132Z","shell.execute_reply":"2025-03-18T06:54:54.279303Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in range(Y_padded.shape[1]):  \n    for j in range(3):  \n        nan_mask = np.isnan(Y_padded[:, i, j])\n        Y_padded[nan_mask, i, j] = np.nanmean(Y_padded[:, :, j])  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T06:54:56.436986Z","iopub.execute_input":"2025-03-18T06:54:56.437271Z","iopub.status.idle":"2025-03-18T06:54:56.442804Z","shell.execute_reply.started":"2025-03-18T06:54:56.437250Z","shell.execute_reply":"2025-03-18T06:54:56.441861Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Shape of X_padded: {X_padded.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T06:56:02.907739Z","iopub.execute_input":"2025-03-18T06:56:02.908049Z","iopub.status.idle":"2025-03-18T06:56:02.912489Z","shell.execute_reply.started":"2025-03-18T06:56:02.908023Z","shell.execute_reply":"2025-03-18T06:56:02.911723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_sequences.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T06:57:23.059673Z","iopub.execute_input":"2025-03-18T06:57:23.060023Z","iopub.status.idle":"2025-03-18T06:57:23.068028Z","shell.execute_reply.started":"2025-03-18T06:57:23.059996Z","shell.execute_reply":"2025-03-18T06:57:23.067150Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"max_seq_length = train_sequences[\"sequence\"].apply(len).max()\nprint(f\"Max sequence length: {max_seq_length}\")\n\nX_padded = pad_sequences(\n    train_sequences[\"encoded_seq\"].tolist(),  \n    maxlen=max_seq_length,  \n    padding=\"post\",  \n    dtype=\"float32\"\n)\n\nprint(f\"Updated X_padded shape: {X_padded.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T06:57:34.269120Z","iopub.execute_input":"2025-03-18T06:57:34.269410Z","iopub.status.idle":"2025-03-18T06:57:34.282134Z","shell.execute_reply.started":"2025-03-18T06:57:34.269388Z","shell.execute_reply":"2025-03-18T06:57:34.281270Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_padded = np.expand_dims(X_padded, axis=-1)  \nprint(f\"Final X_padded shape: {X_padded.shape}\") ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T06:57:47.823097Z","iopub.execute_input":"2025-03-18T06:57:47.823461Z","iopub.status.idle":"2025-03-18T06:57:47.828069Z","shell.execute_reply.started":"2025-03-18T06:57:47.823393Z","shell.execute_reply":"2025-03-18T06:57:47.827049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_labels.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T07:17:55.194190Z","iopub.execute_input":"2025-03-18T07:17:55.194567Z","iopub.status.idle":"2025-03-18T07:17:55.199101Z","shell.execute_reply.started":"2025-03-18T07:17:55.194534Z","shell.execute_reply":"2025-03-18T07:17:55.198301Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv1D, Dropout, Flatten, Dense\n\n# Extract features (RNA sequences)\nX_raw = train_sequences[\"sequence\"].apply(lambda x: [ord(c) for c in x])  # Convert to numerical values\nmax_seq_length = max(X_raw.apply(len))  # Find longest sequence\n\n# Pad RNA sequences to ensure uniform shape\nX_padded = pad_sequences(X_raw, maxlen=max_seq_length, padding=\"post\", dtype=\"float32\")\n\n# Extract labels\nY_padded = np.array(train_labels[[\"x_1\", \"y_1\", \"z_1\"]].fillna(0))  # Ensure correct columns & no NaNs\n\n# Reshape X to fit CNN input\nX_padded = np.expand_dims(X_padded, axis=-1)  # Shape (samples, max_seq_length, 1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T07:20:45.033050Z","iopub.execute_input":"2025-03-18T07:20:45.033336Z","iopub.status.idle":"2025-03-18T07:20:45.068931Z","shell.execute_reply.started":"2025-03-18T07:20:45.033313Z","shell.execute_reply":"2025-03-18T07:20:45.068254Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define CNN Model\nmodel = Sequential([\n    Conv1D(filters=64, kernel_size=3, activation=\"relu\", input_shape=(X_padded.shape[1], 1)),\n    Dropout(0.3),\n    Flatten(),\n    Dense(128, activation=\"relu\"),\n    Dropout(0.3),\n    Dense(3)  # Predicting (x_1, y_1, z_1) only\n])\n\n# Compile Model\nmodel.compile(optimizer=\"adam\", loss=\"mse\", metrics=[\"mae\"])\n\n# Train Model\nmodel.fit(X_padded, Y_padded, epochs=10, batch_size=32, validation_split=0.1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T07:20:57.023141Z","iopub.execute_input":"2025-03-18T07:20:57.023436Z","iopub.status.idle":"2025-03-18T07:21:06.379424Z","shell.execute_reply.started":"2025-03-18T07:20:57.023411Z","shell.execute_reply":"2025-03-18T07:21:06.378768Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Preprocess test data\nX_test_raw = test_sequences[\"sequence\"].apply(lambda x: [ord(c) for c in x])  # Convert sequences to numbers\nX_test_padded = pad_sequences(X_test_raw, maxlen=max_seq_length, padding=\"post\", dtype=\"float32\")\nX_test_padded = np.expand_dims(X_test_padded, axis=-1)  # Shape (samples, max_seq_length, 1)\n\n# Make Predictions\npredictions = model.predict(X_test_padded)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T07:21:25.652975Z","iopub.execute_input":"2025-03-18T07:21:25.653270Z","iopub.status.idle":"2025-03-18T07:21:26.443150Z","shell.execute_reply.started":"2025-03-18T07:21:25.653248Z","shell.execute_reply":"2025-03-18T07:21:26.442492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Ensure submission matches required format\nsubmission_df = pd.DataFrame({\n    \"ID\": test_sequences[\"target_id\"],  \n    \"resname\": \"G\",  # Assuming all residues are 'G'\n    \"resid\": range(1, len(test_sequences) + 1),\n    \"x_1\": predictions[:, 0], \"y_1\": predictions[:, 1], \"z_1\": predictions[:, 2],\n    \"x_2\": predictions[:, 0], \"y_2\": predictions[:, 1], \"z_2\": predictions[:, 2],  # Duplicate for missing data\n    \"x_3\": predictions[:, 0], \"y_3\": predictions[:, 1], \"z_3\": predictions[:, 2],\n    \"x_4\": predictions[:, 0], \"y_4\": predictions[:, 1], \"z_4\": predictions[:, 2],\n    \"x_5\": predictions[:, 0], \"y_5\": predictions[:, 1], \"z_5\": predictions[:, 2],\n})\n\n# Save to CSV\nsubmission_df.to_csv(\"submission.csv\", index=False)\nprint(\"Submission saved successfully!\")\nsubmission_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T07:21:47.245131Z","iopub.execute_input":"2025-03-18T07:21:47.245435Z","iopub.status.idle":"2025-03-18T07:21:47.268854Z","shell.execute_reply.started":"2025-03-18T07:21:47.245409Z","shell.execute_reply":"2025-03-18T07:21:47.267953Z"}},"outputs":[],"execution_count":null}]}