{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":12276181,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<a id=\"2\"></a>\n<h1 style='background:#000000;border:0; color:black;\n    box-shadow: 10px 10px 5px 0px rgba(0,0,0,0.75);\n    transform: rotateX(10deg);\n    '><center style='color: #17E8C4;'>IMPORT IMPORTANT LIBRARIES</center></h1>","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# TensorFlow/Keras for deep learning model\nimport tensorflow as tf\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, Embedding, Conv1D, BatchNormalization, Dropout, Dense, Flatten\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom tensorflow.keras.callbacks import EarlyStopping\n\n# Set random seed for reproducibility\nnp.random.seed(42)\ntf.random.set_seed(42)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-23T04:31:40.326921Z","iopub.execute_input":"2025-05-23T04:31:40.327660Z","iopub.status.idle":"2025-05-23T04:32:00.609727Z","shell.execute_reply.started":"2025-05-23T04:31:40.327629Z","shell.execute_reply":"2025-05-23T04:32:00.608794Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id=\"2\"></a>\n<h1 style='background:#000000;border:0; color:black;\n    box-shadow: 10px 10px 5px 0px rgba(0,0,0,0.75);\n    transform: rotateX(10deg);\n    '><center style='color: #17E8C4;'>LOADING DATASET</center></h1>","metadata":{}},{"cell_type":"code","source":"# Define file paths (Kaggle input paths)\nTRAIN_SEQ_PATH = '/kaggle/input/stanford-rna-3d-folding/train_sequences.csv'\nTRAIN_LABELS_PATH = '/kaggle/input/stanford-rna-3d-folding/train_labels.csv'\nVALID_SEQ_PATH = '/kaggle/input/stanford-rna-3d-folding/validation_sequences.csv'\nVALID_LABELS_PATH = '/kaggle/input/stanford-rna-3d-folding/validation_labels.csv'\nTEST_SEQ_PATH  = '/kaggle/input/stanford-rna-3d-folding/test_sequences.csv'\nSAMPLE_SUB_PATH = '/kaggle/input/stanford-rna-3d-folding/sample_submission.csv'\n\n# Load CSV files\ntrain_sequences = pd.read_csv(TRAIN_SEQ_PATH)\ntrain_labels = pd.read_csv(TRAIN_LABELS_PATH)\nvalid_sequences = pd.read_csv(VALID_SEQ_PATH)\nvalid_labels = pd.read_csv(VALID_LABELS_PATH)\ntest_sequences = pd.read_csv(TEST_SEQ_PATH)\nsample_submission = pd.read_csv(SAMPLE_SUB_PATH)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T04:32:00.611301Z","iopub.execute_input":"2025-05-23T04:32:00.612016Z","iopub.status.idle":"2025-05-23T04:32:01.134238Z","shell.execute_reply.started":"2025-05-23T04:32:00.611989Z","shell.execute_reply":"2025-05-23T04:32:01.133161Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id=\"2\"></a>\n<h1 style='background:#000000;border:0; color:black;\n    box-shadow: 10px 10px 5px 0px rgba(0,0,0,0.75);\n    transform: rotateX(10deg);\n    '><center style='color: #17E8C4;'>EDA</center></h1>","metadata":{}},{"cell_type":"code","source":"# Fill missing values in labels with 0\ntrain_labels.fillna(0, inplace=True)\nvalid_labels.fillna(0, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T04:32:01.135293Z","iopub.execute_input":"2025-05-23T04:32:01.135575Z","iopub.status.idle":"2025-05-23T04:32:01.165711Z","shell.execute_reply.started":"2025-05-23T04:32:01.135546Z","shell.execute_reply":"2025-05-23T04:32:01.164805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Display basic info\nprint(\"Train Sequences Shape:\", train_sequences.shape)\nprint(\"Train Labels Shape:\", train_labels.shape)\nprint(\"Validation Sequences Shape:\", valid_sequences.shape)\nprint(\"Validation Labels Shape:\", valid_labels.shape)\nprint(\"Test Sequences Shape:\", test_sequences.shape)\n\n# Look at a few examples\nprint(\"\\nTrain Sequences Head:\")\nprint(train_sequences.head())\nprint(\"\\nTrain Labels Head:\")\nprint(train_labels.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T04:32:01.167677Z","iopub.execute_input":"2025-05-23T04:32:01.168056Z","iopub.status.idle":"2025-05-23T04:32:01.190677Z","shell.execute_reply.started":"2025-05-23T04:32:01.168031Z","shell.execute_reply":"2025-05-23T04:32:01.189525Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define nucleotide mapping\nnucleotide_map = {'A': 1, 'C': 2, 'G': 3, 'U': 4}\n\ndef encode_sequence(seq):\n    \"\"\"Encodes an RNA sequence into a list of integers based on nucleotide_map.\"\"\"\n    return [nucleotide_map.get(ch, 0) for ch in seq]\n\n# Apply encoding to all sequence files\ntrain_sequences['encoded'] = train_sequences['sequence'].apply(encode_sequence)\nvalid_sequences['encoded'] = valid_sequences['sequence'].apply(encode_sequence)\ntest_sequences['encoded'] = test_sequences['sequence'].apply(encode_sequence)\n\n# Determine the maximum sequence length for padding\nmax_seq_length = max(train_sequences['encoded'].apply(len).max(),\n                     valid_sequences['encoded'].apply(len).max(),\n                     test_sequences['encoded'].apply(len).max())\n\n# Pad sequences\nX_train = pad_sequences(train_sequences['encoded'], maxlen=max_seq_length, padding='post')\nX_valid = pad_sequences(valid_sequences['encoded'], maxlen=max_seq_length, padding='post')\nX_test = pad_sequences(test_sequences['encoded'], maxlen=max_seq_length, padding='post')\n\n# Prepare labels\n# Extract x, y, z coordinates from train_labels\n# Assuming train_labels has columns: ID, resname, resid, x_1, y_1, z_1\n# Group by target_id to align with sequences\ntrain_labels['target_id'] = train_labels['ID'].apply(lambda x: '_'.join(x.split('_')[:-1]))\ngrouped = train_labels.groupby('target_id')\ny_train = []\nfor target_id in train_sequences['target_id']:\n    group = grouped.get_group(target_id)\n    coords = group[['x_1', 'y_1', 'z_1']].values\n    # Pad coordinates if necessary\n    if coords.shape[0] < max_seq_length:\n        padding = np.zeros((max_seq_length - coords.shape[0], 3))\n        coords = np.vstack([coords, padding])\n    y_train.append(coords)\ny_train = np.array(y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T04:32:01.191706Z","iopub.execute_input":"2025-05-23T04:32:01.191965Z","iopub.status.idle":"2025-05-23T04:32:01.951094Z","shell.execute_reply.started":"2025-05-23T04:32:01.191945Z","shell.execute_reply":"2025-05-23T04:32:01.950084Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id=\"2\"></a>\n<h1 style='background:#000000;border:0; color:black;\n    box-shadow: 10px 10px 5px 0px rgba(0,0,0,0.75);\n    transform: rotateX(10deg);\n    '><center style='color: #17E8C4;'>MODEL BUILDING</center></h1>","metadata":{}},{"cell_type":"code","source":"# Build the model\ninput_layer = Input(shape=(max_seq_length,))\nembedding_layer = Embedding(input_dim=5, output_dim=64, input_length=max_seq_length)(input_layer)\nconv1 = Conv1D(filters=128, kernel_size=3, activation='relu', padding='same')(embedding_layer)\nbn1 = BatchNormalization()(conv1)\ndrop1 = Dropout(0.3)(bn1)\nconv2 = Conv1D(filters=64, kernel_size=3, activation='relu', padding='same')(drop1)\nbn2 = BatchNormalization()(conv2)\ndrop2 = Dropout(0.3)(bn2)\nflatten = Flatten()(drop2)\ndense1 = Dense(256, activation='relu')(flatten)\noutput_layer = Dense(max_seq_length * 3)(dense1)\n\nmodel = Model(inputs=input_layer, outputs=output_layer)\nmodel.compile(optimizer='adam', loss='mse')\n\n# Reshape y_train for training\ny_train_reshaped = y_train.reshape(y_train.shape[0], -1)\n\n# Train the model\nearly_stopping = EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True)\nmodel.fit(X_train, y_train_reshaped, validation_split=0.1, epochs=50, batch_size=32, callbacks=[early_stopping])\n\n# Predict on test data\npredictions = model.predict(X_test)\n# Reshape predictions to (num_samples, max_seq_length, 3)\npredictions = predictions.reshape(predictions.shape[0], max_seq_length, 3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T04:32:01.952077Z","iopub.execute_input":"2025-05-23T04:32:01.952349Z","iopub.status.idle":"2025-05-23T04:57:14.146572Z","shell.execute_reply.started":"2025-05-23T04:32:01.952328Z","shell.execute_reply":"2025-05-23T04:57:14.145372Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id=\"2\"></a>\n<h1 style='background:#000000;border:0; color:black;\n    box-shadow: 10px 10px 5px 0px rgba(0,0,0,0.75);\n    transform: rotateX(10deg);\n    '><center style='color: #17E8C4;'>SUBMISSION</center></h1>","metadata":{}},{"cell_type":"code","source":"# Prepare submission\nsubmission = []\nfor idx, target_id in enumerate(test_sequences['target_id']):\n    sequence = test_sequences.loc[test_sequences['target_id'] == target_id, 'sequence'].values[0]\n    for resid, nucleotide in enumerate(sequence, start=1):\n        coords = predictions[idx][resid - 1]\n        row = {\n            'ID': f\"{target_id}_{resid}\",\n            'resname': nucleotide,\n            'resid': resid,\n            'x_1': coords[0],\n            'y_1': coords[1],\n            'z_1': coords[2],\n            # If you have x_2 to z_5, add them here:\n            # 'x_2': coords[3], 'y_2': coords[4], 'z_2': coords[5], ...\n        }\n        submission.append(row)\n\n# Convert to DataFrame and save\nsubmission_df = pd.DataFrame(submission)\nsubmission_df.to_csv('submission.csv', index=False)\n\n# Confirmation\nprint(\"Submission file 'submission.csv' has been saved successfully.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T04:57:14.147515Z","iopub.execute_input":"2025-05-23T04:57:14.147786Z","iopub.status.idle":"2025-05-23T04:57:14.189450Z","shell.execute_reply.started":"2025-05-23T04:57:14.147762Z","shell.execute_reply":"2025-05-23T04:57:14.188040Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(submission_df.head())\nprint(f\"Total rows in submission: {len(submission_df)}\")\nprint(f\"Unique target_ids: {submission_df['ID'].apply(lambda x: x.split('_')[0]).nunique()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T05:36:41.353594Z","iopub.execute_input":"2025-05-23T05:36:41.353993Z","iopub.status.idle":"2025-05-23T05:36:41.371648Z","shell.execute_reply.started":"2025-05-23T05:36:41.353968Z","shell.execute_reply":"2025-05-23T05:36:41.370705Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = []\n\n# Ensure target_id is unique per sequence\nfor idx, target_id in enumerate(test_sequences['target_id'].unique()):\n    # Extract the corresponding sequence\n    sequence_row = test_sequences[test_sequences['target_id'] == target_id]\n    \n    if sequence_row.empty:\n        print(f\"Warning: target_id {target_id} not found in test_sequences.\")\n        continue\n    \n    sequence = sequence_row.iloc[0]['sequence']\n    predicted_coords = predictions[idx]  # shape: (sequence_length, 3)\n\n    for resid, nucleotide in enumerate(sequence, start=1):\n        coords = predicted_coords[resid - 1]\n        \n        row = {\n            'ID': f\"{target_id}_{resid}\",\n            'resname': nucleotide,\n            'resid': resid,\n            'x_1': coords[0],\n            'y_1': coords[1],\n            'z_1': coords[2],\n        }\n        submission.append(row)\n\n# Convert to DataFrame and save\nsubmission_df = pd.DataFrame(submission)\nsubmission_df.to_csv('submission.csv', index=False)\n\nprint(\"Submission file 'submission.csv' has been saved successfully.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T05:37:19.315016Z","iopub.execute_input":"2025-05-23T05:37:19.315474Z","iopub.status.idle":"2025-05-23T05:37:19.359947Z","shell.execute_reply.started":"2025-05-23T05:37:19.315448Z","shell.execute_reply":"2025-05-23T05:37:19.358884Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}