# %% [code] {"execution":{"iopub.status.busy":"2025-04-06T07:22:12.137441Z","iopub.execute_input":"2025-04-06T07:22:12.137671Z","iopub.status.idle":"2025-04-06T07:22:12.565646Z","shell.execute_reply.started":"2025-04-06T07:22:12.137651Z","shell.execute_reply":"2025-04-06T07:22:12.564939Z"}}
import pandas as pd
import numpy as np
import matplotlib.pyplot as plt
import seaborn as sns
from collections import Counter
from mpl_toolkits.mplot3d import Axes3D
from scipy.spatial.distance import cdist
from sklearn.cluster import KMeans
from sklearn.feature_extraction.text import CountVectorizer
from sklearn.preprocessing import StandardScaler
from sklearn.decomposition import PCA
from tensorflow.keras.models import Model
from tensorflow.keras.layers import Input, Embedding, Conv1D, BatchNormalization, Dropout, LSTM, Bidirectional, Dense
from tensorflow.keras.preprocessing.sequence import pad_sequences
from sklearn.metrics import mean_squared_error
import warnings
warnings.simplefilter(action='ignore', category=FutureWarning)

# Load the datasets
train_sequences = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/train_sequences.csv')
train_labels = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/train_labels.csv')
validation_sequences = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/validation_sequences.csv')
validation_labels = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/validation_labels.csv')
test_sequence = pd.read_csv("/kaggle/input/stanford-rna-3d-folding/test_sequences.csv")

# Display basic info
print("Train Sequences Info:")
print(train_sequences.info())
print("\nTrain Labels Info:")
print(train_labels.info())
print("\nValidation Sequences Info:")
print(validation_sequences.info())
print("\nValidation Labels Info:")
print(validation_labels.info())

# %% [code] {"execution":{"iopub.status.busy":"2025-04-06T07:22:12.566970Z","iopub.execute_input":"2025-04-06T07:22:12.567213Z","iopub.status.idle":"2025-04-06T07:22:13.012197Z","shell.execute_reply.started":"2025-04-06T07:22:12.567193Z","shell.execute_reply":"2025-04-06T07:22:13.011338Z"}}
train_sequences['sequence_length'] = train_sequences['sequence'].apply(len)
validation_sequences['sequence_length'] = validation_sequences['sequence'].apply(len)

plt.figure(figsize=(10, 6))
sns.histplot(train_sequences['sequence_length'], bins=50, kde=True, label='Train Sequences', color='blue')
sns.histplot(validation_sequences['sequence_length'], bins=50, kde=True, label='Validation Sequences', color='orange')
plt.title('Distribution of Sequence Lengths')
plt.xlabel('Sequence Length')
plt.ylabel('Frequency')
plt.legend()
plt.show()

# %% [code] {"execution":{"iopub.status.busy":"2025-04-06T07:22:13.013011Z","iopub.execute_input":"2025-04-06T07:22:13.013328Z","iopub.status.idle":"2025-04-06T07:22:13.284851Z","shell.execute_reply.started":"2025-04-06T07:22:13.013298Z","shell.execute_reply":"2025-04-06T07:22:13.284141Z"}}
def nucleotide_composition(sequence):
    return dict(Counter(sequence))

train_sequences['nucleotide_composition'] = train_sequences['sequence'].apply(nucleotide_composition)
validation_sequences['nucleotide_composition'] = validation_sequences['sequence'].apply(nucleotide_composition)

train_nucleotide_counts = pd.DataFrame(train_sequences['nucleotide_composition'].tolist()).fillna(0).sum()
validation_nucleotide_counts = pd.DataFrame(validation_sequences['nucleotide_composition'].tolist()).fillna(0).sum()

plt.figure(figsize=(10, 6))
train_nucleotide_counts.plot(kind='bar', color='blue', label='Train Sequences')
validation_nucleotide_counts.plot(kind='bar', color='orange', label='Validation Sequences', alpha=0.7)
plt.title('Nucleotide Composition')
plt.xlabel('Nucleotide')
plt.ylabel('Count')
plt.legend()
plt.show()

# %% [code] {"execution":{"iopub.status.busy":"2025-04-06T07:22:13.285705Z","iopub.execute_input":"2025-04-06T07:22:13.285913Z","iopub.status.idle":"2025-04-06T07:22:13.712991Z","shell.execute_reply.started":"2025-04-06T07:22:13.285894Z","shell.execute_reply":"2025-04-06T07:22:13.712175Z"}}
train_sequences['temporal_cutoff'] = pd.to_datetime(train_sequences['temporal_cutoff'])
validation_sequences['temporal_cutoff'] = pd.to_datetime(validation_sequences['temporal_cutoff'])

train_sequences['year'] = train_sequences['temporal_cutoff'].dt.year
validation_sequences['year'] = validation_sequences['temporal_cutoff'].dt.year

plt.figure(figsize=(10, 6))
sns.histplot(train_sequences['temporal_cutoff'], bins=50, kde=True, label='Train Sequences')
sns.histplot(validation_sequences['temporal_cutoff'], bins=50, kde=True, label='Validation Sequences', color='orange')
plt.title('Temporal Distribution of Sequences')
plt.xlabel('Temporal Cutoff')
plt.ylabel('Frequency')
plt.legend()
plt.show()

# %% [code] {"execution":{"iopub.status.busy":"2025-04-06T07:22:13.714007Z","iopub.execute_input":"2025-04-06T07:22:13.714334Z","iopub.status.idle":"2025-04-06T07:22:16.758840Z","shell.execute_reply.started":"2025-04-06T07:22:13.714302Z","shell.execute_reply":"2025-04-06T07:22:16.758058Z"}}
# Analyze the distribution of 3D coordinates in train_labels
coordinate_columns = [col for col in train_labels.columns if col.startswith(('x_', 'y_', 'z_'))]
train_labels['num_structures'] = train_labels[coordinate_columns].count(axis=1) // 3

# Plot number of structures per target
plt.figure(figsize=(10, 6))
sns.histplot(train_labels['num_structures'], bins=20, kde=True)
plt.title('Distribution of Number of Structures per Target')
plt.xlabel('Number of Structures')
plt.ylabel('Frequency')
plt.show()

# Plot distribution of coordinates
plt.figure(figsize=(15, 5))
for i, coord in enumerate(['x_1', 'y_1', 'z_1']):
    plt.subplot(1, 3, i+1)
    sns.histplot(train_labels[coord].dropna(), bins=50, kde=True)
    plt.title(f'Distribution of {coord}')
    plt.xlabel(coord)
    plt.ylabel('Frequency')
plt.tight_layout()
plt.show()

# %% [code] {"execution":{"iopub.status.busy":"2025-04-06T07:22:16.759878Z","iopub.execute_input":"2025-04-06T07:22:16.760222Z","iopub.status.idle":"2025-04-06T07:22:17.136146Z","shell.execute_reply.started":"2025-04-06T07:22:16.760191Z","shell.execute_reply":"2025-04-06T07:22:17.135336Z"}}
train_sequences['description_length'] = train_sequences['description'].apply(len)
validation_sequences['description_length'] = validation_sequences['description'].apply(len)

plt.figure(figsize=(10, 6))
sns.histplot(train_sequences['description_length'], bins=50, kde=True, label='Train Sequences')
sns.histplot(validation_sequences['description_length'], bins=50, kde=True, label='Validation Sequences', color='orange')
plt.title('Distribution of Description Lengths')
plt.xlabel('Description Length')
plt.ylabel('Frequency')
plt.legend()
plt.show()

# %% [code] {"execution":{"iopub.status.busy":"2025-04-06T07:22:17.137015Z","iopub.execute_input":"2025-04-06T07:22:17.137336Z","iopub.status.idle":"2025-04-06T07:22:17.160550Z","shell.execute_reply.started":"2025-04-06T07:22:17.137304Z","shell.execute_reply":"2025-04-06T07:22:17.159740Z"}}
duplicate_sequences = train_sequences[train_sequences.duplicated('sequence', keep=False)]
print(f"Number of duplicate sequences in train_sequences: {len(duplicate_sequences)}")

# Check for duplicate target_ids in train_labels
duplicate_targets = train_labels[train_labels.duplicated('ID', keep=False)]
print(f"Number of duplicate targets in train_labels: {len(duplicate_targets)}")

# %% [code] {"execution":{"iopub.status.busy":"2025-04-06T07:22:17.161468Z","iopub.execute_input":"2025-04-06T07:22:17.161779Z","iopub.status.idle":"2025-04-06T07:22:19.636224Z","shell.execute_reply.started":"2025-04-06T07:22:17.161750Z","shell.execute_reply":"2025-04-06T07:22:19.635453Z"}}
# Extract coordinates for a sample target
sample_target = train_labels
x = sample_target['x_1'].values
y = sample_target['y_1'].values
z = sample_target['z_1'].values

# Plot 3D structure
fig = plt.figure(figsize=(10, 8))
ax = fig.add_subplot(111, projection='3d')
ax.scatter(x, y, z, c='blue', marker='o')
ax.set_title('3D RNA Structure')
ax.set_xlabel('X')
ax.set_ylabel('Y')
ax.set_zlabel('Z')
plt.show()

# %% [code] {"execution":{"iopub.status.busy":"2025-04-06T07:22:19.637259Z","iopub.execute_input":"2025-04-06T07:22:19.637596Z","iopub.status.idle":"2025-04-06T07:22:19.780910Z","shell.execute_reply.started":"2025-04-06T07:22:19.637565Z","shell.execute_reply":"2025-04-06T07:22:19.780002Z"}}
# Check for ligand mentions in description
train_sequences['has_ligand'] = train_sequences['description'].str.contains('ligand', case=False)

# Compare sequence lengths for ligand-bound vs. unbound sequences
plt.figure(figsize=(10, 6))
sns.boxplot(x='has_ligand', y='sequence_length', data=train_sequences)
plt.title('Sequence Length for Ligand-Bound vs. Unbound Sequences')
plt.xlabel('Has Ligand')
plt.ylabel('Sequence Length')
plt.show()

# %% [code] {"execution":{"iopub.status.busy":"2025-04-06T07:22:19.781849Z","iopub.execute_input":"2025-04-06T07:22:19.782160Z","iopub.status.idle":"2025-04-06T07:22:20.094698Z","shell.execute_reply.started":"2025-04-06T07:22:19.782127Z","shell.execute_reply":"2025-04-06T07:22:20.093773Z"}}
# Convert sequences to k-mer counts
vectorizer = CountVectorizer(analyzer='char', ngram_range=(3, 3))
kmer_counts = vectorizer.fit_transform(train_sequences['sequence'])

# Perform k-means clustering
kmeans = KMeans(n_clusters=5, random_state=42)
clusters = kmeans.fit_predict(kmer_counts)

# Add clusters to the dataframe
train_sequences['cluster'] = clusters

# Plot cluster distribution
plt.figure(figsize=(10, 6))
sns.countplot(x='cluster', data=train_sequences)
plt.title('Sequence Clusters')
plt.xlabel('Cluster')
plt.ylabel('Count')
plt.show()

# %% [code] {"execution":{"iopub.status.busy":"2025-04-06T07:22:20.097404Z","iopub.execute_input":"2025-04-06T07:22:20.097638Z","iopub.status.idle":"2025-04-06T07:22:20.714812Z","shell.execute_reply.started":"2025-04-06T07:22:20.097618Z","shell.execute_reply":"2025-04-06T07:22:20.713814Z"}}
coordinate_data = train_labels[['x_1', 'y_1', 'z_1']].dropna().values

# Standardizing the coordinates
scaler = StandardScaler()
scaled_coordinates = scaler.fit_transform(coordinate_data)

# Perform PCA to reduce the dimensionality
pca = PCA(n_components=2)
pca_result = pca.fit_transform(scaled_coordinates)

# Plot the first two principal components
plt.figure(figsize=(10, 6))
plt.scatter(pca_result[:, 0], pca_result[:, 1], alpha=0.5)
plt.title('PCA on 3D Coordinates of RNA Sequences')
plt.xlabel('Principal Component 1')
plt.ylabel('Principal Component 2')
plt.colorbar(label='Target ID')
plt.show()

# %% [code] {"execution":{"iopub.status.busy":"2025-04-06T07:22:20.716114Z","iopub.execute_input":"2025-04-06T07:22:20.716343Z","iopub.status.idle":"2025-04-06T07:22:20.737432Z","shell.execute_reply.started":"2025-04-06T07:22:20.716324Z","shell.execute_reply":"2025-04-06T07:22:20.736664Z"}}
# Check for missing values
print("Missing values in train_sequences:")
print(train_sequences.isnull().sum())

print("\nMissing values in train_labels:")
print(train_labels.isnull().sum())

# %% [code] {"execution":{"iopub.status.busy":"2025-04-06T07:22:20.738180Z","iopub.execute_input":"2025-04-06T07:22:20.738457Z","iopub.status.idle":"2025-04-06T07:22:20.996477Z","shell.execute_reply.started":"2025-04-06T07:22:20.738424Z","shell.execute_reply":"2025-04-06T07:22:20.995543Z"}}
# Load Data
train_labels = pd.read_csv("/kaggle/input/stanford-rna-3d-folding/train_labels.csv")
train_sequence = pd.read_csv("/kaggle/input/stanford-rna-3d-folding/train_sequences.csv")
val_labels = pd.read_csv("/kaggle/input/stanford-rna-3d-folding/validation_labels.csv")
val_sequence = pd.read_csv("/kaggle/input/stanford-rna-3d-folding/validation_sequences.csv")
test_sequence = pd.read_csv("/kaggle/input/stanford-rna-3d-folding/test_sequences.csv")

# Fill missing values
train_labels.fillna(0, inplace=True)
validation_labels.fillna(0, inplace=True)

# %% [code] {"execution":{"iopub.status.busy":"2025-04-06T07:22:20.997437Z","iopub.execute_input":"2025-04-06T07:22:20.997761Z","iopub.status.idle":"2025-04-06T07:22:21.018272Z","shell.execute_reply.started":"2025-04-06T07:22:20.997732Z","shell.execute_reply":"2025-04-06T07:22:21.017417Z"}}
# Sequence Encoding
seq_dict = {'A': 1, 'C': 2, 'G': 3, 'U': 4}
def seq_map(seq):
    return [seq_dict.get(char, 0) for char in seq]

train_sequence['encoded_seq'] = train_sequence['sequence'].apply(seq_map)
test_sequence['encoded_seq'] = test_sequence['sequence'].apply(seq_map)
val_sequence['encoded_seq'] = val_sequence['sequence'].apply(seq_map)

# %% [code] {"execution":{"iopub.status.busy":"2025-04-06T07:22:21.018953Z","iopub.execute_input":"2025-04-06T07:22:21.019193Z","iopub.status.idle":"2025-04-06T07:22:32.874193Z","shell.execute_reply.started":"2025-04-06T07:22:21.019170Z","shell.execute_reply":"2025-04-06T07:22:32.873540Z"}}
def generate_label_coord(df):
    result = {}
    df["label"] = df.ID.str.rsplit('_', n=1, expand=True).iloc[:,0]
    for _, row in df.iterrows():
        label = row['label']
        resid = row['resid']
        if label not in result:
            result[label] = []
        
        # Check if all required columns exist
        if all(col in row for col in ['x_1', 'y_1', 'z_1']):
            # If only x_1, y_1, z_1 are available, duplicate them for x_2, y_2, z_2, etc.
            coords = np.array([
                [row['x_1'], row['y_1'], row['z_1']],
                [row['x_1'], row['y_1'], row['z_1']],  # Duplicate for x_2, y_2, z_2
                [row['x_1'], row['y_1'], row['z_1']],  # Duplicate for x_3, y_3, z_3
                [row['x_1'], row['y_1'], row['z_1']],  # Duplicate for x_4, y_4, z_4
                [row['x_1'], row['y_1'], row['z_1']]   # Duplicate for x_5, y_5, z_5
            ], dtype=np.float32)
        else:
            # If no coordinates are available, use zeros
            coords = np.zeros((5, 3), dtype=np.float32)
        
        result[label].append((resid, coords))
    
    for key in result:
        coords = np.stack([c for r, c in result[key]])
        result[key] = coords
    
    return result

train_stacked_coords = generate_label_coord(train_labels)
val_stacked_coords = generate_label_coord(val_labels)

# %% [code] {"execution":{"iopub.status.busy":"2025-04-06T07:22:32.874992Z","iopub.execute_input":"2025-04-06T07:22:32.875228Z","iopub.status.idle":"2025-04-06T07:22:32.915945Z","shell.execute_reply.started":"2025-04-06T07:22:32.875208Z","shell.execute_reply":"2025-04-06T07:22:32.915310Z"}}
def generate_dataset(seq, stacked_coords):
    X, y, tids = [], [], []
    for idx, row in seq.iterrows():
        tid = row['target_id']
        if tid in stacked_coords:
            X.append(row['encoded_seq'])
            y.append(stacked_coords[tid])
            tids.append(tid)
    return X, y, tids

train_X, train_y, train_tids = generate_dataset(train_sequence, train_stacked_coords)
val_X, val_y, val_tids = generate_dataset(val_sequence, val_stacked_coords)

# %% [code] {"execution":{"iopub.status.busy":"2025-04-06T07:22:32.916661Z","iopub.execute_input":"2025-04-06T07:22:32.916847Z","iopub.status.idle":"2025-04-06T07:22:33.232524Z","shell.execute_reply.started":"2025-04-06T07:22:32.916830Z","shell.execute_reply":"2025-04-06T07:22:33.231831Z"}}
# Pad sequences and coordinates
max_len = max(len(seq) for seq in train_X)
train_X_pad = pad_sequences(train_X, maxlen=max_len, padding='post', value=0)
val_X_pad = pad_sequences(val_X, maxlen=max_len, padding='post', value=0)
test_X = test_sequence['encoded_seq'].tolist()
test_X_pad = pad_sequences(test_X, maxlen=max_len, padding='post', value=0)

def pad_coords(coords, max_len):
    L = coords.shape[0]
    if L < max_len:
        pad_width = ((0, max_len-L), (0, 0), (0, 0))
        return np.pad(coords, pad_width, mode='constant', constant_values=0)
    else:
        return coords

train_y_pad = np.array([pad_coords(y, max_len) for y in train_y])
val_y_pad = np.array([pad_coords(y, max_len) for y in val_y])

# %% [code] {"execution":{"iopub.status.busy":"2025-04-06T07:22:33.233203Z","iopub.execute_input":"2025-04-06T07:22:33.233415Z","iopub.status.idle":"2025-04-06T07:22:33.238975Z","shell.execute_reply.started":"2025-04-06T07:22:33.233396Z","shell.execute_reply":"2025-04-06T07:22:33.238164Z"}}
# CNN Model
def build_cnn_model(max_len):
    input_seq = Input(shape=(max_len,), name='input_seq')
    x = Embedding(input_dim=5, output_dim=16, mask_zero=False, name='embedding')(input_seq)
    x = Conv1D(filters=64, kernel_size=3, padding='same', activation='relu', name='conv1')(x)
    x = BatchNormalization(name='norm1')(x)
    x = Dropout(0.2, name='drop1')(x)
    x = Conv1D(filters=64, kernel_size=3, padding='same', activation='relu', name='conv2')(x)
    x = BatchNormalization(name='norm2')(x)
    x = Dropout(0.2, name='drop2')(x)
    # Output 15 values per residue (5 sets of x, y, z coordinates)
    x = Conv1D(filters=15, kernel_size=1, padding='same', activation='linear', name='predicted_coords')(x)
    model = Model(inputs=input_seq, outputs=x)
    model.compile(optimizer='adam', loss='mae')
    return model

# %% [code] {"execution":{"iopub.status.busy":"2025-04-06T07:22:33.239739Z","iopub.execute_input":"2025-04-06T07:22:33.240015Z","iopub.status.idle":"2025-04-06T07:22:33.252809Z","shell.execute_reply.started":"2025-04-06T07:22:33.239995Z","shell.execute_reply":"2025-04-06T07:22:33.252151Z"}}
# BiLSTM Model
def build_bilstm_model(max_len):
    input_seq = Input(shape=(max_len,), name='input_seq')
    x = Embedding(input_dim=5, output_dim=16, mask_zero=False, name='embedding')(input_seq)
    x = Bidirectional(LSTM(64, return_sequences=True, use_cudnn=False))(x)
    x = Dropout(0.2)(x)
    x = Dense(64, activation='relu')(x)
    x = Dropout(0.2)(x)
    # Output 15 values per residue (5 sets of x, y, z coordinates)
    x = Dense(15, activation='linear')(x)
    model = Model(inputs=input_seq, outputs=x)
    model.compile(optimizer='adam', loss='mae')
    return model

# %% [code] {"execution":{"iopub.status.busy":"2025-04-06T07:22:33.253599Z","iopub.execute_input":"2025-04-06T07:22:33.253844Z","iopub.status.idle":"2025-04-06T07:31:17.899691Z","shell.execute_reply.started":"2025-04-06T07:22:33.253816Z","shell.execute_reply":"2025-04-06T07:31:17.899001Z"}}
# Train CNN
cnn_model = build_cnn_model(max_len)
cnn_history = cnn_model.fit(
    train_X_pad, train_y_pad.reshape(train_y_pad.shape[0], train_y_pad.shape[1], -1),
    validation_data=(val_X_pad, val_y_pad.reshape(val_y_pad.shape[0], val_y_pad.shape[1], -1)),
    epochs=50, batch_size=16, verbose=1
)

# Train BiLSTM
bilstm_model = build_bilstm_model(max_len)
bilstm_history = bilstm_model.fit(
    train_X_pad, train_y_pad.reshape(train_y_pad.shape[0], train_y_pad.shape[1], -1),
    validation_data=(val_X_pad, val_y_pad.reshape(val_y_pad.shape[0], val_y_pad.shape[1], -1)),
    epochs=50, batch_size=16, verbose=1
)

# %% [code] {"execution":{"iopub.status.busy":"2025-04-06T07:31:17.900810Z","iopub.execute_input":"2025-04-06T07:31:17.901134Z","iopub.status.idle":"2025-04-06T07:31:18.504967Z","shell.execute_reply.started":"2025-04-06T07:31:17.901097Z","shell.execute_reply":"2025-04-06T07:31:18.504261Z"}}
def evaluate_model(model, X, y):
    preds = model.predict(X)
    preds = preds.reshape(preds.shape[0], preds.shape[1], 5, 3)  # Reshape to [batch_size, seq_len, 5, 3]
    rmse = np.sqrt(mean_squared_error(y.reshape(-1), preds.reshape(-1)))
    return rmse

cnn_rmse = evaluate_model(cnn_model, val_X_pad, val_y_pad)
bilstm_rmse = evaluate_model(bilstm_model, val_X_pad, val_y_pad)
print(f"CNN Validation RMSE: {cnn_rmse}")
print(f"BiLSTM Validation RMSE: {bilstm_rmse}")

# %% [code] {"execution":{"iopub.status.busy":"2025-04-06T07:31:18.505881Z","iopub.execute_input":"2025-04-06T07:31:18.506252Z","iopub.status.idle":"2025-04-06T07:31:18.708642Z","shell.execute_reply.started":"2025-04-06T07:31:18.506217Z","shell.execute_reply":"2025-04-06T07:31:18.707915Z"}}
cnn_preds = cnn_model.predict(test_X_pad)
bilstm_preds = bilstm_model.predict(test_X_pad)

# Ensemble predictions (weighted average)
ensemble_preds = 0.7 * cnn_preds + 0.3 * bilstm_preds
ensemble_preds = ensemble_preds.reshape(ensemble_preds.shape[0], ensemble_preds.shape[1], 5, 3)  

# %% [code] {"execution":{"iopub.status.busy":"2025-04-06T07:31:18.709433Z","iopub.execute_input":"2025-04-06T07:31:18.709677Z","iopub.status.idle":"2025-04-06T07:31:18.785943Z","shell.execute_reply.started":"2025-04-06T07:31:18.709656Z","shell.execute_reply":"2025-04-06T07:31:18.785200Z"}}
submission_rows = []
for idx, row in test_sequence.iterrows():
    target_id = row['target_id']
    coords = ensemble_preds[idx]  # Shape: [sequence_length, 5, 3]
    seq_length = len(row['encoded_seq'])
    coords = coords[:seq_length, :, :]  # Trim to the actual sequence length
    for i in range(seq_length):
        x_coords = coords[i, :, 0]  # x_1, x_2, x_3, x_4, x_5
        y_coords = coords[i, :, 1]  # y_1, y_2, y_3, y_4, y_5
        z_coords = coords[i, :, 2]  # z_1, z_2, z_3, z_4, z_5
        submission_rows.append({
            'ID': f"{target_id}_{i+1}",
            'resname': row['sequence'][i],
            'resid': i+1,
            'x_1': x_coords[0], 'x_2': x_coords[1], 'x_3': x_coords[2], 'x_4': x_coords[3], 'x_5': x_coords[4],
            'y_1': y_coords[0], 'y_2': y_coords[1], 'y_3': y_coords[2], 'y_4': y_coords[3], 'y_5': y_coords[4],
            'z_1': z_coords[0], 'z_2': z_coords[1], 'z_3': z_coords[2], 'z_4': z_coords[3], 'z_5': z_coords[4]})

submission = pd.DataFrame(submission_rows)
submission.to_csv("submission.csv", index=False)
print("Submission file created: submission.csv")