{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":87793,"databundleVersionId":11553390,"isSourceIdPinned":false,"sourceType":"competition"}],"dockerImageVersionId":30919,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom sklearn.preprocessing import MinMaxScaler  # Using MinMaxScaler instead\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Set random seeds for reproducibility\nnp.random.seed(42)\ntf.random.set_seed(42)\n\n# Load data\ndef load_data():\n    train_sequences = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/train_sequences.csv')\n    train_labels = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/train_labels.csv')\n    test_sequences = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/test_sequences.csv')\n    return train_sequences, train_labels, test_sequences\n\n# Preprocessing function with fixed sequence length and scaled data\ndef preprocess_data(train_sequences, train_labels, test_sequences, fixed_seq_length=50):\n    # One-hot encoding mapping\n    nucleotides = {'A': [1,0,0,0], 'C': [0,1,0,0], 'G': [0,0,1,0], 'U': [0,0,0,1]}\n    default_nuc = [0.25, 0.25, 0.25, 0.25]  # For non-standard nucleotides\n    \n    # Process training data\n    X_train = []\n    y_train = []\n    valid_counts = 0\n    \n    for idx, row in train_sequences.iterrows():\n        target_id = row['target_id']\n        seq = row['sequence']\n        \n        # Skip sequences longer than our fixed length to avoid truncation issues\n        if len(seq) > fixed_seq_length * 2:\n            continue\n            \n        # Get labels for this sequence\n        target_labels = train_labels[train_labels['ID'].str.startswith(target_id + '_')]\n        \n        if len(target_labels) > 0:\n            # Check if target_labels has valid numeric data\n            has_valid_coordinates = True\n            for _, label_row in target_labels.iterrows():\n                if (pd.isna(label_row['x_1']) or pd.isna(label_row['y_1']) or pd.isna(label_row['z_1'])):\n                    has_valid_coordinates = False\n                    break\n            \n            if not has_valid_coordinates:\n                continue\n                \n            # Create fixed-length sequence representation\n            seq_encoded = np.zeros((fixed_seq_length, 4))\n            for i in range(min(len(seq), fixed_seq_length)):\n                seq_encoded[i] = nucleotides.get(seq[i], default_nuc)\n            \n            # Create fixed-length coordinate array\n            coords = np.zeros((fixed_seq_length, 3))\n            for _, label_row in target_labels.iterrows():\n                resid = label_row['resid']\n                if 1 <= resid <= fixed_seq_length:\n                    coords[resid-1] = [\n                        float(label_row['x_1']), \n                        float(label_row['y_1']), \n                        float(label_row['z_1'])\n                    ]\n            \n            # Skip if all coordinates are zero (would cause training issues)\n            if np.all(coords == 0):\n                continue\n                \n            X_train.append(seq_encoded)\n            y_train.append(coords)\n            valid_counts += 1\n    \n    print(f\"Using {valid_counts} valid training examples\")\n    \n    # Convert to numpy arrays\n    X_train = np.array(X_train)\n    y_train = np.array(y_train)\n    \n    # Use MinMaxScaler instead of StandardScaler for better numerical stability\n    scaler = MinMaxScaler(feature_range=(-1, 1))  # Range from -1 to 1\n    y_train_reshaped = y_train.reshape(-1, 3)\n    y_train_normalized = scaler.fit_transform(y_train_reshaped)\n    y_train = y_train_normalized.reshape(y_train.shape)\n    \n    # Check for any remaining NaN values and replace them\n    X_train = np.nan_to_num(X_train)\n    y_train = np.nan_to_num(y_train)\n    \n    # Process test data\n    X_test = []\n    test_ids = []\n    test_seq_lengths = []\n    \n    for idx, row in test_sequences.iterrows():\n        target_id = row['target_id']\n        seq = row['sequence']\n        test_seq_lengths.append(len(seq))\n        \n        # Create fixed-length sequence representation\n        seq_encoded = np.zeros((fixed_seq_length, 4))\n        for i in range(min(len(seq), fixed_seq_length)):\n            seq_encoded[i] = nucleotides.get(seq[i], default_nuc)\n        \n        X_test.append(seq_encoded)\n        test_ids.append(target_id)\n    \n    X_test = np.array(X_test)\n    \n    return X_train, y_train, X_test, test_ids, test_seq_lengths, scaler\n\n# Extremely simple model to avoid NaN issues\ndef build_simple_model(seq_length):\n    model = models.Sequential([\n        layers.InputLayer(input_shape=(seq_length, 4)),\n        layers.Flatten(),\n        layers.Dense(128, activation='relu', \n                    kernel_initializer='he_normal',\n                    kernel_regularizer=tf.keras.regularizers.l2(0.001)),\n        layers.Dense(256, activation='relu',\n                    kernel_initializer='he_normal',\n                    kernel_regularizer=tf.keras.regularizers.l2(0.001)),\n        layers.Dense(seq_length * 3),\n        layers.Reshape((seq_length, 3))\n    ])\n    \n    # Use a more robust optimizer with gradient clipping\n    optimizer = tf.keras.optimizers.Adam(\n        learning_rate=0.0001,  # Reduced learning rate\n        clipnorm=1.0  # Gradient clipping\n    )\n    \n    model.compile(optimizer=optimizer, loss='mse')\n    return model\n\n# Generate 5 different predictions\ndef generate_predictions(model, X_test, scaler, test_seq_lengths):\n    # Base prediction\n    base_pred = model.predict(X_test)\n    \n    # Create 5 different predictions\n    all_preds = []\n    \n    # First is the base prediction\n    all_preds.append(base_pred)\n    \n    # Add 4 variations with small noise\n    for i in range(4):\n        noise = np.random.normal(0, 0.02 * (i+1), base_pred.shape)\n        noisy_pred = base_pred + noise\n        all_preds.append(noisy_pred)\n    \n    # Denormalize\n    all_denorm = []\n    for pred in all_preds:\n        pred_flat = pred.reshape(-1, 3)\n        denorm_flat = scaler.inverse_transform(pred_flat)\n        denorm = denorm_flat.reshape(pred.shape)\n        all_denorm.append(denorm)\n    \n    return all_denorm\n\n# Create submission file\ndef create_submission(predictions, test_sequences, test_seq_lengths):\n    sample_submission = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/sample_submission.csv')\n    submission_rows = []\n    \n    for idx, target_id in enumerate(test_sequences['target_id']):\n        seq = test_sequences.loc[test_sequences['target_id'] == target_id, 'sequence'].values[0]\n        seq_length = len(seq)\n        \n        for i in range(seq_length):\n            row = {\n                'ID': f\"{target_id}_{i+1}\",\n                'resname': seq[i],\n                'resid': i+1\n            }\n            \n            # Add coordinates for all 5 predictions\n            for model_idx in range(5):\n                pred = predictions[model_idx][idx]\n                if i < len(pred):\n                    row[f'x_{model_idx+1}'] = pred[i, 0]\n                    row[f'y_{model_idx+1}'] = pred[i, 1] \n                    row[f'z_{model_idx+1}'] = pred[i, 2]\n                else:\n                    # Use last prediction for positions beyond model's sequence length\n                    row[f'x_{model_idx+1}'] = pred[-1, 0]\n                    row[f'y_{model_idx+1}'] = pred[-1, 1]\n                    row[f'z_{model_idx+1}'] = pred[-1, 2]\n            \n            submission_rows.append(row)\n    \n    # Create DataFrame with same columns as sample submission\n    submission = pd.DataFrame(submission_rows)\n    submission = submission[sample_submission.columns]\n    \n    return submission\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-30T19:50:56.456726Z","iopub.execute_input":"2025-03-30T19:50:56.457074Z","iopub.status.idle":"2025-03-30T19:50:56.488613Z","shell.execute_reply.started":"2025-03-30T19:50:56.457050Z","shell.execute_reply":"2025-03-30T19:50:56.487733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Fixed sequence length - using smaller value\nFIXED_SEQ_LENGTH = 50\n\n# 1. Load data\ntrain_sequences, train_labels, test_sequences = load_data()\n\n# 2. Preprocess with fixed length\nX_train, y_train, X_test, test_ids, test_seq_lengths, scaler = preprocess_data(\n    train_sequences, train_labels, test_sequences, FIXED_SEQ_LENGTH\n)\n\nprint(f\"Training data shape: {X_train.shape}, {y_train.shape}\")\nprint(f\"Test data shape: {X_test.shape}\")\n\n# 3. Build and train model\nmodel = build_simple_model(FIXED_SEQ_LENGTH)\nmodel.summary()\n\n# Use early stopping to prevent overfitting and detect NaN\nearly_stopping = tf.keras.callbacks.EarlyStopping(\n    monitor='loss',\n    patience=3,\n    restore_best_weights=True\n)\n\n# Reduced epochs and batch size\nhistory = model.fit(\n    X_train, y_train, \n    epochs=3, \n    batch_size=4, \n    validation_split=0.1,\n    callbacks=[early_stopping],\n    verbose=1\n)\n\n# Check if training was successful\nif np.isnan(history.history['loss'][-1]):\n    print(\"Warning: NaN loss detected. Using fallback prediction method.\")\n    # Create a fallback prediction based on average coordinates\n    avg_coords = np.mean(y_train, axis=0)\n    base_pred = np.tile(avg_coords, (len(X_test), 1, 1))\n    \n    # Manually create 5 predictions with slight variations\n    all_preds = [base_pred]\n    for i in range(4):\n        noise = np.random.normal(0, 0.05 * (i+1), base_pred.shape)\n        noisy_pred = base_pred + noise\n        all_preds.append(noisy_pred)\n        \n    # Denormalize\n    predictions = []\n    for pred in all_preds:\n        pred_flat = pred.reshape(-1, 3)\n        denorm_flat = scaler.inverse_transform(pred_flat)\n        denorm = denorm_flat.reshape(pred.shape)\n        predictions.append(denorm)\nelse:\n    # Normal prediction if training succeeded\n    predictions = generate_predictions(model, X_test, scaler, test_seq_lengths)\n\n# 5. Create submission\nsubmission = create_submission(predictions, test_sequences, test_seq_lengths)\n\n# 6. Save submission\nsubmission.to_csv('submission.csv', index=False)\nprint(\"Submission file created successfully!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-30T19:51:08.979183Z","iopub.execute_input":"2025-03-30T19:51:08.979474Z","iopub.status.idle":"2025-03-30T19:51:33.872873Z","shell.execute_reply.started":"2025-03-30T19:51:08.979451Z","shell.execute_reply":"2025-03-30T19:51:33.872042Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-30T19:51:45.482172Z","iopub.execute_input":"2025-03-30T19:51:45.482457Z","iopub.status.idle":"2025-03-30T19:51:45.501201Z","shell.execute_reply.started":"2025-03-30T19:51:45.482434Z","shell.execute_reply":"2025-03-30T19:51:45.500057Z"}},"outputs":[],"execution_count":null}]}