{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\nprint(\"--- 🔍 Scanning Data Directory to Find True Paths ---\")\n# Loop through the input directory to see exactly how Kaggle named the folder\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-02T19:07:36.451342Z","iopub.execute_input":"2026-06-02T19:07:36.451907Z","iopub.status.idle":"2026-06-02T19:07:36.458070Z","shell.execute_reply.started":"2026-06-02T19:07:36.451883Z","shell.execute_reply":"2026-06-02T19:07:36.457292Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# 1. Load the core Differential Expression dataset using the correct path\nprint(\"--- 🔬 Initializing Digital Lab: Loading Data ---\")\nde_train = pd.read_parquet('/kaggle/input/competitions/open-problems-single-cell-perturbations/de_train.parquet')\n\nprint(f\"✅ Success! Train Dataset Shape: {de_train.shape}\")\n\n# 2. Inspect the first 5 rows and the initial metadata + gene columns\nprint(\"\\n🔍 Exploring the first 5 rows (Metadata + First 3 Genes):\")\nprint(de_train.iloc[:5, :6])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-02T19:08:34.462613Z","iopub.execute_input":"2026-06-02T19:08:34.462867Z","iopub.status.idle":"2026-06-02T19:08:35.775684Z","shell.execute_reply.started":"2026-06-02T19:08:34.462849Z","shell.execute_reply":"2026-06-02T19:08:35.774949Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Separate the metadata (Features) from the gene expression values (Targets)\n# The first 5 columns are metadata, and the rest are the 18,211 genes\nmetadata_cols = ['cell_type', 'sm_name', 'sm_lincs_id', 'SMILES', 'control']\n\nX_metadata = de_train[metadata_cols]\nY_genes = de_train.drop(columns=metadata_cols)\n\nprint(\"--- ⚙️ Feature vs Target Separation ---\")\nprint(f\"X Metadata Shape (Features): {X_metadata.shape}\")\nprint(f\"Y Genes Shape (Targets to predict): {Y_genes.shape}\")\n\n# Check the distribution of cell types in our training data\nprint(\"\\n📊 Distribution of Immunological Cell Types:\")\nprint(X_metadata['cell_type'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-02T19:11:05.222982Z","iopub.execute_input":"2026-06-02T19:11:05.223213Z","iopub.status.idle":"2026-06-02T19:11:05.255311Z","shell.execute_reply.started":"2026-06-02T19:11:05.223194Z","shell.execute_reply":"2026-06-02T19:11:05.254546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import KFold\n\n# 1. One-Hot Encode the categorical features (cell_type and sm_name)\nencoder = OneHotEncoder(sparse_output=False, handle_unknown='ignore')\nX_encoded = encoder.fit_transform(de_train[['cell_type', 'sm_name']])\n\n# Convert to DataFrame just to keep it clean and traceable\nX_encoded_df = pd.DataFrame(X_encoded, columns=encoder.get_feature_names_out(['cell_type', 'sm_name']))\n\nprint(\"--- ⚙️ Feature Encoding Complete ---\")\nprint(f\"Encoded X Features Shape: {X_encoded_df.shape}\")\n\n# 2. Define the Competition Metric: Mean Rowwise Root Mean Squared Error (MRMSE)\ndef mean_rowwise_rmse(y_true, y_pred):\n    # Calculate RMSE for each row independently, then compute the overall mean\n    rowwise_rmse = np.sqrt(np.mean((y_true - y_pred) ** 2, axis=1))\n    return np.mean(rowwise_rmse)\n\n# 3. Cross-Validation Setup (3-Fold CV to validate our model's performance)\nkf = KFold(n_splits=3, shuffle=True, random_state=42)\nmrmse_scores = []\n\nprint(\"\\n🚀 Training Multi-Output Ridge Regression Baseline...\")\nfor fold, (train_idx, val_idx) in enumerate(kf.split(X_encoded_df)):\n    X_train, X_val = X_encoded_df.iloc[train_idx], X_encoded_df.iloc[val_idx]\n    y_train, y_val = Y_genes.iloc[train_idx], Y_genes.iloc[val_idx]\n    \n    # Ridge regression handles 18,211 targets simultaneously with L2 regularization\n    model = Ridge(alpha=1.0)\n    model.fit(X_train, y_train)\n    \n    # Predict and evaluate using MRMSE\n    preds = model.predict(X_val)\n    fold_score = mean_rowwise_rmse(y_val.values, preds)\n    mrmse_scores.append(fold_score)\n    print(f\"Fold {fold + 1} MRMSE Score: {fold_score:.5f}\")\n\nprint(f\"\\n🏆 Overall Baseline CV MRMSE: {np.mean(mrmse_scores):.5f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-02T19:13:33.082599Z","iopub.execute_input":"2026-06-02T19:13:33.082857Z","iopub.status.idle":"2026-06-02T19:13:34.698814Z","shell.execute_reply.started":"2026-06-02T19:13:33.082836Z","shell.execute_reply":"2026-06-02T19:13:34.695463Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.neural_network import MLPRegressor\n\nmrmse_nn_scores = []\n\nprint(\"🚀 Training Non-Linear Neural Network (MLP Regressor) Baseline...\")\n\nfor fold, (train_idx, val_idx) in enumerate(kf.split(X_encoded_df)):\n    X_train, X_val = X_encoded_df.iloc[train_idx], X_encoded_df.iloc[val_idx]\n    y_train, y_val = Y_genes.iloc[train_idx], Y_genes.iloc[val_idx]\n    \n    # Using a 2-layer Multi-Layer Perceptron to capture non-linear drug-cell features\n    # early_stopping=True ensures we don't overfit and saves computation time\n    nn_model = MLPRegressor(\n        hidden_layer_sizes=(128, 64), \n        activation='relu', \n        solver='adam', \n        max_iter=30, \n        early_stopping=True,\n        random_state=42\n    )\n    \n    nn_model.fit(X_train, y_train)\n    \n    # Predict and evaluate\n    nn_preds = nn_model.predict(X_val)\n    fold_score = mean_rowwise_rmse(y_val.values, nn_preds)\n    mrmse_nn_scores.append(fold_score)\n    print(f\"Neural Net Fold {fold + 1} MRMSE Score: {fold_score:.5f}\")\n\nprint(f\"\\n🏆 Overall Neural Network CV MRMSE: {np.mean(mrmse_nn_scores):.5f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-02T19:16:29.661697Z","iopub.execute_input":"2026-06-02T19:16:29.662082Z","iopub.status.idle":"2026-06-02T19:16:38.059967Z","shell.execute_reply.started":"2026-06-02T19:16:29.662062Z","shell.execute_reply":"2026-06-02T19:16:38.059414Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.neural_network import MLPRegressor\n\nmrmse_tuned_nn_scores = []\n\nprint(\"🚀 Training Fully Optimized Non-Linear Neural Network (MLP Regressor)...\")\n\nfor fold, (train_idx, val_idx) in enumerate(kf.split(X_encoded_df)):\n    X_train, X_val = X_encoded_df.iloc[train_idx], X_encoded_df.iloc[val_idx]\n    y_train, y_val = Y_genes.iloc[train_idx], Y_genes.iloc[val_idx]\n    \n    # Increasing max_iter to allow full convergence and tweaking the learning rate\n    tuned_nn = MLPRegressor(\n        hidden_layer_sizes=(128, 64), \n        activation='relu', \n        solver='adam', \n        batch_size=32,\n        learning_rate_init=0.005,\n        max_iter=150, \n        early_stopping=True,\n        n_iter_no_change=10, # Stops automatically if the validation score stops improving\n        random_state=42\n    )\n    \n    tuned_nn.fit(X_train, y_train)\n    \n    # Predict and evaluate\n    tuned_preds = tuned_nn.predict(X_val)\n    fold_score = mean_rowwise_rmse(y_val.values, tuned_preds)\n    mrmse_tuned_nn_scores.append(fold_score)\n    print(f\"Optimized NN Fold {fold + 1} MRMSE Score: {fold_score:.5f}\")\n\nprint(f\"\\n🏆 Overall Tuned Neural Network CV MRMSE: {np.mean(mrmse_tuned_nn_scores):.5f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-02T19:20:17.973058Z","iopub.execute_input":"2026-06-02T19:20:17.973308Z","iopub.status.idle":"2026-06-02T19:20:28.365368Z","shell.execute_reply.started":"2026-06-02T19:20:17.973289Z","shell.execute_reply":"2026-06-02T19:20:28.364925Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.neural_network import MLPRegressor\n\nmrmse_final_nn_scores = []\n\nprint(\"🚀 Training the Corrected & Fully Converged Neural Network...\")\n\nfor fold, (train_idx, val_idx) in enumerate(kf.split(X_encoded_df)):\n    X_train, X_val = X_encoded_df.iloc[train_idx], X_encoded_df.iloc[val_idx]\n    y_train, y_val = Y_genes.iloc[train_idx], Y_genes.iloc[val_idx]\n    \n    # Keeping the stable default learning rate (0.001) and batch size,\n    # but allowing more iterations (150) to fully converge without warnings.\n    final_nn = MLPRegressor(\n        hidden_layer_sizes=(128, 64), \n        activation='relu', \n        solver='adam', \n        learning_rate_init=0.001,  # Safe, stable learning rate\n        max_iter=150,             # Enough time to reach the minimum\n        early_stopping=True,       # Prevents overfitting automatically\n        n_iter_no_change=10, \n        random_state=42\n    )\n    \n    final_nn.fit(X_train, y_train)\n    \n    # Predict and evaluate using our rowwise MRMSE metric\n    final_preds = final_nn.predict(X_val)\n    fold_score = mean_rowwise_rmse(y_val.values, final_preds)\n    mrmse_final_nn_scores.append(fold_score)\n    print(f\"Stable NN Fold {fold + 1} MRMSE Score: {fold_score:.5f}\")\n\nprint(f\"\\n🏆 True Optimized Neural Network CV MRMSE: {np.mean(mrmse_final_nn_scores):.5f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-02T19:24:17.418073Z","iopub.execute_input":"2026-06-02T19:24:17.418344Z","iopub.status.idle":"2026-06-02T19:24:31.225766Z","shell.execute_reply.started":"2026-06-02T19:24:17.418325Z","shell.execute_reply":"2026-06-02T19:24:31.225227Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.neural_network import MLPRegressor\n\nprint(\"--- 🏆 Final Phase: Full Training & Submission Generation ---\")\n\n# 1. Load the Test Map provided by the competition\nid_map = pd.read_csv('/kaggle/input/competitions/open-problems-single-cell-perturbations/id_map.csv')\n\n# 2. Re-fit the encoder on the entire training dataset for maximum coverage\nencoder = OneHotEncoder(sparse_output=False, handle_unknown='ignore')\nX_train_full = encoder.fit_transform(de_train[['cell_type', 'sm_name']])\nX_test_full = encoder.transform(id_map[['cell_type', 'sm_name']])\n\nprint(f\"Full Train Shape: {X_train_full.shape}\")\nprint(f\"Full Test Shape: {X_test_full.shape}\")\n\n# 3. Train our stable Optimized Neural Network on 100% of the training data\nfinal_model = MLPRegressor(\n    hidden_layer_sizes=(128, 64), \n    activation='relu', \n    solver='adam', \n    learning_rate_init=0.001,\n    max_iter=150, \n    early_stopping=True,\n    random_state=42\n)\n\nprint(\"\\n🧠 Training the final production model...\")\nfinal_model.fit(X_train_full, Y_genes)\n\n# 4. Predict the gene expressions for the test set (18,211 genes)\nprint(\"🔮 Generating predictions for the competition test set...\")\ntest_preds = final_model.predict(X_test_full)\n\n# 5. Format the submission exactly as required by Kaggle\nsubmission = pd.DataFrame(test_preds, columns=Y_genes.columns)\nsubmission.insert(0, 'id', id_map['id'])\n\n# Save to CSV in the working directory\nsubmission.to_csv('submission.csv', index=False)\nprint(\"\\n💾 Success! 'submission.csv' generated and saved beautifully.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-02T19:26:23.641074Z","iopub.execute_input":"2026-06-02T19:26:23.641332Z","iopub.status.idle":"2026-06-02T19:26:37.047221Z","shell.execute_reply.started":"2026-06-02T19:26:23.641312Z","shell.execute_reply":"2026-06-02T19:26:37.046611Z"}},"outputs":[],"execution_count":null}]}