{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":59094,"databundleVersionId":7010844,"sourceType":"competition"}],"dockerImageVersionId":30762,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-08-30T06:18:57.102550Z","iopub.execute_input":"2024-08-30T06:18:57.103312Z","iopub.status.idle":"2024-08-30T06:18:57.110496Z","shell.execute_reply.started":"2024-08-30T06:18:57.103270Z","shell.execute_reply":"2024-08-30T06:18:57.109573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Processing","metadata":{}},{"cell_type":"code","source":"!pip install rdkit","metadata":{"execution":{"iopub.status.busy":"2024-08-30T06:18:57.114376Z","iopub.execute_input":"2024-08-30T06:18:57.114761Z","iopub.status.idle":"2024-08-30T06:19:09.781873Z","shell.execute_reply.started":"2024-08-30T06:18:57.114715Z","shell.execute_reply":"2024-08-30T06:19:09.780711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_parquet('../input/open-problems-single-cell-perturbations/de_train.parquet')\ndf.tail()","metadata":{"execution":{"iopub.status.busy":"2024-08-30T06:19:09.784032Z","iopub.execute_input":"2024-08-30T06:19:09.784385Z","iopub.status.idle":"2024-08-30T06:19:10.782541Z","shell.execute_reply.started":"2024-08-30T06:19:09.784348Z","shell.execute_reply":"2024-08-30T06:19:10.781513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2024-08-30T06:19:10.783581Z","iopub.execute_input":"2024-08-30T06:19:10.783894Z","iopub.status.idle":"2024-08-30T06:19:10.790312Z","shell.execute_reply.started":"2024-08-30T06:19:10.783861Z","shell.execute_reply":"2024-08-30T06:19:10.789482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Loader","metadata":{}},{"cell_type":"markdown","source":"# Building Experts","metadata":{}},{"cell_type":"markdown","source":"# final trial","metadata":{}},{"cell_type":"code","source":"feature_cols =['cell_type','sm_name']\ntarget_cols = ['cell_type','sm_name','sm_lincs_id','SMILES','control']\ntargets = df.drop(columns=target_cols)\nfeatures = pd.DataFrame(df,columns=feature_cols)\nfeatures","metadata":{"execution":{"iopub.status.busy":"2024-08-30T06:19:10.792530Z","iopub.execute_input":"2024-08-30T06:19:10.792904Z","iopub.status.idle":"2024-08-30T06:19:10.839387Z","shell.execute_reply.started":"2024-08-30T06:19:10.792871Z","shell.execute_reply":"2024-08-30T06:19:10.838437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from rdkit import Chem\nfrom rdkit.Chem import AllChem\nimport numpy as np\nfrom rdkit import DataStructs\nfrom sklearn.decomposition import PCA\nfrom sklearn.preprocessing import OneHotEncoder, StandardScaler\n\ndef extract_morgan_fingerprint(smiles, radius=2, nBits=2048):\n    if smiles is None or smiles == '':\n        return None\n    \n    mol = Chem.MolFromSmiles(smiles)\n    if mol is None:\n        return None\n    \n    # Generate Morgan fingerprint using MorganGenerator\n    morgan_gen = AllChem.GetMorganGenerator(radius=radius, fpSize=nBits)\n    fp = morgan_gen.GetFingerprint(mol)\n    \n    # Convert to numpy array\n    fp_array = np.zeros((nBits,))\n    DataStructs.ConvertToNumpyArray(fp, fp_array)\n    \n    # Convert numpy array to list\n    fp_list = fp_array.tolist()\n    \n    return fp_list\n\ndef reduce_fingerprint_dimensionality(fingerprints, n_components=64):\n    pca = PCA(n_components=n_components)\n    reduced_fingerprints = pca.fit_transform(fingerprints)\n    return reduced_fingerprints, pca\n\ndef process_data(df, categorical_features, n_components=64):\n    # Extract and reduce Morgan fingerprints\n    morgan_fp_list = df['SMILES'].apply(extract_morgan_fingerprint)\n    morgan_fp_array = np.stack(morgan_fp_list)\n    reduced_fingerprints, pca = reduce_fingerprint_dimensionality(morgan_fp_array, n_components)\n    \n    # Create a DataFrame with reduced fingerprints\n    reduced_fp_df = pd.DataFrame(reduced_fingerprints, columns=[f'MorganFP_PCA_{i}' for i in range(n_components)])\n    \n    # Normalize the reduced fingerprints\n    scaler = StandardScaler()\n    reduced_fp_norm = pd.DataFrame(scaler.fit_transform(reduced_fp_df), \n                                   columns=reduced_fp_df.columns, \n                                   index=reduced_fp_df.index)\n    \n    # One-hot encode categorical features\n    df_categorical = df[categorical_features]\n    encoder = OneHotEncoder(sparse=False)\n    encoded_features = encoder.fit_transform(df_categorical)\n    encoded_df = pd.DataFrame(encoded_features, \n                              columns=encoder.get_feature_names_out(categorical_features),\n                              index=df.index)\n    \n    # Combine normalized fingerprints and encoded categorical features\n    final_df = pd.concat([reduced_fp_norm, encoded_df], axis=1)\n    \n    # Add any other numerical features if needed\n    # numerical_features = ['feature1', 'feature2', ...]  # Add your numerical feature names here\n    # final_df = pd.concat([final_df, df[numerical_features]], axis=1)\n    \n    print(f\"Shape of final dataframe: {final_df.shape}\")\n    return final_df, pca, scaler, encoder","metadata":{"execution":{"iopub.status.busy":"2024-08-30T06:19:10.840623Z","iopub.execute_input":"2024-08-30T06:19:10.840984Z","iopub.status.idle":"2024-08-30T06:19:10.853511Z","shell.execute_reply.started":"2024-08-30T06:19:10.840950Z","shell.execute_reply":"2024-08-30T06:19:10.852496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_features = ['cell_type', 'sm_name']\n    \n# Process data\nfinal_df, pca, scaler, encoder = process_data(df, categorical_features, n_components=64)","metadata":{"execution":{"iopub.status.busy":"2024-08-30T06:19:10.854582Z","iopub.execute_input":"2024-08-30T06:19:10.854889Z","iopub.status.idle":"2024-08-30T06:19:11.625341Z","shell.execute_reply.started":"2024-08-30T06:19:10.854858Z","shell.execute_reply":"2024-08-30T06:19:11.624529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df.shape, df.shape","metadata":{"execution":{"iopub.status.busy":"2024-08-30T06:19:11.630368Z","iopub.execute_input":"2024-08-30T06:19:11.633537Z","iopub.status.idle":"2024-08-30T06:19:11.644200Z","shell.execute_reply.started":"2024-08-30T06:19:11.633469Z","shell.execute_reply":"2024-08-30T06:19:11.643407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df.columns","metadata":{"execution":{"iopub.status.busy":"2024-08-30T06:19:11.649695Z","iopub.execute_input":"2024-08-30T06:19:11.652597Z","iopub.status.idle":"2024-08-30T06:19:11.658887Z","shell.execute_reply.started":"2024-08-30T06:19:11.652564Z","shell.execute_reply":"2024-08-30T06:19:11.658161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prepare data for PyTorch\nimport torch\nX = torch.tensor(final_df.values, dtype=torch.float32)\ny = torch.tensor(targets.values, dtype=torch.float32)","metadata":{"execution":{"iopub.status.busy":"2024-08-30T06:19:11.660306Z","iopub.execute_input":"2024-08-30T06:19:11.661008Z","iopub.status.idle":"2024-08-30T06:19:11.685965Z","shell.execute_reply.started":"2024-08-30T06:19:11.660964Z","shell.execute_reply":"2024-08-30T06:19:11.685050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df.shape, targets.shape","metadata":{"execution":{"iopub.status.busy":"2024-08-30T06:19:11.690017Z","iopub.execute_input":"2024-08-30T06:19:11.690320Z","iopub.status.idle":"2024-08-30T06:19:11.696473Z","shell.execute_reply.started":"2024-08-30T06:19:11.690289Z","shell.execute_reply":"2024-08-30T06:19:11.695509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.utils.data import DataLoader, Dataset, TensorDataset\n# class dataset(Dataset):\n#     def __init__(self,X_train,y_train):\n#         self.X_train=X_train\n#         self.y_train=y_train\n#     def __len__(self):\n#         return len(self.X_train)\n#     def __getitem__(self,idx):\n#         return self.X_train[idx], self.y_train[idx]\n    # Create DataLoader\ndataset = TensorDataset(X, y)\ntrain_size = int(0.8 * len(dataset))\nval_size = len(dataset) - train_size\ntrain_dataset, val_dataset = torch.utils.data.random_split(dataset, [train_size, val_size])\ntrain_loader = DataLoader(train_dataset, batch_size=32, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=32)","metadata":{"execution":{"iopub.status.busy":"2024-08-30T06:19:11.697655Z","iopub.execute_input":"2024-08-30T06:19:11.697982Z","iopub.status.idle":"2024-08-30T06:19:11.705765Z","shell.execute_reply.started":"2024-08-30T06:19:11.697947Z","shell.execute_reply":"2024-08-30T06:19:11.704871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom sklearn.metrics import mean_absolute_error\n\nclass Expert(nn.Module):\n    def __init__(self, input_size, output_size, hidden_size=500, dropout_rate=0.3):\n        super(Expert, self).__init__()\n        self.fc0 = nn.Linear(input_size, hidden_size)\n        self.dropout = nn.Dropout(dropout_rate)\n        self.fc1 = nn.Linear(hidden_size, output_size)\n    \n    def forward(self, x):\n        h1 = F.relu(self.fc0(x))\n        h1 = self.dropout(h1)\n        return self.fc1(h1)\n\nclass GatingNetwork(nn.Module):\n    def __init__(self, input_size, num_experts):\n        super(GatingNetwork, self).__init__()\n        self.linear1 = nn.Linear(input_size, 64)\n        self.linear2 = nn.Linear(64, num_experts)\n    \n    def forward(self, x):\n        x = F.relu(self.linear1(x))\n        return F.softmax(self.linear2(x), dim=-1)\n\nclass MixtureOfExperts(nn.Module):\n    def __init__(self, input_size, output_size, num_experts=5):\n        super(MixtureOfExperts, self).__init__()\n        self.experts = nn.ModuleList([Expert(input_size, output_size) for _ in range(num_experts)])\n        self.gating_network = GatingNetwork(input_size, num_experts)\n    \n    def forward(self, x):\n        expert_outputs = torch.stack([expert(x) for expert in self.experts], dim=1)\n        gating_weights = self.gating_network(x).unsqueeze(-1)\n        return torch.sum(expert_outputs * gating_weights, dim=1) \n\n# RMSE_rowwise_loss function\n\ndef RMSE_rowwise_loss(y_pred, y_true):\n    return torch.sqrt(torch.mean((y_pred - y_true)**2, dim=1)).mean()\n\nclass EarlyStopping:\n    def __init__(self, patience=10, min_delta=0):\n        self.patience = patience\n        self.min_delta = min_delta\n        self.best_loss = None\n        self.counter = 0\n\n    def __call__(self, val_loss):\n        if self.best_loss is None:\n            self.best_loss = val_loss\n        elif val_loss > self.best_loss - self.min_delta:\n            self.counter += 1\n            if self.counter >= self.patience:\n                return True\n        else:\n            self.best_loss = val_loss\n            self.counter = 0\n        return False","metadata":{"execution":{"iopub.status.busy":"2024-08-30T06:19:11.706884Z","iopub.execute_input":"2024-08-30T06:19:11.707361Z","iopub.status.idle":"2024-08-30T06:19:11.725884Z","shell.execute_reply.started":"2024-08-30T06:19:11.707324Z","shell.execute_reply":"2024-08-30T06:19:11.724984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch.optim as optim\nimport matplotlib.pyplot as plt\n\ndef train_model(model, train_loader, val_loader, epochs=200, lr=0.001):\n    optimizer = optim.Adam(model.parameters(), lr=lr, weight_decay=1e-5)\n    scheduler = torch.optim.lr_scheduler.StepLR(optimizer, step_size=10, gamma=0.1)\n    early_stopping = EarlyStopping(patience=10)\n    train_losses = []\n    val_losses = []\n    \n    for epoch in range(epochs):\n        model.train()\n        train_loss = 0\n        for batch, (X, y) in enumerate(train_loader):\n            optimizer.zero_grad()\n            output = model(X)\n            loss = RMSE_rowwise_loss(output, y)\n            loss.backward()\n            optimizer.step()\n            train_loss += loss.item()\n        \n        train_losses.append(train_loss / len(train_loader))\n        \n        model.eval()\n        val_loss = 0\n        with torch.no_grad():\n            for X, y in val_loader:\n                output = model(X)\n                val_loss += RMSE_rowwise_loss(output, y).item()\n        \n        val_losses.append(val_loss / len(val_loader))\n        \n        if epoch % 10 == 0:\n            print(f\"Epoch {epoch}: Train Loss: {train_losses[-1]:.4f}, Val Loss: {val_losses[-1]:.4f}\")\n        \n        # Step the learning rate scheduler at the end of each epoch\n        scheduler.step()\n        \n        # Check early stopping condition\n        if early_stopping(val_loss / len(val_loader)):\n            print(\"Early stopping triggered\")\n            break\n\n    \n    # Plot the losses\n    plt.figure(figsize=(10, 5))\n    plt.plot(train_losses, label='Train Loss')\n    plt.plot(val_losses, label='Val Loss')\n    plt.xlabel('Epochs')\n    plt.ylabel('Loss')\n    plt.legend()\n    plt.title('Training and Validation Losses')\n    plt.show()\n\n    return train_losses, val_losses","metadata":{"execution":{"iopub.status.busy":"2024-08-30T06:19:11.727269Z","iopub.execute_input":"2024-08-30T06:19:11.727576Z","iopub.status.idle":"2024-08-30T06:19:11.741368Z","shell.execute_reply.started":"2024-08-30T06:19:11.727543Z","shell.execute_reply":"2024-08-30T06:19:11.740564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize and train the model\ninput_size = X.shape[1]  # Updated input size including reduced Morgan fingerprints and one-hot encoded categories\noutput_size = y.shape[1]  # Your original output size\nnum_experts = 5\n\nprint(f\"Model input size: {input_size}\")\nprint(f\"Model output size: {output_size}\")\n\nmodel = MixtureOfExperts(input_size, output_size, num_experts)\ntrain_losses, val_losses = train_model(model, train_loader, val_loader)","metadata":{"execution":{"iopub.status.busy":"2024-08-30T06:19:11.742326Z","iopub.execute_input":"2024-08-30T06:19:11.742627Z","iopub.status.idle":"2024-08-30T06:21:28.823782Z","shell.execute_reply.started":"2024-08-30T06:19:11.742594Z","shell.execute_reply":"2024-08-30T06:21:28.822873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map = pd.read_csv(\"/kaggle/input/open-problems-single-cell-perturbations/id_map.csv\")\nid_map","metadata":{"execution":{"iopub.status.busy":"2024-08-30T06:23:33.852512Z","iopub.execute_input":"2024-08-30T06:23:33.853570Z","iopub.status.idle":"2024-08-30T06:23:33.871077Z","shell.execute_reply.started":"2024-08-30T06:23:33.853500Z","shell.execute_reply":"2024-08-30T06:23:33.869866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2024-08-30T06:23:36.373456Z","iopub.execute_input":"2024-08-30T06:23:36.373826Z","iopub.status.idle":"2024-08-30T06:23:36.402699Z","shell.execute_reply.started":"2024-08-30T06:23:36.373792Z","shell.execute_reply":"2024-08-30T06:23:36.401732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"smiles_mapping = df[['sm_name', 'SMILES']].drop_duplicates()\nsmiles_mapping","metadata":{"execution":{"iopub.status.busy":"2024-08-30T06:23:39.262725Z","iopub.execute_input":"2024-08-30T06:23:39.263077Z","iopub.status.idle":"2024-08-30T06:23:39.275794Z","shell.execute_reply.started":"2024-08-30T06:23:39.263044Z","shell.execute_reply":"2024-08-30T06:23:39.274876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testdata = pd.DataFrame(id_map,columns=feature_cols) \ntestdata = testdata.merge(smiles_mapping, on='sm_name', how='left')\ntestdata.head()","metadata":{"execution":{"iopub.status.busy":"2024-08-30T06:23:42.113137Z","iopub.execute_input":"2024-08-30T06:23:42.113532Z","iopub.status.idle":"2024-08-30T06:23:42.127530Z","shell.execute_reply.started":"2024-08-30T06:23:42.113491Z","shell.execute_reply":"2024-08-30T06:23:42.126600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_fingerprint_dimensionality(fingerprints, pca):\n    reduced_fingerprints = pca.transform(fingerprints)\n    return reduced_fingerprints\n\ndef process_test_data(df, categorical_features, pca, scaler, encoder):\n    # Extract Morgan fingerprints\n    morgan_fp_list = df['SMILES'].apply(extract_morgan_fingerprint)\n    morgan_fp_array = np.stack(morgan_fp_list)\n    \n    # Reduce dimensionality using pre-fitted PCA\n    reduced_fingerprints = reduce_fingerprint_dimensionality(morgan_fp_array, pca)\n    \n    # Create a DataFrame with reduced fingerprints\n    reduced_fp_df = pd.DataFrame(reduced_fingerprints, columns=[f'MorganFP_PCA_{i}' for i in range(pca.n_components_)])\n    \n    # Normalize using the pre-fitted scaler\n    reduced_fp_norm = pd.DataFrame(scaler.transform(reduced_fp_df), \n                                   columns=reduced_fp_df.columns, \n                                   index=reduced_fp_df.index)\n    \n    # One-hot encode using the pre-fitted encoder\n    df_categorical = df[categorical_features]\n    encoded_features = encoder.transform(df_categorical)\n    encoded_df = pd.DataFrame(encoded_features, \n                              columns=encoder.get_feature_names_out(categorical_features),\n                              index=df.index)\n    \n    # Combine normalized fingerprints and encoded categorical features\n    final_df = pd.concat([reduced_fp_norm, encoded_df], axis=1)\n    \n    # Add any other numerical features if needed\n    # numerical_features = ['feature1', 'feature2', ...]  # Add your numerical feature names here\n    # final_df = pd.concat([final_df, df[numerical_features]], axis=1)\n    \n    print(f\"Shape of final dataframe: {final_df.shape}\")\n    return final_df","metadata":{"execution":{"iopub.status.busy":"2024-08-30T06:25:08.419567Z","iopub.execute_input":"2024-08-30T06:25:08.420439Z","iopub.status.idle":"2024-08-30T06:25:08.428859Z","shell.execute_reply.started":"2024-08-30T06:25:08.420392Z","shell.execute_reply":"2024-08-30T06:25:08.427849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assume df_test is your testing dataset\n\n# Process the testing data using the fitted pca, scaler, and encoder from training\ntest_final_df = process_test_data(testdata, categorical_features, pca, scaler, encoder)\n\n# Convert the processed test DataFrame to a tensor for model testing\ntest_X = torch.tensor(test_final_df.values, dtype=torch.float32)\n\n# Load the model and set it to evaluation mode\n# model = MixtureOfExperts(input_size, output_size, num_experts)\n# model.load_state_dict(torch.load('mixture_of_experts_model.pth'))\nmodel.eval()\n\n# Test the model\nwith torch.no_grad():\n    test_output = model(test_X)\n    print(test_output)","metadata":{"execution":{"iopub.status.busy":"2024-08-30T06:25:12.371294Z","iopub.execute_input":"2024-08-30T06:25:12.372144Z","iopub.status.idle":"2024-08-30T06:25:12.889517Z","shell.execute_reply.started":"2024-08-30T06:25:12.372102Z","shell.execute_reply":"2024-08-30T06:25:12.888508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv(\"/kaggle/input/open-problems-single-cell-perturbations/sample_submission.csv\")\nsample_submission.columns","metadata":{"execution":{"iopub.status.busy":"2024-08-30T06:25:27.772586Z","iopub.execute_input":"2024-08-30T06:25:27.773335Z","iopub.status.idle":"2024-08-30T06:25:30.358347Z","shell.execute_reply.started":"2024-08-30T06:25:27.773295Z","shell.execute_reply":"2024-08-30T06:25:30.357452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_columns = sample_submission.columns\nsample_columns= sample_columns[1:]\nsubmission_df = pd.DataFrame(test_output.detach().numpy(), columns=sample_columns)","metadata":{"execution":{"iopub.status.busy":"2024-08-30T06:25:51.995303Z","iopub.execute_input":"2024-08-30T06:25:51.995721Z","iopub.status.idle":"2024-08-30T06:25:52.001834Z","shell.execute_reply.started":"2024-08-30T06:25:51.995657Z","shell.execute_reply":"2024-08-30T06:25:52.000709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.insert(0, 'id', range(255))","metadata":{"execution":{"iopub.status.busy":"2024-08-30T06:25:55.219703Z","iopub.execute_input":"2024-08-30T06:25:55.220399Z","iopub.status.idle":"2024-08-30T06:25:55.228400Z","shell.execute_reply.started":"2024-08-30T06:25:55.220358Z","shell.execute_reply":"2024-08-30T06:25:55.227456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission","metadata":{"execution":{"iopub.status.busy":"2024-08-30T06:25:59.819172Z","iopub.execute_input":"2024-08-30T06:25:59.820066Z","iopub.status.idle":"2024-08-30T06:25:59.863569Z","shell.execute_reply.started":"2024-08-30T06:25:59.820022Z","shell.execute_reply":"2024-08-30T06:25:59.862468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df","metadata":{"execution":{"iopub.status.busy":"2024-08-30T06:26:03.291494Z","iopub.execute_input":"2024-08-30T06:26:03.292304Z","iopub.status.idle":"2024-08-30T06:26:03.322222Z","shell.execute_reply.started":"2024-08-30T06:26:03.292262Z","shell.execute_reply":"2024-08-30T06:26:03.321255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-08-30T06:26:08.266774Z","iopub.execute_input":"2024-08-30T06:26:08.267489Z","iopub.status.idle":"2024-08-30T06:26:16.520245Z","shell.execute_reply.started":"2024-08-30T06:26:08.267446Z","shell.execute_reply":"2024-08-30T06:26:16.519439Z"},"trusted":true},"execution_count":null,"outputs":[]}]}