{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59094,"databundleVersionId":7010844,"sourceType":"competition"},{"sourceId":6922726,"sourceType":"datasetVersion","datasetId":3975067}],"dockerImageVersionId":30558,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nfrom tqdm import tqdm\n\nimport lightgbm as lgb\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split, GroupKFold\nfrom sklearn.metrics import mean_squared_error\nfrom lightgbm.callback import early_stopping, log_evaluation","metadata":{"execution":{"iopub.status.busy":"2023-12-03T22:28:20.801211Z","iopub.execute_input":"2023-12-03T22:28:20.801612Z","iopub.status.idle":"2023-12-03T22:28:20.809047Z","shell.execute_reply.started":"2023-12-03T22:28:20.801580Z","shell.execute_reply":"2023-12-03T22:28:20.808053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"POTENCY = 21","metadata":{"execution":{"iopub.status.busy":"2023-12-03T22:31:35.679941Z","iopub.execute_input":"2023-12-03T22:31:35.680455Z","iopub.status.idle":"2023-12-03T22:31:35.685711Z","shell.execute_reply.started":"2023-12-03T22:31:35.680420Z","shell.execute_reply":"2023-12-03T22:31:35.684711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_parquet('/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet')\ndf = df[df.control == 0]   ","metadata":{"execution":{"iopub.status.busy":"2023-12-03T21:20:53.764995Z","iopub.execute_input":"2023-12-03T21:20:53.765404Z","iopub.status.idle":"2023-12-03T21:20:55.129585Z","shell.execute_reply.started":"2023-12-03T21:20:53.765373Z","shell.execute_reply":"2023-12-03T21:20:55.128303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mrrmse_np(y_pred, y_true):\n    return np.sqrt(np.square(y_true - y_pred).mean(axis=1)).mean()","metadata":{"execution":{"iopub.status.busy":"2023-12-03T21:21:03.433900Z","iopub.execute_input":"2023-12-03T21:21:03.434305Z","iopub.status.idle":"2023-12-03T21:21:03.440353Z","shell.execute_reply.started":"2023-12-03T21:21:03.434275Z","shell.execute_reply":"2023-12-03T21:21:03.439023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base = '/kaggle/input/open-problems-single-cell-perturbations/'\nsub = pd.read_csv(base + 'sample_submission.csv', index_col = 0)\nid_map = pd.read_csv(base + 'id_map.csv', index_col = 0)\nsub = sub.merge(id_map, on='id')","metadata":{"execution":{"iopub.status.busy":"2023-12-03T21:21:03.677957Z","iopub.execute_input":"2023-12-03T21:21:03.678958Z","iopub.status.idle":"2023-12-03T21:21:08.406427Z","shell.execute_reply.started":"2023-12-03T21:21:03.678908Z","shell.execute_reply":"2023-12-03T21:21:08.405312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map.cell_type.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-12-03T21:21:08.408330Z","iopub.execute_input":"2023-12-03T21:21:08.408795Z","iopub.status.idle":"2023-12-03T21:21:08.418650Z","shell.execute_reply.started":"2023-12-03T21:21:08.408761Z","shell.execute_reply":"2023-12-03T21:21:08.417471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"code","source":"# Step 1: Melting the DataFrame to long format\nlong_df = df.melt(id_vars=['sm_name', 'cell_type', 'sm_lincs_id', 'SMILES', 'control'], \n                  var_name='gene', value_name='expression_value')\n\n# Step 2: Pivoting the long DataFrame to get the desired format\npivot_df = long_df.pivot_table(index=['gene', 'sm_name', 'sm_lincs_id', 'SMILES', 'control'], \n                               columns='cell_type', values='expression_value')\n\n# Resetting the index and column names for clarity\npivot_df.reset_index(inplace=True)\npivot_df.columns.name = None\npivot_df = pivot_df.drop(columns = 'T cells CD8+')\npivot_clean = pivot_df.drop(columns = ['sm_lincs_id']).dropna()\npivot_clean.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-12-03T21:21:08.419808Z","iopub.execute_input":"2023-12-03T21:21:08.420119Z","iopub.status.idle":"2023-12-03T21:21:31.366159Z","shell.execute_reply.started":"2023-12-03T21:21:08.420093Z","shell.execute_reply":"2023-12-03T21:21:31.365087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Add Donor Cell Counts","metadata":{}},{"cell_type":"code","source":"aom = pd.read_csv('/kaggle/input/open-problems-single-cell-perturbations/adata_obs_meta.csv')","metadata":{"execution":{"iopub.status.busy":"2023-12-03T21:21:31.368806Z","iopub.execute_input":"2023-12-03T21:21:31.369164Z","iopub.status.idle":"2023-12-03T21:21:32.191522Z","shell.execute_reply.started":"2023-12-03T21:21:31.369134Z","shell.execute_reply":"2023-12-03T21:21:32.190376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filtered_data = aom[~aom['cell_type'].isin(['B cells', 'Myeloid cells', 'T cells CD8+'])]\n\n# Now continue with the groupby and pivot_table steps on this filtered data\ncell_counts = filtered_data.groupby(['sm_name', 'donor_id', 'cell_type']).size().reset_index(name='cell_counts')\n\n# Create a pivot table\ncompound_donor = cell_counts.pivot_table(\n    index='sm_name',\n    columns=['cell_type', 'donor_id'],\n    values='cell_counts',\n    fill_value=0  # fill missing values with 0\n)\n\n# Optionally, you can flatten the multi-index columns and concatenate the values to have single-level columns\ncompound_donor.columns = compound_donor.columns.map('_'.join)\n\n# Reset the index if you want 'sm_name' to be a column instead of the index\ncompound_donor.reset_index(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-12-03T21:21:32.193011Z","iopub.execute_input":"2023-12-03T21:21:32.193485Z","iopub.status.idle":"2023-12-03T21:21:32.356821Z","shell.execute_reply.started":"2023-12-03T21:21:32.193443Z","shell.execute_reply":"2023-12-03T21:21:32.355532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## LINCS","metadata":{}},{"cell_type":"code","source":"full_data_lincs_feature = pd.read_csv('/kaggle/input/full-data-lincs/full_data_lincs_feature.csv')\n\n# Handle missing values by filling with -100 and adding an indicator column\nfor column in full_data_lincs_feature.columns:\n    # Add a binary column which indicates whether a replacement took place\n    full_data_lincs_feature[column + '_was_missing'] = full_data_lincs_feature[column].isnull().astype(int)\n    \n    # Now fill the NaNs with -100\n    full_data_lincs_feature[column] = full_data_lincs_feature[column].fillna(-100)\n\n# For categorical columns, use enumeration (label encoding)\ncategorical_cols = full_data_lincs_feature.select_dtypes(include=['object', 'category']).columns\nprint(categorical_cols)\n\nlabel_encoders = {}\nfor column in categorical_cols:\n    if column != 'sm_name':\n        # Create a label encoder for all categorical columns\n        label_encoders[column] = LabelEncoder()\n        # Replace the categorical column with encoded data\n        # Fit and transform the data and add 1 so that -100 does not conflict with 0 encoding.\n        full_data_lincs_feature[column] = label_encoders[column].fit_transform(full_data_lincs_feature[column].astype(str)) + 1\nfull_data_lincs_feature.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-12-03T21:21:32.358510Z","iopub.execute_input":"2023-12-03T21:21:32.359032Z","iopub.status.idle":"2023-12-03T21:21:32.433007Z","shell.execute_reply.started":"2023-12-03T21:21:32.358925Z","shell.execute_reply":"2023-12-03T21:21:32.431846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LGBM","metadata":{}},{"cell_type":"code","source":"lgbm_data = pivot_clean[pivot_clean.control == False]\n\n# Add donor cell_type count for 4 cell types\nlgbm_data = lgbm_data.merge(compound_donor, on='sm_name', how='left')\nlgbm_data = lgbm_data.merge(full_data_lincs_feature, on='sm_name', how='left')","metadata":{"execution":{"iopub.status.busy":"2023-12-03T21:21:32.435024Z","iopub.execute_input":"2023-12-03T21:21:32.435879Z","iopub.status.idle":"2023-12-03T21:21:32.758620Z","shell.execute_reply.started":"2023-12-03T21:21:32.435836Z","shell.execute_reply":"2023-12-03T21:21:32.757612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_splits = 15\ngkf = GroupKFold(n_splits=n_splits)\n\nunique_sm_names = lgbm_data['sm_name'].unique()\ngroups = lgbm_data['sm_name'].map(dict(zip(unique_sm_names, range(len(unique_sm_names)))))\n\nfeatures = ['sm_name', 'NK cells', 'T cells CD4+', 'T regulatory cells'] + ['NK cells_donor_0', 'NK cells_donor_1', 'NK cells_donor_2', \n                'T cells CD4+_donor_0', 'T cells CD4+_donor_1', 'T cells CD4+_donor_2', \n               'T regulatory cells_donor_0', 'T regulatory cells_donor_1', 'T regulatory cells_donor_2']\n\nfeatures += list(full_data_lincs_feature.columns)\n    \nX = lgbm_data[features] \n\nrow_scores = []\nmodels = []\nfor target in ['B cells', 'Myeloid cells']:\n    y = lgbm_data[target]\n    for train_index, test_index in gkf.split(X, y, groups):\n        X_train, X_test = X.iloc[train_index], X.iloc[test_index]\n        y_train, y_test = y.iloc[train_index], y.iloc[test_index]\n\n        # Drop 'sm_name' column from X_train and X_test\n        X_train = X_train.drop(columns=['sm_name'])\n        X_test = X_test.drop(columns=['sm_name'])\n\n        # Creating the LightGBM dataset\n        train_data = lgb.Dataset(X_train, label=y_train)\n        test_data = lgb.Dataset(X_test, label=y_test, reference=train_data)\n\n        # Setting up the parameters for LightGBM\n        params = {\n            'objective': 'regression',\n            'metric': 'rmse',\n            'num_leaves': 30,\n            'learning_rate': 0.2,\n            'verbose': -1,\n            'lambda_l1': 0,\n            'lambda_l2': 0,\n        }\n        callbacks = [\n        early_stopping(stopping_rounds=5, first_metric_only=True),\n        log_evaluation(period=-1)  # Set period to 0 to suppress logging\n        ]\n\n        # Training the model\n        num_round = 100\n        bst = lgb.train(params, train_data, num_round,\n                        valid_sets=[test_data], callbacks=callbacks)\n\n        # Making predictions\n        y_pred = bst.predict(X_test, num_iteration=bst.best_iteration)\n\n        models.append(bst)\n\n        # Evaluating the model\n        row_scores.append(mean_squared_error(y_test, y_pred) ** 0.5)\n    print(np.array(row_scores).mean())\n    print(np.array(sorted(row_scores)[:-12]).mean())\n    print()","metadata":{"execution":{"iopub.status.busy":"2023-12-03T21:23:23.758706Z","iopub.execute_input":"2023-12-03T21:23:23.759100Z","iopub.status.idle":"2023-12-03T21:24:31.280442Z","shell.execute_reply.started":"2023-12-03T21:23:23.759071Z","shell.execute_reply":"2023-12-03T21:24:31.279612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Predict","metadata":{}},{"cell_type":"code","source":"B_models = models[:15]\nM_models = models[15:]   ","metadata":{"execution":{"iopub.status.busy":"2023-12-03T21:14:41.059531Z","iopub.execute_input":"2023-12-03T21:14:41.060117Z","iopub.status.idle":"2023-12-03T21:14:41.067273Z","shell.execute_reply.started":"2023-12-03T21:14:41.060067Z","shell.execute_reply":"2023-12-03T21:14:41.065926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 1: Melting the DataFrame to long format\nlong_df = df.melt(id_vars=['sm_name', 'cell_type', 'sm_lincs_id', 'SMILES', 'control'], \n                  var_name='gene', value_name='expression_value')\n# Step 2: Pivoting the long DataFrame to get the desired format\npivot_df = long_df.pivot_table(index=['gene', 'sm_name', 'sm_lincs_id', 'SMILES', 'control'], \n                               columns='cell_type', values='expression_value')\n# Resetting the index and column names for clarity\npivot_df.reset_index(inplace=True)\npivot_df.columns.name = None\npivot_df = pivot_df.drop(columns = 'T cells CD8+')\npivot_clean = pivot_df.drop(columns = ['sm_lincs_id']) #.dropna()\nlgbm_data = pivot_clean[pivot_clean.control == False]","metadata":{"execution":{"iopub.status.busy":"2023-12-03T21:14:41.069068Z","iopub.execute_input":"2023-12-03T21:14:41.070025Z","iopub.status.idle":"2023-12-03T21:15:08.486508Z","shell.execute_reply.started":"2023-12-03T21:14:41.069977Z","shell.execute_reply":"2023-12-03T21:15:08.484707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgbm_data = pivot_clean[pivot_clean.control == False]\nlgbm_data = lgbm_data.merge(compound_donor, on='sm_name', how='left')\nlgbm_data = lgbm_data.merge(full_data_lincs_feature, on='sm_name', how='left')\nlgbm_data","metadata":{"execution":{"iopub.status.busy":"2023-12-03T21:15:39.396112Z","iopub.execute_input":"2023-12-03T21:15:39.397128Z","iopub.status.idle":"2023-12-03T21:15:44.269000Z","shell.execute_reply.started":"2023-12-03T21:15:39.397047Z","shell.execute_reply":"2023-12-03T21:15:44.267511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check that all data is there\n(lgbm_data.gene.value_counts() != 144).sum()","metadata":{"execution":{"iopub.status.busy":"2023-12-03T21:15:44.273721Z","iopub.execute_input":"2023-12-03T21:15:44.274124Z","iopub.status.idle":"2023-12-03T21:15:44.633420Z","shell.execute_reply.started":"2023-12-03T21:15:44.274093Z","shell.execute_reply":"2023-12-03T21:15:44.631712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predictions\nX = lgbm_data.drop(columns=['gene', 'sm_name', 'SMILES', 'control', 'B cells', 'Myeloid cells'])  \nB_predictions = [model.predict(X) for model in B_models]\nM_predictions = [model.predict(X) for model in M_models]","metadata":{"execution":{"iopub.status.busy":"2023-12-03T21:15:44.635056Z","iopub.execute_input":"2023-12-03T21:15:44.635459Z","iopub.status.idle":"2023-12-03T21:18:06.918418Z","shell.execute_reply.started":"2023-12-03T21:15:44.635423Z","shell.execute_reply":"2023-12-03T21:18:06.916997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Right shape\nB_pred_ugly = np.array(B_predictions).mean(axis=0)  \nFull_B = B_pred_ugly.reshape(-1, 144).transpose()\nM_pred_ugly = np.array(M_predictions).mean(axis=0) \nFull_M = M_pred_ugly.reshape(-1, 144).transpose()","metadata":{"execution":{"iopub.status.busy":"2023-12-03T21:18:06.924283Z","iopub.execute_input":"2023-12-03T21:18:06.924675Z","iopub.status.idle":"2023-12-03T21:18:07.310211Z","shell.execute_reply.started":"2023-12-03T21:18:06.924644Z","shell.execute_reply":"2023-12-03T21:18:07.308973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"template = df[(df.cell_type == 'NK cells') & (df.control == False)].sort_values(by='sm_name')[['cell_type','sm_name']].reset_index(drop=True)\ntemplate","metadata":{"execution":{"iopub.status.busy":"2023-12-03T21:18:07.312367Z","iopub.execute_input":"2023-12-03T21:18:07.312921Z","iopub.status.idle":"2023-12-03T21:18:07.360293Z","shell.execute_reply.started":"2023-12-03T21:18:07.312877Z","shell.execute_reply":"2023-12-03T21:18:07.358941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"B_final = pd.concat([pd.DataFrame(Full_B), template], axis=1)\nB_final['cell_type'] = 'B cells'\nM_final = pd.concat([pd.DataFrame(Full_M), template], axis=1)\nM_final['cell_type'] = 'Myeloid cells'\nB_final.columns = sub.columns\nM_final.columns = sub.columns","metadata":{"execution":{"iopub.status.busy":"2023-12-03T21:18:07.361812Z","iopub.execute_input":"2023-12-03T21:18:07.362165Z","iopub.status.idle":"2023-12-03T21:18:07.403787Z","shell.execute_reply.started":"2023-12-03T21:18:07.362130Z","shell.execute_reply":"2023-12-03T21:18:07.402402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"B_final","metadata":{"execution":{"iopub.status.busy":"2023-12-03T21:18:07.405234Z","iopub.execute_input":"2023-12-03T21:18:07.405639Z","iopub.status.idle":"2023-12-03T21:18:07.447739Z","shell.execute_reply.started":"2023-12-03T21:18:07.405603Z","shell.execute_reply":"2023-12-03T21:18:07.446339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Correlation Aggregation","metadata":{}},{"cell_type":"markdown","source":"Note that some data leakage is here, as the correlation matrix is taking all compounds, not only those that are not currently validated with, if you want to make a fair assesment for the correlation matrix, you should not use the full, but recalculate it for each validation compound!","metadata":{}},{"cell_type":"code","source":"def process_row(row):\n    y_pred = np.array(row[:-2]).astype('float32')\n    \n    # Apply z-score standardization to each row\n    z_standardized_y_pred = (y_pred - mean_for_cell_type) / std_dev_for_cell_type\n\n    # Aggregate with potency\n    z_standardized_y_pred_aggregated = np.matmul(z_standardized_y_pred, potent_arr) / abs(potent_arr).sum(axis=0)\n    \n    # Reverse the z-score standardization\n    y_pred = (z_standardized_y_pred_aggregated * std_dev_for_cell_type) + mean_for_cell_type\n    \n    return y_pred\n\n# Process each row sequentially with a progress bar\nfor target in ['B cells', 'Myeloid cells']:\n    \n    ## Block with z-standardization and array calculation\n    cell_type_base = df[df.cell_type == target].iloc[:, 5:] # rearrange to exclude sm in cv split for fairer assesmen\n    \n    # Calculate correlation matrix shape (18211, 18211)\n    arr = np.array(cell_type_base.corr()).astype('float32')\n    \n    # Calculate the mean and standard deviation for each row\n    mean_for_cell_type = np.array(cell_type_base.mean(axis=0)).astype('float32')\n    std_dev_for_cell_type = np.array(cell_type_base.std(axis=0)).astype('float32')\n    \n    # increase potency of correlations (low ones go towards 0)\n    potent_arr = arr ** POTENCY\n    \n    results = []\n    if target == 'B cells':\n        for _, row in tqdm(B_final.iterrows(), total=len(B_final)):\n            result = process_row(row)\n            results.append(result)\n        # Assuming you want to replace the old values in B_final with the new values\n        for idx, result in enumerate(results):\n            B_final.iloc[idx, :-2] = result\n    elif target == 'Myeloid cells':\n        for _, row in tqdm(M_final.iterrows(), total=len(M_final)):\n            result = process_row(row)\n            results.append(result)\n        # Assuming you want to replace the old values in M_final with the new values\n        for idx, result in enumerate(results):\n            M_final.iloc[idx, :-2] = result","metadata":{"execution":{"iopub.status.busy":"2023-12-03T22:31:41.501700Z","iopub.execute_input":"2023-12-03T22:31:41.502112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Final Part","metadata":{}},{"cell_type":"code","source":"base = '/kaggle/input/open-problems-single-cell-perturbations/'\nsub = pd.read_csv(base + 'sample_submission.csv', index_col = 0)\nid_map = pd.read_csv(base + 'id_map.csv', index_col = 0)\nsub = sub.merge(id_map, on='id')","metadata":{"execution":{"iopub.status.busy":"2023-12-03T21:37:21.259776Z","iopub.execute_input":"2023-12-03T21:37:21.260169Z","iopub.status.idle":"2023-12-03T21:37:26.170390Z","shell.execute_reply.started":"2023-12-03T21:37:21.260138Z","shell.execute_reply":"2023-12-03T21:37:26.169033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map[id_map.cell_type == 'Myeloid cells']","metadata":{"execution":{"iopub.status.busy":"2023-12-03T21:37:26.172073Z","iopub.execute_input":"2023-12-03T21:37:26.173151Z","iopub.status.idle":"2023-12-03T21:37:26.189357Z","shell.execute_reply.started":"2023-12-03T21:37:26.173101Z","shell.execute_reply":"2023-12-03T21:37:26.188074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"compounds_B = sub[sub.cell_type == 'B cells'].sm_name.unique()\ncompounds_M = sub[sub.cell_type == 'Myeloid cells'].sm_name.unique()","metadata":{"execution":{"iopub.status.busy":"2023-12-03T21:37:26.192314Z","iopub.execute_input":"2023-12-03T21:37:26.192838Z","iopub.status.idle":"2023-12-03T21:37:26.220989Z","shell.execute_reply.started":"2023-12-03T21:37:26.192781Z","shell.execute_reply":"2023-12-03T21:37:26.219687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"b_cell_indices = sub.index[sub.cell_type == 'B cells']\nb_final_subset = B_final[B_final.sm_name.isin(compounds_B)]\nb_final_subset.index = b_cell_indices\nsub.loc[b_cell_indices] = b_final_subset\nm_cell_indices = sub.index[sub.cell_type == 'Myeloid cells']\nm_final_subset = M_final[M_final.sm_name.isin(compounds_M)]\nm_final_subset.index = m_cell_indices\nsub.loc[m_cell_indices] = m_final_subset\nsub","metadata":{"execution":{"iopub.status.busy":"2023-12-03T21:37:26.222208Z","iopub.execute_input":"2023-12-03T21:37:26.222627Z","iopub.status.idle":"2023-12-03T21:37:35.261981Z","shell.execute_reply.started":"2023-12-03T21:37:26.222548Z","shell.execute_reply":"2023-12-03T21:37:35.260805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit = sub.drop(columns=['cell_type', 'sm_name'])","metadata":{"execution":{"iopub.status.busy":"2023-12-03T21:37:35.263455Z","iopub.execute_input":"2023-12-03T21:37:35.263881Z","iopub.status.idle":"2023-12-03T21:37:35.282987Z","shell.execute_reply.started":"2023-12-03T21:37:35.263849Z","shell.execute_reply":"2023-12-03T21:37:35.282052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-12-03T21:37:35.284268Z","iopub.execute_input":"2023-12-03T21:37:35.285218Z","iopub.status.idle":"2023-12-03T21:37:49.220998Z","shell.execute_reply.started":"2023-12-03T21:37:35.285185Z","shell.execute_reply":"2023-12-03T21:37:49.219545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit","metadata":{"execution":{"iopub.status.busy":"2023-12-03T21:37:49.222668Z","iopub.execute_input":"2023-12-03T21:37:49.224167Z","iopub.status.idle":"2023-12-03T21:37:49.265942Z","shell.execute_reply.started":"2023-12-03T21:37:49.224124Z","shell.execute_reply":"2023-12-03T21:37:49.264264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}