{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os, gc, pickle\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom colorama import Fore, Back, Style\nfrom matplotlib.ticker import MaxNLocator\n\nfrom sklearn.base import BaseEstimator, TransformerMixin\nfrom sklearn.model_selection import KFold\nfrom sklearn.preprocessing import StandardScaler, scale\nfrom sklearn.decomposition import PCA, TruncatedSVD\nfrom sklearn.dummy import DummyRegressor\nfrom sklearn.pipeline import make_pipeline, Pipeline\nfrom sklearn.linear_model import Ridge, LinearRegression, Lasso\nfrom sklearn.metrics import mean_squared_error\n\nimport scipy\nimport scipy.sparse\n\nimport gc\nimport pickle\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-09-08T05:31:37.379094Z","iopub.execute_input":"2022-09-08T05:31:37.379601Z","iopub.status.idle":"2022-09-08T05:31:37.388297Z","shell.execute_reply.started":"2022-09-08T05:31:37.379561Z","shell.execute_reply":"2022-09-08T05:31:37.386705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def correlation_score(y_true, y_pred):\n    \"\"\"Scores the predictions according to the competition rules. \n    \n    It is assumed that the predictions are not constant.\n    \n    Returns the average of each sample's Pearson correlation coefficient\"\"\"\n    if type(y_true) == pd.DataFrame: y_true = y_true.values\n    if type(y_pred) == pd.DataFrame: y_pred = y_pred.values\n    if y_true.shape != y_pred.shape: raise ValueError(\"Shapes are different.\")\n    corrsum = 0\n    for i in range(len(y_true)):\n        corrsum += np.corrcoef(y_true[i], y_pred[i])[1, 0]\n    return corrsum / len(y_true)\n","metadata":{"execution":{"iopub.status.busy":"2022-09-08T05:31:37.682845Z","iopub.execute_input":"2022-09-08T05:31:37.683536Z","iopub.status.idle":"2022-09-08T05:31:37.689808Z","shell.execute_reply.started":"2022-09-08T05:31:37.683496Z","shell.execute_reply":"2022-09-08T05:31:37.689000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing and cross-validation\n\nWe first load all of the training input data for Multiome. It should take less than a minute.","metadata":{}},{"cell_type":"code","source":"%%time\ntrain_inputs = scipy.sparse.load_npz(\"../input/multimodal-single-cell-as-sparse-matrix/train_multi_inputs_values.sparse.npz\")","metadata":{"execution":{"iopub.status.busy":"2022-09-08T05:31:38.277408Z","iopub.execute_input":"2022-09-08T05:31:38.278114Z","iopub.status.idle":"2022-09-08T05:32:43.632456Z","shell.execute_reply.started":"2022-09-08T05:31:38.278076Z","shell.execute_reply":"2022-09-08T05:32:43.631135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_inputs = train_inputs.astype('float16', copy=False)","metadata":{"execution":{"iopub.status.busy":"2022-09-08T05:32:43.634714Z","iopub.execute_input":"2022-09-08T05:32:43.635090Z","iopub.status.idle":"2022-09-08T05:32:47.631017Z","shell.execute_reply.started":"2022-09-08T05:32:43.635058Z","shell.execute_reply":"2022-09-08T05:32:47.629313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## PCA / TruncatedSVD\nIt is not possible to directly apply PCA to a sparse matrix, because PCA has to first \"center\" the data, which destroys the sparsity. This is why we apply `TruncatedSVD` instead (which is pretty much \"PCA without centering\"). It might be better to normalize the data a bit more here, but we will keep it simple.","metadata":{}},{"cell_type":"code","source":"%%time\npca = TruncatedSVD(n_components=128, random_state=42)\ntrain_inputs = pca.fit_transform(train_inputs)\nprint(pca.explained_variance_ratio_.sum())","metadata":{"execution":{"iopub.status.busy":"2022-09-08T05:32:47.633312Z","iopub.execute_input":"2022-09-08T05:32:47.634144Z","iopub.status.idle":"2022-09-08T05:56:18.751409Z","shell.execute_reply.started":"2022-09-08T05:32:47.634065Z","shell.execute_reply":"2022-09-08T05:56:18.750095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_targets = scipy.sparse.load_npz(\"../input/multimodal-single-cell-as-sparse-matrix/train_multi_targets_values.sparse.npz\")","metadata":{"execution":{"iopub.status.busy":"2022-09-08T05:56:18.754737Z","iopub.execute_input":"2022-09-08T05:56:18.755218Z","iopub.status.idle":"2022-09-08T05:56:43.420704Z","shell.execute_reply.started":"2022-09-08T05:56:18.755184Z","shell.execute_reply":"2022-09-08T05:56:43.419591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\npca2 = TruncatedSVD(n_components=128, random_state=42)\ntrain_target = pca2.fit_transform(train_targets)\nprint(pca2.explained_variance_ratio_.sum())","metadata":{"execution":{"iopub.status.busy":"2022-09-08T05:56:43.422787Z","iopub.execute_input":"2022-09-08T05:56:43.423244Z","iopub.status.idle":"2022-09-08T06:02:21.922679Z","shell.execute_reply.started":"2022-09-08T05:56:43.423200Z","shell.execute_reply":"2022-09-08T06:02:21.921473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def save(name, model):\n    with open(name, 'wb') as f:\n        pickle.dump(model, f)","metadata":{"execution":{"iopub.status.busy":"2022-09-08T06:19:40.996914Z","iopub.execute_input":"2022-09-08T06:19:40.997351Z","iopub.status.idle":"2022-09-08T06:19:41.004006Z","shell.execute_reply.started":"2022-09-08T06:19:40.997308Z","shell.execute_reply":"2022-09-08T06:19:41.002845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save('pca.pkl', pca)\nsave('pca2.pkl', pca2)","metadata":{"execution":{"iopub.status.busy":"2022-09-08T06:19:56.393848Z","iopub.execute_input":"2022-09-08T06:19:56.394315Z","iopub.status.idle":"2022-09-08T06:19:57.422754Z","shell.execute_reply.started":"2022-09-08T06:19:56.394276Z","shell.execute_reply":"2022-09-08T06:19:57.421520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.gaussian_process.kernels import RBF\n# from sklearn.kernel_ridge import KernelRidge\n# kernel = RBF(length_scale = 10)\n# krr = KernelRidge(alpha=0.2, kernel=kernel)\nfrom catboost import CatBoostRegressor\nparams = {'learning_rate': 0.2, \n          'depth': 7, \n          'l2_leaf_reg': 4, \n          'loss_function': 'MultiRMSE', \n          'eval_metric': 'MultiRMSE', \n          'task_type': 'CPU', \n          'iterations': 200,\n          'od_type': 'Iter', \n          'boosting_type': 'Plain', \n          'bootstrap_type': 'Bayesian', \n          'allow_const_label': True, \n          'random_state': 1\n         }\nmodel = CatBoostRegressor(**params)","metadata":{"execution":{"iopub.status.busy":"2022-09-08T06:21:10.270222Z","iopub.execute_input":"2022-09-08T06:21:10.270698Z","iopub.status.idle":"2022-09-08T06:21:10.804649Z","shell.execute_reply.started":"2022-09-08T06:21:10.270658Z","shell.execute_reply":"2022-09-08T06:21:10.803423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n = 1","metadata":{"execution":{"iopub.status.busy":"2022-09-08T06:21:13.940940Z","iopub.execute_input":"2022-09-08T06:21:13.941361Z","iopub.status.idle":"2022-09-08T06:21:13.946940Z","shell.execute_reply.started":"2022-09-08T06:21:13.941327Z","shell.execute_reply":"2022-09-08T06:21:13.945695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(42)\nall_row_indices = np.arange(train_inputs.shape[0])\nnp.random.shuffle(all_row_indices)\n\nkf = KFold(n_splits=5, shuffle=True, random_state=42)\n\nindex = 0\nscore = []\n\n# model = Ridge(copy_X=False)\nd = train_inputs.shape[0]//n\nfor i in range(0, n*d, d):\n    print(f'start [{i}:{i+d}]')\n    ind = all_row_indices[i:i+d]    \n    for idx_tr, idx_va in kf.split(ind):\n        X = train_inputs[ind]\n        Y = train_target[ind] #.todense()\n        Yva = train_targets[ind][idx_va]\n        Xtr, Xva = X[idx_tr], X[idx_va]\n        Ytr = Y[idx_tr]\n        del X, Y\n        gc.collect()\n        print('Train...')\n        model.fit(Xtr, Ytr)\n        del Xtr, Ytr\n        gc.collect()\n        s = correlation_score(Yva.todense(), model.predict(Xva)@pca2.components_)\n        score.append(s)\n        print(index, s)\n        del Xva, Yva\n        gc.collect()\n        pkl_filename = f\"model{index:02d}.pkl\"\n        index += 1\n        with open(pkl_filename, 'wb') as file:\n            pickle.dump(model, file)\n        break\n    break\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-08T06:21:20.906363Z","iopub.execute_input":"2022-09-08T06:21:20.906816Z","iopub.status.idle":"2022-09-08T08:32:17.069487Z","shell.execute_reply.started":"2022-09-08T06:21:20.906773Z","shell.execute_reply":"2022-09-08T08:32:17.067285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_target, train_inputs, train_targets\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-08T08:32:31.973986Z","iopub.execute_input":"2022-09-08T08:32:31.974410Z","iopub.status.idle":"2022-09-08T08:32:32.207509Z","shell.execute_reply.started":"2022-09-08T08:32:31.974374Z","shell.execute_reply":"2022-09-08T08:32:32.206290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predicting","metadata":{}},{"cell_type":"code","source":"%%time\nmulti_test_x = scipy.sparse.load_npz(\"../input/multimodal-single-cell-as-sparse-matrix/test_multi_inputs_values.sparse.npz\")\nmulti_test_x = pca.transform(multi_test_x)","metadata":{"execution":{"iopub.status.busy":"2022-09-08T08:32:43.561128Z","iopub.execute_input":"2022-09-08T08:32:43.561580Z","iopub.status.idle":"2022-09-08T08:33:59.967234Z","shell.execute_reply.started":"2022-09-08T08:32:43.561544Z","shell.execute_reply":"2022-09-08T08:33:59.965788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_sd = np.std(multi_test_x, axis=1).reshape(-1, 1)\n# test_sd[test_sd == 0] = 1\n# test_norm = (multi_test_x - np.mean(multi_test_x, axis=1).reshape(-1, 1)) / test_sd\n# test_norm = test_norm.astype(np.float16)\n# del multi_test_x\n# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-08T08:33:59.970053Z","iopub.execute_input":"2022-09-08T08:33:59.970551Z","iopub.status.idle":"2022-09-08T08:33:59.975902Z","shell.execute_reply.started":"2022-09-08T08:33:59.970505Z","shell.execute_reply":"2022-09-08T08:33:59.974713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_len = multi_test_x.shape[0]\nd = test_len//n\nx = []\nfor i in range(n):\n    x.append(multi_test_x[i*d:i*d+d])\ndel multi_test_x\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-08T08:33:59.977889Z","iopub.execute_input":"2022-09-08T08:33:59.978752Z","iopub.status.idle":"2022-09-08T08:34:00.136352Z","shell.execute_reply.started":"2022-09-08T08:33:59.978698Z","shell.execute_reply":"2022-09-08T08:34:00.135417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"index","metadata":{"execution":{"iopub.status.busy":"2022-09-08T08:34:00.138264Z","iopub.execute_input":"2022-09-08T08:34:00.139200Z","iopub.status.idle":"2022-09-08T08:34:00.149122Z","shell.execute_reply.started":"2022-09-08T08:34:00.139155Z","shell.execute_reply":"2022-09-08T08:34:00.147779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = np.zeros((test_len, 23418), dtype='float16')\nfor i,xx in enumerate(x):\n    for ind in range(index):\n        print(ind, end=' ')\n        with open(f'model{ind:02}.pkl', 'rb') as file:\n            model = pickle.load(file)\n        preds[i*d:i*d+d,:] += (model.predict(xx)@pca2.components_)/index\n        gc.collect()\n    print('')\n    del xx\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-08T08:34:00.150577Z","iopub.execute_input":"2022-09-08T08:34:00.151484Z","iopub.status.idle":"2022-09-08T08:34:57.111893Z","shell.execute_reply.started":"2022-09-08T08:34:00.151412Z","shell.execute_reply":"2022-09-08T08:34:57.110809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del x\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-08T08:40:32.586875Z","iopub.execute_input":"2022-09-08T08:40:32.587375Z","iopub.status.idle":"2022-09-08T08:40:32.632193Z","shell.execute_reply.started":"2022-09-08T08:40:32.587339Z","shell.execute_reply":"2022-09-08T08:40:32.630417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.save('preds.npy', preds)","metadata":{"execution":{"iopub.status.busy":"2022-09-08T08:34:57.253686Z","iopub.execute_input":"2022-09-08T08:34:57.254291Z","iopub.status.idle":"2022-09-08T08:35:01.374351Z","shell.execute_reply.started":"2022-09-08T08:34:57.254244Z","shell.execute_reply":"2022-09-08T08:35:01.373407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating submission\n\nWe load the cells that will have to appear in submission.","metadata":{}},{"cell_type":"code","source":"%%time\n# Read the table of rows and columns required for submission\neval_ids = pd.read_parquet(\"../input/multimodal-single-cell-as-sparse-matrix/evaluation.parquet\")\n# Convert the string columns to more efficient categorical types\n#eval_ids.cell_id = eval_ids.cell_id.apply(lambda s: int(s, base=16))\neval_ids.cell_id = eval_ids.cell_id.astype(pd.CategoricalDtype())\neval_ids.gene_id = eval_ids.gene_id.astype(pd.CategoricalDtype())","metadata":{"execution":{"iopub.status.busy":"2022-09-08T08:58:43.571990Z","iopub.execute_input":"2022-09-08T08:58:43.572461Z","iopub.status.idle":"2022-09-08T08:59:16.758628Z","shell.execute_reply.started":"2022-09-08T08:58:43.572410Z","shell.execute_reply":"2022-09-08T08:59:16.757317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prepare an empty series which will be filled with predictions\nsubmission = pd.Series(name='target',\n                       index=pd.MultiIndex.from_frame(eval_ids), \n                       dtype=np.float32)\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-09-08T08:59:16.760584Z","iopub.execute_input":"2022-09-08T08:59:16.760942Z","iopub.status.idle":"2022-09-08T08:59:41.085805Z","shell.execute_reply.started":"2022-09-08T08:59:16.760911Z","shell.execute_reply":"2022-09-08T08:59:41.084414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We load the `index`  and `columns` of the original dataframe, as we need them to make the submission.","metadata":{}},{"cell_type":"code","source":"%%time\ny_columns = np.load(\"../input/multimodal-single-cell-as-sparse-matrix/train_multi_targets_idxcol.npz\",\n                   allow_pickle=True)[\"columns\"]\n\ntest_index = np.load(\"../input/multimodal-single-cell-as-sparse-matrix/test_multi_inputs_idxcol.npz\",\n                    allow_pickle=True)[\"index\"]","metadata":{"execution":{"iopub.status.busy":"2022-09-08T08:59:41.087923Z","iopub.execute_input":"2022-09-08T08:59:41.088858Z","iopub.status.idle":"2022-09-08T08:59:41.199209Z","shell.execute_reply.started":"2022-09-08T08:59:41.088807Z","shell.execute_reply":"2022-09-08T08:59:41.198300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We assign the predicted values to the correct row in the submission file.","metadata":{}},{"cell_type":"code","source":"cell_dict = dict((k,v) for v,k in enumerate(test_index)) \nassert len(cell_dict)  == len(test_index)\n\ngene_dict = dict((k,v) for v,k in enumerate(y_columns))\nassert len(gene_dict) == len(y_columns)","metadata":{"execution":{"iopub.status.busy":"2022-09-08T08:59:41.201894Z","iopub.execute_input":"2022-09-08T08:59:41.202377Z","iopub.status.idle":"2022-09-08T08:59:41.242022Z","shell.execute_reply.started":"2022-09-08T08:59:41.202340Z","shell.execute_reply":"2022-09-08T08:59:41.240547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eval_ids_cell_num = eval_ids.cell_id.apply(lambda x:cell_dict.get(x, -1))\neval_ids_gene_num = eval_ids.gene_id.apply(lambda x:gene_dict.get(x, -1))\n\nvalid_multi_rows = (eval_ids_gene_num !=-1) & (eval_ids_cell_num!=-1)","metadata":{"execution":{"iopub.status.busy":"2022-09-08T08:59:41.243695Z","iopub.execute_input":"2022-09-08T08:59:41.244124Z","iopub.status.idle":"2022-09-08T08:59:45.479181Z","shell.execute_reply.started":"2022-09-08T08:59:41.244089Z","shell.execute_reply":"2022-09-08T08:59:45.477906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.iloc[valid_multi_rows] = preds[eval_ids_cell_num[valid_multi_rows].to_numpy(),\neval_ids_gene_num[valid_multi_rows].to_numpy()]","metadata":{"execution":{"iopub.status.busy":"2022-09-08T08:59:45.482633Z","iopub.execute_input":"2022-09-08T08:59:45.483117Z","iopub.status.idle":"2022-09-08T08:59:49.622226Z","shell.execute_reply.started":"2022-09-08T08:59:45.483070Z","shell.execute_reply":"2022-09-08T08:59:49.620938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del eval_ids_cell_num, eval_ids_gene_num, valid_multi_rows, eval_ids, test_index, y_columns\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-08T08:59:49.623557Z","iopub.execute_input":"2022-09-08T08:59:49.623885Z","iopub.status.idle":"2022-09-08T08:59:49.883595Z","shell.execute_reply.started":"2022-09-08T08:59:49.623856Z","shell.execute_reply":"2022-09-08T08:59:49.882122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2022-09-08T08:59:49.886645Z","iopub.execute_input":"2022-09-08T08:59:49.890036Z","iopub.status.idle":"2022-09-08T08:59:49.904374Z","shell.execute_reply.started":"2022-09-08T08:59:49.889996Z","shell.execute_reply":"2022-09-08T08:59:49.903502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Merging with CITEseq predictions\n\nWe use the CITEseq predictions from [this notebook](https://www.kaggle.com/code/vuonglam/lgbm-baseline-optuna-drop-constant-cite-task) by VuongLam.","metadata":{}},{"cell_type":"code","source":"submission.reset_index(drop=True, inplace=True)\nsubmission.index.name = 'row_id'","metadata":{"execution":{"iopub.status.busy":"2022-09-08T09:00:00.751794Z","iopub.execute_input":"2022-09-08T09:00:00.752189Z","iopub.status.idle":"2022-09-08T09:00:00.929090Z","shell.execute_reply.started":"2022-09-08T09:00:00.752148Z","shell.execute_reply":"2022-09-08T09:00:00.927588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cite_submission = pd.read_csv(\"../input/msci-citeseq-keras-quickstart/submission.csv\")\ncite_submission = cite_submission.set_index(\"row_id\")\ncite_submission = cite_submission[\"target\"]","metadata":{"execution":{"iopub.status.busy":"2022-09-08T08:57:14.490604Z","iopub.execute_input":"2022-09-08T08:57:14.491228Z","iopub.status.idle":"2022-09-08T08:57:44.325350Z","shell.execute_reply.started":"2022-09-08T08:57:14.491194Z","shell.execute_reply":"2022-09-08T08:57:44.323872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission[submission.isnull()] = cite_submission[submission.isnull()]","metadata":{"execution":{"iopub.status.busy":"2022-09-08T09:00:05.515085Z","iopub.execute_input":"2022-09-08T09:00:05.515523Z","iopub.status.idle":"2022-09-08T09:00:07.642694Z","shell.execute_reply.started":"2022-09-08T09:00:05.515484Z","shell.execute_reply":"2022-09-08T09:00:07.641340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2022-09-08T09:00:09.303377Z","iopub.execute_input":"2022-09-08T09:00:09.304580Z","iopub.status.idle":"2022-09-08T09:00:09.313859Z","shell.execute_reply.started":"2022-09-08T09:00:09.304536Z","shell.execute_reply":"2022-09-08T09:00:09.312925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2022-09-08T09:00:16.155085Z","iopub.execute_input":"2022-09-08T09:00:16.155515Z","iopub.status.idle":"2022-09-08T09:00:16.165035Z","shell.execute_reply.started":"2022-09-08T09:00:16.155477Z","shell.execute_reply":"2022-09-08T09:00:16.164127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.isnull().any()","metadata":{"execution":{"iopub.status.busy":"2022-09-08T09:00:21.714789Z","iopub.execute_input":"2022-09-08T09:00:21.715301Z","iopub.status.idle":"2022-09-08T09:00:21.763650Z","shell.execute_reply.started":"2022-09-08T09:00:21.715255Z","shell.execute_reply":"2022-09-08T09:00:21.762783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-09-08T09:00:22.465913Z","iopub.execute_input":"2022-09-08T09:00:22.466314Z","iopub.status.idle":"2022-09-08T09:02:35.850961Z","shell.execute_reply.started":"2022-09-08T09:00:22.466280Z","shell.execute_reply":"2022-09-08T09:02:35.849565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head submission.csv","metadata":{"execution":{"iopub.status.busy":"2022-09-08T09:02:35.855348Z","iopub.execute_input":"2022-09-08T09:02:35.856027Z","iopub.status.idle":"2022-09-08T09:02:37.161421Z","shell.execute_reply.started":"2022-09-08T09:02:35.855988Z","shell.execute_reply":"2022-09-08T09:02:37.160218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head submission.csv","metadata":{"execution":{"iopub.status.busy":"2022-09-08T08:44:39.175177Z","iopub.execute_input":"2022-09-08T08:44:39.176145Z","iopub.status.idle":"2022-09-08T08:44:40.530109Z","shell.execute_reply.started":"2022-09-08T08:44:39.176079Z","shell.execute_reply":"2022-09-08T08:44:40.528318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}