{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":59094,"databundleVersionId":7010844,"sourceType":"competition"}],"dockerImageVersionId":30616,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"import time\nt0start = time.time()\nfrom fastai.collab import *\nfrom fastai.tabular.all import *\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:03:08.401352Z","iopub.execute_input":"2024-08-28T21:03:08.402312Z","iopub.status.idle":"2024-08-28T21:03:08.407176Z","shell.execute_reply.started":"2024-08-28T21:03:08.402277Z","shell.execute_reply":"2024-08-28T21:03:08.406291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_seed = 42","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:03:08.409243Z","iopub.execute_input":"2024-08-28T21:03:08.409611Z","iopub.status.idle":"2024-08-28T21:03:08.423885Z","shell.execute_reply.started":"2024-08-28T21:03:08.409556Z","shell.execute_reply":"2024-08-28T21:03:08.423163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import set_seed\nset_seed(random_seed)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:03:08.425165Z","iopub.execute_input":"2024-08-28T21:03:08.425514Z","iopub.status.idle":"2024-08-28T21:03:18.500297Z","shell.execute_reply.started":"2024-08-28T21:03:08.425483Z","shell.execute_reply":"2024-08-28T21:03:18.499370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare Train Set","metadata":{}},{"cell_type":"markdown","source":"Here I read the training data and melt it to yield a ```DataFrame``` with three categorical features (```cell_type```, ```sm_name```, and ```gene```) and one target (```value```).","metadata":{}},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet'\ndf_de_train = pd.read_parquet(fn)# , index_col = 0)\ntrain_df = df_de_train.drop(columns=['sm_lincs_id', 'SMILES', 'control'])\ntrdf = train_df.melt(id_vars=['cell_type', 'sm_name'], value_vars=train_df.iloc[:,2:].columns, var_name='gene', value_name='value')","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:03:18.501476Z","iopub.execute_input":"2024-08-28T21:03:18.502052Z","iopub.status.idle":"2024-08-28T21:03:23.719843Z","shell.execute_reply.started":"2024-08-28T21:03:18.502025Z","shell.execute_reply":"2024-08-28T21:03:23.718864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The next line samples 5 % of the training data to demonstrate model training in a reasonable time frame:","metadata":{}},{"cell_type":"code","source":"trdf = trdf.sample(frac=0.05).reset_index(drop=true)\ndisplay(trdf)","metadata":{"execution":{"iopub.status.busy":"2024-08-28T21:03:23.722489Z","iopub.execute_input":"2024-08-28T21:03:23.722764Z","iopub.status.idle":"2024-08-28T21:03:24.504080Z","shell.execute_reply.started":"2024-08-28T21:03:23.722740Z","shell.execute_reply":"2024-08-28T21:03:24.503168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare Test Set","metadata":{}},{"cell_type":"markdown","source":"Similar procedure for the test set.","metadata":{}},{"cell_type":"code","source":"fn = '/kaggle/input/open-problems-single-cell-perturbations/id_map.csv'\ndf_id_map = pd.read_csv(fn)\nfn = '/kaggle/input/open-problems-single-cell-perturbations/sample_submission.csv'\ndf = pd.read_csv(fn, index_col = 0)\n\ncols_to_add = df_de_train.iloc[:,5:].columns\ndisplay(cols_to_add)\n\ndf_zeros = pd.DataFrame(0.0, columns=cols_to_add, index=df_id_map.index)\ndisplay(df_zeros)\n\ndf_id_map_preds = pd.concat([df_id_map, df_zeros], axis=1)\ntsdf = df_id_map_preds.melt(id_vars=['cell_type', 'sm_name'], value_vars=df_id_map_preds.iloc[:,3:].columns, var_name='gene', value_name='value')\ndisplay(tsdf)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:03:24.505257Z","iopub.execute_input":"2024-08-28T21:03:24.505607Z","iopub.status.idle":"2024-08-28T21:03:29.344874Z","shell.execute_reply.started":"2024-08-28T21:03:24.505578Z","shell.execute_reply":"2024-08-28T21:03:29.343767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inspect Train and Test Set","metadata":{}},{"cell_type":"code","source":"trdf","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:03:29.345883Z","iopub.execute_input":"2024-08-28T21:03:29.346158Z","iopub.status.idle":"2024-08-28T21:03:29.358600Z","shell.execute_reply.started":"2024-08-28T21:03:29.346133Z","shell.execute_reply":"2024-08-28T21:03:29.357692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tsdf","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:03:29.360142Z","iopub.execute_input":"2024-08-28T21:03:29.361935Z","iopub.status.idle":"2024-08-28T21:03:29.430426Z","shell.execute_reply.started":"2024-08-28T21:03:29.361869Z","shell.execute_reply":"2024-08-28T21:03:29.429577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare Transformer Input","metadata":{}},{"cell_type":"code","source":"trdf['input'] = \"Estimate the −log(p-value) confidence for change in gene expression of \" + trdf.gene + \" in \" + trdf.cell_type + \" when treated with \" + trdf.sm_name + \" compared to DMSO.\"\nprint(trdf.input[0])\ntrdf.input.head()","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:03:29.431525Z","iopub.execute_input":"2024-08-28T21:03:29.431820Z","iopub.status.idle":"2024-08-28T21:03:30.390009Z","shell.execute_reply.started":"2024-08-28T21:03:29.431795Z","shell.execute_reply":"2024-08-28T21:03:30.389104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tsdf['input'] = \"Estimate the −log(p-value) confidence for change in gene expression of \" + tsdf.gene + \" in \" + tsdf.cell_type + \" when treated with \" + tsdf.sm_name + \" compared to DMSO.\"\nprint(trdf.input[0])\nprint(tsdf.input[0])","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:03:30.391379Z","iopub.execute_input":"2024-08-28T21:03:30.391768Z","iopub.status.idle":"2024-08-28T21:03:38.813281Z","shell.execute_reply.started":"2024-08-28T21:03:30.391732Z","shell.execute_reply":"2024-08-28T21:03:38.812345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from datasets import Dataset, DatasetDict\ntrds = Dataset.from_pandas(trdf)\ntrds","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:03:38.814592Z","iopub.execute_input":"2024-08-28T21:03:38.815307Z","iopub.status.idle":"2024-08-28T21:03:39.230646Z","shell.execute_reply.started":"2024-08-28T21:03:38.815271Z","shell.execute_reply":"2024-08-28T21:03:39.229749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tsds = Dataset.from_pandas(tsdf)\ntsds","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:03:39.236126Z","iopub.execute_input":"2024-08-28T21:03:39.236430Z","iopub.status.idle":"2024-08-28T21:03:42.862589Z","shell.execute_reply.started":"2024-08-28T21:03:39.236386Z","shell.execute_reply":"2024-08-28T21:03:42.861510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare Model","metadata":{}},{"cell_type":"markdown","source":"### Tokenize","metadata":{}},{"cell_type":"code","source":"from transformers import AutoModelForSequenceClassification, AutoTokenizer\nmodel_nm = 'nlpie/tiny-biobert'\ntokz = AutoTokenizer.from_pretrained(model_nm)\n\n# @misc{https://doi.org/10.48550/arxiv.2209.03182,\n#   doi = {10.48550/ARXIV.2209.03182},\n#   url = {https://arxiv.org/abs/2209.03182},\n#   author = {Rohanian, Omid and Nouriborji, Mohammadmahdi and Kouchaki, Samaneh and Clifton, David A.},\n#   keywords = {Computation and Language (cs.CL), Machine Learning (cs.LG), FOS: Computer and information sciences, FOS: Computer and information sciences, 68T50},\n#   title = {On the Effectiveness of Compact Biomedical Transformers},\n#   publisher = {arXiv},\n#   year = {2022}, \n#   copyright = {arXiv.org perpetual, non-exclusive license}\n# }\n","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:03:42.864046Z","iopub.execute_input":"2024-08-28T21:03:42.864427Z","iopub.status.idle":"2024-08-28T21:03:44.470060Z","shell.execute_reply.started":"2024-08-28T21:03:42.864377Z","shell.execute_reply":"2024-08-28T21:03:44.469275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"L(tokz.tokenize(trds['input'][0]))","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:03:44.471082Z","iopub.execute_input":"2024-08-28T21:03:44.471334Z","iopub.status.idle":"2024-08-28T21:03:45.521931Z","shell.execute_reply.started":"2024-08-28T21:03:44.471312Z","shell.execute_reply":"2024-08-28T21:03:45.520863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Started tokenization...')\ndef tok_func(x): return tokz(x[\"input\"])\n\ntok_trds = trds.map(tok_func, batched = True)\ntok_tsds = tsds.map(tok_func, batched = True)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:03:45.523124Z","iopub.execute_input":"2024-08-28T21:03:45.523491Z","iopub.status.idle":"2024-08-28T21:12:02.038807Z","shell.execute_reply.started":"2024-08-28T21:03:45.523463Z","shell.execute_reply":"2024-08-28T21:12:02.037629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(tok_trds[0]['input']), L(tok_trds[0]['input_ids'])","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:12:02.040144Z","iopub.execute_input":"2024-08-28T21:12:02.040457Z","iopub.status.idle":"2024-08-28T21:12:02.049650Z","shell.execute_reply.started":"2024-08-28T21:12:02.040431Z","shell.execute_reply":"2024-08-28T21:12:02.048771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(tok_tsds[0]['input']), L(tok_tsds[0]['input_ids'])","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:12:02.050901Z","iopub.execute_input":"2024-08-28T21:12:02.051238Z","iopub.status.idle":"2024-08-28T21:12:02.064645Z","shell.execute_reply.started":"2024-08-28T21:12:02.051205Z","shell.execute_reply":"2024-08-28T21:12:02.063730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(tokz)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:12:02.065916Z","iopub.execute_input":"2024-08-28T21:12:02.066246Z","iopub.status.idle":"2024-08-28T21:12:02.082218Z","shell.execute_reply.started":"2024-08-28T21:12:02.066215Z","shell.execute_reply":"2024-08-28T21:12:02.081456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in tok_tsds[0]['input_ids']:\n    print(tokz.decode(i))","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:12:02.083472Z","iopub.execute_input":"2024-08-28T21:12:02.083814Z","iopub.status.idle":"2024-08-28T21:12:02.096179Z","shell.execute_reply.started":"2024-08-28T21:12:02.083783Z","shell.execute_reply":"2024-08-28T21:12:02.095348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tok_trds = tok_trds.rename_columns({\"value\":\"labels\"})\ntok_trds","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:12:02.097235Z","iopub.execute_input":"2024-08-28T21:12:02.097532Z","iopub.status.idle":"2024-08-28T21:12:02.112564Z","shell.execute_reply.started":"2024-08-28T21:12:02.097508Z","shell.execute_reply":"2024-08-28T21:12:02.111771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Split Train Data","metadata":{}},{"cell_type":"code","source":"dds = tok_trds.train_test_split(0.20, seed=random_seed)\ndds","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:12:02.113553Z","iopub.execute_input":"2024-08-28T21:12:02.113922Z","iopub.status.idle":"2024-08-28T21:12:02.318532Z","shell.execute_reply.started":"2024-08-28T21:12:02.113898Z","shell.execute_reply":"2024-08-28T21:12:02.317600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Define Model","metadata":{}},{"cell_type":"code","source":"bs = 256\nepochs = 5\nlr = 8e-5","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:12:02.319490Z","iopub.execute_input":"2024-08-28T21:12:02.319739Z","iopub.status.idle":"2024-08-28T21:12:02.323885Z","shell.execute_reply.started":"2024-08-28T21:12:02.319716Z","shell.execute_reply":"2024-08-28T21:12:02.323088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"steps = len(dds['train']) // bs\nsteps","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:12:02.324965Z","iopub.execute_input":"2024-08-28T21:12:02.325224Z","iopub.status.idle":"2024-08-28T21:12:02.336207Z","shell.execute_reply.started":"2024-08-28T21:12:02.325200Z","shell.execute_reply":"2024-08-28T21:12:02.335380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import TrainingArguments, Trainer\n\nargs = TrainingArguments('outputs', save_steps=steps, learning_rate=lr, warmup_ratio=0.1, lr_scheduler_type='cosine', fp16=True,\n    evaluation_strategy=\"epoch\", per_device_train_batch_size=bs, per_device_eval_batch_size=bs*2,\n    num_train_epochs=epochs, weight_decay=0.01, report_to='none', seed=random_seed) # resume_from_checkpoint=\"/kaggle/working/20231005/outputs/checkpoint-104826\", fp16=True","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:12:02.337221Z","iopub.execute_input":"2024-08-28T21:12:02.337530Z","iopub.status.idle":"2024-08-28T21:12:02.624831Z","shell.execute_reply.started":"2024-08-28T21:12:02.337493Z","shell.execute_reply":"2024-08-28T21:12:02.624089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import EvalPrediction\n\ndef compute_metrics(p: EvalPrediction):\n    preds = p.predictions\n    labels = p.label_ids\n    rmse = np.sqrt(((preds - labels) ** 2).mean())\n    return {\"rmse\": rmse}","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:12:02.626037Z","iopub.execute_input":"2024-08-28T21:12:02.626730Z","iopub.status.idle":"2024-08-28T21:12:02.632071Z","shell.execute_reply.started":"2024-08-28T21:12:02.626694Z","shell.execute_reply":"2024-08-28T21:12:02.631142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = AutoModelForSequenceClassification.from_pretrained(\"/kaggle/working/tinybiobert_tasked_one_sentence_5eps42\", num_labels=1) # for checkpoint use e.g. \"/kaggle/working/20231005/outputs/checkpoint-104826\" instead of model_nm\ntrainer = Trainer(model, args, train_dataset=dds['train'], eval_dataset=dds['test'],\n                  tokenizer=tokz)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:12:02.633379Z","iopub.execute_input":"2024-08-28T21:12:02.633688Z","iopub.status.idle":"2024-08-28T21:12:03.459514Z","shell.execute_reply.started":"2024-08-28T21:12:02.633662Z","shell.execute_reply":"2024-08-28T21:12:03.458712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training","metadata":{}},{"cell_type":"code","source":"print('Started training...')\ntrainer.train()","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:12:03.460653Z","iopub.execute_input":"2024-08-28T21:12:03.460955Z","iopub.status.idle":"2024-08-28T21:49:30.008239Z","shell.execute_reply.started":"2024-08-28T21:12:03.460928Z","shell.execute_reply":"2024-08-28T21:49:30.007299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_string = 'opxtinybiobert_5eps'\ntrainer.save_model(f'./{model_string}/')","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:49:30.009493Z","iopub.execute_input":"2024-08-28T21:49:30.009792Z","iopub.status.idle":"2024-08-28T21:49:30.190488Z","shell.execute_reply.started":"2024-08-28T21:49:30.009765Z","shell.execute_reply":"2024-08-28T21:49:30.189679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import AutoModelForSequenceClassification, AutoTokenizer\nmodel_nm = 'nlpie/distil-biobert'\ntokz = AutoTokenizer.from_pretrained(model_nm)","metadata":{"execution":{"iopub.status.busy":"2024-08-28T21:49:30.191626Z","iopub.execute_input":"2024-08-28T21:49:30.191909Z","iopub.status.idle":"2024-08-28T21:49:31.224290Z","shell.execute_reply.started":"2024-08-28T21:49:30.191884Z","shell.execute_reply":"2024-08-28T21:49:31.223422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_loss_history = [log['eval_loss'] for log in trainer.state.log_history if 'eval_loss' in log]\nvalid_step_history = [log['step'] for log in trainer.state.log_history if 'eval_loss' in log]\ntrain_loss_history = [log['loss'] for log in trainer.state.log_history if 'loss' in log]\ntrain_step_history = [log['step'] for log in trainer.state.log_history if 'loss' in log]\nlrate_history = [log['learning_rate'] for log in trainer.state.log_history if 'loss' in log]\nepoch_history = [log['epoch'] for log in trainer.state.log_history if 'eval_loss' in log]\n\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 5))\n\nax1.plot(valid_step_history, valid_loss_history)\nax1.plot(train_step_history, train_loss_history)\nax1.set_xlabel('Step')\nax1.set_ylabel('Loss')\nax1.set_title('Loss History')\n\nax2.plot(train_step_history, lrate_history)\nax2.set_xlabel('Step')\nax2.set_ylabel('Learning Rate')\nax2.set_title('Learning Rate History')\n\nplt.savefig(f'{model_string}_training.pdf', format='pdf')\nplt.tight_layout\nplt.show()","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:49:31.225508Z","iopub.execute_input":"2024-08-28T21:49:31.225800Z","iopub.status.idle":"2024-08-28T21:49:32.252604Z","shell.execute_reply.started":"2024-08-28T21:49:31.225774Z","shell.execute_reply":"2024-08-28T21:49:32.251628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(f'{model_string}_training_history.txt', 'w') as file:\n    file.write(str(trainer.state.log_history))","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:49:32.253818Z","iopub.execute_input":"2024-08-28T21:49:32.254426Z","iopub.status.idle":"2024-08-28T21:49:32.259362Z","shell.execute_reply.started":"2024-08-28T21:49:32.254381Z","shell.execute_reply":"2024-08-28T21:49:32.258470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predict","metadata":{}},{"cell_type":"code","source":"print('Started inference...')\npreds = trainer.predict(tok_tsds).predictions.astype(float)\npreds","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-08-28T21:49:32.260540Z","iopub.execute_input":"2024-08-28T21:49:32.260831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds.max(), preds.min()","metadata":{"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_rounded = np.round_(preds, 4)\npreds_rounded","metadata":{"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In the last step I reshape the predictions back into a 255 x 18211 tensor for submission:","metadata":{}},{"cell_type":"code","source":"to_submit = preds_rounded.reshape(18211, -1).T","metadata":{"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit = pd.DataFrame(to_submit, columns=df_de_train.iloc[:,5:].columns)\nsubmit.index.name = 'id'\nsubmit","metadata":{"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit.to_csv(f'submission.csv')","metadata":{"tags":[],"trusted":true},"execution_count":null,"outputs":[]}]}