{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Tabular Playground Oct 2022 - TabNet","metadata":{}},{"cell_type":"markdown","source":"In this notebook, we will use Tabnet to fit a model.","metadata":{}},{"cell_type":"code","source":"!pip install pytorch-tabnet","metadata":{"execution":{"iopub.status.busy":"2022-10-03T13:51:10.652430Z","iopub.execute_input":"2022-10-03T13:51:10.653033Z","iopub.status.idle":"2022-10-03T13:51:23.828565Z","shell.execute_reply.started":"2022-10-03T13:51:10.652894Z","shell.execute_reply":"2022-10-03T13:51:23.827044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pytorch_tabnet.tab_model import TabNetClassifier","metadata":{"execution":{"iopub.status.busy":"2022-10-03T13:56:14.068451Z","iopub.execute_input":"2022-10-03T13:56:14.068949Z","iopub.status.idle":"2022-10-03T13:56:16.308197Z","shell.execute_reply.started":"2022-10-03T13:56:14.068904Z","shell.execute_reply":"2022-10-03T13:56:16.307056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport gc\nfrom itertools import chain\nfrom sklearn.metrics import log_loss\nimport seaborn as sns\nimport torch","metadata":{"execution":{"iopub.status.busy":"2022-10-03T05:59:12.764573Z","iopub.execute_input":"2022-10-03T05:59:12.765381Z","iopub.status.idle":"2022-10-03T05:59:12.845162Z","shell.execute_reply.started":"2022-10-03T05:59:12.765341Z","shell.execute_reply":"2022-10-03T05:59:12.843592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DAT_DIR = '../input/tabular-playground-series-oct-2022'","metadata":{"execution":{"iopub.status.busy":"2022-10-03T05:59:12.848747Z","iopub.execute_input":"2022-10-03T05:59:12.849243Z","iopub.status.idle":"2022-10-03T05:59:12.855106Z","shell.execute_reply.started":"2022-10-03T05:59:12.849194Z","shell.execute_reply":"2022-10-03T05:59:12.853464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(os.path.join(DAT_DIR, \"train_0.csv\"))","metadata":{"execution":{"iopub.status.busy":"2022-10-03T05:59:12.857249Z","iopub.execute_input":"2022-10-03T05:59:12.857882Z","iopub.status.idle":"2022-10-03T05:59:54.745722Z","shell.execute_reply.started":"2022-10-03T05:59:12.857821Z","shell.execute_reply":"2022-10-03T05:59:54.744552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we will train the model by batch, we will divide the training files into categories of training, validation, and (out-of-sample) testing. The last category is to assess the model performance free of biases as they have never been seen by the model before. ","metadata":{}},{"cell_type":"code","source":"train_files = list(range(7))\nvalid_files = list(range(7,8))\ntest_files = list(range(8, 10))","metadata":{"execution":{"iopub.status.busy":"2022-10-03T05:59:54.747581Z","iopub.execute_input":"2022-10-03T05:59:54.748075Z","iopub.status.idle":"2022-10-03T05:59:54.754913Z","shell.execute_reply.started":"2022-10-03T05:59:54.748027Z","shell.execute_reply":"2022-10-03T05:59:54.753284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'train_files = {train_files}')\nprint(f'valid_files = {valid_files}')\nprint(f'test_files = {test_files}')","metadata":{"execution":{"iopub.status.busy":"2022-10-03T05:59:54.757032Z","iopub.execute_input":"2022-10-03T05:59:54.757527Z","iopub.status.idle":"2022-10-03T05:59:54.770950Z","shell.execute_reply.started":"2022-10-03T05:59:54.757462Z","shell.execute_reply":"2022-10-03T05:59:54.769694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid = pd.read_csv(os.path.join(DAT_DIR, f\"train_{valid_files[0]}.csv\"))","metadata":{"execution":{"iopub.status.busy":"2022-10-03T05:59:54.774852Z","iopub.execute_input":"2022-10-03T05:59:54.775243Z","iopub.status.idle":"2022-10-03T06:00:38.639456Z","shell.execute_reply.started":"2022-10-03T05:59:54.775210Z","shell.execute_reply":"2022-10-03T06:00:38.636333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid","metadata":{"execution":{"iopub.status.busy":"2022-10-03T06:00:38.641877Z","iopub.execute_input":"2022-10-03T06:00:38.642396Z","iopub.status.idle":"2022-10-03T06:00:39.080204Z","shell.execute_reply.started":"2022-10-03T06:00:38.642352Z","shell.execute_reply":"2022-10-03T06:00:39.078914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature Engineering","metadata":{}},{"cell_type":"markdown","source":"We want to add the distance between the ball and players as features. We add 3-d Euclidean distances.","metadata":{}},{"cell_type":"code","source":"dist_p_ball_cols = [f'dist_p_ball_{i}' for i in range(6)]","metadata":{"execution":{"iopub.status.busy":"2022-10-03T06:00:39.084711Z","iopub.execute_input":"2022-10-03T06:00:39.085222Z","iopub.status.idle":"2022-10-03T06:00:39.091817Z","shell.execute_reply.started":"2022-10-03T06:00:39.085189Z","shell.execute_reply":"2022-10-03T06:00:39.090459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'dist_p_ball_cols = {dist_p_ball_cols}')","metadata":{"execution":{"iopub.status.busy":"2022-10-03T06:00:39.093201Z","iopub.execute_input":"2022-10-03T06:00:39.093587Z","iopub.status.idle":"2022-10-03T06:00:39.104287Z","shell.execute_reply.started":"2022-10-03T06:00:39.093553Z","shell.execute_reply":"2022-10-03T06:00:39.102584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_dist_cols(df):\n    for i in range(6):\n        df[f'dist_p_ball_{i}'] = np.sqrt((df.ball_pos_x - df[f'p{i}_pos_x'])**2 + (df.ball_pos_y - df[f'p{i}_pos_y'])**2 + (df.ball_pos_z - df[f'p{i}_pos_z'])**2)\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-10-03T06:00:39.105732Z","iopub.execute_input":"2022-10-03T06:00:39.106126Z","iopub.status.idle":"2022-10-03T06:00:39.115679Z","shell.execute_reply.started":"2022-10-03T06:00:39.106095Z","shell.execute_reply":"2022-10-03T06:00:39.114105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train = add_dist_cols(train)","metadata":{"execution":{"iopub.status.busy":"2022-10-03T06:00:39.117527Z","iopub.execute_input":"2022-10-03T06:00:39.117936Z","iopub.status.idle":"2022-10-03T06:00:39.125592Z","shell.execute_reply.started":"2022-10-03T06:00:39.117905Z","shell.execute_reply":"2022-10-03T06:00:39.124379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#valid = add_dist_cols(valid)","metadata":{"execution":{"iopub.status.busy":"2022-10-03T06:00:39.126934Z","iopub.execute_input":"2022-10-03T06:00:39.127303Z","iopub.status.idle":"2022-10-03T06:00:39.136366Z","shell.execute_reply.started":"2022-10-03T06:00:39.127268Z","shell.execute_reply":"2022-10-03T06:00:39.135361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ball_pos_cols = [f'ball_pos_{i}' for i in ['x', 'y', 'z']]\nball_vel_cols = [f'ball_vel_{i}' for i in ['x', 'y', 'z']]\nplayer_pos_cols = list(chain.from_iterable([[f'p{i}_pos_{j}' for j in ['x', 'y', 'z']] for i in range(6)]))\nplayer_vel_cols = list(chain.from_iterable([[f'p{i}_vel_{j}' for j in ['x', 'y', 'z']] for i in range(6)]))\nboost_cols = [f'p{i}_boost' for i in range(6)]\nboost_timer_cols = [f'boost{i}_timer' for i in range(6)]","metadata":{"execution":{"iopub.status.busy":"2022-10-03T06:00:39.138101Z","iopub.execute_input":"2022-10-03T06:00:39.139270Z","iopub.status.idle":"2022-10-03T06:00:39.149351Z","shell.execute_reply.started":"2022-10-03T06:00:39.139108Z","shell.execute_reply":"2022-10-03T06:00:39.148328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'ball_pos_cols = {ball_pos_cols}')\nprint(f'ball_vel_cols = {ball_vel_cols}')\nprint(f'player_pos_cols = {player_pos_cols}')\nprint(f'boost_cols = {boost_cols}')\nprint(f'boost_timer_cols = {boost_timer_cols}')","metadata":{"execution":{"iopub.status.busy":"2022-10-03T06:00:39.150453Z","iopub.execute_input":"2022-10-03T06:00:39.150852Z","iopub.status.idle":"2022-10-03T06:00:39.162164Z","shell.execute_reply.started":"2022-10-03T06:00:39.150814Z","shell.execute_reply":"2022-10-03T06:00:39.160852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_cols = ball_pos_cols + ball_vel_cols + player_pos_cols + player_vel_cols + boost_cols + boost_timer_cols \n#+ dist_p_ball_cols\nprint(f'x_cols = {x_cols}')","metadata":{"execution":{"iopub.status.busy":"2022-10-03T06:00:39.163615Z","iopub.execute_input":"2022-10-03T06:00:39.164366Z","iopub.status.idle":"2022-10-03T06:00:39.174240Z","shell.execute_reply.started":"2022-10-03T06:00:39.164325Z","shell.execute_reply":"2022-10-03T06:00:39.172462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_cols = ['team_A_scoring_within_10sec', 'team_B_scoring_within_10sec']\nprint(f'y_cols = {y_cols}')","metadata":{"execution":{"iopub.status.busy":"2022-10-03T06:00:39.176052Z","iopub.execute_input":"2022-10-03T06:00:39.176440Z","iopub.status.idle":"2022-10-03T06:00:39.186221Z","shell.execute_reply.started":"2022-10-03T06:00:39.176406Z","shell.execute_reply":"2022-10-03T06:00:39.184949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Fitting and Predicting Team A","metadata":{}},{"cell_type":"markdown","source":"We will use the first training file to train the model from scratch. Subsequent files are fed to the fitted model from the previous iteration, i.e., boosting. We keep all the hyperparameter constant from each iterations.  We set max iteration high and rely on early termination to terminate boosting.","metadata":{}},{"cell_type":"code","source":"MAX_ITER = 10\nPATIENCE = 3\nSEED = 42","metadata":{"execution":{"iopub.status.busy":"2022-10-03T06:00:39.188043Z","iopub.execute_input":"2022-10-03T06:00:39.188412Z","iopub.status.idle":"2022-10-03T06:00:39.196706Z","shell.execute_reply.started":"2022-10-03T06:00:39.188370Z","shell.execute_reply":"2022-10-03T06:00:39.195545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MODEL_PARAMS = {\n    'n_d': 8,\n    'n_a': 8,\n    'n_steps': 3,\n    'gamma': 1.3,                      \n    'n_independent': 2,\n    'n_shared': 2,\n    'momentum': 0.02,\n    'clip_value': None,\n    'lambda_sparse': 1e-5,\n    'optimizer_fn': torch.optim.Adam,\n    'scheduler_fn': torch.optim.lr_scheduler.CosineAnnealingLR,\n    'scheduler_params': {\"T_max\" : 6},\n    'mask_type': 'sparsemax',\n    'seed': SEED}\n\nmdl_A = TabNetClassifier(**MODEL_PARAMS)","metadata":{"execution":{"iopub.status.busy":"2022-10-03T06:00:39.198644Z","iopub.execute_input":"2022-10-03T06:00:39.199562Z","iopub.status.idle":"2022-10-03T06:00:39.226177Z","shell.execute_reply.started":"2022-10-03T06:00:39.199520Z","shell.execute_reply":"2022-10-03T06:00:39.224975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mdl_A.fit(train[x_cols].fillna(-1).values, train[y_cols[0]].values,\n          eval_set=[(valid[x_cols].fillna(-1).values, valid[y_cols[0]].values)],\n          eval_metric=['logloss'],\n          max_epochs = MAX_ITER,\n          patience = PATIENCE)","metadata":{"execution":{"iopub.status.busy":"2022-10-03T06:00:39.227529Z","iopub.execute_input":"2022-10-03T06:00:39.227875Z","iopub.status.idle":"2022-10-03T06:37:43.335110Z","shell.execute_reply.started":"2022-10-03T06:00:39.227844Z","shell.execute_reply":"2022-10-03T06:37:43.333588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in train_files[1:]:\n    print(f'Iteration {i}, Reading train_{i}.csv')\n    train = pd.read_csv(os.path.join(DAT_DIR, f'train_{i}.csv'))\n    #train = add_dist_cols(train)\n    print('Fitting model')\n    mdl_A.fit(train[x_cols].fillna(-1).values, \n              train[y_cols[0]].values,\n              eval_set=[(valid[x_cols].fillna(-1).values, valid[y_cols[0]].values)],\n              eval_metric=['logloss'],\n              max_epochs = MAX_ITER,\n              patience = PATIENCE,\n              warm_start = True)","metadata":{"execution":{"iopub.status.busy":"2022-10-03T06:37:43.336928Z","iopub.execute_input":"2022-10-03T06:37:43.337424Z","iopub.status.idle":"2022-10-03T09:47:06.056644Z","shell.execute_reply.started":"2022-10-03T06:37:43.337375Z","shell.execute_reply":"2022-10-03T09:47:06.054938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in test_files:\n    test = pd.read_csv(os.path.join(DAT_DIR, f\"train_{i}.csv\"))\n    #test = add_dist_cols(test)\n    pred_probs = mdl_A.predict_proba(test[x_cols].fillna(-1).values)[:, 1]\n    test['pred_probs'] = pred_probs\n    print(f'LogLoss of test on train_{i}.csv: {round(log_loss(test[\"team_A_scoring_within_10sec\"], test[\"pred_probs\"]), 3)}')","metadata":{"execution":{"iopub.status.busy":"2022-10-03T09:47:06.058881Z","iopub.execute_input":"2022-10-03T09:47:06.059282Z","iopub.status.idle":"2022-10-03T09:50:24.156638Z","shell.execute_reply.started":"2022-10-03T09:47:06.059249Z","shell.execute_reply":"2022-10-03T09:50:24.155360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let us check the importance of variables.","metadata":{}},{"cell_type":"code","source":"mdl_A_imp = pd.DataFrame({\n    'feature_importance': mdl_A.feature_importances_, \n    'feature_names': x_cols}).sort_values(by=['feature_importance'], ascending=False)\n\nsns.barplot(data=mdl_A_imp.iloc[:20, :], y='feature_names', x='feature_importance')","metadata":{"execution":{"iopub.status.busy":"2022-10-03T09:50:24.158320Z","iopub.execute_input":"2022-10-03T09:50:24.158804Z","iopub.status.idle":"2022-10-03T09:50:24.610651Z","shell.execute_reply.started":"2022-10-03T09:50:24.158756Z","shell.execute_reply":"2022-10-03T09:50:24.609418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It is quite interesting to note that the feature importances are not what I have assumed in my [EDA notebook](https://www.kaggle.com/code/stautxie/tabular-challenge-oct-2022-eda). Therefore, I had assumed $y$-coordinates should be the most relevant. However, the lightgbm model seems to suggest the $x$-coordinates of ball as well as boost items are the most important.","metadata":{}},{"cell_type":"markdown","source":"### Fitting and Predicting Team B","metadata":{}},{"cell_type":"markdown","source":"We apply the same approach to predict scoring probability of team B in the next 10 seconds.","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(os.path.join(DAT_DIR, \"train_0.csv\"))\n#train = add_dist_cols(train)","metadata":{"execution":{"iopub.status.busy":"2022-10-03T09:50:24.612526Z","iopub.execute_input":"2022-10-03T09:50:24.613019Z","iopub.status.idle":"2022-10-03T09:51:07.634782Z","shell.execute_reply.started":"2022-10-03T09:50:24.612972Z","shell.execute_reply":"2022-10-03T09:51:07.633283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid = pd.read_csv(os.path.join(DAT_DIR, f\"train_{valid_files[0]}.csv\"))\n#valid = add_dist_cols(valid)","metadata":{"execution":{"iopub.status.busy":"2022-10-03T09:51:07.636659Z","iopub.execute_input":"2022-10-03T09:51:07.637753Z","iopub.status.idle":"2022-10-03T09:51:52.739556Z","shell.execute_reply.started":"2022-10-03T09:51:07.637710Z","shell.execute_reply":"2022-10-03T09:51:52.737983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mdl_B = TabNetClassifier(**MODEL_PARAMS)\n\nmdl_B.fit(train[x_cols].fillna(-1).values, train[y_cols[1]].values,\n          eval_set=[(valid[x_cols].fillna(-1).values, valid[y_cols[1]].values)],\n          eval_metric=['logloss'],\n          max_epochs = MAX_ITER,\n          patience = PATIENCE)","metadata":{"execution":{"iopub.status.busy":"2022-10-03T09:51:52.741556Z","iopub.execute_input":"2022-10-03T09:51:52.741917Z","iopub.status.idle":"2022-10-03T09:51:52.899808Z","shell.execute_reply.started":"2022-10-03T09:51:52.741886Z","shell.execute_reply":"2022-10-03T09:51:52.897120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in train_files[1:]:\n    print(f'Iteration {i}, Reading train_{i}.csv')\n    train = pd.read_csv(os.path.join(DAT_DIR, f'train_{i}.csv'))\n    #train = add_dist_cols(train)\n    print('Fitting model')\n    mdl_B.fit(train[x_cols].fillna(-1).values, \n              train[y_cols[1]].values,\n              eval_set=[(valid[x_cols].fillna(-1).values, valid[y_cols[1]].values)],\n              eval_metric=['logloss'],\n              max_epochs = MAX_ITER,\n              patience = PATIENCE,\n              warm_start = True)","metadata":{"execution":{"iopub.status.busy":"2022-10-03T09:51:52.900791Z","iopub.status.idle":"2022-10-03T09:51:52.901205Z","shell.execute_reply.started":"2022-10-03T09:51:52.901010Z","shell.execute_reply":"2022-10-03T09:51:52.901029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in test_files:\n    test = pd.read_csv(os.path.join(DAT_DIR, f\"train_{i}.csv\"))\n    #test = add_dist_cols(test)\n    pred_probs = mdl_B.predict_proba(test[x_cols].fillna(-1).values)[:, 1]\n    test['pred_probs'] = pred_probs\n    print(f'LogLoss of test on train_{i}.csv: {round(log_loss(test[\"team_B_scoring_within_10sec\"], test[\"pred_probs\"]), 3)}')","metadata":{"execution":{"iopub.status.busy":"2022-10-03T09:51:52.903626Z","iopub.status.idle":"2022-10-03T09:51:52.904080Z","shell.execute_reply.started":"2022-10-03T09:51:52.903874Z","shell.execute_reply":"2022-10-03T09:51:52.903895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mdl_B_imp = pd.DataFrame({\n    'feature_importance': mdl_B.feature_importances_, \n    'feature_names': x_cols}).sort_values(by=['feature_importance'], ascending=False)\n\nsns.barplot(data=mdl_B_imp.iloc[:20, :], y='feature_names', x='feature_importance')","metadata":{"execution":{"iopub.status.busy":"2022-10-03T09:51:52.905588Z","iopub.status.idle":"2022-10-03T09:51:52.905992Z","shell.execute_reply.started":"2022-10-03T09:51:52.905803Z","shell.execute_reply":"2022-10-03T09:51:52.905822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We see that the prediction quality and variable importance of the team B's are very similar to that of the team As.","metadata":{}},{"cell_type":"markdown","source":"### Predicting on Test","metadata":{}},{"cell_type":"code","source":"final_test = pd.read_csv(os.path.join(DAT_DIR, \"test.csv\"))\nsample_submit = pd.read_csv(os.path.join(DAT_DIR, \"sample_submission.csv\"))","metadata":{"execution":{"iopub.status.busy":"2022-10-03T09:51:52.908403Z","iopub.status.idle":"2022-10-03T09:51:52.908879Z","shell.execute_reply.started":"2022-10-03T09:51:52.908670Z","shell.execute_reply":"2022-10-03T09:51:52.908692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#final_test = add_dist_cols(final_test)\nfinal_test","metadata":{"execution":{"iopub.status.busy":"2022-10-03T09:51:52.911050Z","iopub.status.idle":"2022-10-03T09:51:52.911587Z","shell.execute_reply.started":"2022-10-03T09:51:52.911327Z","shell.execute_reply":"2022-10-03T09:51:52.911350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submit[\"team_A_scoring_within_10sec\"] = mdl_A.predict_proba(final_test[x_cols].fillna(-1).values)[:,1]\nsample_submit[\"team_B_scoring_within_10sec\"] = mdl_B.predict_proba(final_test[x_cols].fillna(-1).values)[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-10-03T09:51:52.913283Z","iopub.status.idle":"2022-10-03T09:51:52.913742Z","shell.execute_reply.started":"2022-10-03T09:51:52.913535Z","shell.execute_reply":"2022-10-03T09:51:52.913556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submit","metadata":{"execution":{"iopub.status.busy":"2022-10-03T09:51:52.914736Z","iopub.status.idle":"2022-10-03T09:51:52.915144Z","shell.execute_reply.started":"2022-10-03T09:51:52.914954Z","shell.execute_reply":"2022-10-03T09:51:52.914973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submit.to_csv(\"submission.csv\", index=None)","metadata":{"execution":{"iopub.status.busy":"2022-10-03T09:51:52.916704Z","iopub.status.idle":"2022-10-03T09:51:52.917085Z","shell.execute_reply.started":"2022-10-03T09:51:52.916903Z","shell.execute_reply":"2022-10-03T09:51:52.916921Z"},"trusted":true},"execution_count":null,"outputs":[]}]}