{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Tabular Playground Oct 2022 - Baseline","metadata":{}},{"cell_type":"markdown","source":"In this notebook, we will use catboost to fit a model.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport gc\nfrom itertools import chain\nimport catboost\nfrom sklearn.metrics import log_loss\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-10-02T19:56:27.787299Z","iopub.execute_input":"2022-10-02T19:56:27.787704Z","iopub.status.idle":"2022-10-02T19:56:28.461195Z","shell.execute_reply.started":"2022-10-02T19:56:27.787672Z","shell.execute_reply":"2022-10-02T19:56:28.460141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DAT_DIR = '../input/tabular-playground-series-oct-2022'","metadata":{"execution":{"iopub.status.busy":"2022-10-02T19:56:28.462912Z","iopub.execute_input":"2022-10-02T19:56:28.463451Z","iopub.status.idle":"2022-10-02T19:56:28.467710Z","shell.execute_reply.started":"2022-10-02T19:56:28.463418Z","shell.execute_reply":"2022-10-02T19:56:28.466930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(os.path.join(DAT_DIR, \"train_0.csv\"))","metadata":{"execution":{"iopub.status.busy":"2022-10-02T19:56:28.468884Z","iopub.execute_input":"2022-10-02T19:56:28.469423Z","iopub.status.idle":"2022-10-02T19:57:02.001941Z","shell.execute_reply.started":"2022-10-02T19:56:28.469391Z","shell.execute_reply":"2022-10-02T19:57:02.000536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we will train the model by batch, we will divide the training files into categories of training, validation, and (out-of-sample) testing. The last category is to assess the model performance free of biases as they have never been seen by the model before. ","metadata":{}},{"cell_type":"code","source":"train_files = list(range(7))\nvalid_files = list(range(7,8))\ntest_files = list(range(8, 10))","metadata":{"execution":{"iopub.status.busy":"2022-10-02T19:57:02.004170Z","iopub.execute_input":"2022-10-02T19:57:02.004656Z","iopub.status.idle":"2022-10-02T19:57:02.010324Z","shell.execute_reply.started":"2022-10-02T19:57:02.004596Z","shell.execute_reply":"2022-10-02T19:57:02.009048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'train_files = {train_files}')\nprint(f'valid_files = {valid_files}')\nprint(f'test_files = {test_files}')","metadata":{"execution":{"iopub.status.busy":"2022-10-02T19:57:02.013010Z","iopub.execute_input":"2022-10-02T19:57:02.014027Z","iopub.status.idle":"2022-10-02T19:57:02.023335Z","shell.execute_reply.started":"2022-10-02T19:57:02.013988Z","shell.execute_reply":"2022-10-02T19:57:02.022051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid = pd.read_csv(os.path.join(DAT_DIR, f\"train_{valid_files[0]}.csv\"))","metadata":{"execution":{"iopub.status.busy":"2022-10-02T19:57:02.025547Z","iopub.execute_input":"2022-10-02T19:57:02.026520Z","iopub.status.idle":"2022-10-02T19:57:35.071744Z","shell.execute_reply.started":"2022-10-02T19:57:02.026333Z","shell.execute_reply":"2022-10-02T19:57:35.069761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid","metadata":{"execution":{"iopub.status.busy":"2022-10-02T19:57:35.073983Z","iopub.execute_input":"2022-10-02T19:57:35.074485Z","iopub.status.idle":"2022-10-02T19:57:35.487025Z","shell.execute_reply.started":"2022-10-02T19:57:35.074444Z","shell.execute_reply":"2022-10-02T19:57:35.486169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature Engineering","metadata":{}},{"cell_type":"markdown","source":"We want to add the distance between the ball and players as features. We add 3-d Euclidean distances.","metadata":{}},{"cell_type":"code","source":"dist_p_ball_cols = [f'dist_p_ball_{i}' for i in range(6)]","metadata":{"execution":{"iopub.status.busy":"2022-10-02T19:57:35.488741Z","iopub.execute_input":"2022-10-02T19:57:35.489436Z","iopub.status.idle":"2022-10-02T19:57:35.494588Z","shell.execute_reply.started":"2022-10-02T19:57:35.489395Z","shell.execute_reply":"2022-10-02T19:57:35.493434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'dist_p_ball_cols = {dist_p_ball_cols}')","metadata":{"execution":{"iopub.status.busy":"2022-10-02T19:57:35.495776Z","iopub.execute_input":"2022-10-02T19:57:35.496134Z","iopub.status.idle":"2022-10-02T19:57:35.511456Z","shell.execute_reply.started":"2022-10-02T19:57:35.496062Z","shell.execute_reply":"2022-10-02T19:57:35.510157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_dist_cols(df):\n    for i in range(6):\n        df[f'dist_p_ball_{i}'] = np.sqrt((df.ball_pos_x - df[f'p{i}_pos_x'])**2 + (df.ball_pos_y - df[f'p{i}_pos_y'])**2 + (df.ball_pos_z - df[f'p{i}_pos_z'])**2)\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-10-02T19:57:35.513194Z","iopub.execute_input":"2022-10-02T19:57:35.514398Z","iopub.status.idle":"2022-10-02T19:57:35.521640Z","shell.execute_reply.started":"2022-10-02T19:57:35.514342Z","shell.execute_reply":"2022-10-02T19:57:35.520618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = add_dist_cols(train)","metadata":{"execution":{"iopub.status.busy":"2022-10-02T19:57:35.522998Z","iopub.execute_input":"2022-10-02T19:57:35.523588Z","iopub.status.idle":"2022-10-02T19:57:36.106894Z","shell.execute_reply.started":"2022-10-02T19:57:35.523553Z","shell.execute_reply":"2022-10-02T19:57:36.106061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid = add_dist_cols(valid)","metadata":{"execution":{"iopub.status.busy":"2022-10-02T19:57:36.108032Z","iopub.execute_input":"2022-10-02T19:57:36.108586Z","iopub.status.idle":"2022-10-02T19:57:36.659710Z","shell.execute_reply.started":"2022-10-02T19:57:36.108549Z","shell.execute_reply":"2022-10-02T19:57:36.658801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ball_pos_cols = [f'ball_pos_{i}' for i in ['x', 'y', 'z']]\nball_vel_cols = [f'ball_vel_{i}' for i in ['x', 'y', 'z']]\nplayer_pos_cols = list(chain.from_iterable([[f'p{i}_pos_{j}' for j in ['x', 'y', 'z']] for i in range(6)]))\nplayer_vel_cols = list(chain.from_iterable([[f'p{i}_vel_{j}' for j in ['x', 'y', 'z']] for i in range(6)]))\nboost_cols = [f'p{i}_boost' for i in range(6)]\nboost_timer_cols = [f'boost{i}_timer' for i in range(6)]","metadata":{"execution":{"iopub.status.busy":"2022-10-02T19:57:36.661060Z","iopub.execute_input":"2022-10-02T19:57:36.661395Z","iopub.status.idle":"2022-10-02T19:57:36.669638Z","shell.execute_reply.started":"2022-10-02T19:57:36.661365Z","shell.execute_reply":"2022-10-02T19:57:36.668405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'ball_pos_cols = {ball_pos_cols}')\nprint(f'ball_vel_cols = {ball_vel_cols}')\nprint(f'player_pos_cols = {player_pos_cols}')\nprint(f'boost_cols = {boost_cols}')\nprint(f'boost_timer_cols = {boost_timer_cols}')","metadata":{"execution":{"iopub.status.busy":"2022-10-02T19:57:36.675742Z","iopub.execute_input":"2022-10-02T19:57:36.676557Z","iopub.status.idle":"2022-10-02T19:57:36.682577Z","shell.execute_reply.started":"2022-10-02T19:57:36.676514Z","shell.execute_reply":"2022-10-02T19:57:36.681404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_cols = ball_pos_cols + ball_vel_cols + player_pos_cols + player_vel_cols + boost_cols + boost_timer_cols + dist_p_ball_cols\nprint(f'x_cols = {x_cols}')","metadata":{"execution":{"iopub.status.busy":"2022-10-02T19:57:36.684011Z","iopub.execute_input":"2022-10-02T19:57:36.685044Z","iopub.status.idle":"2022-10-02T19:57:36.694711Z","shell.execute_reply.started":"2022-10-02T19:57:36.685002Z","shell.execute_reply":"2022-10-02T19:57:36.693772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_cols = ['team_A_scoring_within_10sec', 'team_B_scoring_within_10sec']\nprint(f'y_cols = {y_cols}')","metadata":{"execution":{"iopub.status.busy":"2022-10-02T19:57:36.696136Z","iopub.execute_input":"2022-10-02T19:57:36.696688Z","iopub.status.idle":"2022-10-02T19:57:36.705952Z","shell.execute_reply.started":"2022-10-02T19:57:36.696654Z","shell.execute_reply":"2022-10-02T19:57:36.704885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Fitting and Predicting Team A","metadata":{}},{"cell_type":"markdown","source":"We will use the first training file to train the model from scratch. Subsequent files are fed to the fitted model from the previous iteration, i.e., boosting. We keep all the hyperparameter constant from each iterations.  We set max iteration high and rely on early termination to terminate boosting.","metadata":{}},{"cell_type":"code","source":"MAX_ITER = 1000\nPATIENCE = 100\nDISPLAY_FREQ = 100","metadata":{"execution":{"iopub.status.busy":"2022-10-02T19:57:36.707430Z","iopub.execute_input":"2022-10-02T19:57:36.708166Z","iopub.status.idle":"2022-10-02T19:57:36.717833Z","shell.execute_reply.started":"2022-10-02T19:57:36.708131Z","shell.execute_reply":"2022-10-02T19:57:36.716528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MODEL_PARAMS = {'random_seed': 1234,    \n                'learning_rate': 0.05,                \n                'iterations': MAX_ITER,\n                'early_stopping_rounds': PATIENCE,\n                'metric_period': DISPLAY_FREQ,\n                'use_best_model': True,\n                'eval_metric': 'Logloss'}\n\nmdl_A = catboost.CatBoostClassifier(**MODEL_PARAMS)","metadata":{"execution":{"iopub.status.busy":"2022-10-02T20:08:37.961828Z","iopub.execute_input":"2022-10-02T20:08:37.962316Z","iopub.status.idle":"2022-10-02T20:08:37.968503Z","shell.execute_reply.started":"2022-10-02T20:08:37.962274Z","shell.execute_reply":"2022-10-02T20:08:37.967474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mdl_A.fit(X=train[x_cols], y=train[y_cols[0]],\n          eval_set=[(valid[x_cols], valid[y_cols[0]])],\n          early_stopping_rounds = PATIENCE,\n          metric_period = DISPLAY_FREQ)","metadata":{"execution":{"iopub.status.busy":"2022-10-02T20:08:44.557101Z","iopub.execute_input":"2022-10-02T20:08:44.557537Z","iopub.status.idle":"2022-10-02T20:17:07.301648Z","shell.execute_reply.started":"2022-10-02T20:08:44.557504Z","shell.execute_reply":"2022-10-02T20:17:07.300317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in train_files[1:]:\n    print(f'Iteration {i}, Reading train_{i}.csv')\n    train = pd.read_csv(os.path.join(DAT_DIR, f'train_{i}.csv'))\n    train = add_dist_cols(train)\n    print('Fitting model')\n    mdl_A.fit(X=train[x_cols], \n              y=train[y_cols[0]],\n              eval_set=[(valid[x_cols], valid[y_cols[0]])],\n             init_model=mdl_A)","metadata":{"execution":{"iopub.status.busy":"2022-10-02T20:19:16.351507Z","iopub.execute_input":"2022-10-02T20:19:16.352018Z","iopub.status.idle":"2022-10-02T20:36:42.878444Z","shell.execute_reply.started":"2022-10-02T20:19:16.351980Z","shell.execute_reply":"2022-10-02T20:36:42.876376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in test_files:\n    test = pd.read_csv(os.path.join(DAT_DIR, f\"train_{i}.csv\"))\n    test = add_dist_cols(test)\n    pred_probs = mdl_A.predict_proba(test[x_cols])[:, 1]\n    test['pred_probs'] = pred_probs\n    print(f'LogLoss of test on train_{i}.csv: {round(log_loss(test[\"team_A_scoring_within_10sec\"], test[\"pred_probs\"]), 3)}')","metadata":{"execution":{"iopub.status.busy":"2022-10-02T20:47:06.071460Z","iopub.execute_input":"2022-10-02T20:47:06.072079Z","iopub.status.idle":"2022-10-02T20:48:29.322556Z","shell.execute_reply.started":"2022-10-02T20:47:06.072040Z","shell.execute_reply":"2022-10-02T20:48:29.321079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let us check the importance of variables.","metadata":{}},{"cell_type":"code","source":"mdl_A_imp = pd.DataFrame({\n    'feature_importance': mdl_A.get_feature_importance(), \n    'feature_names': x_cols}).sort_values(by=['feature_importance'], ascending=False)\n\nsns.barplot(data=mdl_A_imp.iloc[:20, :], y='feature_names', x='feature_importance')","metadata":{"execution":{"iopub.status.busy":"2022-10-02T20:49:20.271846Z","iopub.execute_input":"2022-10-02T20:49:20.272341Z","iopub.status.idle":"2022-10-02T20:49:20.997244Z","shell.execute_reply.started":"2022-10-02T20:49:20.272294Z","shell.execute_reply":"2022-10-02T20:49:20.996037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It is quite interesting to note that the feature importances are not what I have assumed in my [EDA notebook](https://www.kaggle.com/code/stautxie/tabular-challenge-oct-2022-eda). Therefore, I had assumed $y$-coordinates should be the most relevant. However, the lightgbm model seems to suggest the $x$-coordinates of ball as well as boost items are the most important.","metadata":{}},{"cell_type":"markdown","source":"### Fitting and Predicting Team B","metadata":{}},{"cell_type":"markdown","source":"We apply the same approach to predict scoring probability of team B in the next 10 seconds.","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(os.path.join(DAT_DIR, \"train_0.csv\"))\ntrain = add_dist_cols(train)","metadata":{"execution":{"iopub.status.busy":"2022-10-02T20:49:57.791595Z","iopub.execute_input":"2022-10-02T20:49:57.792148Z","iopub.status.idle":"2022-10-02T20:50:31.412379Z","shell.execute_reply.started":"2022-10-02T20:49:57.792075Z","shell.execute_reply":"2022-10-02T20:50:31.411085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid = pd.read_csv(os.path.join(DAT_DIR, f\"train_{valid_files[0]}.csv\"))\nvalid = add_dist_cols(valid)","metadata":{"execution":{"iopub.status.busy":"2022-10-02T20:50:31.415112Z","iopub.execute_input":"2022-10-02T20:50:31.415524Z","iopub.status.idle":"2022-10-02T20:51:03.838639Z","shell.execute_reply.started":"2022-10-02T20:50:31.415484Z","shell.execute_reply":"2022-10-02T20:51:03.837238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mdl_B = catboost.CatBoostClassifier(**MODEL_PARAMS)\n\nmdl_B.fit(X=train[x_cols], y=train[y_cols[1]],\n          eval_set=[(valid[x_cols], valid[y_cols[1]])],\n          early_stopping_rounds = PATIENCE,\n          metric_period = DISPLAY_FREQ)","metadata":{"execution":{"iopub.status.busy":"2022-10-02T20:59:19.962349Z","iopub.execute_input":"2022-10-02T20:59:19.962790Z","iopub.status.idle":"2022-10-02T21:07:15.229244Z","shell.execute_reply.started":"2022-10-02T20:59:19.962754Z","shell.execute_reply":"2022-10-02T21:07:15.227697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in train_files[1:]:\n    print(f'Iteration {i}, Reading train_{i}.csv')\n    train = pd.read_csv(os.path.join(DAT_DIR, f'train_{i}.csv'))\n    train = add_dist_cols(train)\n    print('Fitting model')\n    mdl_B.fit(X=train[x_cols], \n              y=train[y_cols[1]],\n              eval_set=[(valid[x_cols], valid[y_cols[1]])],\n             init_model=mdl_B)","metadata":{"execution":{"iopub.status.busy":"2022-10-02T21:08:58.802486Z","iopub.execute_input":"2022-10-02T21:08:58.802986Z","iopub.status.idle":"2022-10-02T21:25:59.466407Z","shell.execute_reply.started":"2022-10-02T21:08:58.802944Z","shell.execute_reply":"2022-10-02T21:25:59.464627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in test_files:\n    test = pd.read_csv(os.path.join(DAT_DIR, f\"train_{i}.csv\"))\n    test = add_dist_cols(test)\n    pred_probs = mdl_B.predict_proba(test[x_cols])[:, 1]\n    test['pred_probs'] = pred_probs\n    print(f'LogLoss of test on train_{i}.csv: {round(log_loss(test[\"team_B_scoring_within_10sec\"], test[\"pred_probs\"]), 3)}')","metadata":{"execution":{"iopub.status.busy":"2022-10-02T21:27:28.947603Z","iopub.execute_input":"2022-10-02T21:27:28.948145Z","iopub.status.idle":"2022-10-02T21:28:44.495660Z","shell.execute_reply.started":"2022-10-02T21:27:28.948073Z","shell.execute_reply":"2022-10-02T21:28:44.494800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mdl_B_imp = pd.DataFrame({\n    'feature_importance': mdl_B.get_feature_importance(), \n    'feature_names': x_cols}).sort_values(by=['feature_importance'], ascending=False)\n\nsns.barplot(data=mdl_B_imp.iloc[:20, :], y='feature_names', x='feature_importance')","metadata":{"execution":{"iopub.status.busy":"2022-10-02T21:28:54.644046Z","iopub.execute_input":"2022-10-02T21:28:54.644507Z","iopub.status.idle":"2022-10-02T21:28:54.970753Z","shell.execute_reply.started":"2022-10-02T21:28:54.644468Z","shell.execute_reply":"2022-10-02T21:28:54.969395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We see that the prediction quality and variable importance of the team B's are very similar to that of the team As.","metadata":{}},{"cell_type":"markdown","source":"### Predicting on Test","metadata":{}},{"cell_type":"code","source":"final_test = pd.read_csv(os.path.join(DAT_DIR, \"test.csv\"))\nsample_submit = pd.read_csv(os.path.join(DAT_DIR, \"sample_submission.csv\"))","metadata":{"execution":{"iopub.status.busy":"2022-10-02T21:28:59.003284Z","iopub.execute_input":"2022-10-02T21:28:59.003737Z","iopub.status.idle":"2022-10-02T21:29:09.700988Z","shell.execute_reply.started":"2022-10-02T21:28:59.003698Z","shell.execute_reply":"2022-10-02T21:29:09.699550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_test = add_dist_cols(final_test)\nfinal_test","metadata":{"execution":{"iopub.status.busy":"2022-10-02T21:29:33.229723Z","iopub.execute_input":"2022-10-02T21:29:33.230776Z","iopub.status.idle":"2022-10-02T21:29:33.468982Z","shell.execute_reply.started":"2022-10-02T21:29:33.230717Z","shell.execute_reply":"2022-10-02T21:29:33.467777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submit[\"team_A_scoring_within_10sec\"] = mdl_A.predict_proba(final_test[x_cols])[:,1]\nsample_submit[\"team_B_scoring_within_10sec\"] = mdl_B.predict_proba(final_test[x_cols])[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-10-02T21:29:41.948765Z","iopub.execute_input":"2022-10-02T21:29:41.949210Z","iopub.status.idle":"2022-10-02T21:29:44.174563Z","shell.execute_reply.started":"2022-10-02T21:29:41.949175Z","shell.execute_reply":"2022-10-02T21:29:44.173166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submit","metadata":{"execution":{"iopub.status.busy":"2022-10-02T21:29:50.484648Z","iopub.execute_input":"2022-10-02T21:29:50.485179Z","iopub.status.idle":"2022-10-02T21:29:50.500960Z","shell.execute_reply.started":"2022-10-02T21:29:50.485067Z","shell.execute_reply":"2022-10-02T21:29:50.499273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submit.to_csv(\"submission.csv\", index=None)","metadata":{"execution":{"iopub.status.busy":"2022-10-02T21:30:01.154805Z","iopub.execute_input":"2022-10-02T21:30:01.155220Z","iopub.status.idle":"2022-10-02T21:30:04.117910Z","shell.execute_reply.started":"2022-10-02T21:30:01.155185Z","shell.execute_reply":"2022-10-02T21:30:04.115155Z"},"trusted":true},"execution_count":null,"outputs":[]}]}