{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# TPS Oct 2022\nThe Oct edition of the 2022 Tabular Playground Series has as argument Rocket League matchs!!!. Given a snapshot from a Rocket League match, we need to predict the probability of each team scoring within the next 10 seconds of the game.\n\n<img src=\"https://i.postimg.cc/cJZcdTGn/Tournaments-Prematch-Lobby-v2.webp\"/>","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n!pip install mrmr_selection\n\nfrom mrmr import mrmr_classif\n\nimport matplotlib.pyplot as plt\nimport random \nfrom numpy import dtype\nimport matplotlib.pyplot as plt\nfrom pandas import DataFrame\nimport seaborn as sns\nimport gc\nfrom lightgbm import LGBMClassifier\nfrom catboost import CatBoostClassifier\nfrom sklearn.ensemble import ExtraTreesClassifier\nfrom xgboost import XGBClassifier\nfrom sklearn.metrics import accuracy_score, roc_auc_score, log_loss\nfrom sklearn.model_selection import KFold, StratifiedKFold, GroupKFold\n\n#ignore warning messages \nimport warnings\nwarnings.filterwarnings('ignore') \n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-10-14T05:42:02.734499Z","iopub.execute_input":"2022-10-14T05:42:02.734826Z","iopub.status.idle":"2022-10-14T05:42:11.004784Z","shell.execute_reply.started":"2022-10-14T05:42:02.734800Z","shell.execute_reply":"2022-10-14T05:42:11.003627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train files\n- we've 10 big train files. In order to speed up the EDA we are going to use only the first one.\n- instead the **model that we are going to train will use ALL available files**","metadata":{}},{"cell_type":"code","source":"dtypes_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_dtypes.csv')\ndtypes = {k: v for (k, v) in zip(dtypes_df.column, dtypes_df.dtype)}\nprint(f\"Loading train set 0\")\ntrain0_df = pd.read_csv(f\"/kaggle/input/tabular-playground-series-oct-2022/train_0.csv\", dtype=dtypes)\ntrain_df = train0_df\nprint(f\"Loading test set\")\ntest_df = pd.read_csv(f'/kaggle/input/tabular-playground-series-oct-2022/test.csv', dtype=dtypes)\nsubmission_df = pd.read_csv(f'/kaggle/input/tabular-playground-series-oct-2022/sample_submission.csv')\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:11.008269Z","iopub.execute_input":"2022-10-14T05:42:11.008553Z","iopub.status.idle":"2022-10-14T05:42:24.276217Z","shell.execute_reply.started":"2022-10-14T05:42:11.008525Z","shell.execute_reply":"2022-10-14T05:42:24.275289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:24.277344Z","iopub.execute_input":"2022-10-14T05:42:24.278026Z","iopub.status.idle":"2022-10-14T05:42:24.302031Z","shell.execute_reply.started":"2022-10-14T05:42:24.277999Z","shell.execute_reply":"2022-10-14T05:42:24.301129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:24.304570Z","iopub.execute_input":"2022-10-14T05:42:24.305286Z","iopub.status.idle":"2022-10-14T05:42:24.312763Z","shell.execute_reply.started":"2022-10-14T05:42:24.305257Z","shell.execute_reply":"2022-10-14T05:42:24.311955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:24.314516Z","iopub.execute_input":"2022-10-14T05:42:24.314784Z","iopub.status.idle":"2022-10-14T05:42:24.339343Z","shell.execute_reply.started":"2022-10-14T05:42:24.314757Z","shell.execute_reply":"2022-10-14T05:42:24.338480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:24.340408Z","iopub.execute_input":"2022-10-14T05:42:24.340663Z","iopub.status.idle":"2022-10-14T05:42:24.346505Z","shell.execute_reply.started":"2022-10-14T05:42:24.340640Z","shell.execute_reply":"2022-10-14T05:42:24.345577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Columns description:\n\n- **game_num** (train only): Unique identifier for the game from which the event was taken.\n- **event_id** (train only): Unique identifier for the sequence of consecutive frames.\n- **event_time** (train only): Time in seconds before the event ended, either by a goal being scored or simply when we decided to truncate the timeseries if a goal was not scored.\n- **ball_pos_[xyz]**: Ball's position as a 3d vector.\n- **ball_vel_[xyz]**: Ball's velocity as a 3d vector.\n- For i in [0, 6]:\n    - **p{i}_pos_[xyz]**: Player i's position as a 3d vector.\n    - **p{i}_vel_[xyz]**: Player i's velocity as a 3d vector.\n    - **p{i}_boost**: Player i's boost remaining, in [0, 100]. A player can consume boost to substantially increase their speed, and is required to fly up into the z dimension (besides driving up a wall, or the small air gained by a jump).\n    - **boost{i}_timer**: Time in seconds until big boost orb i respawns, or 0 if it's available. Big boost orbs grant a full 100 boost to a player driving over it. The orb (x, y) locations are roughly [ (-61.4, -81.9), (61.4, -81.9), (-71.7, 0), (71.7, 0), (-61.4, 81.9), (61.4, 81.9) ] with z = 0. (Players can also gain boost from small boost pads across the map, but we do not capture those pads in this dataset).\n- **player_scoring_next** (train only): Which player scores at the end of the current event, in [0, 6], or -1 if the event does not end in a goal.\n- **team_scoring_next** (train only): Which team scores at the end of the current event (A or B), or NaN if the event does not end in a goal.\n- **team_[A|B]_scoring_within_10sec** (train only): [Target columns] Value of 1 if team_scoring_next == [A|B] and time_before_event is in [-10, 0], otherwise 0.\n- **id** (test and submission only): Unique identifier for each test row. Your submission should be a pair of team_A_scoring_within_10sec and team_B_scoring_within_10sec probability predictions for each id, where your predictions can range the real numbers from [0, 1].\n- Players 0, 1, and 2 make up team A and players 3, 4, and 5 make up team B.\n- The orientation vector of the player's car (which way the car is facing) does not necessarily match the player's velocity vector, and this dataset does not capture orientation data.\n\n\n## Note\n- There are two features (UNAVAILABLE_FEATURES_ON_TEST) on train set not available on test set. for the moment we are going to skip them\n","metadata":{}},{"cell_type":"code","source":"UNAVAILABLE_FEATURES_ON_TEST = ['player_scoring_next', 'team_scoring_next']\ntrain_df = train_df.drop(UNAVAILABLE_FEATURES_ON_TEST, axis=1)\ntrain0_df = train0_df.drop(UNAVAILABLE_FEATURES_ON_TEST, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:24.347884Z","iopub.execute_input":"2022-10-14T05:42:24.348159Z","iopub.status.idle":"2022-10-14T05:42:24.569725Z","shell.execute_reply.started":"2022-10-14T05:42:24.348134Z","shell.execute_reply":"2022-10-14T05:42:24.568881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Seed all\nseed = 12\nrandom.seed(seed)\nos.environ[\"PYTHONHASHSEED\"] = str(seed)\nnp.random.seed(seed)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:24.571085Z","iopub.execute_input":"2022-10-14T05:42:24.571322Z","iopub.status.idle":"2022-10-14T05:42:24.771534Z","shell.execute_reply.started":"2022-10-14T05:42:24.571298Z","shell.execute_reply":"2022-10-14T05:42:24.770621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA - Exploratory Data Analysis","metadata":{}},{"cell_type":"code","source":"train_df.info(null_counts=True)","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:24.772688Z","iopub.execute_input":"2022-10-14T05:42:24.773055Z","iopub.status.idle":"2022-10-14T05:42:24.942869Z","shell.execute_reply.started":"2022-10-14T05:42:24.773019Z","shell.execute_reply":"2022-10-14T05:42:24.941965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# NaN Values\nThere are some features with NaN values; \nAll p{i} columns will be NaN if and only if the player is demolished (destroyed by an enemy player; will respawn within a few seconds).\n\n**How Manage them?**\n- we could impute them in some way like: mean, constant value, knn, linear regression or other\n- we could use a Tree-Model in order to manage them without impute\n\nWe choose the second one, so we'll train a LGBM Model","metadata":{}},{"cell_type":"markdown","source":"## Events distribution for single Games","metadata":{}},{"cell_type":"code","source":"f, axes = plt.subplots(nrows = 1, ncols = 2, figsize = (25, 10))\nlabels = [f\"Games with {i+1} events\" for i in range(10)]\ntrain_df.groupby(['game_num'])['event_id'].nunique().value_counts().plot(kind=\"pie\", ax=axes[0], labels=labels)\ntrain_df.groupby(['game_num'])['event_id'].nunique().value_counts().plot(kind=\"bar\", ax=axes[1])","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:24.946551Z","iopub.execute_input":"2022-10-14T05:42:24.947046Z","iopub.status.idle":"2022-10-14T05:42:25.343798Z","shell.execute_reply.started":"2022-10-14T05:42:24.947020Z","shell.execute_reply":"2022-10-14T05:42:25.342967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Frames count distribution for a Game","metadata":{}},{"cell_type":"code","source":"train_df.groupby(['game_num'])['event_id'].count().plot(kind=\"hist\", figsize=(30,5), bins=200, color=\"b\")","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:25.344876Z","iopub.execute_input":"2022-10-14T05:42:25.345137Z","iopub.status.idle":"2022-10-14T05:42:25.800762Z","shell.execute_reply.started":"2022-10-14T05:42:25.345111Z","shell.execute_reply":"2022-10-14T05:42:25.800188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Frames count distribution for Game/Event","metadata":{}},{"cell_type":"code","source":"train_df.groupby(['game_num'])['event_id'].value_counts().plot(kind=\"hist\", figsize=(30,5), bins=500, color=\"b\")","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:25.801757Z","iopub.execute_input":"2022-10-14T05:42:25.802143Z","iopub.status.idle":"2022-10-14T05:42:26.708350Z","shell.execute_reply.started":"2022-10-14T05:42:25.802119Z","shell.execute_reply":"2022-10-14T05:42:26.707546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Average and Minimum Event Time Duration","metadata":{}},{"cell_type":"code","source":"train_df.groupby(['game_num'])['event_time'].min().plot(kind=\"hist\", figsize=(30,5), bins=200, color=\"b\")\ntrain_df.groupby(['game_num'])['event_time'].mean().plot(kind=\"hist\", figsize=(30,5), bins=200, color=\"r\")","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:26.709206Z","iopub.execute_input":"2022-10-14T05:42:26.709439Z","iopub.status.idle":"2022-10-14T05:42:27.516227Z","shell.execute_reply.started":"2022-10-14T05:42:26.709416Z","shell.execute_reply":"2022-10-14T05:42:27.515346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Average Event Time difference between events_id on the same Game","metadata":{}},{"cell_type":"code","source":"train_df['event_time_diff'] = abs(train_df['event_time'].shift(1).fillna(method='ffill')-train_df['event_time'])\naverage_event_time_diff_by_game = train_df.groupby(['game_num'])['event_time_diff'].mean()\naverage_event_time_diff_by_game.plot(kind=\"hist\") ","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:27.517459Z","iopub.execute_input":"2022-10-14T05:42:27.518022Z","iopub.status.idle":"2022-10-14T05:42:27.718516Z","shell.execute_reply.started":"2022-10-14T05:42:27.517997Z","shell.execute_reply":"2022-10-14T05:42:27.717879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Ball Position in the field when Team A or B make a goal within 10 sec.","metadata":{}},{"cell_type":"code","source":"# Team A field vs Team B Field\nf, axes = plt.subplots(nrows = 1, ncols = 2, figsize = (15, 10))\naxes[0].title.set_text(f'Ball Position when Team A scoring within 10 sec')\naxes[0].scatter(train_df[train_df['team_A_scoring_within_10sec']==1]['ball_pos_x'], train_df[train_df['team_A_scoring_within_10sec']==1]['ball_pos_y'], s=0.1)\naxes[1].title.set_text(f'Ball Position when Team B scoring within 10 sec')\naxes[1].scatter(train_df[train_df['team_B_scoring_within_10sec']==1]['ball_pos_x'], train_df[train_df['team_B_scoring_within_10sec']==1]['ball_pos_y'], s=0.1, c=\"red\")\n","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:27.719762Z","iopub.execute_input":"2022-10-14T05:42:27.720524Z","iopub.status.idle":"2022-10-14T05:42:28.532239Z","shell.execute_reply.started":"2022-10-14T05:42:27.720486Z","shell.execute_reply":"2022-10-14T05:42:28.531482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Team A and B Position in the field when itself score within 10 sec","metadata":{}},{"cell_type":"code","source":"f, axes = plt.subplots(nrows = 1, ncols = 4, figsize = (20, 7))\nf.suptitle(\"Team A positions distribution when itself scoring within 10 sec.\")\nfor p in range(3):\n    axes[p].title.set_text(f'Player {p}')\n    axes[p].scatter(train_df[train_df['team_A_scoring_within_10sec']==1][f'p{p}_pos_x'], train_df[train_df['team_A_scoring_within_10sec']==1][f'p{p}_pos_y'], s=0.05)\naxes[3].title.set_text(f'Ball Position')\naxes[3].scatter(train_df[train_df['team_A_scoring_within_10sec']==1]['ball_pos_x'], train_df[train_df['team_A_scoring_within_10sec']==1]['ball_pos_y'], s=0.05)\n    \nf, axes = plt.subplots(nrows = 1, ncols = 4, figsize = (20, 7))\nf.suptitle(\"Team B positions distribution when itself scoring within 10 sec.\")\nfor p in range(3):\n    axes[p].title.set_text(f'Player {3+p}')\n    axes[p].scatter(train_df[train_df['team_B_scoring_within_10sec']==1][f'p{3+p}_pos_x'], train_df[train_df['team_B_scoring_within_10sec']==1][f'p{3+p}_pos_y'], s=0.05, c=\"red\")\naxes[3].title.set_text(f'Ball Position')\naxes[3].scatter(train_df[train_df['team_B_scoring_within_10sec']==1]['ball_pos_x'], train_df[train_df['team_B_scoring_within_10sec']==1]['ball_pos_y'], s=0.05, c=\"red\")","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:28.533266Z","iopub.execute_input":"2022-10-14T05:42:28.533581Z","iopub.status.idle":"2022-10-14T05:42:30.984720Z","shell.execute_reply.started":"2022-10-14T05:42:28.533547Z","shell.execute_reply":"2022-10-14T05:42:30.983889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Team A and B Position in the field when the adversarial score within 10 sec","metadata":{}},{"cell_type":"code","source":"f, axes = plt.subplots(nrows = 1, ncols = 4, figsize = (20, 7))\nf.suptitle(\"Team A positions distribution when Team B scoring within 10 sec.\")\nfor p in range(3):\n    axes[p].title.set_text(f'Player {p}')\n    axes[p].scatter(train_df[train_df['team_B_scoring_within_10sec']==1][f'p{p}_pos_x'], train_df[train_df['team_B_scoring_within_10sec']==1][f'p{p}_pos_y'], s=0.02)\naxes[3].title.set_text(f'Ball Position')\naxes[3].scatter(train_df[train_df['team_B_scoring_within_10sec']==1]['ball_pos_x'], train_df[train_df['team_B_scoring_within_10sec']==1]['ball_pos_y'], s=0.05, c=\"red\")\n    \nf, axes = plt.subplots(nrows = 1, ncols = 4, figsize = (20, 7))\nf.suptitle(\"Team B positions distribution when Team A scoring within 10 sec.\")\nfor p in range(3):\n    axes[p].title.set_text(f'Player {3+p}')\n    axes[p].scatter(train_df[train_df['team_A_scoring_within_10sec']==1][f'p{3+p}_pos_x'], train_df[train_df['team_A_scoring_within_10sec']==1][f'p{3+p}_pos_y'], s=0.05, c=\"red\")\naxes[3].title.set_text(f'Ball Position')\naxes[3].scatter(train_df[train_df['team_A_scoring_within_10sec']==1]['ball_pos_x'], train_df[train_df['team_A_scoring_within_10sec']==1]['ball_pos_y'], s=0.05)","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:30.985972Z","iopub.execute_input":"2022-10-14T05:42:30.986411Z","iopub.status.idle":"2022-10-14T05:42:33.680103Z","shell.execute_reply.started":"2022-10-14T05:42:30.986381Z","shell.execute_reply":"2022-10-14T05:42:33.679236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Players Positions Analysis\nThe attack of a team that makes a goal within 10 sec. is probably brought ahead by individuals players (attack team positions distributed randomly in the field) instead the defense is managed by all players of a team (defense team distributed near its gate)\n\n**Idea.** This behavior could be translated as a new feature engineering","metadata":{}},{"cell_type":"markdown","source":"# Player Positions Analysis in the Test set\nwe try to visualize the players position in the test set","metadata":{}},{"cell_type":"code","source":"f, axes = plt.subplots(nrows = 1, ncols = 7, figsize = (40, 9))\nf.suptitle(\"Team A, B and Ball positions on Test set\")\nfor p in range(3):\n    axes[p].title.set_text(f'Player {p}')\n    axes[p].scatter(test_df[f'p{p}_pos_x'], test_df[f'p{p}_pos_y'], s=0.002)\naxes[3].title.set_text(f'Ball Position')\naxes[3].scatter(test_df['ball_pos_x'], test_df['ball_pos_y'], s=0.002, c=\"green\")\nfor p in range(3):\n    axes[4+p].title.set_text(f'Player {3+p}')\n    axes[4+p].scatter(test_df[f'p{3+p}_pos_x'], test_df[f'p{3+p}_pos_y'], s=0.002, c=\"red\")\n","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:33.681256Z","iopub.execute_input":"2022-10-14T05:42:33.681679Z","iopub.status.idle":"2022-10-14T05:42:38.166471Z","shell.execute_reply.started":"2022-10-14T05:42:33.681645Z","shell.execute_reply":"2022-10-14T05:42:38.165828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Targets Distribution (imbalanced)","metadata":{}},{"cell_type":"code","source":"f, (ax1, ax2) = plt.subplots(1, 2, figsize = (15, 7))\nlabels_a = [f\"{p:.2f}%\" for p in train_df['team_A_scoring_within_10sec'].value_counts()/train_df['team_A_scoring_within_10sec'].value_counts().sum()*100]\nlabels_b = [f\"{p:.2f}%\" for p in train_df['team_B_scoring_within_10sec'].value_counts()/train_df['team_A_scoring_within_10sec'].value_counts().sum()*100]\ntrain_df['team_A_scoring_within_10sec'].value_counts().plot(kind='pie', ax=ax1, labels=labels_a, startangle=-140)\ntrain_df['team_B_scoring_within_10sec'].value_counts().plot(kind='pie', ax=ax2, labels=labels_b, startangle=-140)","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:38.167400Z","iopub.execute_input":"2022-10-14T05:42:38.168384Z","iopub.status.idle":"2022-10-14T05:42:38.415071Z","shell.execute_reply.started":"2022-10-14T05:42:38.168333Z","shell.execute_reply":"2022-10-14T05:42:38.414326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## StratifiedKFold and imbalanced datasets\n\nIn unbalanced datasets, the minority class, often of greatest interest and whose predictions are most valuable, is more difficult to predict because in practice there are few examples available.\n\nFurthermore, most machine learning algorithms for classification assume an equal distribution of classes. This means that in an unbalanced scenario the model focuses only on learning the characteristics of the most abundant observations, neglecting the examples of the minority class.\n\nThe most common approaches used for model evaluation are train / test splitting and the k-fold cross-validation procedure. Both approaches can be very effective in general, although they can lead to misleading results and potentially fail when used on classification problems with severe class imbalance. The techniques that must be used to stratify the sampling according to the class label are: the subdivision of the stratified train test and the stratified **k-fold cross-validation**\n\n### How to manage imbalanced dataset?\n- the first option is to check if your model algo support class weight option\n- another option is to use https://imbalanced-learn.org algorithms in order to under-sampling or over-sampling your dataset\n- use ensemble/bagging techniques help to reduce the effect of imbalanced dataset\n- naturally we need to validate our model using a cross-validation or train/test split which maintains the same percentage of target weight (ex. Stratified)\n\n#### Note for sub-sampling\n- The problem with subsampling techniques is that you may strip out some valuable information and alter the overall distribution of the dataset that is representative in a specific domain\n- **Idea to Test** : trying to sub-sampling 0 label group in order to reduce the complete dataset and train on that","metadata":{}},{"cell_type":"code","source":"# imbalanced ratio factor about 16\ntrain_df['team_A_scoring_within_10sec'].value_counts()[0]/train_df['team_A_scoring_within_10sec'].value_counts()[1]","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:38.416176Z","iopub.execute_input":"2022-10-14T05:42:38.416480Z","iopub.status.idle":"2022-10-14T05:42:38.451665Z","shell.execute_reply.started":"2022-10-14T05:42:38.416451Z","shell.execute_reply":"2022-10-14T05:42:38.450652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"=> Our dataset has an imbalance ratio of about 16, so we might consider our case as medium unbalanced.\n\n### Note\nIn real-world applications such as fraud detection we can also find imbalance ratios ranging from 1: 1000 up to 1: 5000. So, these would have a serious imbalance","metadata":{}},{"cell_type":"markdown","source":"## Features and Columns considerations\n\n### Games and Events\n- **game_num** (train only): 737x10 games.\n- **event_id** (train only): 3000 average events in each game.\n- **event_time** (train only): Time in seconds before the event ended, either by a goal being scored or simply when we decided to truncate the timeseries if a goal was not scored.\n\n### Ball and Players position\n- **[xyz] position (ball and players)**: Ball's position as a 3d vector.\n    * => **Match field measures** \n        - X : [-80:80]\n        - Y : [-100:100]\n        - Z : [0:40]\n        \n### Some Important Features are not available on Test set, so we are going to drop also on train set\n- **player_scoring_next** (train only): Important feature we need to transform with others (Which player scores at the end of the current event, in [0, 6], or -1 if the event does not end in a goal.)\n- **team_scoring_next** (train only): Important feature (Which team scores at the end of the current event (A or B), or NaN if the event does not end in a goal.)\n","metadata":{}},{"cell_type":"markdown","source":"# Features Correlation and HeatMap","metadata":{}},{"cell_type":"code","source":"# features correlations values\nmatrix = train_df.corr()","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:38.453052Z","iopub.execute_input":"2022-10-14T05:42:38.453366Z","iopub.status.idle":"2022-10-14T05:42:52.079865Z","shell.execute_reply.started":"2022-10-14T05:42:38.453338Z","shell.execute_reply":"2022-10-14T05:42:52.079154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\n# we use the absolute because negative correlation are good in the same way\nmatrix_correlations = matrix.abs()\nmask = np.triu(np.ones_like(matrix_correlations, dtype=bool))\nmatrix_correlations = matrix_correlations.mask(mask)\n\n# visualize correlation through heatmap chart\nplt.figure(figsize=(30, 20))\nmask_tri = np.triu(np.ones_like(matrix_correlations, dtype=bool))\nsns.heatmap(matrix_correlations, cmap=\"RdYlBu_r\", annot=True, fmt=\".1f\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:52.081263Z","iopub.execute_input":"2022-10-14T05:42:52.081638Z","iopub.status.idle":"2022-10-14T05:42:57.173414Z","shell.execute_reply.started":"2022-10-14T05:42:52.081601Z","shell.execute_reply":"2022-10-14T05:42:57.172730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Correlation beetween features and the target","metadata":{}},{"cell_type":"code","source":"# correlation beetween features and the target\ncorr=matrix_correlations.round(2)  \ncorr=corr.iloc[-1,:-1].sort_values(ascending=False)\npal=sns.color_palette(\"RdYlBu\",32).as_hex()\ntitles=[i for i in corr.index]\ncorr.index=titles\ncorr.plot.bar(color=pal, figsize=(20,5))","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:57.174510Z","iopub.execute_input":"2022-10-14T05:42:57.174959Z","iopub.status.idle":"2022-10-14T05:42:57.865576Z","shell.execute_reply.started":"2022-10-14T05:42:57.174930Z","shell.execute_reply":"2022-10-14T05:42:57.864591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Features Engineering","metadata":{}},{"cell_type":"code","source":"TARGETS = ['team_A_scoring_within_10sec', 'team_B_scoring_within_10sec']\nDROP_FEATURES = ['id', 'game_num', 'event_id', 'event_time', 'team_scoring_next', 'player_scoring_next']\n","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:57.868536Z","iopub.execute_input":"2022-10-14T05:42:57.868787Z","iopub.status.idle":"2022-10-14T05:42:57.872742Z","shell.execute_reply.started":"2022-10-14T05:42:57.868763Z","shell.execute_reply":"2022-10-14T05:42:57.871967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def columns_exchanger(df_in, a, b):\n    df = df_in.copy()\n    tmp = df[a].copy()\n    df[a] = df[b].copy()\n    df[b] = tmp.copy()\n    return df\n    \ndef feature_engineering(df_in, flip=False):\n    \"\"\"\n    Work in progress\n    \"\"\"\n    df = df_in.copy()\n    # flipping the ball and the players position for \"Data Augmentation\" how suggest by @pietromaldini1 in discussion https://www.kaggle.com/competitions/tabular-playground-series-oct-2022/discussion/357577\n    # for the moment I'm going to try to flip X position and changing players in the same Team\n    # Mirroring all (x|y-axis and teams) doesn't fit well (I'm investigationg why)\n    if flip:\n        # flipping the ball only X\n        df[\"ball_pos_x\"] = -df[\"ball_pos_x\"]\n        #df[\"ball_pos_y\"] = -df[\"ball_pos_y\"]\n        df[\"ball_vel_x\"] = -df[\"ball_vel_x\"]\n        #df[\"ball_vel_y\"] = -df[\"ball_vel_y\"]\n        # flipping the player position\n        for player in range(6):\n            df[f\"p{player}_pos_x\"] = -df[f\"p{player}_pos_x\"]\n            #df[f\"p{player}_pos_y\"] = -df[f\"p{player}_pos_y\"]\n            df[f\"p{player}_vel_x\"] = -df[f\"p{player}_vel_x\"]\n            #df[f\"p{player}_vel_y\"] = -df[f\"p{player}_vel_y\"]\n        # exchange players [0,2] to [3,5] in order to avoid to flip the targets \n        #for player in range(3):\n        for template in ['p{PLAYER}_pos_x', 'p{PLAYER}_pos_y', 'p{PLAYER}_pos_z', 'p{PLAYER}_vel_x', 'p{PLAYER}_vel_y', 'p{PLAYER}_vel_z', 'p{PLAYER}_boost', 'boost{PLAYER}_timer']:\n            df = columns_exchanger(df, template.format(PLAYER=0), template.format(PLAYER=2))\n            df = columns_exchanger(df, template.format(PLAYER=3), template.format(PLAYER=5))\n    # simgle player missing in the field\n    for player in range(6):        \n        df[f\"p{player}_missing\"] = df[f\"p{player}_pos_x\"].isna().astype('int8')\n    df.fillna(0, inplace=True)\n    # two simple new features rapresenting the distance between the ball and the gate of a team\n    df[f\"goal_a_distance\"] = ((df[\"ball_pos_x\"]-0)**2 + (df[\"ball_pos_y\"]-100)**2 + (df[\"ball_pos_z\"]-20)**2)**0.5\n    df[f\"goal_b_distance\"] = ((df[\"ball_pos_x\"]-0)**2 + (df[\"ball_pos_y\"]+100)**2 + (df[\"ball_pos_z\"]-20)**2)**0.5\n    # two other features rapresenting the distance between the team and its gate indicates when the team is defending (see analysis above)\n    df[f\"team_a_defending\"] = pd.Series(np.zeros(df.shape[0]), index=df.index)\n    for player in range(3):\n        df[f\"team_a_defending\"] += ((df[f\"p{player}_pos_x\"]-0)**2 + (df[f\"p{player}_pos_y\"]+100)**2 + (df[f\"p{player}_pos_z\"]-20)**2)**0.5\n    df[f\"team_b_defending\"] = pd.Series(np.zeros(df.shape[0]), index=df.index)\n    for player in range(3):\n        df[f\"team_b_defending\"] += ((df[f\"p{3+player}_pos_x\"]-0)**2 + (df[f\"p{3+player}_pos_y\"]-100)**2 + (df[f\"p{3+player}_pos_z\"]-20)**2)**0.5\n    # we calcolate 3 new features as the position [x,y,z] in the next future considering the velocity [x,y,z] and the average Event Time difference between events_id on the same Game see specific plot above\n    df['ball_pos_x_next'] = df['ball_pos_x'] + df['ball_vel_x']*0.2\n    df['ball_pos_y_next'] = df['ball_pos_y'] + df['ball_vel_y']*0.2\n    df['ball_pos_z_next'] = df['ball_pos_z'] + df['ball_vel_z']*0.2\n    df = df.drop(DROP_FEATURES, axis=1, errors=\"ignore\")   \n    return df\n","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:57.873765Z","iopub.execute_input":"2022-10-14T05:42:57.874035Z","iopub.status.idle":"2022-10-14T05:42:57.889978Z","shell.execute_reply.started":"2022-10-14T05:42:57.874006Z","shell.execute_reply":"2022-10-14T05:42:57.889242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Hybrid Model Class\n\nThe following class permit us to encapsulate the logic of ensembling N models making a weighted average of their predictions (Soft-Voting)\n\nWe are going to choose several models in order to ensemble them and try to get better results\n\n## Hybrid Model Schema\n\n![https://i.postimg.cc/prVWrkCg/hybrid-model-drawio.png](https://i.postimg.cc/prVWrkCg/hybrid-model-drawio.png)\n\n### Note\nFor a models comparison see my notebook [\"TPS Oct 2022 | PyCaret Model Analysis\"](https://www.kaggle.com/code/infrarosso/tps-oct-2022-pycaret-model-analysis)","metadata":{}},{"cell_type":"code","source":"class EnsembleHybrid:\n   def __init__(self, models=[], weights=[]):\n       self.models = models\n       self.weights = weights\n\n   def fit(self, X, y):\n       # Train models\n       for m in self.models:\n           print(f\"Training {m}...\")\n           m.fit(X, y)\n\n   def predict_proba(self, X_test):\n       y_pred = pd.Series(np.zeros(X_test.shape[0]), index=X_test.index)\n       for i, m in enumerate(self.models):\n           y_pred += pd.Series(m.predict_proba(X_test)[:,1], index=X_test.index) * self.weights[i]\n       return y_pred","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:57.890859Z","iopub.execute_input":"2022-10-14T05:42:57.891577Z","iopub.status.idle":"2022-10-14T05:42:57.903483Z","shell.execute_reply.started":"2022-10-14T05:42:57.891549Z","shell.execute_reply":"2022-10-14T05:42:57.902526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Cross-validation with StratifiedKFold\n\nI'm going to use StratifiedKFold in order to better fit imbalanced dataset\nWe ensemble N models","metadata":{}},{"cell_type":"code","source":"FOLDS = 5\n    \ndef model_it(X_train, Y_train, X_test):\n    ZERO_AS_MISSING = True\n    SCALE_WEIGHT=16\n    Y_validations, model_val_preds, model_test_preds, ensemble_models, scores=[],[],[],[],[]\n    feat_importance=pd.DataFrame(index=X_train.columns)\n        \n    lgbm = LGBMClassifier(objective='binary',\n                      metric='logloss',\n                      importance_type='gain',\n                      random_state=seed,\n                      zero_as_missing=ZERO_AS_MISSING,\n                      learning_rate=0.1,\n                      max_depth=10,\n                      min_child_samples=340,\n                      min_child_weight=1e-05,\n                      n_estimators=100,\n                      num_leaves=130,\n                      reg_alpha=50,\n                      reg_lambda=50,\n                      subsample=0.7865007820901366)   \n    \n    cat_boost = CatBoostClassifier(random_seed=seed,\n                               eval_metric='Logloss',\n                               logging_level='Silent',\n                               learning_rate=0.05,\n                               iterations=100)\n    \n    xgbm = XGBClassifier(objective='binary:logistic',\n                     random_state=seed,\n                     learning_rate=0.1,\n                     n_estimators=100,\n                     max_depth=8, \n                     #tree_method='gpu_hist')\n                     tree_method='hist')\n    \n    #models = [lgbm, xgbm]\n    #weights=[0.7, 0.3]\n    models = [lgbm]\n    weights=[1]\n    \n    k_fold = StratifiedKFold(n_splits=FOLDS, shuffle=True, random_state=seed)\n    for fold, (train_idx, val_idx) in enumerate(k_fold.split(X_train, Y_train)):\n        print(\"\\nFold {}\".format(fold+1))\n        X_fold_train, Y_fold_train = X_train.iloc[train_idx,:], Y_train[train_idx]\n        X_fold_val, Y_fold_val = X_train.iloc[val_idx,:], Y_train[val_idx]\n        print(\"Train shape: {}, {}, Valid shape: {}, {}\".format(\n            X_fold_train.shape, Y_fold_train.shape, X_fold_val.shape, Y_fold_val.shape))\n        ensemble_model = EnsembleHybrid(models=models, weights=weights)\n        ensemble_model.fit(X_fold_train, Y_fold_train)\n        model_prob = ensemble_model.predict_proba(X_fold_val)\n        Y_validations.append(Y_fold_val)\n        model_val_preds.append(model_prob)\n        model_test_preds.append(ensemble_model.predict_proba(X_test))\n        ensemble_models.append(ensemble_model)\n        feat_importance[\"Importance_Fold\"+str(fold)]=lgbm.feature_importances_\n        score=log_loss(Y_fold_val, model_prob)\n        scores.append(score)\n        print(\"Validation Log Loss = {:.4f}\".format(score))\n        del X_fold_train, Y_fold_train, X_fold_val, Y_fold_val\n        gc.collect()\n    return model_test_preds, feat_importance, ensemble_models\n","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:57.904379Z","iopub.execute_input":"2022-10-14T05:42:57.904602Z","iopub.status.idle":"2022-10-14T05:42:57.916495Z","shell.execute_reply.started":"2022-10-14T05:42:57.904580Z","shell.execute_reply":"2022-10-14T05:42:57.915709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# set to 10 to use all train dataset\n# I'm going to use 9 for train set and I leave 1 for final validation\nVALIDATION=True\nif VALIDATION:\n    TRAIN_SET = 1\n    final_validation_df = pd.read_csv(\"/kaggle/input/tabular-playground-series-oct-2022/train_9.csv\", dtype=dtypes)\n    print(\"Loading train 9 as Validation set\")\nelse:\n    TRAIN_SET = 10","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:42:57.920564Z","iopub.execute_input":"2022-10-14T05:42:57.920801Z","iopub.status.idle":"2022-10-14T05:43:18.725495Z","shell.execute_reply.started":"2022-10-14T05:42:57.920779Z","shell.execute_reply":"2022-10-14T05:43:18.724718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Selection with MRMR (optional)\nMaximum Relevance — Minimum Redundancy” (aka MRMR) is an algorithm used by Uber’s machine learning platform for finding the “minimal-optimal” subset of features.\n- Enabling FEATURE_SELECTION_ENABLED the train will be done with the features selection algorithm\n- Disabling FEATURE_SELECTION_ENABLED the train will be done with all features availables in the train set (all columns)","metadata":{}},{"cell_type":"code","source":"FEATURE_SELECTION_ENABLED = False\n# FEATURE SELECTION MRMR\ndef feature_selection(X, y):\n    if not FEATURE_SELECTION_ENABLED:\n        return X.columns\n    out = mrmr_classif(X, y, K=40)\n    print(\"Features selection:\", out)\n    return out","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:43:18.726535Z","iopub.execute_input":"2022-10-14T05:43:18.726758Z","iopub.status.idle":"2022-10-14T05:43:18.731294Z","shell.execute_reply.started":"2022-10-14T05:43:18.726735Z","shell.execute_reply":"2022-10-14T05:43:18.730371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Ensembling Hybrid Models \n\nThe following schema describe how we are going to use single train_df file as a fold in order to train several hybrid models and soft voting their predictions for the final submission \n\n## Ensemble Schema\n\n![https://i.postimg.cc/RhMkBk8r/tps-2022-10-drawio.png](https://i.postimg.cc/RhMkBk8r/tps-2022-10-drawio.png)\n\n","metadata":{}},{"cell_type":"markdown","source":"## Training All","metadata":{}},{"cell_type":"code","source":"X_test = feature_engineering(test_df)\nX_test.drop(TARGETS+DROP_FEATURES, axis=1, errors='ignore', inplace=True)\nmodel_test_preds_a_all = []\nmodel_test_preds_b_all = []\nmodels_a = []\nmodels_b = []\n#FLIP_ITERATIONS_VALUES=[False, True]\nFLIP_ITERATIONS_VALUES=[False]\nfor flip in FLIP_ITERATIONS_VALUES:\n    print(f\"Elaborating dataset flipped:{flip}\")\n    for i in range(TRAIN_SET):\n        print(f\"Loading train set {i}\")\n        if i==0:\n            train_df = train0_df\n        else:\n            train_df = pd.read_csv(f\"/kaggle/input/tabular-playground-series-oct-2022/train_{i}.csv\", dtype=dtypes)\n        # FEATURE ENGINEERING\n        y_train_a = train_df[TARGETS[0]]\n        y_train_b = train_df[TARGETS[1]]\n        X_train = feature_engineering(train_df, flip)\n        X_train.drop(TARGETS, axis=1, errors='ignore', inplace=True)\n        X_train.reset_index().drop(\"index\", axis=1, inplace=True)\n        y_train_a = y_train_a.reset_index()['team_A_scoring_within_10sec']\n        y_train_b = y_train_b.reset_index()['team_B_scoring_within_10sec']\n        # FEATURE SELECTION\n        features_selection_a = feature_selection(X_train, y_train_a)\n        features_selection_b = feature_selection(X_train, y_train_b)\n        # MODELLING\n        model_test_preds_a, feat_importance_a, model_a_folds = model_it(X_train[features_selection_a], y_train_a, X_test[features_selection_a])\n        model_test_preds_b, feat_importance_b, model_a_folds = model_it(X_train[features_selection_b], y_train_b, X_test[features_selection_b])\n        model_test_preds_a_all.append(model_test_preds_a)\n        model_test_preds_b_all.append(model_test_preds_b)\n        models_a.append(model_a_folds)\n        models_b.append(model_a_folds)\n        gc.collect()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-10-14T05:43:18.732333Z","iopub.execute_input":"2022-10-14T05:43:18.732545Z","iopub.status.idle":"2022-10-14T05:54:32.939266Z","shell.execute_reply.started":"2022-10-14T05:43:18.732524Z","shell.execute_reply":"2022-10-14T05:54:32.938188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Final Validation ","metadata":{}},{"cell_type":"code","source":"# when flip is enabled we have 2 x folds and trainset\nFLIP_FOLDS = len(FLIP_ITERATIONS_VALUES)\n\nif VALIDATION:\n    X_validation = feature_engineering(final_validation_df)\n    y_validation_a = final_validation_df[TARGETS[0]]\n    y_validation_b = final_validation_df[TARGETS[1]]\n    y_validations = [y_validation_a, y_validation_b]\n    teams = ['a', 'b']\n    X_validation.drop(TARGETS+DROP_FEATURES, axis=1, errors='ignore', inplace=True)\n    for i, models in enumerate([models_a, models_b]):\n        model_fold_validation_preds = np.zeros(X_validation.shape[0])\n        for folds_models in models:\n            for fold_model in folds_models:\n                model_fold_validation_preds += fold_model.predict_proba(X_validation)\n        model_fold_validation_preds = model_fold_validation_preds/(FOLDS*TRAIN_SET*FLIP_FOLDS)\n        score=log_loss(y_validations[i], model_fold_validation_preds)\n        print(f\"Final Validation Log Loss for Team {teams[i]} = {score:.4f}\")","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:54:32.940552Z","iopub.execute_input":"2022-10-14T05:54:32.940919Z","iopub.status.idle":"2022-10-14T05:55:45.278578Z","shell.execute_reply.started":"2022-10-14T05:54:32.940884Z","shell.execute_reply":"2022-10-14T05:55:45.277741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Importance by LGBM","metadata":{}},{"cell_type":"code","source":"def feature_importance(feat_importance):\n    feat_importance['avg']=feat_importance.mean(axis=1)\n    feat_importance=feat_importance.sort_values(by='avg',ascending=False)\n\n    pal=sns.color_palette(\"RdYlBu\",32).as_hex()\n    titles=[i for i in feat_importance.index]\n    feat_importance.index=titles\n    feat_importance['avg'].plot.bar(color=pal, figsize=(20,5))","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:55:45.279621Z","iopub.execute_input":"2022-10-14T05:55:45.279898Z","iopub.status.idle":"2022-10-14T05:55:45.285628Z","shell.execute_reply.started":"2022-10-14T05:55:45.279871Z","shell.execute_reply":"2022-10-14T05:55:45.284785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_importance(feat_importance_a)","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:55:45.286858Z","iopub.execute_input":"2022-10-14T05:55:45.287564Z","iopub.status.idle":"2022-10-14T05:55:46.057828Z","shell.execute_reply.started":"2022-10-14T05:55:45.287529Z","shell.execute_reply":"2022-10-14T05:55:46.056768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_importance(feat_importance_b)","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:55:46.059199Z","iopub.execute_input":"2022-10-14T05:55:46.059510Z","iopub.status.idle":"2022-10-14T05:55:46.823218Z","shell.execute_reply.started":"2022-10-14T05:55:46.059483Z","shell.execute_reply":"2022-10-14T05:55:46.822222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"submission_a = np.zeros(test_df.shape[0])\nsubmission_b = np.zeros(test_df.shape[0])\nfor i in range(len(model_test_preds_a_all)):\n    for j in range(FOLDS):\n        submission_a += model_test_preds_a_all[i][j]\n        submission_b += model_test_preds_b_all[i][j]\n# 2 if we train with flipped version of train data\nsubmission_a = submission_a/(FOLDS*TRAIN_SET*FLIP_FOLDS)\nsubmission_b = submission_b/(FOLDS*TRAIN_SET*FLIP_FOLDS)","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:55:46.825966Z","iopub.execute_input":"2022-10-14T05:55:46.826248Z","iopub.status.idle":"2022-10-14T05:55:46.844320Z","shell.execute_reply.started":"2022-10-14T05:55:46.826222Z","shell.execute_reply":"2022-10-14T05:55:46.843308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = pd.read_csv(f'/kaggle/input/tabular-playground-series-oct-2022/sample_submission.csv')\nsubmission_df['team_A_scoring_within_10sec'] = submission_a\nsubmission_df['team_B_scoring_within_10sec'] = submission_b\nsubmission_df.to_csv('submission_lgbm.csv', index=False)\nsubmission_df","metadata":{"execution":{"iopub.status.busy":"2022-10-14T05:55:46.845635Z","iopub.execute_input":"2022-10-14T05:55:46.846089Z","iopub.status.idle":"2022-10-14T05:55:48.749102Z","shell.execute_reply.started":"2022-10-14T05:55:46.846063Z","shell.execute_reply":"2022-10-14T05:55:48.747914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"-------------------------------------------------------------------------\n<div style=\"text-align: center;\">\n    <h3>Thanks for watching till the end ;)  </h3>\n    <h2>If you liked this notebook, upvote it ! </h2>\n<img src=\"https://i.postimg.cc/SsChkSJv/upvote.png\" width=\"70\"/>\n</div>\n\n\n\n\n---\n","metadata":{}}]}