{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport gc\nfrom lightgbm import LGBMClassifier\nfrom lightgbm import early_stopping, log_evaluation\nimport lightgbm as lgb\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nsns.set_style ('darkgrid')\nsns.palplot(sns.color_palette('rainbow'))\nsns.set_palette('rainbow')\n\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)\n\nclass config:\n    SEED = 777\n    test_size = 0.20","metadata":{"execution":{"iopub.status.busy":"2022-10-20T09:44:12.282257Z","iopub.execute_input":"2022-10-20T09:44:12.283011Z","iopub.status.idle":"2022-10-20T09:44:14.494246Z","shell.execute_reply.started":"2022-10-20T09:44:12.282882Z","shell.execute_reply":"2022-10-20T09:44:14.492868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TARGETS = ['team_A_scoring_within_10sec', 'team_B_scoring_within_10sec']\nDROP_FEATURES = ['id', 'game_num', 'event_id', 'event_time', 'team_scoring_next', 'player_scoring_next']\n\ndef feature_engineering(df):\n    df = df.drop(DROP_FEATURES, axis=1, errors=\"ignore\")    \n    return df\n\n# https://www.kaggle.com/code/mattop/rocket-league-tps-eda\ndef add_features(df):\n    df[f\"goal_a_distance\"] = ((df[\"ball_pos_x\"]-0)**2 + (df[\"ball_pos_y\"]-100)**2 + (df[\"ball_pos_z\"]-20)**2)**0.5\n    df[f\"goal_b_distance\"] = ((df[\"ball_pos_x\"]-0)**2 + (df[\"ball_pos_y\"]+100)**2 + (df[\"ball_pos_z\"]-20)**2)**0.5\n    df = df.drop(DROP_FEATURES, axis=1, errors=\"ignore\")   \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-10-20T09:44:14.498033Z","iopub.execute_input":"2022-10-20T09:44:14.499353Z","iopub.status.idle":"2022-10-20T09:44:14.513612Z","shell.execute_reply.started":"2022-10-20T09:44:14.499298Z","shell.execute_reply":"2022-10-20T09:44:14.512305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/code/gazu468/feather-to-compress-your-data-8x-faster\n\ntest = pd.read_feather(\"../input/tpsoct22-feather-files/test.feather\")\ntest = feature_engineering(test)\ntest = add_features(test)","metadata":{"execution":{"iopub.status.busy":"2022-10-20T09:44:14.515465Z","iopub.execute_input":"2022-10-20T09:44:14.516591Z","iopub.status.idle":"2022-10-20T09:44:19.558204Z","shell.execute_reply.started":"2022-10-20T09:44:14.516521Z","shell.execute_reply":"2022-10-20T09:44:19.557108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data Descriptions \n*   **`game_num`** _(train only)_: Unique identifier for the game from which the event was taken.\n    \n*   **`event_id`** _(train only)_: Unique identifier for the sequence of consecutive frames.\n    \n*   **`event_time`** _(train only)_: Time in seconds before the event ended, either by a goal being scored or simply when we decided to truncate the timeseries if a goal was not scored.\n    \n*   **`ball_pos_[xyz]`**: Ball's position as a 3d vector.\n    \n*   **`ball_vel_[xyz]`**: Ball's velocity as a 3d vector.\n    \n*   For `i` in `[0, 6)`:\n    \n    *   **`p{i}_pos_[xyz]`**: Player `i`'s position as a 3d vector.\n    *   **`p{i}_vel_[xyz]`**: Player `i`'s velocity as a 3d vector.\n    *   **`p{i}_boost`**: Player `i`'s boost remaining, in `[0, 100]`. A player can consume boost to substantially increase their speed, and is required to fly up into the `z` dimension (besides driving up a wall, or the small air gained by a jump).\n    *   All `p{i}` columns will be `NaN` if and only if the player is demolished (destroyed by an enemy player; will respawn within a few seconds).\n    *   Players 0, 1, and 2 make up team `A` and players 3, 4, and 5 make up team `B`.\n    *   The orientation vector of the player's car (which way the car is facing) does not necessarily match the player's velocity vector, and this dataset does not capture orientation data.\n*   For `i` in `[0, 6)`:\n    \n    *   **`boost{i}_timer`**: Time in seconds until big boost orb `i` respawns, or `0` if it's available. Big boost orbs grant a full 100 boost to a player driving over it. The orb `(x, y)` locations are roughly `[ (-61.4, -81.9), (61.4, -81.9), (-71.7, 0), (71.7, 0), (-61.4, 81.9), (61.4, 81.9) ]` with `z = 0`. (Players can also gain boost from small boost pads across the map, but we do not capture those pads in this dataset).\n*   **`player_scoring_next`** _(train only)_: Which player scores at the end of the current event, in `[0, 6)`, or `-1` if the event does not end in a goal.\n    \n*   **`team_scoring_next`** _(train only)_: Which team scores at the end of the current event (`A` or `B`), or `NaN` if the event does not end in a goal.\n    \n*   **`team_[A|B]_scoring_within_10sec`** _(train only)_: **\\[Target columns\\]** Value of `1` if `team_scoring_next == [A|B]` and `time_before_event` is in `[-10, 0]`, otherwise `0`.\n    \n*   **`id`** _(test and submission only)_: Unique identifier for each test row. Your submission should be a pair of `team_A_scoring_within_10sec` and `team_B_scoring_within_10sec` probability predictions for each `id`, where your predictions can range the real numbers from `[0, 1]`.","metadata":{}},{"cell_type":"code","source":"dtypes_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_dtypes.csv')\ndtypes = {k: v for (k, v) in zip(dtypes_df.column, dtypes_df.dtype)}\n\ntrain = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_0.csv',nrows=10000, dtype = dtypes)\n#train = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_0.csv', dtype = dtypes).sample(10000)\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-20T09:44:19.560468Z","iopub.execute_input":"2022-10-20T09:44:19.560848Z","iopub.status.idle":"2022-10-20T09:44:19.742478Z","shell.execute_reply.started":"2022-10-20T09:44:19.560815Z","shell.execute_reply":"2022-10-20T09:44:19.741422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-20T09:44:19.744034Z","iopub.execute_input":"2022-10-20T09:44:19.744469Z","iopub.status.idle":"2022-10-20T09:44:19.779489Z","shell.execute_reply.started":"2022-10-20T09:44:19.744428Z","shell.execute_reply":"2022-10-20T09:44:19.778698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **<span style=\"color:#e76f51;\">Missings</span>**","metadata":{}},{"cell_type":"code","source":"print('Number of missing values in training set:',train.isna().sum().sum())\nprint('')\nprint('Number of missing values in test set:',test.isna().sum().sum())","metadata":{"execution":{"iopub.status.busy":"2022-10-20T09:44:19.780687Z","iopub.execute_input":"2022-10-20T09:44:19.781186Z","iopub.status.idle":"2022-10-20T09:44:19.873217Z","shell.execute_reply.started":"2022-10-20T09:44:19.781156Z","shell.execute_reply":"2022-10-20T09:44:19.872118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trn_null = train.isnull().sum()\ntst_null = test.isnull().sum()\n\nprint('Train columns with null values:\\n', trn_null[trn_null>0])\nprint(\"-\"*10)\n\nprint('Test/Validation columns with null values:\\n', tst_null[tst_null>0])\nprint(\"-\"*10)","metadata":{"execution":{"iopub.status.busy":"2022-10-20T09:44:19.875905Z","iopub.execute_input":"2022-10-20T09:44:19.876659Z","iopub.status.idle":"2022-10-20T09:44:19.971107Z","shell.execute_reply.started":"2022-10-20T09:44:19.876597Z","shell.execute_reply":"2022-10-20T09:44:19.969317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# per sample\nplt.figure(figsize=(10, 8))\nplt.imshow(train.isna(), aspect=\"auto\", interpolation=\"nearest\", cmap=\"gray\")\nplt.xlabel(\"Column Number\")\nplt.ylabel(\"Sample Number\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-20T09:44:19.972415Z","iopub.execute_input":"2022-10-20T09:44:19.972788Z","iopub.status.idle":"2022-10-20T09:44:20.205917Z","shell.execute_reply.started":"2022-10-20T09:44:19.972747Z","shell.execute_reply":"2022-10-20T09:44:20.204779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# per feature\ntrain.isna().mean().sort_values().plot(\n    kind=\"bar\", figsize=(15, 4),\n    title=\"Percentage of missing values per feature\",\n    ylabel=\"Ratio of missing values per feature\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-20T09:44:20.207164Z","iopub.execute_input":"2022-10-20T09:44:20.207478Z","iopub.status.idle":"2022-10-20T09:44:21.969488Z","shell.execute_reply.started":"2022-10-20T09:44:20.207450Z","shell.execute_reply":"2022-10-20T09:44:21.968418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **<span style=\"color:#e76f51;\">Duplicates</span>**","metadata":{}},{"cell_type":"code","source":"ignore_cols = ['game_num','event_id','event_time']\nn_duplicates = train.drop(labels=ignore_cols, axis=1).duplicated().sum()\nn_duplicates_test = test.duplicated().sum()\nprint(f\"You have {n_duplicates} duplicates in train.\")\nprint(f\"You have {n_duplicates_test} duplicates in test.\")","metadata":{"execution":{"iopub.status.busy":"2022-10-20T09:44:21.974794Z","iopub.execute_input":"2022-10-20T09:44:21.975155Z","iopub.status.idle":"2022-10-20T09:44:26.402739Z","shell.execute_reply.started":"2022-10-20T09:44:21.975125Z","shell.execute_reply":"2022-10-20T09:44:26.401542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **<span style=\"color:#e76f51;\">EDA</span>**","metadata":{}},{"cell_type":"code","source":"unique_values = train.select_dtypes(include=\"number\").nunique().sort_values()\n\n# Plot information with y-axis in log-scale\nunique_values.plot.bar(logy=True, figsize=(15, 4), title=\"Unique values per feature\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-20T09:44:26.404059Z","iopub.execute_input":"2022-10-20T09:44:26.404418Z","iopub.status.idle":"2022-10-20T09:44:28.848770Z","shell.execute_reply.started":"2022-10-20T09:44:26.404387Z","shell.execute_reply":"2022-10-20T09:44:28.847958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.plot(lw=0, marker=\".\", subplots=True, layout=(-1, 4),\n          figsize=(15, 30), markersize=1);","metadata":{"execution":{"iopub.status.busy":"2022-10-20T09:44:28.849993Z","iopub.execute_input":"2022-10-20T09:44:28.850491Z","iopub.status.idle":"2022-10-20T09:44:41.859012Z","shell.execute_reply.started":"2022-10-20T09:44:28.850461Z","shell.execute_reply":"2022-10-20T09:44:41.858094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Feature distribution\ntrain.hist(bins=25, figsize=(15, 25), layout=(-1, 5), edgecolor=\"black\")\nplt.tight_layout();","metadata":{"execution":{"iopub.status.busy":"2022-10-20T09:44:41.860004Z","iopub.execute_input":"2022-10-20T09:44:41.860318Z","iopub.status.idle":"2022-10-20T09:44:55.626269Z","shell.execute_reply.started":"2022-10-20T09:44:41.860290Z","shell.execute_reply":"2022-10-20T09:44:55.625155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Most frequent entries\nmost_frequent_entry = train.mode()\n\ndf_freq = train.eq(most_frequent_entry.values, axis=1)\ndf_freq = df_freq.mean().sort_values(ascending=False)\n\n\ndisplay(df_freq.head())\n\ndf_freq.plot.bar(figsize=(15, 4));","metadata":{"execution":{"iopub.status.busy":"2022-10-20T09:44:55.628027Z","iopub.execute_input":"2022-10-20T09:44:55.628681Z","iopub.status.idle":"2022-10-20T09:44:57.852500Z","shell.execute_reply.started":"2022-10-20T09:44:55.628616Z","shell.execute_reply":"2022-10-20T09:44:57.851337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **<span style=\"color:#e76f51;\">Correlation</span>**","metadata":{}},{"cell_type":"code","source":"drop_cols = ['game_num','event_id','event_time']\ndf_corr = train.drop(columns=drop_cols).corr()\nlabels = np.where(np.abs(df_corr)>0.75, \"S\",\n                  np.where(np.abs(df_corr)>0.5, \"M\",\n                           np.where(np.abs(df_corr)>0.25, \"W\", \"\")))\n\nplt.figure(figsize=(30, 20))\nsns.heatmap(df_corr, mask=np.eye(len(df_corr)), square=True,\n            center=0, annot=labels, fmt='', linewidths=.5,\n            cmap=\"YlGnBu\", cbar_kws={\"shrink\": 0.8});","metadata":{"execution":{"iopub.status.busy":"2022-10-20T09:44:57.854038Z","iopub.execute_input":"2022-10-20T09:44:57.854739Z","iopub.status.idle":"2022-10-20T09:45:03.754083Z","shell.execute_reply.started":"2022-10-20T09:44:57.854692Z","shell.execute_reply":"2022-10-20T09:45:03.752823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train\ndel dtypes_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-20T09:45:03.755532Z","iopub.execute_input":"2022-10-20T09:45:03.755963Z","iopub.status.idle":"2022-10-20T09:45:04.046743Z","shell.execute_reply.started":"2022-10-20T09:45:03.755921Z","shell.execute_reply":"2022-10-20T09:45:04.045234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **<span style=\"color:#e76f51;\">Continue Training LightGBM Model</span>**\n\nUsing LightGBM model or Booster instance for continue training with **init_model** option. **init_model** will perform gradient boosting for num_iterations additional rounds. This will allow us to use all 10 train files with LightGBM model training.\n\nReference: https://lightgbm.readthedocs.io/en/latest/pythonapi/lightgbm.train.html","metadata":{}},{"cell_type":"code","source":"params = {\n    'force_col_wise' : True,\n    'objective': 'binary',\n    'random_state' : config.SEED,\n    #'importance_type': 'gain',\n    'metric': 'logloss',\n    'n_jobs': -1,\n}\n\n\n\nes = early_stopping(stopping_rounds=10, first_metric_only=False)\nle = log_evaluation(-1)\n\nlr = lgb.reset_parameter(learning_rate=lambda iter: 0.05 * (0.99 ** iter))\n#lr = lgb.reset_parameter(learning_rate=lambda iter: 0.1 - 0.001 * iter)","metadata":{"execution":{"iopub.status.busy":"2022-10-20T09:45:04.048584Z","iopub.execute_input":"2022-10-20T09:45:04.049425Z","iopub.status.idle":"2022-10-20T09:45:04.058126Z","shell.execute_reply.started":"2022-10-20T09:45:04.049381Z","shell.execute_reply":"2022-10-20T09:45:04.056708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = [f for f in test.columns]\ninit = 1\n\nfor split in range(10):\n    gc.collect()\n\n    filename = \"../input/tpsoct22-feather-files/train_\"+str(split)+\".feather\"\n    train = pd.read_feather(filename)\n    # drop unwanted features\n    train = feature_engineering(train)\n    # add new features\n    train = add_features(train)\n    \n    label_a = 'team_A_scoring_within_10sec'\n    label_b = 'team_B_scoring_within_10sec'\n\n    train[label_a] = train[label_a].astype('int8')\n    train[label_b] = train[label_b].astype('int8')\n       \n    X_train, X_val, y_train_a, y_val_a = train_test_split(train[features], train[label_a], test_size = config.test_size, random_state = config.SEED)\n    X_train, X_val, y_train_b, y_val_b = train_test_split(train[features], train[label_b], test_size = config.test_size, random_state = config.SEED)\n    \n    if (init):\n        print(f'Start a new training job....split {split} with file {filename}')\n              \n        # Team A\n        lgb_a = LGBMClassifier(**params)\n        lgb_a.fit(X_train,y_train_a, eval_set=[(X_val, y_val_a)], eval_metric = ['logloss'],  callbacks=[es,le])\n        \n        # Team B\n        lgb_b = LGBMClassifier(**params)\n        lgb_b.fit(X_train,y_train_b, eval_set=[(X_val, y_val_b)], eval_metric = ['logloss'],  callbacks=[es,le])\n        \n        init = 0\n        #break\n    else:\n        print(f'continue training....split {split} with file {filename}')\n        lgb_train_a = lgb.Dataset(X_train, y_train_a)\n        lgb_eval_a = lgb.Dataset(X_val, y_val_a, reference=lgb_train_a)\n        \n        lgb_train_b = lgb.Dataset(X_train, y_train_b)\n        lgb_eval_b = lgb.Dataset(X_val, y_val_b, reference=lgb_train_b)\n        \n        # Team A\n        lgb_a = lgb.train(params,lgb_train_a, num_boost_round=100, valid_sets=lgb_eval_a, callbacks=[es,le,lr], init_model=lgb_a)\n        \n        # Team B\n        lgb_b = lgb.train(params,lgb_train_b, num_boost_round=100, valid_sets=lgb_eval_b, callbacks=[es,le,lr], init_model=lgb_b) ","metadata":{"execution":{"iopub.status.busy":"2022-10-20T09:45:04.060315Z","iopub.execute_input":"2022-10-20T09:45:04.060698Z","iopub.status.idle":"2022-10-20T10:07:09.115262Z","shell.execute_reply.started":"2022-10-20T09:45:04.060667Z","shell.execute_reply":"2022-10-20T10:07:09.114170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(20, 10))\nlgb.plot_importance(lgb_a,ax=ax)\nplt.title('Feature Importance - Model A')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-20T10:07:09.117401Z","iopub.execute_input":"2022-10-20T10:07:09.118184Z","iopub.status.idle":"2022-10-20T10:07:10.173966Z","shell.execute_reply.started":"2022-10-20T10:07:09.118134Z","shell.execute_reply":"2022-10-20T10:07:10.172860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(20, 10))\nlgb.plot_importance(lgb_b,ax=ax)\nplt.title('Feature Importance - Model B')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-20T10:07:10.175405Z","iopub.execute_input":"2022-10-20T10:07:10.175770Z","iopub.status.idle":"2022-10-20T10:07:11.245768Z","shell.execute_reply.started":"2022-10-20T10:07:10.175738Z","shell.execute_reply":"2022-10-20T10:07:11.244842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shap\n\nshap.initjs()\n\ndf_shap = X_val[:1000]\nexplainer = shap.TreeExplainer(lgb_a)\nshap_values = explainer.shap_values(df_shap)","metadata":{"execution":{"iopub.status.busy":"2022-10-20T10:07:11.247471Z","iopub.execute_input":"2022-10-20T10:07:11.247852Z","iopub.status.idle":"2022-10-20T10:07:21.322117Z","shell.execute_reply.started":"2022-10-20T10:07:11.247820Z","shell.execute_reply":"2022-10-20T10:07:21.321105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shap.summary_plot(shap_values, df_shap)","metadata":{"execution":{"iopub.status.busy":"2022-10-20T10:07:21.323135Z","iopub.execute_input":"2022-10-20T10:07:21.323748Z","iopub.status.idle":"2022-10-20T10:07:21.929661Z","shell.execute_reply.started":"2022-10-20T10:07:21.323713Z","shell.execute_reply":"2022-10-20T10:07:21.928545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# visualize a single prediction\nshap.force_plot(explainer.expected_value[1], shap_values[1][0,:], df_shap.iloc[0,:])","metadata":{"execution":{"iopub.status.busy":"2022-10-20T10:07:21.930965Z","iopub.execute_input":"2022-10-20T10:07:21.931298Z","iopub.status.idle":"2022-10-20T10:07:21.939704Z","shell.execute_reply.started":"2022-10-20T10:07:21.931267Z","shell.execute_reply":"2022-10-20T10:07:21.938789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shap.force_plot(explainer.expected_value[1], shap_values[1][:1000,:], df_shap.iloc[:1000,:])","metadata":{"execution":{"iopub.status.busy":"2022-10-20T10:07:21.941250Z","iopub.execute_input":"2022-10-20T10:07:21.941655Z","iopub.status.idle":"2022-10-20T10:07:23.634880Z","shell.execute_reply.started":"2022-10-20T10:07:21.941597Z","shell.execute_reply":"2022-10-20T10:07:23.633525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# visualize tree 0 for lgb_a\nfig, ax = plt.subplots(figsize=(20, 10))\nlgb.plot_tree(lgb_a, ax=ax, show_info = ['internal_value','leaf_count'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-20T10:22:09.122750Z","iopub.execute_input":"2022-10-20T10:22:09.123226Z","iopub.status.idle":"2022-10-20T10:22:12.029806Z","shell.execute_reply.started":"2022-10-20T10:22:09.123191Z","shell.execute_reply":"2022-10-20T10:22:12.028338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **<span style=\"color:#e76f51;\">Submission</span>**","metadata":{}},{"cell_type":"code","source":"dtypes_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/test_dtypes.csv')\ndtypes = {k: v for (k, v) in zip(dtypes_df.column, dtypes_df.dtype)}\n\nsubmission = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/sample_submission.csv',dtype = dtypes)","metadata":{"execution":{"iopub.status.busy":"2022-10-20T10:07:23.645815Z","iopub.execute_input":"2022-10-20T10:07:23.646676Z","iopub.status.idle":"2022-10-20T10:07:23.874200Z","shell.execute_reply.started":"2022-10-20T10:07:23.646605Z","shell.execute_reply":"2022-10-20T10:07:23.873165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_a = lgb_a.predict(test[features])\npredictions_b = lgb_b.predict(test[features])\n\nsubmission['team_A_scoring_within_10sec'] = predictions_a\nsubmission['team_B_scoring_within_10sec'] = predictions_b","metadata":{"execution":{"iopub.status.busy":"2022-10-20T10:07:23.875669Z","iopub.execute_input":"2022-10-20T10:07:23.876035Z","iopub.status.idle":"2022-10-20T10:08:03.961454Z","shell.execute_reply.started":"2022-10-20T10:07:23.876001Z","shell.execute_reply":"2022-10-20T10:08:03.960429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index = False)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-20T10:08:03.966416Z","iopub.execute_input":"2022-10-20T10:08:03.967089Z","iopub.status.idle":"2022-10-20T10:08:06.660753Z","shell.execute_reply.started":"2022-10-20T10:08:03.967054Z","shell.execute_reply":"2022-10-20T10:08:06.659478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **<span style=\"color:#e76f51;\">Work in progress</span>**\n\n**TODO:**\n- Dynamic learning rate\n- more EDA and Feature Engineering","metadata":{}}]}