{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Set Parameters","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-05T16:55:48.152803Z","iopub.execute_input":"2022-10-05T16:55:48.153465Z","iopub.status.idle":"2022-10-05T16:55:48.181734Z","shell.execute_reply.started":"2022-10-05T16:55:48.153332Z","shell.execute_reply":"2022-10-05T16:55:48.180363Z"}}},{"cell_type":"code","source":"do_kfold = True","metadata":{"execution":{"iopub.status.busy":"2022-10-05T19:28:11.989224Z","iopub.execute_input":"2022-10-05T19:28:11.989743Z","iopub.status.idle":"2022-10-05T19:28:11.997525Z","shell.execute_reply.started":"2022-10-05T19:28:11.989704Z","shell.execute_reply":"2022-10-05T19:28:11.995873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n!pip install datatable\nimport datatable as dt\n\nfrom lightgbm import LGBMClassifier\n\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.metrics import log_loss\nfrom sklearn.model_selection import StratifiedGroupKFold\n\nimport gc","metadata":{"execution":{"iopub.status.busy":"2022-10-05T19:28:12.002920Z","iopub.execute_input":"2022-10-05T19:28:12.003563Z","iopub.status.idle":"2022-10-05T19:28:26.226354Z","shell.execute_reply.started":"2022-10-05T19:28:12.003505Z","shell.execute_reply":"2022-10-05T19:28:26.225281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"# Read data\n# Thanks @Sergey Saharovskiy (https://www.kaggle.com/competitions/tabular-playground-series-oct-2022/discussion/356544)\ndtypes_dict = {\n    'game_num': 'int8',\n    'ball_pos_x': 'float16', 'ball_pos_y': 'float16', 'ball_pos_z': 'float16',\n    'ball_vel_x': 'float16', 'ball_vel_y': 'float16', 'ball_vel_z': 'float16',\n    'p0_pos_x': 'float16', 'p0_pos_y': 'float16', 'p0_pos_z': 'float16',\n    'p0_vel_x': 'float16', 'p0_vel_y': 'float16', 'p0_vel_z': 'float16',\n    'p0_boost': 'float16', 'p1_pos_x': 'float16', 'p1_pos_y': 'float16',\n    'p1_pos_z': 'float16', 'p1_vel_x': 'float16', 'p1_vel_y': 'float16',\n    'p1_vel_z': 'float16', 'p1_boost': 'float16', 'p2_pos_x': 'float16',\n    'p2_pos_y': 'float16', 'p2_pos_z': 'float16', 'p2_vel_x': 'float16',\n    'p2_vel_y': 'float16', 'p2_vel_z': 'float16', 'p2_boost': 'float16',\n    'p3_pos_x': 'float16', 'p3_pos_y': 'float16', 'p3_pos_z': 'float16',\n    'p3_vel_x': 'float16', 'p3_vel_y': 'float16', 'p3_vel_z': 'float16',\n    'p3_boost': 'float16', 'p4_pos_x': 'float16', 'p4_pos_y': 'float16',\n    'p4_pos_z': 'float16', 'p4_vel_x': 'float16', 'p4_vel_y': 'float16',\n    'p4_vel_z': 'float16', 'p4_boost': 'float16', 'p5_pos_x': 'float16',\n    'p5_pos_y': 'float16', 'p5_pos_z': 'float16', 'p5_vel_x': 'float16',\n    'p5_vel_y': 'float16', 'p5_vel_z': 'float16', 'p5_boost': 'float16',\n    'boost0_timer': 'float16', 'boost1_timer': 'float16', 'boost2_timer': 'float16',\n    'boost3_timer': 'float16', 'boost4_timer': 'float16', 'boost5_timer': 'float16',\n    'team_A_scoring_within_10sec': 'O',\n    'team_B_scoring_within_10sec': 'O'\n}\n\n\npath_to_data = '../input/tabular-playground-series-oct-2022'\ndf = pd.DataFrame({}, columns=dtypes_dict.keys())\nfor i in range(3):\n    dt_read = dt.fread(f'{path_to_data}/train_{i}.csv').to_pandas()\n    dt_read = dt_read.drop([\"player_scoring_next\", \"team_scoring_next\", \"event_id\", \"event_time\"], axis=1)\n    dt_read = dt_read.astype(dtypes_dict)\n    df = pd.concat([df, dt_read])\n    del dt_read\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-05T19:28:26.228368Z","iopub.execute_input":"2022-10-05T19:28:26.228688Z","iopub.status.idle":"2022-10-05T19:28:56.491799Z","shell.execute_reply.started":"2022-10-05T19:28:26.228655Z","shell.execute_reply":"2022-10-05T19:28:56.490501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.index = range(len(df.index))\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-05T19:28:56.493407Z","iopub.execute_input":"2022-10-05T19:28:56.494231Z","iopub.status.idle":"2022-10-05T19:28:56.639206Z","shell.execute_reply.started":"2022-10-05T19:28:56.494173Z","shell.execute_reply":"2022-10-05T19:28:56.637821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Mirror Data to create additional training data","metadata":{}},{"cell_type":"code","source":"# Mirrors team A and Team B\n# Thanks @Maher el Ouahabi for th clarification of the axis (https://www.kaggle.com/competitions/tabular-playground-series-oct-2022/discussion/356789)\ndef change_teams(df):\n    # What needs to be mirrored: y-position and y-velocity \n    # What not needs to be mirrored: z-psoition and z-velocity\n    \n    # Mirror Ball\n    df[\"ball_pos_y\"] = df[\"ball_pos_y\"] * -1\n    df[\"ball_vel_y\"] = df[\"ball_vel_y\"] * -1\n    df[\"ball_pos_x\"]= df[\"ball_pos_x\"] * -1\n    df[\"ball_vel_x\"] = df[\"ball_vel_x\"] * -1\n    \n    # Mirror player\n    for p in range(3):\n        df = mirror_player(df, p, \"y\")\n        df = mirror_player(df, p, \"x\")     \n    \n    # Mirror Booster \n    # Booster position:  [ (-61.4, -81.9), (61.4, -81.9), (-71.7, 0), (71.7, 0), (-61.4, 81.9), (61.4, 81.9) ]\n    s = df[\"boost0_timer\"].copy()\n    df[\"boost0_timer\"] = df[\"boost5_timer\"]\n    df[\"boost5_timer\"] = s\n    \n    s = df[\"boost1_timer\"].copy()\n    df[\"boost1_timer\"] = df[\"boost4_timer\"]\n    df[\"boost4_timer\"] = s\n    \n    s = df[\"boost2_timer\"].copy()\n    df[\"boost2_timer\"] = df[\"boost3_timer\"]\n    df[\"boost3_timer\"] = s\n    \n    # columns from train data\n    if 'team_A_scoring_within_10sec' in df.columns:\n        s = df[\"team_A_scoring_within_10sec\"].copy()\n        df[\"team_A_scoring_within_10sec\"] = df[\"team_B_scoring_within_10sec\"]\n        df[\"team_B_scoring_within_10sec\"] = s\n    \n    return df\n    \n    \n# p is player 0,1 or 2\n# a is axis. a=y for mirroring on x axis\ndef mirror_player(df, p, a):\n    df[f\"p{p}_pos_{a}\"] = df[f\"p{p}_pos_{a}\"] * -1\n    df[f\"p{p+3}_pos_{a}\"] = df[f\"p{p+3}_pos_{a}\"] * -1\n    df[f\"p{p}_vel_{a}\"] = df[f\"p{p}_vel_{a}\"] * -1\n    df[f\"p{p+3}_vel_{a}\"] = df[f\"p{p+3}_vel_{a}\"] * -1\n    \n    s = df[f\"p{p}_pos_{a}\"].copy()\n    df[f\"p{p}_pos_{a}\"] = df[f\"p{p+3}_pos_{a}\"]\n    df[f\"p{p+3}_pos_{a}\"] = s\n    \n    s = df[f\"p{p}_vel_{a}\"].copy()\n    df[f\"p{p}_vel_{a}\"] = df[f\"p{p+3}_vel_{a}\"]\n    df[f\"p{p+3}_vel_{a}\"] = s\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-10-05T19:28:56.641186Z","iopub.execute_input":"2022-10-05T19:28:56.641898Z","iopub.status.idle":"2022-10-05T19:28:56.654750Z","shell.execute_reply.started":"2022-10-05T19:28:56.641848Z","shell.execute_reply":"2022-10-05T19:28:56.653435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mirror = change_teams(df)\ndf = pd.concat([df, mirror])\ndel(mirror)\ndf.index = range(len(df.index))\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-05T19:28:56.658758Z","iopub.execute_input":"2022-10-05T19:28:56.659162Z","iopub.status.idle":"2022-10-05T19:29:01.189798Z","shell.execute_reply.started":"2022-10-05T19:28:56.659129Z","shell.execute_reply":"2022-10-05T19:29:01.188447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Creation and data cleaning","metadata":{}},{"cell_type":"code","source":"def feature_creation(df):\n    \n    # Distance to the Ball for each player\n    for i in range(6):\n        df[f\"p{i}_ball_distance\"] = ((df[\"ball_pos_x\"]-df[f\"p{i}_pos_x\"])**2 + (df[\"ball_pos_y\"]-df[f\"p{i}_pos_y\"])**2 + (df[\"ball_pos_z\"]-df[f\"p{i}_pos_z\"])**2)**0.5\n            \n#     # Player missing\n#     df[\"teamA_less_players\"] = np.isnan(df[\"p0_pos_x\"]) | np.isnan(df[\"p1_pos_x\"]) | np.isnan(df[\"p2_pos_x\"])\n#     df[\"teamB_less_players\"] = np.isnan(df[\"p3_pos_x\"]) | np.isnan(df[\"p4_pos_x\"]) | np.isnan(df[\"p5_pos_x\"])\n    \n#     # https://www.kaggle.com/code/infrarosso/tps-oct-2022-eda-lgbm-model-ensemble/notebook#Features-Engineering\n#     # two simple new features rapresenting the distance between the ball and the gate of a team\n#     df[\"goal_a_distance\"] = ((df[\"ball_pos_x\"]-0)**2 + (df[\"ball_pos_y\"]-100)**2 + (df[\"ball_pos_z\"]-20)**2)**0.5\n#     df[\"goal_b_distance\"] = ((df[\"ball_pos_x\"]-0)**2 + (df[\"ball_pos_y\"]+100)**2 + (df[\"ball_pos_z\"]-20)**2)**0.5\n    \n    return df\n\ndf = feature_creation(df)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T19:29:01.191660Z","iopub.execute_input":"2022-10-05T19:29:01.192447Z","iopub.status.idle":"2022-10-05T19:29:10.291548Z","shell.execute_reply.started":"2022-10-05T19:29:01.192398Z","shell.execute_reply":"2022-10-05T19:29:10.290282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Remove null values\ndef replace_null_values(df):\n    # Missing player positions should be replaced with their spawning position\n    # Team A pos x -> -56\n    # Team A pos y -> --92\n    # Team A pos z -> 0.5\n    # Team B pos x -> -54\n    # Team B pos y -> 92\n    \n    # Velocity z = 33\n    \n    \n    # z position and velocity should be 0\n    df = df.fillna(0)\n    return df\n\ndf = replace_null_values(df)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T19:29:10.293453Z","iopub.execute_input":"2022-10-05T19:29:10.293844Z","iopub.status.idle":"2022-10-05T19:29:21.962017Z","shell.execute_reply.started":"2022-10-05T19:29:10.293807Z","shell.execute_reply":"2022-10-05T19:29:21.960689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"def prepare_df(df):\n    # Save group\n    groups = df[\"game_num\"]\n    \n    remove_colums = [\"game_num\",'team_B_scoring_within_10sec']\n                     # \"event_id\",\"event_time\",\n                     # \"player_scoring_next\",\"team_scoring_next\",\n                    # \"p0_vel_z\", \"p1_vel_z\", \"p2_vel_z\", \"p3_vel_z\", \"p4_vel_z\", \"p5_vel_z\",\n                    # \"p0_pos_z\", \"p1_pos_z\", \"p2_pos_z\", \"p3_pos_z\", \"p4_pos_z\", \"p5_pos_z\"\n                    \n\n    for col in remove_colums:\n        del df[col]\n        gc.collect()\n     \n    yA = df[\"team_A_scoring_within_10sec\"].astype('int8')\n    \n    X = df.copy()\n    del X[\"team_A_scoring_within_10sec\"]\n    del df\n    \n    return X, yA, groups\n\nX, yA, groups = prepare_df(df)\ndel(df)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T19:29:50.447400Z","iopub.execute_input":"2022-10-05T19:29:50.447918Z","iopub.status.idle":"2022-10-05T19:29:52.775832Z","shell.execute_reply.started":"2022-10-05T19:29:50.447876Z","shell.execute_reply":"2022-10-05T19:29:52.774833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train\n# Didnt found most of these parameter myself, but the notebook i gt it from is not available anymore\nparams = {\n    'histogram_pool_size': 4096,\n    'objective': 'binary',\n    # 'seed': 42,\n    'num_leaves': 128,\n    'n_estimators': 100,\n    'max_depth': 10, # was 10\n    'learning_rate': 0.1,\n    'feature_fraction': 0.75, # was 0.75\n    'subsample': 0.7,\n    'subsample_freq': 8,\n    'n_jobs': -1,\n    'reg_alpha': 1,\n    'reg_lambda': 2,\n    'min_child_samples': 90,\n}\n\n# eval_setA = [(X_test, yA_test), (X_train, yA_train)]\n# eval_setB = [(X_test, yB_test), (X_train, yB_train)]","metadata":{"execution":{"iopub.status.busy":"2022-10-05T19:29:53.817007Z","iopub.execute_input":"2022-10-05T19:29:53.817762Z","iopub.status.idle":"2022-10-05T19:29:53.825065Z","shell.execute_reply.started":"2022-10-05T19:29:53.817711Z","shell.execute_reply":"2022-10-05T19:29:53.824059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pipeline\npipeA = Pipeline([\n    ('ss', StandardScaler()),\n    ('lgbm_model', LGBMClassifier(**params))\n])","metadata":{"execution":{"iopub.status.busy":"2022-10-05T19:29:54.260402Z","iopub.execute_input":"2022-10-05T19:29:54.260851Z","iopub.status.idle":"2022-10-05T19:29:54.266328Z","shell.execute_reply.started":"2022-10-05T19:29:54.260805Z","shell.execute_reply":"2022-10-05T19:29:54.264648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Group Kfold\n# ROB MULLA (https://www.kaggle.com/code/robikscube/cross-validation-visualized-youtube-tutorial/notebook)\n\n\nif do_kfold:\n    sgk = StratifiedGroupKFold(n_splits=3, shuffle=True)\n\n    fold = 0\n    log_losses = []\n    for train_idx, val_idx in sgk.split(X, yA, groups):\n        X_tr = X.loc[train_idx]\n        y_tr = yA.loc[train_idx]\n\n        X_val = X.loc[val_idx]\n        y_val = yA.loc[val_idx]\n\n        # Fit Model on Train\n        pipeA.fit(X_tr, y_tr)\n        pred_prob = pipeA.predict_proba(X_val)[:, 1]\n        logloss_score = log_loss(y_val, pred_prob)\n\n        print(f\"======= Fold {fold} ========\")\n        print(\n            f\"Our log loss on the validation set is {logloss_score:0.4f}\"\n        )\n        fold += 1\n        log_losses.append(logloss_score)\n\n\n    average_log_loss = np.mean(log_losses)\n    print(f'Our average log loss out of {fold:0.4f} folds is {average_log_loss:0.4f}')\n\n    del(X_tr)\n    del(y_tr)\n    del(X_val)\n    del(y_val)\n    del(pred_prob)\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-05T19:29:54.885122Z","iopub.execute_input":"2022-10-05T19:29:54.885582Z","iopub.status.idle":"2022-10-05T19:43:20.099962Z","shell.execute_reply.started":"2022-10-05T19:29:54.885543Z","shell.execute_reply":"2022-10-05T19:43:20.098371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# scores with first file\n# feature_fraction=0.6 -> 0.2028\n# feature_fraction=0.75 -> 0.2024\n# feature_fraction=0.8 -> 0.2027\n# feature_fraction=0.9 -> 0.2028\n# feature_fraction=1 -> 0.2029\n\n# max_depth=9 -> 0.2031\n# max_depth=11 -> 0.2025\n# max_depth=13 -> 0.2028\n\n# feature_fraction=1 and no Scaler -> 0.2023","metadata":{"execution":{"iopub.status.busy":"2022-10-05T19:43:20.102691Z","iopub.execute_input":"2022-10-05T19:43:20.103614Z","iopub.status.idle":"2022-10-05T19:43:20.108486Z","shell.execute_reply.started":"2022-10-05T19:43:20.103566Z","shell.execute_reply":"2022-10-05T19:43:20.106982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train with all data\npipeA.fit(X, yA)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T19:43:20.109952Z","iopub.execute_input":"2022-10-05T19:43:20.110406Z","iopub.status.idle":"2022-10-05T19:49:33.788636Z","shell.execute_reply.started":"2022-10-05T19:43:20.110372Z","shell.execute_reply":"2022-10-05T19:49:33.787043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Importance","metadata":{}},{"cell_type":"code","source":"# https://www.kaggle.com/code/ashishpatel26/feature-importance-of-lightgbm#Feature-importance\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\n# sorted(zip(clf.feature_importances_, X.columns), reverse=True)\nfeature_imp = pd.DataFrame(sorted(zip(pipeA[1].feature_importances_,X.columns)), columns=['Value','Feature'])\n\nplt.figure(figsize=(20, 15))\nsns.barplot(x=\"Value\", y=\"Feature\", data=feature_imp.sort_values(by=\"Value\", ascending=False))\nplt.title('LightGBM Features')\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-05T19:49:33.791848Z","iopub.execute_input":"2022-10-05T19:49:33.792242Z","iopub.status.idle":"2022-10-05T19:49:35.322712Z","shell.execute_reply.started":"2022-10-05T19:49:33.792203Z","shell.execute_reply":"2022-10-05T19:49:35.321196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del(X)\ndel(yA)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-05T19:49:35.325173Z","iopub.execute_input":"2022-10-05T19:49:35.325667Z","iopub.status.idle":"2022-10-05T19:49:35.528476Z","shell.execute_reply.started":"2022-10-05T19:49:35.325618Z","shell.execute_reply":"2022-10-05T19:49:35.526729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict Team A and B","metadata":{"execution":{"iopub.status.busy":"2022-10-04T17:59:13.514728Z","iopub.execute_input":"2022-10-04T17:59:13.515051Z","iopub.status.idle":"2022-10-04T17:59:13.520644Z","shell.execute_reply.started":"2022-10-04T17:59:13.515017Z","shell.execute_reply":"2022-10-04T17:59:13.519510Z"}}},{"cell_type":"code","source":"# Read Data\ndtypes_dict = {\n    'id': 'int64',\n    'ball_pos_x': 'float16', 'ball_pos_y': 'float16', 'ball_pos_z': 'float16',\n    'ball_vel_x': 'float16', 'ball_vel_y': 'float16', 'ball_vel_z': 'float16',\n    'p0_pos_x': 'float16', 'p0_pos_y': 'float16', 'p0_pos_z': 'float16',\n    'p0_vel_x': 'float16', 'p0_vel_y': 'float16', 'p0_vel_z': 'float16',\n    'p0_boost': 'float16', 'p1_pos_x': 'float16', 'p1_pos_y': 'float16',\n    'p1_pos_z': 'float16', 'p1_vel_x': 'float16', 'p1_vel_y': 'float16',\n    'p1_vel_z': 'float16', 'p1_boost': 'float16', 'p2_pos_x': 'float16',\n    'p2_pos_y': 'float16', 'p2_pos_z': 'float16', 'p2_vel_x': 'float16',\n    'p2_vel_y': 'float16', 'p2_vel_z': 'float16', 'p2_boost': 'float16',\n    'p3_pos_x': 'float16', 'p3_pos_y': 'float16', 'p3_pos_z': 'float16',\n    'p3_vel_x': 'float16', 'p3_vel_y': 'float16', 'p3_vel_z': 'float16',\n    'p3_boost': 'float16', 'p4_pos_x': 'float16', 'p4_pos_y': 'float16',\n    'p4_pos_z': 'float16', 'p4_vel_x': 'float16', 'p4_vel_y': 'float16',\n    'p4_vel_z': 'float16', 'p4_boost': 'float16', 'p5_pos_x': 'float16',\n    'p5_pos_y': 'float16', 'p5_pos_z': 'float16', 'p5_vel_x': 'float16',\n    'p5_vel_y': 'float16', 'p5_vel_z': 'float16', 'p5_boost': 'float16',\n    'boost0_timer': 'float16', 'boost1_timer': 'float16', 'boost2_timer': 'float16',\n    'boost3_timer': 'float16', 'boost4_timer': 'float16', 'boost5_timer': 'float16',\n}\n\n\npath_to_data = '../input/tabular-playground-series-oct-2022'\ntest_df = pd.DataFrame({}, columns=dtypes_dict.keys())\ndt_read = dt.fread(f'{path_to_data}/test.csv').to_pandas()\ndt_read = dt_read.astype(dtypes_dict)\ntest_df = pd.concat([test_df, dt_read])\ndel dt_read\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-05T19:49:35.531464Z","iopub.execute_input":"2022-10-05T19:49:35.532072Z","iopub.status.idle":"2022-10-05T19:49:38.970409Z","shell.execute_reply.started":"2022-10-05T19:49:35.532023Z","shell.execute_reply":"2022-10-05T19:49:38.968724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids = test_df[\"id\"]\ntest_df = test_df.drop([\"id\"], axis=1)\ntest_df","metadata":{"execution":{"iopub.status.busy":"2022-10-05T19:49:38.972169Z","iopub.execute_input":"2022-10-05T19:49:38.972635Z","iopub.status.idle":"2022-10-05T19:49:39.153877Z","shell.execute_reply.started":"2022-10-05T19:49:38.972586Z","shell.execute_reply":"2022-10-05T19:49:39.152697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prepare test data\ntest_df = feature_creation(test_df)\ntest_df = replace_null_values(test_df)\n\n# predict Team A\npredictionA = pipeA.predict_proba(test_df)[:, 1]\n\n# Predict Team B\ntest_df = change_teams(test_df)\npredictionB = pipeA.predict_proba(test_df)[:, 1]","metadata":{"execution":{"iopub.status.busy":"2022-10-05T19:49:39.155457Z","iopub.execute_input":"2022-10-05T19:49:39.155836Z","iopub.status.idle":"2022-10-05T19:49:43.857103Z","shell.execute_reply.started":"2022-10-05T19:49:39.155792Z","shell.execute_reply":"2022-10-05T19:49:43.854995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create Submission","metadata":{}},{"cell_type":"code","source":"df_submission = pd.DataFrame(\n    {\n        \"id\": ids,\n        \"team_A_scoring_within_10sec\": predictionA,\n        \"team_B_scoring_within_10sec\": predictionB\n    }\n)\ndf_submission","metadata":{"execution":{"iopub.status.busy":"2022-10-05T19:49:43.858162Z","iopub.status.idle":"2022-10-05T19:49:43.858632Z","shell.execute_reply.started":"2022-10-05T19:49:43.858416Z","shell.execute_reply":"2022-10-05T19:49:43.858435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T19:49:43.860448Z","iopub.status.idle":"2022-10-05T19:49:43.860892Z","shell.execute_reply.started":"2022-10-05T19:49:43.860661Z","shell.execute_reply":"2022-10-05T19:49:43.860679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}