{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# CatBoost baseline + online learning model","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport gc # garbage collector\nimport random\n\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import log_loss\nfrom sklearn.model_selection import train_test_split, StratifiedKFold\nfrom catboost import CatBoostClassifier\nimport seaborn as sns\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-23T13:52:39.283385Z","iopub.execute_input":"2022-10-23T13:52:39.287048Z","iopub.status.idle":"2022-10-23T13:52:40.732622Z","shell.execute_reply.started":"2022-10-23T13:52:39.286908Z","shell.execute_reply":"2022-10-23T13:52:40.731388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading dataset in parquet format\n\nI have create this dataset in parquet (float16) format to speed up the data loading time [here](https://www.kaggle.com/datasets/alvinleenh/tps-rocket-league-data-float16-parquet-format)\n","metadata":{"execution":{"iopub.status.busy":"2022-10-02T07:33:20.146427Z","iopub.execute_input":"2022-10-02T07:33:20.146891Z","iopub.status.idle":"2022-10-02T07:33:20.151487Z","shell.execute_reply.started":"2022-10-02T07:33:20.146851Z","shell.execute_reply":"2022-10-02T07:33:20.150657Z"}}},{"cell_type":"code","source":"#train0_df = pd.read_parquet('/kaggle/input/tps-rocket-league-data-float16-parquet-format/train_0.parquet.gzip')","metadata":{"execution":{"iopub.status.busy":"2022-10-23T13:52:40.734454Z","iopub.execute_input":"2022-10-23T13:52:40.734829Z","iopub.status.idle":"2022-10-23T13:52:40.743673Z","shell.execute_reply.started":"2022-10-23T13:52:40.734792Z","shell.execute_reply":"2022-10-23T13:52:40.740010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing\n**CREDIT:** Marcociav for 3D euclidian norm feature [here](https://www.kaggle.com/code/chazzer/rocket-league-xgboost-feat-engineering-cv/notebook?scriptVersionId=107113700)\n\nUpdated feature engineering:\n* fill NA for player position and velocity\n* 3D euclidian norm of distances between player and ball\n* if player hit the ball\n* 3D euclidian norm of distances between player and goal\n* 3D euclidian norm of distances between ball and goal\n* magnitude of player's velocity\n* magnitude of ball's velocity","metadata":{}},{"cell_type":"code","source":"def fillnaPlayer(data):    \n    for i in range(6):\n        data[f'p{i}_pos_x'] = data[f'p{i}_pos_x'].fillna(0)\n        data[f'p{i}_pos_y'] = data[f'p{i}_pos_y'].fillna(0)\n        data[f'p{i}_pos_z'] = data[f'p{i}_pos_z'].fillna(0)\n        data[f'p{i}_vel_x'] = data[f'p{i}_vel_x'].fillna(0)\n        data[f'p{i}_vel_y'] = data[f'p{i}_vel_y'].fillna(0)\n        data[f'p{i}_vel_z'] = data[f'p{i}_vel_z'].fillna(0)\n    return data\n\ndef generatePlayerBallDiff(data):    \n    for i in range(6):\n        data[f'p{i}_ball_x_diff'] = abs(data['ball_pos_x']-data[f'p{i}_pos_x'])\n        data[f'p{i}_ball_y_diff'] = abs(data['ball_pos_y']-data[f'p{i}_pos_y'])\n        data[f'p{i}_ball_z_diff'] = abs(data['ball_pos_z']-data[f'p{i}_pos_z'])\n        data[f'p{i}_ball'] = np.sqrt(data[f'p{i}_ball_x_diff']**2 +\n                                     data[f'p{i}_ball_y_diff']**2 +\n                                     data[f'p{i}_ball_z_diff']**2)\n        data[f'p{i}_hit_ball'] = 0\n        data.loc[(data[f'p{i}_ball_x_diff']+data[f'p{i}_ball_y_diff']+data[f'p{i}_ball_z_diff']) < 10\n                 ,f'p{i}_hit_ball'] = 1\n    return data\n\ndef generatePlayerGoalDiff(data):    \n    goalx = [0,0]\n    goaly = [-100,100]\n    goalz = [6.8,6.8]\n    for i in range(6):\n        for count, j in enumerate(['A','B']):\n            data[f'p{i}_goal_x_diff'] = abs(data[f'p{i}_pos_x']-goalx[count])\n            data[f'p{i}_goal_y_diff'] = abs(data[f'p{i}_pos_y']-goalx[count])\n            data[f'p{i}_goal_z_diff'] = abs(data[f'p{i}_pos_z']-goalx[count])\n            data[f'p{i}_goal{j}'] = np.sqrt(data[f'p{i}_goal_x_diff']**2 +\n                                            data[f'p{i}_goal_y_diff']**2 +\n                                            data[f'p{i}_goal_z_diff']**2)\n    return data\n\ndef generateBallGoalDiff(data):    \n    goalx = [0,0]\n    goaly = [-100,100]\n    goalz = [6.8,6.8]\n    for count, j in enumerate(['A','B']):\n        data['ball_goal_x_diff'] = abs(data['ball_pos_x']-goalx[count])\n        data['ball_goal_y_diff'] = abs(data['ball_pos_y']-goalx[count])\n        data['ball_goal_z_diff'] = abs(data['ball_pos_z']-goalx[count])\n        data[f'ball_goal{j}'] = np.sqrt(data['ball_goal_x_diff']**2 +\n                                        data['ball_goal_y_diff']**2 +\n                                        data['ball_goal_z_diff']**2)\n    return data\n\ndef generatePlayerVel(data): \n    for i in range(6):\n        data[f'p{i}_vel'] = np.sqrt(data[f'p{i}_vel_x']**2 +\n                                    data[f'p{i}_vel_y']**2 +\n                                    data[f'p{i}_vel_z']**2)\n    return data\n\ndef generateBallVel(data):    \n    data['ball_vel_mag'] = np.sqrt(data['ball_vel_x']**2 +\n                                   data['ball_vel_y']**2 +\n                                   data['ball_vel_z']**2)\n    return data","metadata":{"execution":{"iopub.status.busy":"2022-10-23T13:52:40.745489Z","iopub.execute_input":"2022-10-23T13:52:40.745875Z","iopub.status.idle":"2022-10-23T13:52:40.770726Z","shell.execute_reply.started":"2022-10-23T13:52:40.745833Z","shell.execute_reply":"2022-10-23T13:52:40.769632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Mirror the board\n\n**CREDIT:** Nigel A. R. Henry's discussion [here](https://www.kaggle.com/competitions/tabular-playground-series-oct-2022/discussion/357339), ŞAFAK TÜRKELI's notebook [here](https://www.kaggle.com/code/sfktrkl/tps-oct-2022#Feature-Engineering) and \nSPYROW's notebook [here](https://www.kaggle.com/code/spyrow/playground-oct-2022-lgbmclassifier?scriptVersionId=107206095)\n\nBased on Nigel's finding \"Team A won 4.5% more games than Team B\"\nTherefore, we have to generalize the dataset, but mirroring the board, as suggested by SAFAK and SPYROW\n* Mirror the ball\n* Mirror the player\n* Mirror boosts\n* Mirror Team A/B goals","metadata":{}},{"cell_type":"code","source":"def mirror_ball(data):\n    # Mirror the coordinates\n    data[\"ball_pos_y\"] = data[\"ball_pos_y\"] * -1\n    data[\"ball_vel_y\"] = data[\"ball_vel_y\"] * -1\n    data[\"ball_pos_x\"] = data[\"ball_pos_x\"] * -1\n    data[\"ball_vel_x\"] = data[\"ball_vel_x\"] * -1\n    return data\n    \ndef mirror_players(data):\n    # Mirror the coordinates\n    def mirror(data, p, a):\n        data[f\"p{p}_pos_{a}\"] = data[f\"p{p}_pos_{a}\"] * -1\n        data[f\"p{p+3}_pos_{a}\"] = data[f\"p{p+3}_pos_{a}\"] * -1\n        data[f\"p{p}_vel_{a}\"] = data[f\"p{p}_vel_{a}\"] * -1\n        data[f\"p{p+3}_vel_{a}\"] = data[f\"p{p+3}_vel_{a}\"] * -1\n        \n        tmp= data[f\"p{p}_pos_{a}\"].copy()\n        data[f\"p{p}_pos_{a}\"] = data[f\"p{p+3}_pos_{a}\"]\n        data[f\"p{p+3}_pos_{a}\"] = tmp\n\n        tmp= data[f\"p{p}_vel_{a}\"].copy()\n        data[f\"p{p}_vel_{a}\"] = data[f\"p{p+3}_vel_{a}\"]\n        data[f\"p{p+3}_vel_{a}\"] = tmp\n        return data\n    \n    for p in range(3):\n        data = mirror(data, p, \"y\")\n        data = mirror(data, p, \"x\")\n    return data\n\ndef mirror_others(data): \n    tmp= data[\"boost0_timer\"].copy()\n    data[\"boost0_timer\"] = data[\"boost3_timer\"]\n    data[\"boost3_timer\"] = tmp\n    \n    tmp= data[\"boost1_timer\"].copy()\n    data[\"boost1_timer\"] = data[\"boost4_timer\"]\n    data[\"boost4_timer\"] = tmp\n    \n    tmp= data[\"boost2_timer\"].copy()\n    data[\"boost2_timer\"] = data[\"boost5_timer\"]\n    data[\"boost5_timer\"] = tmp\n    \n    tmp= data[\"team_A_scoring_within_10sec\"].copy()\n    data[\"team_A_scoring_within_10sec\"] = data[\"team_B_scoring_within_10sec\"]\n    data[\"team_B_scoring_within_10sec\"] = tmp\n        \n    return data\n\ndef mirror_board(data):\n    mirrordata = data.copy()\n    mirrordata = mirror_ball(mirrordata)\n    mirrordata = mirror_players(mirrordata)\n    mirrordata = mirror_others(mirrordata)\n    \n    data = pd.concat([data, mirrordata])\n    data = data.reset_index(drop=True)\n    return data","metadata":{"execution":{"iopub.status.busy":"2022-10-23T13:52:40.777023Z","iopub.execute_input":"2022-10-23T13:52:40.778182Z","iopub.status.idle":"2022-10-23T13:52:40.807242Z","shell.execute_reply.started":"2022-10-23T13:52:40.778141Z","shell.execute_reply":"2022-10-23T13:52:40.806291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocessing(data):  \n    #if 'game_num' in data.columns:\n    #    data = mirror_board(data)\n    \n    if 'game_num' in data.columns:\n        data = data.drop(columns=['event_id', 'event_time','player_scoring_next','team_scoring_next'])\n    if 'id' in data.columns:\n        data = data.drop(columns='id')\n    fillnaPlayer(data)\n    generatePlayerBallDiff(data)\n    generatePlayerGoalDiff(data)\n    generateBallGoalDiff(data)\n    generatePlayerVel(data)\n    generateBallVel(data)\n    \n    return data","metadata":{"execution":{"iopub.status.busy":"2022-10-23T13:52:40.808823Z","iopub.execute_input":"2022-10-23T13:52:40.816690Z","iopub.status.idle":"2022-10-23T13:52:40.831672Z","shell.execute_reply.started":"2022-10-23T13:52:40.809599Z","shell.execute_reply":"2022-10-23T13:52:40.830632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train/test split without target leakage\n\n**CREDIT:** Alex from discussion [here](https://www.kaggle.com/competitions/tabular-playground-series-oct-2022/discussion/359714) and Patrick from notebook [here](https://www.kaggle.com/code/paddykb/tps-2022-10-fastai)","metadata":{}},{"cell_type":"code","source":"%%time\ntrain0_df = pd.read_parquet('/kaggle/input/tps-rocket-league-data-float16-parquet-format/train_0.parquet.gzip')\n\ntrain = preprocessing(train0_df)\n# split train & validation\n\ngame_nums = train['game_num'].unique()\ntrain_game_nums = random.sample(list(game_nums), int(len(game_nums) * 0.80))\n\nfeatures = train.columns.values\nfeatures = np.delete(features,np.argwhere(features=='team_A_scoring_within_10sec'))\nfeatures = np.delete(features,np.argwhere(features=='team_B_scoring_within_10sec'))\nfeatures = np.delete(features,np.argwhere(features=='game_num'))\n\nTARGET = ['team_A_scoring_within_10sec','team_B_scoring_within_10sec']\nX_train = train.query(\"game_num in @train_game_nums\")[features]\ny_train  = train.query(\"game_num in @train_game_nums\")[TARGET]\nX_val = train.query(\"game_num not in @train_game_nums\")[features]\ny_val  = train.query(\"game_num not in @train_game_nums\")[TARGET]\n\ndel train0_df, train; \ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-23T13:52:40.833214Z","iopub.execute_input":"2022-10-23T13:52:40.833876Z","iopub.status.idle":"2022-10-23T13:52:53.302527Z","shell.execute_reply.started":"2022-10-23T13:52:40.833841Z","shell.execute_reply":"2022-10-23T13:52:53.301508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CTB baseline model prediction","metadata":{}},{"cell_type":"code","source":"def trainBaseModel(model):\n    for i, feature in enumerate(TARGET):\n        model[i].fit(X_train,y_train.iloc[:,i],eval_set=(X_val,y_val.iloc[:,i]),verbose=False)    \n        y_pred = model[i].predict_proba(X_val)[:,1]\n        loss = log_loss(y_val.iloc[:,i],y_pred)\n        print(f\"LogLoss for {feature} = {loss}\")\n    return","metadata":{"execution":{"iopub.status.busy":"2022-10-23T13:52:53.304041Z","iopub.execute_input":"2022-10-23T13:52:53.304666Z","iopub.status.idle":"2022-10-23T13:52:53.312388Z","shell.execute_reply.started":"2022-10-23T13:52:53.304628Z","shell.execute_reply":"2022-10-23T13:52:53.310579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clfA = CatBoostClassifier(metric_period = 100,verbose=0,task_type='GPU')\nclfB = CatBoostClassifier(metric_period = 100,verbose=0,task_type='GPU')\nbase_model = [clfA, clfB]\n\ntrainBaseModel(base_model)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T13:52:53.314227Z","iopub.execute_input":"2022-10-23T13:52:53.314689Z","iopub.status.idle":"2022-10-23T13:55:26.783881Z","shell.execute_reply.started":"2022-10-23T13:52:53.314652Z","shell.execute_reply":"2022-10-23T13:55:26.781958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Batch training\n\nBlend tress and counters of trained CatBoost model into a new model with sum_model method","metadata":{}},{"cell_type":"code","source":"def batchTraining(batch,y,model,k):\n    for i, feature in enumerate(TARGET):\n        model[i].fit(batch[i],verbose=False)    \n        y_pred = model[i].predict_proba(x)[:,1]\n        loss = log_loss(y[i],y_pred)\n        print(f\"LogLoss for {feature} = {loss}\")\n        \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-10-23T13:55:26.785441Z","iopub.execute_input":"2022-10-23T13:55:26.785754Z","iopub.status.idle":"2022-10-23T13:55:26.791401Z","shell.execute_reply.started":"2022-10-23T13:55:26.785726Z","shell.execute_reply":"2022-10-23T13:55:26.790347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from catboost import Pool, sum_models, to_classifier\n\nallmodelA = []\nallmodelB = []\nfor i in range(0,10):\n    train_df = pd.read_parquet(f'/kaggle/input/tps-rocket-league-data-float16-parquet-format/train_{i}.parquet.gzip')\n    train = preprocessing(train_df)\n    del train_df\n    gc.collect()\n    TARGET = ['team_A_scoring_within_10sec','team_B_scoring_within_10sec']\n    \n    x = train.drop(columns=TARGET)\n    x = x.drop(columns='game_num')\n    yA = train['team_A_scoring_within_10sec']\n    yB = train['team_B_scoring_within_10sec']\n    ytarget = [yA, yB]\n    clfA = CatBoostClassifier(metric_period = 100,verbose=0,task_type='GPU')\n    clfB = CatBoostClassifier(metric_period = 100,verbose=0,task_type='GPU')\n    base_model = [clfA, clfB]\n    batch = [Pool(x,label=yA), Pool(x,label=yB)]\n    print(f'\\n\\nOnline training with train_{i} dataset\\n')\n    model = batchTraining(batch,ytarget,base_model,i)\n    gc.collect()\n    allmodelA.append(model[0])\n    allmodelB.append(model[1])","metadata":{"execution":{"iopub.status.busy":"2022-10-23T13:55:26.795262Z","iopub.execute_input":"2022-10-23T13:55:26.795958Z","iopub.status.idle":"2022-10-23T14:11:31.719987Z","shell.execute_reply.started":"2022-10-23T13:55:26.795920Z","shell.execute_reply":"2022-10-23T14:11:31.718952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_modelA = sum_models(allmodelA,weights=[1.0/len(allmodelA)] * len(allmodelA))\nfinal_modelB = sum_models(allmodelB,weights=[1.0/len(allmodelB)] * len(allmodelB))\nfinalmodelA = to_classifier(final_modelA)\nfinalmodelB = to_classifier(final_modelB)\nfinalmodelA.save_model('modelA.ctb')\nfinalmodelB.save_model('modelB.ctb')","metadata":{"execution":{"iopub.status.busy":"2022-10-23T14:11:31.721428Z","iopub.execute_input":"2022-10-23T14:11:31.722060Z","iopub.status.idle":"2022-10-23T14:11:31.838777Z","shell.execute_reply.started":"2022-10-23T14:11:31.722022Z","shell.execute_reply":"2022-10-23T14:11:31.837621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature importance","metadata":{}},{"cell_type":"code","source":"# feature_imp = pd.Series(finalmodelA.feature_importances_,index=x.columns).sort_values(ascending=False)\n# f,ax = plt.subplots(figsize=(30,24))\n# ax = sns.barplot(x=feature_imp, y=feature_imp.index)\n# ax.set_title('Feature importance for model A')\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-23T14:11:31.840196Z","iopub.execute_input":"2022-10-23T14:11:31.840796Z","iopub.status.idle":"2022-10-23T14:11:31.847405Z","shell.execute_reply.started":"2022-10-23T14:11:31.840756Z","shell.execute_reply":"2022-10-23T14:11:31.846234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# feature_imp = pd.Series(final_modelB.feature_importances_,index=x.columns).sort_values(ascending=False)\n# f,ax = plt.subplots(figsize=(30,24))\n# ax = sns.barplot(x=feature_imp, y=feature_imp.index)\n# ax.set_title('Feature importance for model B')\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-23T14:11:31.849110Z","iopub.execute_input":"2022-10-23T14:11:31.849506Z","iopub.status.idle":"2022-10-23T14:11:31.856156Z","shell.execute_reply.started":"2022-10-23T14:11:31.849470Z","shell.execute_reply":"2022-10-23T14:11:31.854999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Ensemble all predictions","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_parquet('/kaggle/input/tps-rocket-league-data-float16-parquet-format/test.parquet.gzip')\nsubmitData = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-10-23T14:11:31.857978Z","iopub.execute_input":"2022-10-23T14:11:31.858412Z","iopub.status.idle":"2022-10-23T14:11:34.154572Z","shell.execute_reply.started":"2022-10-23T14:11:31.858376Z","shell.execute_reply":"2022-10-23T14:11:34.153612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testx = preprocessing(test_df)\npredTest = submitData.copy()\npredTest.loc[:,'team_A_scoring_within_10sec'] = finalmodelA.predict_proba(testx)[:,1]\npredTest.loc[:,'team_B_scoring_within_10sec'] = finalmodelB.predict_proba(testx)[:,1]\npredTest.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T14:11:34.156205Z","iopub.execute_input":"2022-10-23T14:11:34.156615Z","iopub.status.idle":"2022-10-23T14:11:50.518824Z","shell.execute_reply.started":"2022-10-23T14:11:34.156575Z","shell.execute_reply":"2022-10-23T14:11:50.517765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Shuffling player","metadata":{}},{"cell_type":"code","source":"col = test_df.columns\nhead = test_df.iloc[:,0:7]\np0 = test_df.iloc[:,7:14]\np1 = test_df.iloc[:,14:21]\np2 = test_df.iloc[:,21:28]\np3 = test_df.iloc[:,28:35]\np4 = test_df.iloc[:,35:42]\np5 = test_df.iloc[:,42:49]\ntail = test_df.iloc[:,49:]\npA = pd.concat([p1,p2,p0],axis=1)\npB = pd.concat([p4,p5,p3],axis=1)\ntest2 = pd.concat([head,pA,pB,tail],axis=1)\ntest2.columns = col\ntestx = preprocessing(test2)\npredTest2 = submitData.copy()\npredTest2.loc[:,'team_A_scoring_within_10sec'] = finalmodelA.predict_proba(testx)[:,1]\npredTest2.loc[:,'team_B_scoring_within_10sec'] = finalmodelB.predict_proba(testx)[:,1]\npredTest2.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T14:34:51.101410Z","iopub.execute_input":"2022-10-23T14:34:51.102504Z","iopub.status.idle":"2022-10-23T14:35:07.842726Z","shell.execute_reply.started":"2022-10-23T14:34:51.102460Z","shell.execute_reply":"2022-10-23T14:35:07.841809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pA = pd.concat([p2,p0,p1],axis=1)\npB = pd.concat([p5,p3,p4],axis=1)\ntest3 = pd.concat([head,pA,pB,tail],axis=1)\ntest3.columns = col\ntestx = preprocessing(test3)\npredTest3 = submitData.copy()\npredTest3.loc[:,'team_A_scoring_within_10sec'] = finalmodelA.predict_proba(testx)[:,1]\npredTest3.loc[:,'team_B_scoring_within_10sec'] = finalmodelB.predict_proba(testx)[:,1]\npredTest3.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T14:36:23.474349Z","iopub.execute_input":"2022-10-23T14:36:23.474744Z","iopub.status.idle":"2022-10-23T14:36:40.496547Z","shell.execute_reply.started":"2022-10-23T14:36:23.474708Z","shell.execute_reply":"2022-10-23T14:36:40.495589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pA = pd.concat([p2,p1,p0],axis=1)\npB = pd.concat([p5,p4,p3],axis=1)\ntest4 = pd.concat([head,pA,pB,tail],axis=1)\ntest4.columns = col\ntestx = preprocessing(test4)\npredTest4 = submitData.copy()\npredTest4.loc[:,'team_A_scoring_within_10sec'] = finalmodelA.predict_proba(testx)[:,1]\npredTest4.loc[:,'team_B_scoring_within_10sec'] = finalmodelB.predict_proba(testx)[:,1]\npredTest4.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T14:38:00.263454Z","iopub.execute_input":"2022-10-23T14:38:00.264159Z","iopub.status.idle":"2022-10-23T14:38:17.592381Z","shell.execute_reply.started":"2022-10-23T14:38:00.264119Z","shell.execute_reply":"2022-10-23T14:38:17.591385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pA = pd.concat([p0,p2,p1],axis=1)\npB = pd.concat([p3,p5,p4],axis=1)\ntest5 = pd.concat([head,pA,pB,tail],axis=1)\ntest5.columns = col\ntestx = preprocessing(test5)\npredTest5 = submitData.copy()\npredTest5.loc[:,'team_A_scoring_within_10sec'] = finalmodelA.predict_proba(testx)[:,1]\npredTest5.loc[:,'team_B_scoring_within_10sec'] = finalmodelB.predict_proba(testx)[:,1]\npredTest5.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T14:39:24.980399Z","iopub.execute_input":"2022-10-23T14:39:24.981096Z","iopub.status.idle":"2022-10-23T14:39:41.638017Z","shell.execute_reply.started":"2022-10-23T14:39:24.981056Z","shell.execute_reply":"2022-10-23T14:39:41.637033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predTestAll = (predTest + predTest2 + predTest3 + predTest4 + predTest5)/6","metadata":{"execution":{"iopub.status.busy":"2022-10-23T14:40:23.716326Z","iopub.execute_input":"2022-10-23T14:40:23.716714Z","iopub.status.idle":"2022-10-23T14:40:23.745557Z","shell.execute_reply.started":"2022-10-23T14:40:23.716677Z","shell.execute_reply":"2022-10-23T14:40:23.744619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predTestAll.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T14:40:42.188087Z","iopub.execute_input":"2022-10-23T14:40:42.188546Z","iopub.status.idle":"2022-10-23T14:40:42.212031Z","shell.execute_reply.started":"2022-10-23T14:40:42.188500Z","shell.execute_reply":"2022-10-23T14:40:42.210931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Final submission","metadata":{}},{"cell_type":"code","source":"submitData = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/sample_submission.csv')\noutput = pd.DataFrame({'id': submitData.id, \n                       'team_A_scoring_within_10sec': predTestAll['team_A_scoring_within_10sec'],\n                       'team_B_scoring_within_10sec': predTestAll['team_B_scoring_within_10sec']})\noutput.to_csv('submission.csv', index=False)\nprint(\"Your submission was successfully saved!\")","metadata":{"execution":{"iopub.status.busy":"2022-10-23T14:41:32.579944Z","iopub.execute_input":"2022-10-23T14:41:32.580352Z","iopub.status.idle":"2022-10-23T14:41:34.873955Z","shell.execute_reply.started":"2022-10-23T14:41:32.580316Z","shell.execute_reply":"2022-10-23T14:41:34.872830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T14:41:36.212932Z","iopub.execute_input":"2022-10-23T14:41:36.213287Z","iopub.status.idle":"2022-10-23T14:41:36.223260Z","shell.execute_reply.started":"2022-10-23T14:41:36.213255Z","shell.execute_reply":"2022-10-23T14:41:36.222163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}