{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#import libraies\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nfrom xgboost import XGBClassifier\nfrom xgboost import plot_importance\nfrom sklearn.model_selection import KFold, StratifiedKFold, GroupKFold\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.metrics import log_loss\nimport os\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-10-23T11:10:55.592472Z","iopub.execute_input":"2022-10-23T11:10:55.592890Z","iopub.status.idle":"2022-10-23T11:10:56.310175Z","shell.execute_reply.started":"2022-10-23T11:10:55.592806Z","shell.execute_reply":"2022-10-23T11:10:56.309130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir='../input/tpsoct22-feather-files'","metadata":{"execution":{"iopub.status.busy":"2022-10-23T11:10:57.747454Z","iopub.execute_input":"2022-10-23T11:10:57.747830Z","iopub.status.idle":"2022-10-23T11:10:57.752701Z","shell.execute_reply.started":"2022-10-23T11:10:57.747797Z","shell.execute_reply":"2022-10-23T11:10:57.751473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import train and test_data\ntrain_0=pd.read_feather(data_dir + \"/train_0.feather\")\ntest = pd.read_feather(data_dir + \"/test.feather\")\ntrain_0.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-23T11:10:59.999549Z","iopub.execute_input":"2022-10-23T11:10:59.999943Z","iopub.status.idle":"2022-10-23T11:11:08.345529Z","shell.execute_reply.started":"2022-10-23T11:10:59.999908Z","shell.execute_reply":"2022-10-23T11:11:08.344149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-23T10:49:07.036937Z","iopub.execute_input":"2022-10-23T10:49:07.037305Z","iopub.status.idle":"2022-10-23T10:49:07.066091Z","shell.execute_reply.started":"2022-10-23T10:49:07.037274Z","shell.execute_reply":"2022-10-23T10:49:07.064863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_0.info()","metadata":{"execution":{"iopub.status.busy":"2022-10-20T17:15:04.139503Z","iopub.execute_input":"2022-10-20T17:15:04.140490Z","iopub.status.idle":"2022-10-20T17:15:04.173799Z","shell.execute_reply.started":"2022-10-20T17:15:04.140456Z","shell.execute_reply":"2022-10-20T17:15:04.172400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*Those data are float64  and we can save memory to converting this into float32 without losing any information*","metadata":{}},{"cell_type":"code","source":"train_0['p1_pos_y']=train_0['p1_pos_y'].astype('float32')\ntrain_0['p1_pos_y'].dtypes","metadata":{"execution":{"iopub.status.busy":"2022-10-23T10:49:07.067726Z","iopub.execute_input":"2022-10-23T10:49:07.068991Z","iopub.status.idle":"2022-10-23T10:49:07.553927Z","shell.execute_reply.started":"2022-10-23T10:49:07.068953Z","shell.execute_reply":"2022-10-23T10:49:07.553021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Number of Coulumns in dataset\ntrain_0.columns","metadata":{"execution":{"iopub.status.busy":"2022-10-23T10:49:07.555488Z","iopub.execute_input":"2022-10-23T10:49:07.556319Z","iopub.status.idle":"2022-10-23T10:49:07.563049Z","shell.execute_reply.started":"2022-10-23T10:49:07.556286Z","shell.execute_reply":"2022-10-23T10:49:07.561965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets, Understand the data First,\n- In data section, `game_num` defines the event and `event_id` shows you the consucutive frame means which event happend after current event.\n- `event_time` work on 10 second gap which identified goal happen in that time or not.\n- `ball_pos_[xyz]` Ball position identifier in 3d vector ball could be x,y and z positions.\n- `ball_vel_[xyz]` Same as ball position, here tell about the speed of ball.\n- `p{i}_pos_[xyz]` Player position in 3d vector, we know there have six player so position could be six types.\n- `p{i}_vel_[xyz]` What is the speed of each player .\n- `p{i}_boost` Boost could be 100 and that is the hight limit.\n- `boost{i}_timer` Timer to go boost from which time to and slow speed time.\n- `player_scoring_next` which player score a goal among six if no goal happen in current event then -1\n- `team_scoring_next` Which team scor next A or B.\n- `team_[A|B]_scoring_within_10sec` Goal happen then 1 otherwise 0 in 1- sec.","metadata":{}},{"cell_type":"markdown","source":"*Now Check the Target Column*","metadata":{"execution":{"iopub.status.busy":"2022-10-02T11:33:18.053476Z","iopub.execute_input":"2022-10-02T11:33:18.053776Z","iopub.status.idle":"2022-10-02T11:33:18.062751Z","shell.execute_reply.started":"2022-10-02T11:33:18.053751Z","shell.execute_reply":"2022-10-02T11:33:18.061619Z"}}},{"cell_type":"code","source":"plt.figure(figsize=(12,6))\nsns.countplot(x='team_scoring_next', data=train_0)\nplt.title(\"Which team scroing Next \")","metadata":{"execution":{"iopub.status.busy":"2022-10-20T17:15:04.693805Z","iopub.execute_input":"2022-10-20T17:15:04.694314Z","iopub.status.idle":"2022-10-20T17:15:05.995716Z","shell.execute_reply.started":"2022-10-20T17:15:04.694266Z","shell.execute_reply":"2022-10-20T17:15:05.994938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,6))\nsns.countplot(x='player_scoring_next', data=train_0)\nplt.title('which player score next')","metadata":{"execution":{"iopub.status.busy":"2022-10-20T17:15:05.997039Z","iopub.execute_input":"2022-10-20T17:15:05.998001Z","iopub.status.idle":"2022-10-20T17:15:06.409588Z","shell.execute_reply.started":"2022-10-20T17:15:05.997967Z","shell.execute_reply":"2022-10-20T17:15:06.408505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Insight:\n- Class imbalance dataset\n- Which team score next in B but at the goal with 10 second in B is 0.\n- Player scoring next is -1 because No goal happen in maximum 10 seconds event.\n- From A team 0 number Player Score highest GOAL\n- From B team 4 Number player score Most GOAL","metadata":{}},{"cell_type":"markdown","source":"We can see the co-relation map to identified which features are related to the target variable and we should drop which one.","metadata":{}},{"cell_type":"code","source":"matrix_correlations=train_0.corr()\nplt.figure(figsize=(30, 20))\nsns.heatmap(matrix_correlations, cmap=\"YlGnBu\", annot=True, fmt=\".1f\")","metadata":{"execution":{"iopub.status.busy":"2022-10-20T17:15:06.413839Z","iopub.execute_input":"2022-10-20T17:15:06.414187Z","iopub.status.idle":"2022-10-20T17:15:44.323946Z","shell.execute_reply.started":"2022-10-20T17:15:06.414155Z","shell.execute_reply":"2022-10-20T17:15:44.322767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,6))\nsns.countplot(x='team_A_scoring_within_10sec', data=train_0)\nplt.title(\"Team A Score within 10 Second\")","metadata":{"execution":{"iopub.status.busy":"2022-10-20T17:15:44.325727Z","iopub.execute_input":"2022-10-20T17:15:44.326066Z","iopub.status.idle":"2022-10-20T17:15:44.721927Z","shell.execute_reply.started":"2022-10-20T17:15:44.326038Z","shell.execute_reply":"2022-10-20T17:15:44.720810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Insight:\n- In classification section 0 is more than 1.\n- With hugely Imbalance dataset","metadata":{}},{"cell_type":"markdown","source":"# Class Imbalance","metadata":{}},{"cell_type":"markdown","source":"For class Imbalance we can create duplicate one or Used <a href=\"https://scikit-learn.org/stable/modules/generated/sklearn.model_selection.StratifiedKFold.html\">StratifiedKFold</a> cross validation. More information about class Imbalance <a href=\"https://machinelearningmastery.com/what-is-imbalanced-classification/\"> visit that.</a>","metadata":{}},{"cell_type":"markdown","source":"There are several technique to deal with Class imbalance:\n1. **Random Under-Sampling:** Undersampling can be defined as removing some observations of the majority class. This is done until the majority and minority class is balanced out.\n2. **Random Over-Sampling:** Oversampling can be defined as adding more copies to the minority class. Oversampling can be a good choice when you don’t have a ton of data to work with.\n3. **Under-sampling: Tomek links:** Tomek links are pairs of very close instances but of opposite classes. Removing the instances of the majority class of each pair increases the space between the two classes, facilitating the classification process.\n4. **Synthetic Minority Oversampling Technique (SMOTE):** SMOTE (Synthetic Minority Oversampling Technique) works by randomly picking a point from the minority class and computing the k-nearest neighbors for this point. The synthetic points are added between the chosen point and its neighbors.\n5. **Change the performance metric:** Accuracy is not the best metric to use when evaluating imbalanced datasets as it can be misleading.\n\nMore about class Imbalance Visit: <a href=\"https://www.analyticsvidhya.com/blog/2020/07/10-techniques-to-deal-with-class-imbalance-in-machine-learning/\">See this turorial</a>","metadata":{"execution":{"iopub.status.busy":"2022-10-03T11:12:51.186211Z","iopub.execute_input":"2022-10-03T11:12:51.186569Z","iopub.status.idle":"2022-10-03T11:12:51.194362Z","shell.execute_reply.started":"2022-10-03T11:12:51.186538Z","shell.execute_reply":"2022-10-03T11:12:51.193000Z"}}},{"cell_type":"markdown","source":"##  Modeling ","metadata":{}},{"cell_type":"markdown","source":"* Lets concat those data for training, In thia section we only used half 4 subset of full dataset","metadata":{"execution":{"iopub.status.busy":"2022-10-02T18:19:45.857867Z","iopub.execute_input":"2022-10-02T18:19:45.858198Z","iopub.status.idle":"2022-10-02T18:19:45.865736Z","shell.execute_reply.started":"2022-10-02T18:19:45.858169Z","shell.execute_reply":"2022-10-02T18:19:45.864375Z"}}},{"cell_type":"code","source":"# try to uf you have enough memory\n# train_df = pd.concat([train_0, train_3]) #memory capacity we take three of them\n# train_df.head()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-10-20T17:15:44.723388Z","iopub.execute_input":"2022-10-20T17:15:44.723673Z","iopub.status.idle":"2022-10-20T17:15:44.727450Z","shell.execute_reply.started":"2022-10-20T17:15:44.723646Z","shell.execute_reply":"2022-10-20T17:15:44.726258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_A =train_0['team_A_scoring_within_10sec']\ny_B=train_0['team_B_scoring_within_10sec']","metadata":{"execution":{"iopub.status.busy":"2022-10-23T11:11:24.502688Z","iopub.execute_input":"2022-10-23T11:11:24.503422Z","iopub.status.idle":"2022-10-23T11:11:24.511166Z","shell.execute_reply.started":"2022-10-23T11:11:24.503378Z","shell.execute_reply":"2022-10-23T11:11:24.510149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"featurs = train_0.columns[5:-18] \ndf_train=train_0[featurs].copy()\ndf_test=test[featurs].copy()\nfeaturs","metadata":{"execution":{"iopub.status.busy":"2022-10-23T11:11:25.885359Z","iopub.execute_input":"2022-10-23T11:11:25.885734Z","iopub.status.idle":"2022-10-23T11:11:26.511921Z","shell.execute_reply.started":"2022-10-23T11:11:25.885699Z","shell.execute_reply":"2022-10-23T11:11:26.510313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# del train_0","metadata":{"execution":{"iopub.status.busy":"2022-10-23T10:49:53.573207Z","iopub.execute_input":"2022-10-23T10:49:53.574362Z","iopub.status.idle":"2022-10-23T10:49:53.580977Z","shell.execute_reply.started":"2022-10-23T10:49:53.574322Z","shell.execute_reply":"2022-10-23T10:49:53.580022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training with StratifiedKFold for Secor_A and Score_B","metadata":{}},{"cell_type":"code","source":"params = {\n            'objective':'binary:logistic',\n            'tree_method': 'gpu_hist',\n            'booster' : 'gbtree',\n            'subsample' : 0.8326,\n            'gamma' : 0.48,\n            'max_depth': 7,\n            'alpha': 10,\n            'learning_rate': .027,\n            'n_estimators':51000, #3000\n            'predictor': 'gpu_predictor'\n        }       \n           ","metadata":{"execution":{"iopub.status.busy":"2022-10-23T11:11:29.942285Z","iopub.execute_input":"2022-10-23T11:11:29.942694Z","iopub.status.idle":"2022-10-23T11:11:29.948705Z","shell.execute_reply.started":"2022-10-23T11:11:29.942660Z","shell.execute_reply":"2022-10-23T11:11:29.947782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"group = train_0['game_num'].copy().tolist()","metadata":{"execution":{"iopub.status.busy":"2022-10-23T11:11:52.111800Z","iopub.execute_input":"2022-10-23T11:11:52.112185Z","iopub.status.idle":"2022-10-23T11:11:52.181690Z","shell.execute_reply.started":"2022-10-23T11:11:52.112149Z","shell.execute_reply":"2022-10-23T11:11:52.180555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-23T11:11:55.205481Z","iopub.execute_input":"2022-10-23T11:11:55.205842Z","iopub.status.idle":"2022-10-23T11:11:55.212954Z","shell.execute_reply.started":"2022-10-23T11:11:55.205811Z","shell.execute_reply":"2022-10-23T11:11:55.211554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FOLDS = 2\nseed=42\npredA,y_pred,xgb_val,scores=[],[],[],[]\ndef model_fit(df_train, y_A, X_test):\n#     predA,y_pred,xgb_val,scores=[],[],[],[]\n    k_fold = StratifiedGroupKFold(n_splits=FOLDS, shuffle=True, random_state=seed)\n    for train_idx, val_idx in k_fold.split(df_train, y_A,group):\n        X_fold_train, Y_fold_train = df_train.iloc[train_idx,:], y_A[train_idx]\n        X_fold_val, Y_fold_val = df_train.iloc[val_idx,:], y_A[val_idx]\n#         print(f\"--------FOLD-{fold+1}--------\")\n    \n        #start model\n        xgb = XGBClassifier(**params)\n        eval_set = [(X_fold_train, Y_fold_train), (X_fold_val, Y_fold_val)]\n        xgb.fit(X_fold_train, Y_fold_train,\n              early_stopping_rounds=150,\n        #           eval_set=[(test_X,y_test)],\n            eval_set=eval_set,\n             eval_metric=[\"error\", \"logloss\"],\n              verbose=500)\n        #check score\n        eval_set = xgb.predict_proba(X_fold_val)[:,1]\n        y_pred.append(Y_fold_val)\n        xgb_val.append(eval_set)\n        predA = xgb.predict_proba(X_test)[:,1]\n        score=log_loss(Y_fold_val, eval_set)\n        print(f\"Validation Logloss  = {score:.4f}\")\n        scores.append(score)\n        del X_fold_train, Y_fold_train, X_fold_val, Y_fold_val\n    return predA","metadata":{"execution":{"iopub.status.busy":"2022-10-23T11:18:42.230910Z","iopub.execute_input":"2022-10-23T11:18:42.231299Z","iopub.status.idle":"2022-10-23T11:18:42.241612Z","shell.execute_reply.started":"2022-10-23T11:18:42.231267Z","shell.execute_reply":"2022-10-23T11:18:42.240458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model fit for A\nmodel_fit(df_train, y_A, df_test)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-10-23T11:18:42.723815Z","iopub.execute_input":"2022-10-23T11:18:42.724478Z","iopub.status.idle":"2022-10-23T11:19:17.515676Z","shell.execute_reply.started":"2022-10-23T11:18:42.724426Z","shell.execute_reply":"2022-10-23T11:19:17.514649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'score:{np.mean(scores):0.4f}')\nscores","metadata":{"execution":{"iopub.status.busy":"2022-10-23T11:19:17.520372Z","iopub.execute_input":"2022-10-23T11:19:17.521057Z","iopub.status.idle":"2022-10-23T11:19:17.536904Z","shell.execute_reply.started":"2022-10-23T11:19:17.521018Z","shell.execute_reply":"2022-10-23T11:19:17.535921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model fit for B\nmodel_fit(df_train, y_B, df_test)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-10-23T11:19:17.538638Z","iopub.execute_input":"2022-10-23T11:19:17.539777Z","iopub.status.idle":"2022-10-23T11:19:50.202686Z","shell.execute_reply.started":"2022-10-23T11:19:17.539606Z","shell.execute_reply":"2022-10-23T11:19:50.200859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'score:{np.mean(scores):0.4f}')\nscores","metadata":{"execution":{"iopub.status.busy":"2022-10-23T11:19:50.209172Z","iopub.execute_input":"2022-10-23T11:19:50.211561Z","iopub.status.idle":"2022-10-23T11:19:50.227774Z","shell.execute_reply.started":"2022-10-23T11:19:50.211530Z","shell.execute_reply":"2022-10-23T11:19:50.226504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Now lets go to submission","metadata":{}},{"cell_type":"markdown","source":" Again set the parameter based on the ovefiting knowledge from StratifiedKFold training, Look after which  `n_estimators` training goes into overfit.","metadata":{}},{"cell_type":"code","source":"params = {\n            'objective':'binary:logistic',\n            'tree_method': 'gpu_hist',\n            'booster' : 'gbtree',\n            'seed': 42,\n            'subsample' : 0.8326,\n            'gamma' : 0.48,\n            'max_depth': 12,\n            'alpha': 10,\n            'learning_rate': .027,\n            'n_estimators':300,\n            'predictor': 'gpu_predictor',\n        }       \n           ","metadata":{"execution":{"iopub.status.busy":"2022-10-23T11:20:14.352662Z","iopub.execute_input":"2022-10-23T11:20:14.353032Z","iopub.status.idle":"2022-10-23T11:20:14.358818Z","shell.execute_reply.started":"2022-10-23T11:20:14.352999Z","shell.execute_reply":"2022-10-23T11:20:14.357560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#predict for score A\nxgb = XGBClassifier(**params)\neval_set = [(df_train, y_A)]\nxgb.fit(df_train, y_A,\n          early_stopping_rounds=150,\n#           eval_set=[(test_X,y_test)],\n        eval_set=eval_set,\n         eval_metric=[\"error\", \"logloss\"],\n          verbose=True)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-10-23T11:20:15.355894Z","iopub.execute_input":"2022-10-23T11:20:15.356280Z","iopub.status.idle":"2022-10-23T11:21:03.732209Z","shell.execute_reply.started":"2022-10-23T11:20:15.356247Z","shell.execute_reply":"2022-10-23T11:21:03.731070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predA = xgb.predict_proba(df_test)[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-10-23T11:21:23.145355Z","iopub.execute_input":"2022-10-23T11:21:23.145736Z","iopub.status.idle":"2022-10-23T11:21:24.236979Z","shell.execute_reply.started":"2022-10-23T11:21:23.145702Z","shell.execute_reply":"2022-10-23T11:21:24.235928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predA","metadata":{"execution":{"iopub.status.busy":"2022-10-23T11:21:25.411848Z","iopub.execute_input":"2022-10-23T11:21:25.412232Z","iopub.status.idle":"2022-10-23T11:21:25.419786Z","shell.execute_reply.started":"2022-10-23T11:21:25.412197Z","shell.execute_reply":"2022-10-23T11:21:25.418799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,1,figsize=(20,12))\nplot_importance(xgb,ax=ax, xlabel=None)\nplt.title('XGB Feature importance for Score_A')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-23T11:21:28.326952Z","iopub.execute_input":"2022-10-23T11:21:28.328039Z","iopub.status.idle":"2022-10-23T11:21:28.930463Z","shell.execute_reply.started":"2022-10-23T11:21:28.328000Z","shell.execute_reply":"2022-10-23T11:21:28.929466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Now for Score_B for Predicion","metadata":{}},{"cell_type":"code","source":"xgb = XGBClassifier(**params)\neval_set = [(df_train, y_B)]\nxgb.fit(df_train, y_B,\n          early_stopping_rounds=150,\n#           eval_set=[(test_X,y_test)],\n        eval_set=eval_set,\n         eval_metric=[\"error\", \"logloss\"],\n          verbose=True)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-10-23T11:21:34.362981Z","iopub.execute_input":"2022-10-23T11:21:34.363674Z","iopub.status.idle":"2022-10-23T11:22:20.548318Z","shell.execute_reply.started":"2022-10-23T11:21:34.363636Z","shell.execute_reply":"2022-10-23T11:22:20.547340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predB = xgb.predict_proba(df_test)[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-10-23T11:22:20.550168Z","iopub.execute_input":"2022-10-23T11:22:20.550451Z","iopub.status.idle":"2022-10-23T11:22:21.626820Z","shell.execute_reply.started":"2022-10-23T11:22:20.550425Z","shell.execute_reply":"2022-10-23T11:22:21.625651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predB ","metadata":{"execution":{"iopub.status.busy":"2022-10-23T11:22:33.319802Z","iopub.execute_input":"2022-10-23T11:22:33.320347Z","iopub.status.idle":"2022-10-23T11:22:33.327635Z","shell.execute_reply.started":"2022-10-23T11:22:33.320305Z","shell.execute_reply":"2022-10-23T11:22:33.326610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,1,figsize=(20,12))\nplot_importance(xgb,ax=ax, xlabel=None)\nplt.title('XGB Feature importance for Score_B')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-23T11:22:34.695644Z","iopub.execute_input":"2022-10-23T11:22:34.697392Z","iopub.status.idle":"2022-10-23T11:22:35.538090Z","shell.execute_reply.started":"2022-10-23T11:22:34.697344Z","shell.execute_reply":"2022-10-23T11:22:35.536987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Merge Score_A and Score_B for Submission","metadata":{}},{"cell_type":"code","source":"sub = pd.read_csv(\"../input/tabular-playground-series-oct-2022/sample_submission.csv\")\n\nsub[\"team_A_scoring_within_10sec\"] = predA\nsub[\"team_B_scoring_within_10sec\"] = predB\n\nsub.to_csv(\"submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T11:22:39.430777Z","iopub.execute_input":"2022-10-23T11:22:39.431165Z","iopub.status.idle":"2022-10-23T11:22:41.278083Z","shell.execute_reply.started":"2022-10-23T11:22:39.431128Z","shell.execute_reply":"2022-10-23T11:22:41.277111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T11:22:44.202890Z","iopub.execute_input":"2022-10-23T11:22:44.203655Z","iopub.status.idle":"2022-10-23T11:22:44.215631Z","shell.execute_reply.started":"2022-10-23T11:22:44.203608Z","shell.execute_reply":"2022-10-23T11:22:44.214587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Update Could happend tuining:\n- Feature Engineering \n- Feature Selection\n- Class imbalance handale and cheeck other method\n- Parameter Tuning and Apply other Model Like LGBM and also NN.\n- Other dataset Train and merge the result","metadata":{}}]}