{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook is structured as follows\n1. Constants. Here the data and model parameters defined, some geometric properties of the football field and the features that will be used.\n2. Feature Engineering. Some geometric quantities that can be computed from the initial features like velocities and angles.\n3. Loading and transforming of the train data.\n4. Import train data.\n5. Fitting the data.\n6. Submitting the data.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import os\nimport gc\nimport joblib\nimport pickle\n\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import (StandardScaler,\n                                   LabelEncoder\n                                  )\n                         \nfrom tensorflow.config import list_physical_devices\n\n\nimport lightgbm as lgb\nfrom lightgbm import plot_importance \n","metadata":{"execution":{"iopub.status.busy":"2022-10-30T18:34:57.844546Z","iopub.execute_input":"2022-10-30T18:34:57.845403Z","iopub.status.idle":"2022-10-30T18:35:04.343714Z","shell.execute_reply.started":"2022-10-30T18:34:57.845317Z","shell.execute_reply":"2022-10-30T18:35:04.342815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. CONSTANTS","metadata":{}},{"cell_type":"markdown","source":"## Data and Model constants","metadata":{}},{"cell_type":"code","source":"\nGPU = list_physical_devices('GPU') != []\n\n\n####### DATA LOADING ################################################################\nDEBUG = False\nSAMPLE =1#0.25 #percentage of train file used 0.3\nSIZE = 10 #number of used train files\nSEED = 1234\nPATH_TO_DATA = '/kaggle/input/tabular-playground-series-oct-2022/'\n\n\n\n####################### MODEL ##############################\nparams = {\n    'objective':'binary',\n    'learning_rate':0.15, #0.025\n    'lambda_l2':2,\n    'path_smooth':1,\n    'boosting_type':'dart', #gbdt\n    'num_leaves':2**6,\n    'max_depth':6,\n    'extra_trees':True,\n    'bagging_freq':10,\n    'bagging_fraction':0.7,\n    'num_iterations':1000,#300,#1200,#2000, #250\n    'force_col_wise':True,\n}","metadata":{"execution":{"iopub.status.busy":"2022-10-30T18:39:08.986045Z","iopub.execute_input":"2022-10-30T18:39:08.986488Z","iopub.status.idle":"2022-10-30T18:39:08.998765Z","shell.execute_reply.started":"2022-10-30T18:39:08.986450Z","shell.execute_reply":"2022-10-30T18:39:08.997695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## The field, the teams and the ball","metadata":{}},{"cell_type":"code","source":"GOAL_A=np.array([0,-104.31,1.2])\nGOAL_B=np.array([0,-104.31,1.2])\nGOAL_WITDH= 32.6\n\nTEAM_A=[0,1,2]\nTEAM_B=[3,4,5]\n\nBALL_POS = ['ball_pos_x','ball_pos_y', 'ball_pos_z']\nBALL_VEL = ['ball_vel_x','ball_vel_y', 'ball_vel_z']\n\nPLAYER_VEL = {\n    f\"{j}\": [f'p{j}_vel_x', f'p{j}_vel_y', f'p{j}_vel_z']\n    for j in [f'{i}' for i in range(6)]\n}\nPLAYER_POS = {\n    f\"{j}\": [f'p{j}_pos_x', f'p{j}_pos_y', f'p{j}_pos_z']\n    for j in [f'{i}' for i in range(6)]\n}","metadata":{"execution":{"iopub.status.busy":"2022-10-30T18:39:05.681082Z","iopub.execute_input":"2022-10-30T18:39:05.681505Z","iopub.status.idle":"2022-10-30T18:39:05.689279Z","shell.execute_reply.started":"2022-10-30T18:39:05.681470Z","shell.execute_reply":"2022-10-30T18:39:05.688107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train and test data columns\n\nThe train and the test data have the following columns. Some of them are leakage.","metadata":{}},{"cell_type":"code","source":"train= ['game_num', 'event_id', 'event_time', 'ball_pos_x', 'ball_pos_y',\n       'ball_pos_z', 'ball_vel_x', 'ball_vel_y', 'ball_vel_z', 'p0_pos_x',\n       'p0_pos_y', 'p0_pos_z', 'p0_vel_x', 'p0_vel_y', 'p0_vel_z', 'p0_boost',\n       'p1_pos_x', 'p1_pos_y', 'p1_pos_z', 'p1_vel_x', 'p1_vel_y', 'p1_vel_z',\n       'p1_boost', 'p2_pos_x', 'p2_pos_y', 'p2_pos_z', 'p2_vel_x', 'p2_vel_y',\n       'p2_vel_z', 'p2_boost', 'p3_pos_x', 'p3_pos_y', 'p3_pos_z', 'p3_vel_x',\n       'p3_vel_y', 'p3_vel_z', 'p3_boost', 'p4_pos_x', 'p4_pos_y', 'p4_pos_z',\n       'p4_vel_x', 'p4_vel_y', 'p4_vel_z', 'p4_boost', 'p5_pos_x', 'p5_pos_y',\n       'p5_pos_z', 'p5_vel_x', 'p5_vel_y', 'p5_vel_z', 'p5_boost',\n       'boost0_timer', 'boost1_timer', 'boost2_timer', 'boost3_timer',\n       'boost4_timer', 'boost5_timer', 'player_scoring_next',\n       'team_scoring_next','team_A_scoring_within_10sec', 'team_B_scoring_within_10sec']\n\ntest= ['id', 'ball_pos_x', 'ball_pos_y',\n       'ball_pos_z', 'ball_vel_x', 'ball_vel_y', 'ball_vel_z', 'p0_pos_x',\n       'p0_pos_y', 'p0_pos_z', 'p0_vel_x', 'p0_vel_y', 'p0_vel_z', 'p0_boost',\n       'p1_pos_x', 'p1_pos_y', 'p1_pos_z', 'p1_vel_x', 'p1_vel_y', 'p1_vel_z',\n       'p1_boost', 'p2_pos_x', 'p2_pos_y', 'p2_pos_z', 'p2_vel_x', 'p2_vel_y',\n       'p2_vel_z', 'p2_boost', 'p3_pos_x', 'p3_pos_y', 'p3_pos_z', 'p3_vel_x',\n       'p3_vel_y', 'p3_vel_z', 'p3_boost', 'p4_pos_x', 'p4_pos_y', 'p4_pos_z',\n       'p4_vel_x', 'p4_vel_y', 'p4_vel_z', 'p4_boost', 'p5_pos_x', 'p5_pos_y',\n       'p5_pos_z', 'p5_vel_x', 'p5_vel_y', 'p5_vel_z', 'p5_boost',\n       'boost0_timer', 'boost1_timer', 'boost2_timer', 'boost3_timer',\n       'boost4_timer', 'boost5_timer']\n\nINIT_FEATS=[ 'ball_pos_x', 'ball_pos_y',\n       'ball_pos_z', 'ball_vel_x', 'ball_vel_y', 'ball_vel_z', 'p0_pos_x',\n       'p0_pos_y', 'p0_pos_z', 'p0_vel_x', 'p0_vel_y', 'p0_vel_z', 'p0_boost',\n       'p1_pos_x', 'p1_pos_y', 'p1_pos_z', 'p1_vel_x', 'p1_vel_y', 'p1_vel_z',\n       'p1_boost', 'p2_pos_x', 'p2_pos_y', 'p2_pos_z', 'p2_vel_x', 'p2_vel_y',\n       'p2_vel_z', 'p2_boost', 'p3_pos_x', 'p3_pos_y', 'p3_pos_z', 'p3_vel_x',\n       'p3_vel_y', 'p3_vel_z', 'p3_boost', 'p4_pos_x', 'p4_pos_y', 'p4_pos_z',\n       'p4_vel_x', 'p4_vel_y', 'p4_vel_z', 'p4_boost', 'p5_pos_x', 'p5_pos_y',\n       'p5_pos_z', 'p5_vel_x', 'p5_vel_y', 'p5_vel_z', 'p5_boost',\n       'boost0_timer', 'boost1_timer', 'boost2_timer', 'boost3_timer',\n       'boost4_timer', 'boost5_timer']","metadata":{"execution":{"iopub.status.busy":"2022-10-30T18:39:03.198748Z","iopub.execute_input":"2022-10-30T18:39:03.199140Z","iopub.status.idle":"2022-10-30T18:39:03.209776Z","shell.execute_reply.started":"2022-10-30T18:39:03.199105Z","shell.execute_reply":"2022-10-30T18:39:03.208811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Features and Targets\nThe following features will be used in the model. Features that are commented out, where not impactful. Velocity of player in $x$ and $y$ direction, and boosting features are removed.","metadata":{}},{"cell_type":"code","source":"###########DROPPED FEATURES ################################################\nDROP_COLS_TRAIN=['game_num', 'event_id', 'event_time','player_scoring_next',\n       'team_scoring_next']\n\nDROP_COLS_TEST=['id']\n###########INIT FEATURES ################################################\n\nTARGETS=['team_A_scoring_within_10sec', 'team_B_scoring_within_10sec']\n\nBALL=['ball_pos_x', 'ball_pos_y','ball_pos_z',\n       'ball_vel_x', 'ball_vel_y', 'ball_vel_z', \n          ]\nPOS_X=[]\nfor i in range(6):\n    POS_X.append(f'p{i}_pos_x')\nPOS_Y=[]\nfor i in range(6):\n    POS_Y.append(f'p{i}_pos_y')\nPOS_Z=[]\nfor i in range(6):\n    POS_Z.append(f'p{i}_pos_z')\n\nVEL_X=[]\nfor i in range(6):\n    VEL_X.append(f'p{i}_vel_x')\nVEL_Y=[]\nfor i in range(6):\n    VEL_Y.append(f'p{i}_vel_y')\nVEL_Z=[]\nfor i in range(6):\n    VEL_Z.append(f'p{i}_vel_z')\n    \nPLAYER_BOOST= ['p0_boost','p1_boost','p2_boost','p3_boost', 'p4_boost','p5_boost']\nBOOST=['boost0_timer', 'boost1_timer', 'boost2_timer', 'boost3_timer',\n       'boost4_timer', 'boost5_timer']\n\n####### ENGINEERED FEATURES##################################################\nBALL_FEATS=['ball_vel']\nfor goal in ['A','B']:\n    BALL_FEATS.append(f'ball_vis_goal_{goal}')\n    BALL_FEATS.append(f'ball_dis_goal_{goal}')\n    BALL_FEATS.append(f'ball_move_goal_{goal}')\n#    BALL_FEATS.append(f'ball_speed_in_dir_goal_{goal}')\n\n\nPLAYER_FEATS=[]\nfor i in range(6):\n    PLAYER_FEATS.append(f'p{i}_vel')\n#    PLAYER_FEATS.append(f'p{i}_pos')\n    PLAYER_FEATS.append(f'p{i}_dis_ball')\n    for goal in ['A','B']:\n#        PLAYER_FEATS.append(f'p{i}_vis_goal_{goal}')\n        PLAYER_FEATS.append(f'p{i}_dis_goal_{goal}')\n#        PLAYER_FEATS.append(f'p{i}_move_goal_{goal}')\n#        PLAYER_FEATS.append(f'p{i}_speed_in_dir_goal_{goal}')\n#     for j in range(i):\n#         PLAYER_FEATS.append(f'p{i}_p{j}_dist')                    ##### this feature wasn't very helpful\n\nNUM_FEATS= (BALL\n            +POS_X+POS_Y+POS_Z\n            +VEL_Y   +VEL_X+VEL_Z\n            +PLAYER_BOOST#############vel_x vel_z removed! PLAYER_BOOST; BOOST REMOVED\n            +BALL_FEATS+PLAYER_FEATS\n            +['Ball_possession']\n           )\n\n\n#+ ['Team_A_missing', 'Team_B_missing']\n\nDROP_COLS_AFTER_FE=BOOST #+PLAYER_BOOST\n#for i in range(6):\n    #DROP_COLS_AFTER_FE.append(f'p{i}_vel_x')\n    #DROP_COLS_AFTER_FE.append(f'p{i}_vel_z')\n    #DROP_COLS_AFTER_FE.append(f'p{i}_pos_z')\n    \nTEAM_A_COLS=NUM_FEATS.copy()\nTEAM_A_COLS.remove('ball_vis_goal_A')\nTEAM_A_COLS.remove('ball_dis_goal_A')\nTEAM_A_COLS.remove(f'ball_move_goal_{\"A\"}')\nfor i in range(6):\n    TEAM_A_COLS.remove(f'p{i}_dis_goal_A')\n\n\nTEAM_B_COLS=NUM_FEATS.copy()\nTEAM_B_COLS.remove('ball_vis_goal_B')\nTEAM_B_COLS.remove('ball_dis_goal_B')\nTEAM_B_COLS.remove(f'ball_move_goal_{\"B\"}')\nfor i in range(6):\n    TEAM_B_COLS.remove(f'p{i}_dis_goal_B')\n    \n    \nprint(f'We have {len(NUM_FEATS)} features.')    \nprint(f'We have {len(TEAM_A_COLS)} Team A features.')\nprint(f'We have {len(TEAM_B_COLS)} Team B features.')","metadata":{"execution":{"iopub.status.busy":"2022-10-30T18:38:58.893820Z","iopub.execute_input":"2022-10-30T18:38:58.894216Z","iopub.status.idle":"2022-10-30T18:38:58.911121Z","shell.execute_reply.started":"2022-10-30T18:38:58.894180Z","shell.execute_reply":"2022-10-30T18:38:58.910003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Feature engineering\n\nThe geometric functions involved are defined in the subsequent cell.","metadata":{}},{"cell_type":"code","source":"def feature_eng(df):\n    df['ball_vel']=euclidean_norm(df[BALL_VEL].values)\n#    df['ball_pos']=euclidean_norm(df[BALL_POS].values)\n    for goal in ['A','B']:\n        GOAL= GOAL_A if (goal=='A') else GOAL_B\n        df[f'ball_vis_goal_{goal}']= view_on_goal(df[BALL_POS[:2]].values,\n                                                     GOAL[:2])\n        df[f'ball_dis_goal_{goal}']=euclidean_norm(df[BALL_POS].values\n                                                  -GOAL)\n        df[f'ball_move_goal_{goal}']=cosine_law(GOAL-df[BALL_POS].values,\n                                                df[BALL_VEL])\n #       df[f'ball_speed_in_dir_goal_{goal}']= np.sign(df[f'ball_vel_y'].values)*df[f'ball_move_goal_{goal}']*df[f'ball_vel']\n    \n    for i in range(6):\n        df[f'p{i}_vel']=euclidean_norm(df[PLAYER_VEL[f'{i}']].values)\n#        df[f'p{i}_pos']=euclidean_norm(df[PLAYER_POS[f'{i}']].values)\n        df[f'p{i}_dis_ball']=euclidean_norm(df[PLAYER_POS[f'{i}']].values-df[BALL_POS].values)\n        for goal in ['A','B']:\n            GOAL= GOAL_A if (goal=='A') else GOAL_B\n#            df[f'p{i}_vis_goal_{goal}']=view_on_goal(df[PLAYER_POS[f'{i}'][:2]].values,\n#                                                     GOAL[:2])\n            df[f'p{i}_dis_goal_{goal}']=euclidean_norm(df[PLAYER_POS[f'{i}']].values\n                                                  -GOAL)\n#            df[f'p{i}_move_goal_{goal}']=cosine_law(GOAL-df[PLAYER_POS[f'{i}']].values,\n#                                                df[PLAYER_VEL[f'{i}']].values)\n#            df[f'p{i}_speed_in_dir_goal_{goal}']= np.sign(df[f'p{i}_vel_y'].values)*df[f'p{i}_move_goal_{goal}']*df[f'p{i}_vel']\n#        for j in range(i):\n#            df[f'p{i}_p{j}_dist']=euclidean_norm(df[PLAYER_POS[f'{i}']].values\n#                                                  -df[PLAYER_POS[f'{j}']].values)\n            \n    df['Ball_possession']=(np.minimum(df[f'p{0}_dis_ball'].values,df[f'p{1}_dis_ball'].values,df[f'p{2}_dis_ball'].values)\n                           -np.minimum(df[f'p{3}_dis_ball'].values,df[f'p{4}_dis_ball'].values,df[f'p{5}_dis_ball'].values)\n                          )\n    df=df.drop(DROP_COLS_AFTER_FE,axis=1)\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-10-30T18:38:55.918863Z","iopub.execute_input":"2022-10-30T18:38:55.919836Z","iopub.status.idle":"2022-10-30T18:38:55.931013Z","shell.execute_reply.started":"2022-10-30T18:38:55.919796Z","shell.execute_reply":"2022-10-30T18:38:55.929859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Geometric functions and field functions\n\nEnginered features involve the movement of an object in direction of the goals (cosine law) $\\frac{<x,y>}{|x||y|}$,\nthe norm $ |x|  $ and the two-dimensionl angle between an object the goal posts, $\\text{atan2}$.","metadata":{}},{"cell_type":"code","source":"def euclidean_norm(X):\n    return np.linalg.norm(X, axis=1)\n\n#+1.e-5 to ensure not deviding by zero\ndef normalize(X):\n    return X/((euclidean_norm(X)+1.e-5)[:, None])\n\n#rounding errors\ndef cosine_law(X,Y):\n    return np.sum(normalize(X)*normalize(Y),axis=1)\n\n\ndef movement_in_dir_goal(pos,vel,goal):\n    return cosine_law(goal-pos,vel) \n\n#I am aware that the signs on this are wrong but since I take it for both goals it should still work.\ndef view_on_goal(X,goal):\n    A=goal-np.array([-16.3,0])-X\n    B=goal+np.array([16.3,0])-X\n    return np.arctan2(A[:, 0]*B[:, 1]- B[:, 0]*A[:, 1],A[:, 0]*A[:, 1]+B[:, 1]*B[:, 1])\n\n#if the angle version does not work so well\n#sign is movement direction X is vel goal is A or B\n# def view_on_goal(sign,X,GOAL):\n#     a=euclidean_norm(X-GOAL-np.array([GOAL_WITDH/2,0]))\n#     b=euclidean_norm(X-GOAL+np.array([GOAL_WITDH/2,0]))\n#     return sign*triangle_alpha(a,b,GOAL_WITDH)","metadata":{"execution":{"iopub.status.busy":"2022-10-30T18:38:53.313654Z","iopub.execute_input":"2022-10-30T18:38:53.314045Z","iopub.status.idle":"2022-10-30T18:38:53.324055Z","shell.execute_reply.started":"2022-10-30T18:38:53.314011Z","shell.execute_reply":"2022-10-30T18:38:53.323063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data augmentation\n\nWe rotate the players within the team and the teams themselves.","metadata":{}},{"cell_type":"code","source":"def rotate_players(df):\n    df['p0_boost'] , df['p1_boost'] = df['p1_boost'], df['p2_boost']\n    df['p1_boost'] , df['p2_boost'] = df['p2_boost'] , df['p1_boost']\n    for dim in ['x','y','z']:\n        df[f'p0_pos_{dim}'],df[f'p1_pos_{dim}']=df[f'p1_pos_{dim}'].copy().values,df[f'p0_pos_{dim}'].copy().values\n        df[f'p1_pos_{dim}'],df[f'p2_pos_{dim}']=df[f'p2_pos_{dim}'].copy().values,df[f'p1_pos_{dim}'].copy().values\n        \n        df[f'p0_vel_{dim}'],df[f'p1_vel_{dim}']=df[f'p1_vel_{dim}'].copy().values,df[f'p0_vel_{dim}'].copy().values\n        df[f'p1_vel_{dim}'],df[f'p2_vel_{dim}']=df[f'p2_vel_{dim}'].copy().values,df[f'p1_vel_{dim}'].copy().values\n        \n        \n        df[f'p3_pos_{dim}'],df[f'p4_pos_{dim}']=df[f'p4_pos_{dim}'].copy().values,df[f'p3_pos_{dim}'].copy().values\n        df[f'p4_pos_{dim}'],df[f'p5_pos_{dim}']=df[f'p5_pos_{dim}'].copy().values,df[f'p4_pos_{dim}'].copy().values\n        \n        df[f'p3_vel_{dim}'],df[f'p4_vel_{dim}']=df[f'p4_vel_{dim}'].copy().values,df[f'p3_vel_{dim}'].copy().values\n        df[f'p4_vel_{dim}'],df[f'p5_vel_{dim}']=df[f'p5_vel_{dim}'].copy().values,df[f'p4_vel_{dim}'].copy().values\n\n    return df\n\n\n\n\n# Mirrors team A and Team B\ndef mirror_match_x(df):\n    # What needs to be mirrored: y-position and y-velocity \n    # What not needs to be mirrored: z-psoition and z-velocity\n    \n    # Mirror Ball\n    df[\"ball_pos_y\"] = df[\"ball_pos_y\"] * -1\n    df[\"ball_vel_y\"] = df[\"ball_vel_y\"] * -1\n    df[\"ball_pos_x\"]= df[\"ball_pos_x\"] * -1\n    df[\"ball_vel_x\"] = df[\"ball_vel_x\"] * -1\n    \n    # Mirror player\n    for p in range(3):\n        df = mirror_player(df, p, \"y\")\n        df = mirror_player(df, p, \"x\")     \n    \n    # Mirror Booster \n    # Booster position:  [ (-61.4, -81.9), (61.4, -81.9), (-71.7, 0), (71.7, 0), (-61.4, 81.9), (61.4, 81.9) ]\n    df[\"boost0_timer\"], df[\"boost5_timer\"]  = df[\"boost5_timer\"].copy(), df[\"boost0_timer\"].copy()\n    df[\"boost1_timer\"], df[\"boost4_timer\"]  = df[\"boost4_timer\"].copy(), df[\"boost1_timer\"].copy()\n    df[\"boost2_timer\"], df[\"boost3_timer\"]  = df[\"boost3_timer\"].copy(), df[\"boost2_timer\"].copy()\n    \n    # columns from train data\n    if 'team_A_scoring_within_10sec' in df.columns:\n        df[\"team_A_scoring_within_10sec\"] , df[\"team_B_scoring_within_10sec\"] = df[\"team_B_scoring_within_10sec\"].copy() , df[\"team_A_scoring_within_10sec\"].copy()\n    \n    return df\n    \n    \n# p is player 0,1 or 2\n# a is axis. a=y for mirroring on x axis\ndef mirror_player(df, p, a):\n    df[f\"p{p}_pos_{a}\"] = df[f\"p{p}_pos_{a}\"] * -1\n    df[f\"p{p+3}_pos_{a}\"] = df[f\"p{p+3}_pos_{a}\"] * -1\n    df[f\"p{p}_vel_{a}\"] = df[f\"p{p}_vel_{a}\"] * -1\n    df[f\"p{p+3}_vel_{a}\"] = df[f\"p{p+3}_vel_{a}\"] * -1\n    \n    s = df[f\"p{p}_pos_{a}\"].copy()\n    df[f\"p{p}_pos_{a}\"] = df[f\"p{p+3}_pos_{a}\"]\n    df[f\"p{p+3}_pos_{a}\"] = s\n    \n    s = df[f\"p{p}_vel_{a}\"].copy()\n    df[f\"p{p}_vel_{a}\"] = df[f\"p{p+3}_vel_{a}\"]\n    df[f\"p{p+3}_vel_{a}\"] = s\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-10-30T18:38:50.652097Z","iopub.execute_input":"2022-10-30T18:38:50.653297Z","iopub.status.idle":"2022-10-30T18:38:50.672478Z","shell.execute_reply.started":"2022-10-30T18:38:50.653181Z","shell.execute_reply":"2022-10-30T18:38:50.671453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Cleaning the data\n\nFilling the missing numbers with $0$ and setting dtypes to save some space.","metadata":{}},{"cell_type":"code","source":"def impute(df):\n#     df['Team_A_missing']=df['p0_pos_x'].isnull().values | df['p1_pos_x'].isnull().values | df['p2_pos_x'].isnull().values              ####adding missing player columns was no improvement\n#     df['Team_B_missing']=df['p3_pos_x'].isnull().values | df['p4_pos_x'].isnull().values | df['p5_pos_x'].isnull().values\n    df=df.fillna(0)\n    return df\n\ndef set_dtypes(df):\n    for col in INIT_FEATS:\n        df[col]=df[col].astype('float16')\n    return df\n\n","metadata":{"execution":{"iopub.status.busy":"2022-10-30T18:38:41.522732Z","iopub.execute_input":"2022-10-30T18:38:41.523492Z","iopub.status.idle":"2022-10-30T18:38:41.529504Z","shell.execute_reply.started":"2022-10-30T18:38:41.523453Z","shell.execute_reply":"2022-10-30T18:38:41.528282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Loading and preparing the train data\n\nTraining data is loaded. Then rotated playerwise and mirrored versions are concatenated.","metadata":{}},{"cell_type":"code","source":"def load_train():\n    X_train=pd.DataFrame()\n    y=pd.DataFrame()\n    for i in range(SIZE):#[5,8,1,0,4,3,9,2,7,6]: #\n        X_train_temp = pd.read_csv(PATH_TO_DATA+f'train_{i}.csv')\n               \n#         X_train_temp_rot = rotate_players(X_train_temp.copy())\n#         X_train_temp= pd.concat([X_train_temp,X_train_temp_rot])\n#         del X_train_temp_rot\n#         gc.collect()        \n#         X_train_mirror = mirror_match_x(X_train_temp.copy())\n#         X_train_temp= pd.concat([X_train_temp,X_train_mirror]) \n#         del X_train_mirror\n#         gc.collect()\n        \n        \n        X_train_temp = X_train_temp.sample(frac=SAMPLE, random_state=SEED)\n        X_train_temp=X_train_temp.drop(DROP_COLS_TRAIN,axis=1)\n\n        \n        \n        y_temp=X_train_temp[TARGETS]\n        X_train_temp= X_train_temp.drop(TARGETS,axis=1)\n\n\n        \n        X_train_temp=impute(X_train_temp)\n        X_train_temp=set_dtypes(X_train_temp)\n        \n        X_train = pd.concat([X_train,X_train_temp])\n        y = pd.concat([y,y_temp])\n\n        \n        del X_train_temp, y_temp\n        gc.collect()\n\n        print(f'Completed importing train_{i}.')\n    return X_train, y\n        ","metadata":{"execution":{"iopub.status.busy":"2022-10-30T18:38:45.335024Z","iopub.execute_input":"2022-10-30T18:38:45.335438Z","iopub.status.idle":"2022-10-30T18:38:45.344281Z","shell.execute_reply.started":"2022-10-30T18:38:45.335405Z","shell.execute_reply":"2022-10-30T18:38:45.343138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading the data","metadata":{}},{"cell_type":"code","source":"%%time\nX_train, y =load_train()\nprint(X_train.shape,y.shape )\nX_train=feature_eng(X_train)\n#X_train=X_train.drop(DROP_COLS_AFTER_FE,axis=1)\n# for col in X_train.columns:\n#     if col not in NUM_FEATS:\n#         print(col)\n\n\nfor col in NUM_FEATS:\n        X_train[col]=X_train[col].astype('float16')\n        \nscaler= StandardScaler()\nscaler = scaler.fit(X_train)\nX_train=  pd.DataFrame(scaler.transform(X_train), columns = NUM_FEATS)\n\n\nfor col in NUM_FEATS:\n        X_train[col]=X_train[col].astype('float16')\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-10-30T14:32:05.186067Z","iopub.execute_input":"2022-10-30T14:32:05.186552Z","iopub.status.idle":"2022-10-30T14:44:03.882329Z","shell.execute_reply.started":"2022-10-30T14:32:05.186518Z","shell.execute_reply":"2022-10-30T14:44:03.880830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Fitting the models\n\nSimple train and test sets are generated. Then for each team, a LGBM-model is being fitted. The feature importance is plotted to see whether the importance of the features corresponds to intuition.","metadata":{}},{"cell_type":"code","source":"# from sklearn.model_selection import StratifiedKFold\n# import lightgbm as lgbm\n# from sklearn.metrics import log_loss\n# models_A=[]\n# N_SPLITS = 3\n# strat_kf = StratifiedKFold(n_splits=N_SPLITS, shuffle=True, random_state=1121218)\n\n# scores = np.empty(N_SPLITS)\n# for idx, (train_idx, test_idx) in enumerate(strat_kf.split(X, yA)):\n#     print(\"=\" * 12 + f\"Training fold {idx}\" + 12 * \"=\")\n\n#     X_tr, X_val = X.iloc[train_idx].copy(), X.iloc[test_idx].copy()\n#     y_tr, y_val = yA[train_idx].copy(), yA[test_idx].copy()\n#     eval_set = [(X_val, y_val)]\n\n#     lgbm_clf = lgbm.LGBMClassifier(n_estimators=100)\n#     lgbm_clf.fit(\n#         X_tr,\n#         y_tr,\n#         eval_set=eval_set,\n#         early_stopping_rounds=10,\n#         eval_metric=\"binary_logloss\",\n#         verbose=False,\n#     )\n\n#     preds = lgbm_clf.predict_proba(X_val)\n#     loss = log_loss(y_val, preds)\n#     scores[idx] = loss\n#     models_A.append(lgbm_clf)","metadata":{"execution":{"iopub.status.busy":"2022-10-30T06:51:45.583530Z","iopub.execute_input":"2022-10-30T06:51:45.585444Z","iopub.status.idle":"2022-10-30T06:51:45.591449Z","shell.execute_reply.started":"2022-10-30T06:51:45.585395Z","shell.execute_reply":"2022-10-30T06:51:45.590419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Team A","metadata":{}},{"cell_type":"code","source":"# %%time\n\n# model_A = lgb.LGBMClassifier(**params)\n\n# yA = y.pop(\"team_A_scoring_within_10sec\")\n\n# train_XA,valid_XA,train_yA,valid_yA = train_test_split(X_train[TEAM_A_COLS],yA,test_size=0.3,random_state=42)\n\n# model_A.fit(train_XA,train_yA,eval_set = [(valid_XA,valid_yA),(train_XA,train_yA)])\n\n# del train_XA,train_yA\n# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-30T06:51:45.592992Z","iopub.execute_input":"2022-10-30T06:51:45.593684Z","iopub.status.idle":"2022-10-30T14:10:27.738324Z","shell.execute_reply.started":"2022-10-30T06:51:45.593647Z","shell.execute_reply":"2022-10-30T14:10:27.736676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Team A feature importance","metadata":{}},{"cell_type":"code","source":"\n# joblib.dump(model_A, 'model_A.pkl')\nmodel_A= joblib.load('../input/models/model_A.pkl')\nplot_importance(model_A, figsize=(40, 30))","metadata":{"execution":{"iopub.status.busy":"2022-10-30T14:30:58.938238Z","iopub.execute_input":"2022-10-30T14:30:58.938756Z","iopub.status.idle":"2022-10-30T14:31:01.056178Z","shell.execute_reply.started":"2022-10-30T14:30:58.938712Z","shell.execute_reply":"2022-10-30T14:31:01.055292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Team B","metadata":{}},{"cell_type":"code","source":"%%time\n\nparams = {\n    'objective':'binary',\n    'learning_rate':0.15, #0.025\n    'lambda_l2':2,\n    'path_smooth':1,\n    'boosting_type':'dart', #gbdt\n    'num_leaves':2**6,\n    'max_depth':6,\n    'extra_trees':True,\n    'bagging_freq':10,\n    'bagging_fraction':0.7,\n    'num_iterations':750,#300,#1200,#2000, #250\n    'force_col_wise':True,\n}\n\n\n\nmodel_B = lgb.LGBMClassifier(**params)\n\nyB = y.pop(\"team_B_scoring_within_10sec\")\n\ntrain_XB,valid_XB,train_yB,valid_yB = train_test_split(X_train[TEAM_B_COLS],yB,test_size=0.3,random_state=42)\n\nmodel_B.fit(train_XB,train_yB,eval_set = [(valid_XB,valid_yB),(train_XB,train_yB)])\n\ndel train_XB,train_yB\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-30T14:44:28.695912Z","iopub.execute_input":"2022-10-30T14:44:28.697135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Tean B feature importance","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"joblib.dump(model_B, 'model_B.pkl')\nplot_importance(model_B, figsize=(40, 30))","metadata":{"execution":{"iopub.status.busy":"2022-10-30T14:30:28.530384Z","iopub.execute_input":"2022-10-30T14:30:28.530821Z","iopub.status.idle":"2022-10-30T14:30:28.637780Z","shell.execute_reply.started":"2022-10-30T14:30:28.530785Z","shell.execute_reply":"2022-10-30T14:30:28.635269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submit","metadata":{}},{"cell_type":"code","source":"def load_test():\n    X_test = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/test.csv')\n    df_id=X_test.pop('id')\n    X_test=impute(X_test)\n    X_test=set_dtypes(X_test)\n    return X_test, df_id\n\n        ","metadata":{"execution":{"iopub.status.busy":"2022-10-30T18:38:17.001081Z","iopub.execute_input":"2022-10-30T18:38:17.002217Z","iopub.status.idle":"2022-10-30T18:38:17.007553Z","shell.execute_reply.started":"2022-10-30T18:38:17.002176Z","shell.execute_reply":"2022-10-30T18:38:17.006313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test, df_id=load_test() #r\n\n#X_test_rot=rotate_players(X_test.copy()) #r\n\nX_test_mir=mirror_match_x(X_test.copy()) #l\n\n#X_test_mir_rot=mirror_match_x(X_test_mir.copy()) #l\n\nX_test=feature_eng(X_test)\n#X_test_rot=feature_eng(X_test_rot)\nX_test_mir=feature_eng(X_test_mir)\n#X_test_mir_rot=feature_eng(X_test_mir_rot)\n\n\n#X_test=X_test.drop(DROP_COLS_AFTER_FE,axis=1)\n\nX_test= pd.DataFrame(scaler.transform(X_test), columns = NUM_FEATS)\n#X_test_rot= pd.DataFrame(scaler.transform(X_test_rot), columns = NUM_FEATS)\nX_test_mir= pd.DataFrame(scaler.transform(X_test_mir), columns = NUM_FEATS)\n#X_test_mir_rot= pd.DataFrame(scaler.transform(X_test_mir_rot), columns = NUM_FEATS)\n\n\nfor col in NUM_FEATS:\n        X_test[col]=X_test[col].astype('float16')\n        #X_test_rot[col]=X_test_rot[col].astype('float16')\n        X_test_mir[col]=X_test_mir[col].astype('float16')\n        #X_test_mir_rot[col]=X_test_mir_rot[col].astype('float16')\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-10-30T18:39:19.144264Z","iopub.execute_input":"2022-10-30T18:39:19.145313Z","iopub.status.idle":"2022-10-30T18:39:36.315953Z","shell.execute_reply.started":"2022-10-30T18:39:19.145274Z","shell.execute_reply":"2022-10-30T18:39:36.314592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Make predictions","metadata":{}},{"cell_type":"code","source":"predict_a = (model_A.predict_proba(X_test[TEAM_A_COLS])[:,1] \n             #+model_A.predict_proba(X_test_rot[TEAM_A_COLS])[:,1]\n             +model_B.predict_proba(X_test_mir[TEAM_B_COLS])[:,1]\n             #+model_B.predict_proba(X_test_mir_rot[TEAM_B_COLS])[:,1])/4\n            )/2\npredict_b = (model_B.predict_proba(X_test[TEAM_B_COLS])[:,1]\n             #+model_B.predict_proba(X_test[TEAM_B_COLS])[:,1]\n             +model_A.predict_proba(X_test_mir[TEAM_A_COLS])[:,1]\n             #+model_A.predict_proba(X_test_mir_rot[TEAM_A_COLS])[:,1])/4\n            )/2","metadata":{"execution":{"iopub.status.busy":"2022-10-30T14:30:28.643786Z","iopub.status.idle":"2022-10-30T14:30:28.644221Z","shell.execute_reply.started":"2022-10-30T14:30:28.643998Z","shell.execute_reply":"2022-10-30T14:30:28.644034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Generate submission","metadata":{}},{"cell_type":"code","source":"df_submission = pd.DataFrame(\n    {\n        \"id\": df_id,\n        \"team_A_scoring_within_10sec\": predict_a,\n        \"team_B_scoring_within_10sec\": predict_b\n    }\n)\ndf_submission.to_csv('submission123.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-10-30T14:30:28.646643Z","iopub.status.idle":"2022-10-30T14:30:28.647945Z","shell.execute_reply.started":"2022-10-30T14:30:28.647617Z","shell.execute_reply":"2022-10-30T14:30:28.647655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import cPickle\n# # save the classifier\n# with open('my_dumped_classifier.pkl', 'wb') as fid:\n#     cPickle.dump(gnb, fid)    \n\n# # load it again\n# with open('my_dumped_classifier.pkl', 'rb') as fid:\n#     gnb_loaded = cPickle.load(fid)","metadata":{"execution":{"iopub.status.busy":"2022-10-30T14:30:28.649718Z","iopub.status.idle":"2022-10-30T14:30:28.650656Z","shell.execute_reply.started":"2022-10-30T14:30:28.650323Z","shell.execute_reply":"2022-10-30T14:30:28.650354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Retrain model","metadata":{}},{"cell_type":"code","source":"\n# save model\n# joblib.dump(model_A, 'model_A.pkl')\n# joblib.dump(model_B, 'model_B.pkl')\n# # load model\n\n\n# import joblib\n# import pickle\n# model_A_saved = joblib.load('model_A.pkl')\n# model_B_saved = joblib.load('model_B.pkl')","metadata":{"execution":{"iopub.status.busy":"2022-10-30T14:30:28.652381Z","iopub.status.idle":"2022-10-30T14:30:28.653562Z","shell.execute_reply.started":"2022-10-30T14:30:28.653333Z","shell.execute_reply":"2022-10-30T14:30:28.653358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# model_A_saved = joblib.load('../input/models/model_A.pkl')","metadata":{"execution":{"iopub.status.busy":"2022-10-30T18:37:34.080049Z","iopub.execute_input":"2022-10-30T18:37:34.080426Z","iopub.status.idle":"2022-10-30T18:37:34.240972Z","shell.execute_reply.started":"2022-10-30T18:37:34.080395Z","shell.execute_reply":"2022-10-30T18:37:34.240158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model_A_saved","metadata":{"execution":{"iopub.status.busy":"2022-10-30T18:37:52.154313Z","iopub.execute_input":"2022-10-30T18:37:52.155139Z","iopub.status.idle":"2022-10-30T18:37:52.166369Z","shell.execute_reply.started":"2022-10-30T18:37:52.155092Z","shell.execute_reply":"2022-10-30T18:37:52.165296Z"},"trusted":true},"execution_count":null,"outputs":[]}]}