{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nfrom sklearn.linear_model import SGDClassifier\nimport random\nimport pandas as pd\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\nimport dask.dataframe as dd\nfrom sklearn.base import BaseEstimator, TransformerMixin","metadata":{"execution":{"iopub.status.busy":"2022-10-26T13:09:46.710883Z","iopub.execute_input":"2022-10-26T13:09:46.711484Z","iopub.status.idle":"2022-10-26T13:09:49.059464Z","shell.execute_reply.started":"2022-10-26T13:09:46.711437Z","shell.execute_reply":"2022-10-26T13:09:49.057732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load data","metadata":{}},{"cell_type":"code","source":"dtypes_train_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_dtypes.csv')\ndtypes_test_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/test_dtypes.csv')\ndtypes_train = {col: dt for (col, dt) in zip(dtypes_train_df.column, dtypes_train_df.dtype)}\ndtypes_test = {col: dt for (col, dt) in zip(dtypes_test_df.column, dtypes_test_df.dtype)}\ndel dtypes_train_df\ndel dtypes_test_df","metadata":{"execution":{"iopub.status.busy":"2022-10-26T13:09:49.062007Z","iopub.execute_input":"2022-10-26T13:09:49.063175Z","iopub.status.idle":"2022-10-26T13:09:49.111807Z","shell.execute_reply.started":"2022-10-26T13:09:49.063099Z","shell.execute_reply":"2022-10-26T13:09:49.109917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = dd.read_csv('../input/tabular-playground-series-oct-2022/train_*.csv', dtype = dtypes_train)\ntrain_df = train_df.compute()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/test.csv', dtype = dtypes_test)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T13:10:49.476340Z","iopub.status.idle":"2022-10-26T13:10:49.477447Z","shell.execute_reply.started":"2022-10-26T13:10:49.476852Z","shell.execute_reply":"2022-10-26T13:10:49.476997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature engineering","metadata":{}},{"cell_type":"markdown","source":"## Dropping unnecessary columns","metadata":{}},{"cell_type":"code","source":"cols_to_drop = [\n    'game_num', 'event_id', 'event_time', 'player_scoring_next', 'team_scoring_next',\n    'column', 'dtype'\n]\ntrain_df = train_df.drop(columns=cols_to_drop)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T13:10:49.480985Z","iopub.status.idle":"2022-10-26T13:10:49.482905Z","shell.execute_reply.started":"2022-10-26T13:10:49.482316Z","shell.execute_reply":"2022-10-26T13:10:49.482364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.dropna(inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T13:10:49.487549Z","iopub.status.idle":"2022-10-26T13:10:49.489838Z","shell.execute_reply.started":"2022-10-26T13:10:49.489363Z","shell.execute_reply":"2022-10-26T13:10:49.489440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Missing values","metadata":{}},{"cell_type":"code","source":"def euclidian_norm(x):\n    return np.linalg.norm(x, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T13:10:49.493434Z","iopub.status.idle":"2022-10-26T13:10:49.494345Z","shell.execute_reply.started":"2022-10-26T13:10:49.493941Z","shell.execute_reply":"2022-10-26T13:10:49.493980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vel_groups = {\n    f\"{el}_vel\": [f'{el}_vel_x', f'{el}_vel_y', f'{el}_vel_z']\n    for el in ['ball'] + [f'p{i}' for i in range(6)]\n}\npos_groups = {\n    f\"{el}_pos\": [f'{el}_pos_x', f'{el}_pos_y', f'{el}_pos_z']\n    for el in ['ball'] + [f'p{i}' for i in range(6)]\n}","metadata":{"execution":{"iopub.status.busy":"2022-10-26T13:10:49.498373Z","iopub.status.idle":"2022-10-26T13:10:49.502166Z","shell.execute_reply.started":"2022-10-26T13:10:49.501149Z","shell.execute_reply":"2022-10-26T13:10:49.501234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Calculate new columns","metadata":{}},{"cell_type":"code","source":"class CombinedAttributesAdder(BaseEstimator, TransformerMixin):\n    def __init__(self, add_velocity_magnitude = True, add_distance_from_ball = True, \n                 add_distance_from_goal = True): \n        self.add_velocity_magnitude = add_velocity_magnitude\n        self.add_distance_from_ball = add_distance_from_ball\n        self.add_distance_from_goal = add_distance_from_goal\n    def fit(self, X, y=None):\n        cols = X.columns.tolist()\n        if self.add_velocity_magnitude:\n            cols.extend([col for col, vec in vel_groups.items()])\n        if self.add_distance_from_ball:\n            cols.extend([col + \"_ball_dist\" for col, vec in pos_groups.items()])\n        if self.add_distance_from_goal:\n            cols.extend([col + \"_goal_a_dist\" for col, vec in pos_groups.items()])\n            cols.extend([col + \"_goal_b_dist\" for col, vec in pos_groups.items()])\n        self.columns = cols\n        self.initialized = True\n        return self \n    def get_feature_names(self): \n        return self.columns\n    def transform(self, X):\n        if self.add_velocity_magnitude: \n            # add columns to df: velocity magnitude for ball and players\n            for col, vec in vel_groups.items(): \n                X[col] = euclidian_norm(X[vec].values.astype(float))\n        if self.add_distance_from_ball: \n            # calculate distance of each player from the ball\n            for col, vec in pos_groups.items():\n                if col != 'ball_pos':\n                    X[col + \"_ball_dist\"] = euclidian_norm(X[vec].values.astype(float) - X[pos_groups[\"ball_pos\"]].values.astype(float))\n        if self.add_distance_from_goal: \n            # calculate distance of ball and each player from the goal\n            goal_a = np.array( [0, -120, 1.2], dtype='float16')\n            goal_b = np.array( [0,  120, 1.2], dtype='float16')\n            for col, vec in pos_groups.items():\n                X[col + \"_goal_a_dist\"] = euclidian_norm(X[vec].values.astype(float) - goal_a)\n                X[col + \"_goal_b_dist\"] = euclidian_norm(X[vec].values.astype(float) - goal_b)\n        return X\n\n# attr_adder = CombinedAttributesAdder()\n# train_df = attr_adder.transform(train_df)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T13:10:49.507798Z","iopub.status.idle":"2022-10-26T13:10:49.510125Z","shell.execute_reply.started":"2022-10-26T13:10:49.509341Z","shell.execute_reply":"2022-10-26T13:10:49.509404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.columns","metadata":{"execution":{"iopub.status.busy":"2022-10-26T13:10:49.517683Z","iopub.status.idle":"2022-10-26T13:10:49.519102Z","shell.execute_reply.started":"2022-10-26T13:10:49.518528Z","shell.execute_reply":"2022-10-26T13:10:49.518571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_y = train_df[['team_A_scoring_within_10sec','team_B_scoring_within_10sec']]\ntrain_X = train_df.drop(['team_A_scoring_within_10sec','team_B_scoring_within_10sec'], axis = 1)\ndel train_df","metadata":{"execution":{"iopub.status.busy":"2022-10-26T13:10:49.524155Z","iopub.status.idle":"2022-10-26T13:10:49.527205Z","shell.execute_reply.started":"2022-10-26T13:10:49.526374Z","shell.execute_reply":"2022-10-26T13:10:49.526446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# calculating new columns\nattr_adder = CombinedAttributesAdder()\ntrain_X = attr_adder.fit_transform(train_X)\n\n# dropping unnecessary columns\ncolumns_to_drop = ['ball_pos_x', 'ball_pos_z', 'ball_vel_x', 'ball_vel_z',\n                  'p0_pos_x', 'p0_pos_z', 'p1_pos_x', 'p1_pos_z',\n                  'p2_pos_x', 'p2_pos_z','p3_pos_x', 'p3_pos_z',\n                  'p4_pos_x', 'p4_pos_z', 'p5_pos_x', 'p5_pos_z',\n                  'p0_vel_x', 'p0_vel_z', 'p1_vel_x', 'p1_vel_z',\n                  'p2_vel_x', 'p2_vel_z', 'p3_vel_x', 'p3_vel_z',\n                  'p4_vel_x', 'p4_vel_z', 'p5_vel_x', 'p5_vel_z']\ntrain_X.drop(columns=columns_to_drop, inplace = True)\n\n# scaling\nscaler = StandardScaler()\ntrain_X[train_X.columns] = scaler.fit_transform(train_X)\n  ","metadata":{"execution":{"iopub.status.busy":"2022-10-26T13:10:49.529755Z","iopub.status.idle":"2022-10-26T13:10:49.533345Z","shell.execute_reply.started":"2022-10-26T13:10:49.532763Z","shell.execute_reply":"2022-10-26T13:10:49.532807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Online learning","metadata":{}},{"cell_type":"code","source":"train_X.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-26T13:10:49.536242Z","iopub.status.idle":"2022-10-26T13:10:49.537699Z","shell.execute_reply.started":"2022-10-26T13:10:49.537242Z","shell.execute_reply":"2022-10-26T13:10:49.537305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def batches(l, n):\n    for i in range(0, len(l), n):\n        yield l[i:i+n]","metadata":{"execution":{"iopub.status.busy":"2022-10-26T13:10:49.540512Z","iopub.status.idle":"2022-10-26T13:10:49.541819Z","shell.execute_reply.started":"2022-10-26T13:10:49.541406Z","shell.execute_reply":"2022-10-26T13:10:49.541457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X.shape","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training binary classifier for team A scoring within 10 seconds\nclfA = SGDClassifier(loss='log') \nshuffledRange = list(range(len(train_X)))\nn_iter = 3\nfor n in range(n_iter):\n    print(f'Shuffling...  niter = {n}/{n_iter} \\n')\n    random.shuffle(shuffledRange)\n    shuffledX = train_X.iloc[shuffledRange]\n    shuffledY = train_y.iloc[shuffledRange,0]\n    for batch in batches(range(len(shuffledX)), 500000):\n        clfA.partial_fit(shuffledX[batch[0]:batch[-1]+1], shuffledY[batch[0]:batch[-1]+1], classes=np.unique(train_y.iloc[:,0]))\n        print(f'Finished: {batch[0]} - {batch[-1]+1} rows trained \\n')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training binary classifier for team B scoring within 10 seconds\nclfB = SGDClassifier(loss='log') \nshuffledRange = list(range(len(train_X)))\nn_iter = 3\nfor n in range(n_iter):\n    print(f'Shuffling...  niter = {n}/{n_iter} \\n')\n    random.shuffle(shuffledRange)\n    shuffledX = train_X.iloc[shuffledRange]\n    shuffledY = train_y.iloc[shuffledRange,1]\n    for batch in batches(range(len(shuffledX)), 500000):\n        clfB.partial_fit(shuffledX[batch[0]:batch[-1]+1], shuffledY[batch[0]:batch[-1]+1], classes=np.unique(train_y.iloc[:,1]))\n        print(f'Finished: {batch[0]} - {batch[-1]+1} rows trained \\n')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test set","metadata":{}},{"cell_type":"code","source":"test_df.columns","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids = test_df['id']\ntest_df = test_df.drop(columns=['id'])","metadata":{"execution":{"iopub.status.busy":"2022-10-26T12:20:50.120423Z","iopub.execute_input":"2022-10-26T12:20:50.120908Z","iopub.status.idle":"2022-10-26T12:20:50.199538Z","shell.execute_reply.started":"2022-10-26T12:20:50.120852Z","shell.execute_reply":"2022-10-26T12:20:50.198258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.isna().sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# imputing missing values\nimputer = SimpleImputer(strategy=\"mean\")\ntest_df[test_df.columns] = imputer.fit_transform(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T12:20:55.971539Z","iopub.execute_input":"2022-10-26T12:20:55.971989Z","iopub.status.idle":"2022-10-26T12:20:56.922480Z","shell.execute_reply.started":"2022-10-26T12:20:55.971950Z","shell.execute_reply":"2022-10-26T12:20:56.920539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# calculating new columns\ntest_df = attr_adder.transform(test_df)\n\n# dropping unnecessary columns\ncolumns_to_drop = ['ball_pos_x', 'ball_pos_z', 'ball_vel_x', 'ball_vel_z',\n                  'p0_pos_x', 'p0_pos_z', 'p1_pos_x', 'p1_pos_z',\n                  'p2_pos_x', 'p2_pos_z','p3_pos_x', 'p3_pos_z',\n                  'p4_pos_x', 'p4_pos_z', 'p5_pos_x', 'p5_pos_z',\n                  'p0_vel_x', 'p0_vel_z', 'p1_vel_x', 'p1_vel_z',\n                  'p2_vel_x', 'p2_vel_z', 'p3_vel_x', 'p3_vel_z',\n                  'p4_vel_x', 'p4_vel_z', 'p5_vel_x', 'p5_vel_z']\ntest_df.drop(columns=columns_to_drop, inplace = True)\n\n# scaling\ntest_df[test_df.columns] = scaler.transform(test_df)\n  ","metadata":{"execution":{"iopub.status.busy":"2022-10-26T12:21:00.134675Z","iopub.execute_input":"2022-10-26T12:21:00.135162Z","iopub.status.idle":"2022-10-26T12:21:03.320310Z","shell.execute_reply.started":"2022-10-26T12:21:00.135124Z","shell.execute_reply":"2022-10-26T12:21:03.319231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prediction","metadata":{}},{"cell_type":"code","source":"predictionA = clfA.predict_proba(test_df)[:,1] \npredictionB = clfB.predict_proba(test_df)[:,1] ","metadata":{"execution":{"iopub.status.busy":"2022-10-26T12:21:04.531632Z","iopub.execute_input":"2022-10-26T12:21:04.532103Z","iopub.status.idle":"2022-10-26T12:21:04.914240Z","shell.execute_reply.started":"2022-10-26T12:21:04.532065Z","shell.execute_reply":"2022-10-26T12:21:04.912706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"df_submission = pd.DataFrame(\n    {\n        \"id\": ids,\n        \"team_A_scoring_within_10sec\": predictionA,\n        \"team_B_scoring_within_10sec\": predictionB\n    }\n)\ndf_submission","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T12:21:14.538533Z","iopub.execute_input":"2022-10-26T12:21:14.539011Z","iopub.status.idle":"2022-10-26T12:21:17.275331Z","shell.execute_reply.started":"2022-10-26T12:21:14.538973Z","shell.execute_reply":"2022-10-26T12:21:17.273927Z"},"trusted":true},"execution_count":null,"outputs":[]}]}