{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Using the TFDF library that was shown in the Google Machine Learning Bootcamp\n! pip install tensorflow_decision_forests","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-10-23T20:27:51.230135Z","iopub.execute_input":"2022-10-23T20:27:51.230801Z","iopub.status.idle":"2022-10-23T20:28:27.854846Z","shell.execute_reply.started":"2022-10-23T20:27:51.230758Z","shell.execute_reply":"2022-10-23T20:28:27.853357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport tensorflow_decision_forests as tfdf\nimport tensorflow as tf\nimport gc","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-23T20:38:09.836146Z","iopub.execute_input":"2022-10-23T20:38:09.836558Z","iopub.status.idle":"2022-10-23T20:38:09.842610Z","shell.execute_reply.started":"2022-10-23T20:38:09.836524Z","shell.execute_reply":"2022-10-23T20:38:09.841329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Read data","metadata":{}},{"cell_type":"code","source":"# Modify the dtypes to make the dataset compatible with TFDF (float16 -> float32)\ndtypes = {'game_num': 'int32',\n 'event_id': 'int32',\n 'event_time': 'float32',\n 'ball_pos_x': 'float32',\n 'ball_pos_y': 'float32',\n 'ball_pos_z': 'float32',\n 'ball_vel_x': 'float32',\n 'ball_vel_y': 'float32',\n 'ball_vel_z': 'float32',\n 'p0_pos_x': 'float32',\n 'p0_pos_y': 'float32',\n 'p0_pos_z': 'float32',\n 'p0_vel_x': 'float32',\n 'p0_vel_y': 'float32',\n 'p0_vel_z': 'float32',\n 'p0_boost': 'float32',\n 'p1_pos_x': 'float32',\n 'p1_pos_y': 'float32',\n 'p1_pos_z': 'float32',\n 'p1_vel_x': 'float32',\n 'p1_vel_y': 'float32',\n 'p1_vel_z': 'float32',\n 'p1_boost': 'float32',\n 'p2_pos_x': 'float32',\n 'p2_pos_y': 'float32',\n 'p2_pos_z': 'float32',\n 'p2_vel_x': 'float32',\n 'p2_vel_y': 'float32',\n 'p2_vel_z': 'float32',\n 'p2_boost': 'float32',\n 'p3_pos_x': 'float32',\n 'p3_pos_y': 'float32',\n 'p3_pos_z': 'float32',\n 'p3_vel_x': 'float32',\n 'p3_vel_y': 'float32',\n 'p3_vel_z': 'float32',\n 'p3_boost': 'float32',\n 'p4_pos_x': 'float32',\n 'p4_pos_y': 'float32',\n 'p4_pos_z': 'float32',\n 'p4_vel_x': 'float32',\n 'p4_vel_y': 'float32',\n 'p4_vel_z': 'float32',\n 'p4_boost': 'float32',\n 'p5_pos_x': 'float32',\n 'p5_pos_y': 'float32',\n 'p5_pos_z': 'float32',\n 'p5_vel_x': 'float32',\n 'p5_vel_y': 'float32',\n 'p5_vel_z': 'float32',\n 'p5_boost': 'float32',\n 'boost0_timer': 'float32',\n 'boost1_timer': 'float32',\n 'boost2_timer': 'float32',\n 'boost3_timer': 'float32',\n 'boost4_timer': 'float32',\n 'boost5_timer': 'float32',\n 'player_scoring_next': 'str',\n 'team_scoring_next': 'object',\n 'team_A_scoring_within_10sec': 'int32',\n 'team_B_scoring_within_10sec': 'int32'}\ndf1 = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_1.csv', dtype=dtypes)\ndf2 = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_2.csv', dtype=dtypes)\ndf3 = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_3.csv', dtype=dtypes)\n\ndf = pd.concat([df1,df2,df3], axis=0)\n\n# test_df = pd.read_csv(\"/kaggle/input/tabular-playground-series-oct-2022/test.csv\", dtype=dtypes)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T20:40:05.043935Z","iopub.execute_input":"2022-10-23T20:40:05.044395Z","iopub.status.idle":"2022-10-23T20:40:58.386964Z","shell.execute_reply.started":"2022-10-23T20:40:05.044358Z","shell.execute_reply":"2022-10-23T20:40:58.385843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-23T20:40:58.417679Z","iopub.execute_input":"2022-10-23T20:40:58.418085Z","iopub.status.idle":"2022-10-23T20:40:58.607428Z","shell.execute_reply.started":"2022-10-23T20:40:58.418049Z","shell.execute_reply":"2022-10-23T20:40:58.606069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Add target data as a single column","metadata":{}},{"cell_type":"code","source":"def get_scorer(x):\n    t = {\n        (0,0): \"Neither\",\n        (1,0): \"A\",\n        (0,1): \"B\"\n    }\n    return t[tuple(x.to_list())]\n\n\ndf['target'] = df[['team_A_scoring_within_10sec', 'team_B_scoring_within_10sec']].apply(get_scorer, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T20:40:58.610226Z","iopub.execute_input":"2022-10-23T20:40:58.610705Z","iopub.status.idle":"2022-10-23T20:41:52.924314Z","shell.execute_reply.started":"2022-10-23T20:40:58.610658Z","shell.execute_reply":"2022-10-23T20:41:52.923063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check balance of data\ndf.target.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-10-23T20:43:17.798113Z","iopub.execute_input":"2022-10-23T20:43:17.798613Z","iopub.status.idle":"2022-10-23T20:43:18.078879Z","shell.execute_reply.started":"2022-10-23T20:43:17.798576Z","shell.execute_reply":"2022-10-23T20:43:18.077765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Split the data by game","metadata":{}},{"cell_type":"code","source":"# get a val set splitting by game_num\ngame_nums = df.game_num.unique()\nn_games = len(game_nums)\nn_games_val = int(n_games*.01)\nprint(n_games, n_games_val)\nnp.random.shuffle(game_nums)\n\ngame_nums_val = game_nums[:n_games_val]\ngame_nums_train = game_nums[n_games_val:]\n\nval_df = df[df.game_num.isin(game_nums_val)].drop('game_num', axis=1)\ntrain_df = df[df.game_num.isin(game_nums_train)].drop(\"game_num\", axis=1)\n\ndrop_cols = [\n    'team_scoring_next', 'team_A_scoring_within_10sec', \n    'team_B_scoring_within_10sec', 'event_id', \n    'event_time', 'player_scoring_next'\n]\n\nds_train = tfdf.keras.pd_dataframe_to_tf_dataset(train_df.drop(drop_cols, axis=1), label='target').prefetch(1)\nds_val = tfdf.keras.pd_dataframe_to_tf_dataset(val_df.drop(drop_cols, axis=1), label='target').prefetch(1)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T20:45:42.410943Z","iopub.execute_input":"2022-10-23T20:45:42.412177Z","iopub.status.idle":"2022-10-23T20:45:53.341920Z","shell.execute_reply.started":"2022-10-23T20:45:42.412121Z","shell.execute_reply":"2022-10-23T20:45:53.340791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create GBT model","metadata":{}},{"cell_type":"code","source":"import os\ncpus = os.cpu_count()\nprint(cpus)\nmodel = tfdf.keras.GradientBoostedTreesModel(num_threads=cpus)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T21:09:08.224820Z","iopub.execute_input":"2022-10-23T21:09:08.225622Z","iopub.status.idle":"2022-10-23T21:09:08.239648Z","shell.execute_reply.started":"2022-10-23T21:09:08.225582Z","shell.execute_reply":"2022-10-23T21:09:08.238659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(x=ds_train, validation_data=ds_val)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T21:09:08.378717Z","iopub.execute_input":"2022-10-23T21:09:08.379890Z","iopub.status.idle":"2022-10-23T22:15:52.488628Z","shell.execute_reply.started":"2022-10-23T21:09:08.379827Z","shell.execute_reply":"2022-10-23T22:15:52.487160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save(\"GBTv2\")","metadata":{"execution":{"iopub.status.busy":"2022-10-23T22:16:07.606757Z","iopub.execute_input":"2022-10-23T22:16:07.607177Z","iopub.status.idle":"2022-10-23T22:16:11.159427Z","shell.execute_reply.started":"2022-10-23T22:16:07.607140Z","shell.execute_reply":"2022-10-23T22:16:11.158099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Obtain val performance","metadata":{}},{"cell_type":"code","source":"val_preds = model.predict(ds_val)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T22:16:11.162159Z","iopub.execute_input":"2022-10-23T22:16:11.163002Z","iopub.status.idle":"2022-10-23T22:16:16.471268Z","shell.execute_reply.started":"2022-10-23T22:16:11.162951Z","shell.execute_reply":"2022-10-23T22:16:16.470307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_df['A_pred'] = val_preds[:,0]\nval_df['B_pred'] = val_preds[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-10-23T22:18:06.984521Z","iopub.execute_input":"2022-10-23T22:18:06.985015Z","iopub.status.idle":"2022-10-23T22:18:06.993305Z","shell.execute_reply.started":"2022-10-23T22:18:06.984956Z","shell.execute_reply":"2022-10-23T22:18:06.991863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_df.A_pred.hist()","metadata":{"execution":{"iopub.status.busy":"2022-10-23T22:18:09.076363Z","iopub.execute_input":"2022-10-23T22:18:09.076818Z","iopub.status.idle":"2022-10-23T22:18:09.335202Z","shell.execute_reply.started":"2022-10-23T22:18:09.076775Z","shell.execute_reply":"2022-10-23T22:18:09.333873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import metrics\nimport matplotlib.pyplot as plt\nfpr_b, tpr_b, thr_b = metrics.roc_curve(val_df.team_B_scoring_within_10sec,  val_df.B_pred)\nfpr_a, tpr_a, thr_a = metrics.roc_curve(val_df.team_A_scoring_within_10sec,  val_df.A_pred)\n\n#create ROC curve\nplt.plot(fpr_a,tpr_a, label='A')\nplt.plot(fpr_b,tpr_b, label='B')\n\nplt.ylabel('True Positive Rate')\nplt.xlabel('False Positive Rate')\nplt.show()\n\npos_a = np.argmax(tpr_a-fpr_a)\npos_b = np.argmax(tpr_b-fpr_b)\nbest_thr_a = thr_a[pos_a]\nbest_thr_b = thr_b[pos_b]\nprint(\"A: Threshold with biggest difference\", best_thr_a, \"@ FPR:\", fpr_a[pos_a], \"TPR:\",tpr_a[pos_a])\nprint(\"A: AUC\", metrics.auc(fpr_a, tpr_a))\nprint(\"B: Threshold with biggest difference\", best_thr_b, \"@ FPR:\", fpr_b[pos_b], \"TPR:\",tpr_b[pos_b])\nprint(\"B: AUC\", metrics.auc(fpr_b, tpr_b))\nbest_thr = (best_thr_a + best_thr_b) / 2","metadata":{"execution":{"iopub.status.busy":"2022-10-23T22:18:10.972866Z","iopub.execute_input":"2022-10-23T22:18:10.973313Z","iopub.status.idle":"2022-10-23T22:19:02.210652Z","shell.execute_reply.started":"2022-10-23T22:18:10.973272Z","shell.execute_reply":"2022-10-23T22:19:02.209770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare submission","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_csv(\"/kaggle/input/tabular-playground-series-oct-2022/test.csv\", dtype=dtypes)\nds_test = tfdf.keras.pd_dataframe_to_tf_dataset(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T22:19:34.257466Z","iopub.execute_input":"2022-10-23T22:19:34.258081Z","iopub.status.idle":"2022-10-23T22:19:41.116219Z","shell.execute_reply.started":"2022-10-23T22:19:34.258029Z","shell.execute_reply":"2022-10-23T22:19:41.114858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model.predict(ds_test)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T22:19:41.118790Z","iopub.execute_input":"2022-10-23T22:19:41.119174Z","iopub.status.idle":"2022-10-23T22:20:36.284134Z","shell.execute_reply.started":"2022-10-23T22:19:41.119140Z","shell.execute_reply":"2022-10-23T22:20:36.282945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = test_df[['id']]","metadata":{"execution":{"iopub.status.busy":"2022-10-23T22:20:36.285852Z","iopub.execute_input":"2022-10-23T22:20:36.286412Z","iopub.status.idle":"2022-10-23T22:20:36.298265Z","shell.execute_reply.started":"2022-10-23T22:20:36.286363Z","shell.execute_reply":"2022-10-23T22:20:36.296899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df['team_A_scoring_within_10sec'] = tf.keras.activations.sigmoid(pred[:,0] - best_thr).numpy()\nsubmission_df['team_B_scoring_within_10sec'] = tf.keras.activations.sigmoid(pred[:,1] - best_thr).numpy()","metadata":{"execution":{"iopub.status.busy":"2022-10-23T22:20:36.300720Z","iopub.execute_input":"2022-10-23T22:20:36.301089Z","iopub.status.idle":"2022-10-23T22:20:36.322644Z","shell.execute_reply.started":"2022-10-23T22:20:36.301055Z","shell.execute_reply":"2022-10-23T22:20:36.321339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv(\"submissionv2.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T22:20:36.324306Z","iopub.execute_input":"2022-10-23T22:20:36.324784Z","iopub.status.idle":"2022-10-23T22:20:38.191071Z","shell.execute_reply.started":"2022-10-23T22:20:36.324722Z","shell.execute_reply":"2022-10-23T22:20:38.189660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}