{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-18T14:58:41.809767Z","iopub.execute_input":"2022-10-18T14:58:41.810186Z","iopub.status.idle":"2022-10-18T14:58:41.820899Z","shell.execute_reply.started":"2022-10-18T14:58:41.810151Z","shell.execute_reply":"2022-10-18T14:58:41.819126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Reference:\nhttps://www.kaggle.com/code/cv13j0/tps-oct22-gbdt-classifier\n\nhttps://www.kaggle.com/code/samuelcortinhas/tps-oct-22-rocket-league-eda","metadata":{}},{"cell_type":"markdown","source":"# Import Libary","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport missingno as msno\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom xgboost import XGBClassifier","metadata":{"execution":{"iopub.status.busy":"2022-10-18T14:58:41.823413Z","iopub.execute_input":"2022-10-18T14:58:41.824177Z","iopub.status.idle":"2022-10-18T14:58:41.833258Z","shell.execute_reply.started":"2022-10-18T14:58:41.824137Z","shell.execute_reply":"2022-10-18T14:58:41.832220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data","metadata":{}},{"cell_type":"code","source":"df_train_types = pd.read_csv('../input/tabular-playground-series-oct-2022/train_dtypes.csv')\ndf_train_types","metadata":{"execution":{"iopub.status.busy":"2022-10-18T14:58:41.892552Z","iopub.execute_input":"2022-10-18T14:58:41.892921Z","iopub.status.idle":"2022-10-18T14:58:41.909546Z","shell.execute_reply.started":"2022-10-18T14:58:41.892890Z","shell.execute_reply":"2022-10-18T14:58:41.908457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('../input/tabular-playground-series-oct-2022/train_0.csv', dtype=dict(df_train_types.values))\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-18T14:58:41.966514Z","iopub.execute_input":"2022-10-18T14:58:41.968395Z","iopub.status.idle":"2022-10-18T14:58:57.757792Z","shell.execute_reply.started":"2022-10-18T14:58:41.968345Z","shell.execute_reply":"2022-10-18T14:58:57.756717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test_types = pd.read_csv('../input/tabular-playground-series-oct-2022/test_dtypes.csv')\ndf_test = pd.read_csv('../input/tabular-playground-series-oct-2022/test.csv', dtype=dict(df_test_types.values)) ","metadata":{"execution":{"iopub.status.busy":"2022-10-18T14:58:57.760146Z","iopub.execute_input":"2022-10-18T14:58:57.760587Z","iopub.status.idle":"2022-10-18T14:59:02.424741Z","shell.execute_reply.started":"2022-10-18T14:58:57.760545Z","shell.execute_reply":"2022-10-18T14:59:02.423693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-18T14:59:02.426421Z","iopub.execute_input":"2022-10-18T14:59:02.426813Z","iopub.status.idle":"2022-10-18T14:59:02.455388Z","shell.execute_reply.started":"2022-10-18T14:59:02.426773Z","shell.execute_reply":"2022-10-18T14:59:02.454134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('train:', df_train.shape)\nprint('test:', df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-10-18T14:59:02.459612Z","iopub.execute_input":"2022-10-18T14:59:02.460080Z","iopub.status.idle":"2022-10-18T14:59:02.467595Z","shell.execute_reply.started":"2022-10-18T14:59:02.460016Z","shell.execute_reply":"2022-10-18T14:59:02.466261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# drop the columns which not in test data\ncols = list(df_test.columns) + [\"team_A_scoring_within_10sec\", \"team_B_scoring_within_10sec\"]\ndf_train.drop(columns=df_train.columns.difference(cols), axis=1, inplace=True)\ndf_train = df_train.sample(frac=0.4, random_state=44)\ndf_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-18T14:59:02.469391Z","iopub.execute_input":"2022-10-18T14:59:02.470405Z","iopub.status.idle":"2022-10-18T14:59:03.161411Z","shell.execute_reply.started":"2022-10-18T14:59:02.470356Z","shell.execute_reply":"2022-10-18T14:59:03.160220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Drop the columns not in test data except the target columns","metadata":{}},{"cell_type":"markdown","source":"### Duplicated Data","metadata":{}},{"cell_type":"code","source":"print('train:', df_train.duplicated().sum())\nprint('test:', df_test.duplicated().sum())","metadata":{"execution":{"iopub.status.busy":"2022-10-18T14:59:03.163077Z","iopub.execute_input":"2022-10-18T14:59:03.163735Z","iopub.status.idle":"2022-10-18T14:59:10.741817Z","shell.execute_reply.started":"2022-10-18T14:59:03.163694Z","shell.execute_reply":"2022-10-18T14:59:10.740740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Mising Data","metadata":{}},{"cell_type":"code","source":"missing_data_cols = df_train.columns[df_train.isna().any()]\ndf_train[missing_data_cols].isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-10-18T14:59:10.743557Z","iopub.execute_input":"2022-10-18T14:59:10.743979Z","iopub.status.idle":"2022-10-18T14:59:10.906194Z","shell.execute_reply.started":"2022-10-18T14:59:10.743941Z","shell.execute_reply":"2022-10-18T14:59:10.904969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"msno.matrix(df_train[missing_data_cols], labels=True, figsize=(16,10))","metadata":{"execution":{"iopub.status.busy":"2022-10-18T14:59:10.907553Z","iopub.execute_input":"2022-10-18T14:59:10.908212Z","iopub.status.idle":"2022-10-18T14:59:23.661282Z","shell.execute_reply.started":"2022-10-18T14:59:10.908174Z","shell.execute_reply":"2022-10-18T14:59:23.660120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- There are small amount of missing data. It should relate with player destroy and respawns","metadata":{}},{"cell_type":"code","source":"df_train.fillna(0, inplace=True)\ndf_test.fillna(0, inplace=True)\n\ndf_train[missing_data_cols].isna().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2022-10-18T14:59:23.662780Z","iopub.execute_input":"2022-10-18T14:59:23.663385Z","iopub.status.idle":"2022-10-18T14:59:23.911745Z","shell.execute_reply.started":"2022-10-18T14:59:23.663343Z","shell.execute_reply":"2022-10-18T14:59:23.910617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ny_a = df_train['team_A_scoring_within_10sec']\ny_b = df_train['team_B_scoring_within_10sec']\n\ndf_train = df_train.drop(['team_A_scoring_within_10sec', 'team_B_scoring_within_10sec'], axis=1)\nfeatures = df_train.columns\n\nscaler = StandardScaler()\ndf_train = scaler.fit_transform(df_train)\ndf_test[features] = scaler.transform(df_test[features])\n\nX_train_a, X_val_a, y_train_a, y_val_a = train_test_split(df_train, y_a, test_size=0.3, random_state=44)\nX_train_b, X_val_b, y_train_b, y_val_b = train_test_split(df_train, y_b, test_size=0.3, random_state=44)","metadata":{"execution":{"iopub.status.busy":"2022-10-18T15:00:16.112651Z","iopub.execute_input":"2022-10-18T15:00:16.113065Z","iopub.status.idle":"2022-10-18T15:00:18.944020Z","shell.execute_reply.started":"2022-10-18T15:00:16.113019Z","shell.execute_reply":"2022-10-18T15:00:18.943037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"code","source":"# from https://www.kaggle.com/code/cv13j0/tps-oct22-gbdt-classifier?scriptVersionId=107510679&cellId=42\nxgb_params = {\n              'objective'       : 'binary:logistic',\n              'tree_method'     : 'gpu_hist',\n             }\n\nclf = XGBClassifier(**xgb_params)","metadata":{"execution":{"iopub.status.busy":"2022-10-18T15:17:52.469092Z","iopub.execute_input":"2022-10-18T15:17:52.470246Z","iopub.status.idle":"2022-10-18T15:17:52.475651Z","shell.execute_reply.started":"2022-10-18T15:17:52.470201Z","shell.execute_reply":"2022-10-18T15:17:52.474407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Team A","metadata":{}},{"cell_type":"code","source":"clf.fit(X_train_a, \n        y_train_a, \n        eval_set = [(X_val_a, y_val_a)], \n        eval_metric = ['logloss'], \n        early_stopping_rounds = 128, \n        verbose = 32)","metadata":{"execution":{"iopub.status.busy":"2022-10-18T15:18:26.307124Z","iopub.execute_input":"2022-10-18T15:18:26.307520Z","iopub.status.idle":"2022-10-18T15:18:28.667166Z","shell.execute_reply.started":"2022-10-18T15:18:26.307489Z","shell.execute_reply":"2022-10-18T15:18:28.666005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf.classes_","metadata":{"execution":{"iopub.status.busy":"2022-10-18T15:18:51.662933Z","iopub.execute_input":"2022-10-18T15:18:51.663980Z","iopub.status.idle":"2022-10-18T15:18:51.672152Z","shell.execute_reply.started":"2022-10-18T15:18:51.663934Z","shell.execute_reply":"2022-10-18T15:18:51.671058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# take True probability\npredict_a = clf.predict_proba(df_test[features])[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-10-18T15:18:57.796818Z","iopub.execute_input":"2022-10-18T15:18:57.797226Z","iopub.status.idle":"2022-10-18T15:18:59.566995Z","shell.execute_reply.started":"2022-10-18T15:18:57.797195Z","shell.execute_reply":"2022-10-18T15:18:59.565659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Team B","metadata":{}},{"cell_type":"code","source":"clf.fit(X_train_b, \n        y_train_b, \n        eval_set = [(X_val_b, y_val_b)], \n        eval_metric = ['logloss'], \n        early_stopping_rounds = 128, \n        verbose = 32)","metadata":{"execution":{"iopub.status.busy":"2022-10-18T15:19:15.985926Z","iopub.execute_input":"2022-10-18T15:19:15.986644Z","iopub.status.idle":"2022-10-18T15:19:18.614162Z","shell.execute_reply.started":"2022-10-18T15:19:15.986605Z","shell.execute_reply":"2022-10-18T15:19:18.612994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict_b = clf.predict_proba(df_test[features])[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-10-18T15:19:21.264245Z","iopub.execute_input":"2022-10-18T15:19:21.264745Z","iopub.status.idle":"2022-10-18T15:19:23.152438Z","shell.execute_reply.started":"2022-10-18T15:19:21.264698Z","shell.execute_reply":"2022-10-18T15:19:23.151594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"df_submit = pd.read_csv('../input/tabular-playground-series-oct-2022/sample_submission.csv')\ndf_submit['team_A_scoring_within_10sec'] = predict_a\ndf_submit['team_B_scoring_within_10sec'] = predict_b","metadata":{"execution":{"iopub.status.busy":"2022-10-18T15:19:24.534693Z","iopub.execute_input":"2022-10-18T15:19:24.535100Z","iopub.status.idle":"2022-10-18T15:19:24.632102Z","shell.execute_reply.started":"2022-10-18T15:19:24.535059Z","shell.execute_reply":"2022-10-18T15:19:24.631107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submit.to_csv('./submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-10-18T15:19:33.881279Z","iopub.execute_input":"2022-10-18T15:19:33.881655Z","iopub.status.idle":"2022-10-18T15:19:36.418782Z","shell.execute_reply.started":"2022-10-18T15:19:33.881626Z","shell.execute_reply":"2022-10-18T15:19:36.416099Z"},"trusted":true},"execution_count":null,"outputs":[]}]}