{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-03T22:17:47.133705Z","iopub.execute_input":"2022-10-03T22:17:47.134963Z","iopub.status.idle":"2022-10-03T22:17:47.145142Z","shell.execute_reply.started":"2022-10-03T22:17:47.134920Z","shell.execute_reply":"2022-10-03T22:17:47.143985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_paths = {i: f'/kaggle/input/tpsoct22-feather-files/train_{i}.feather' for i in range(10)}\ntest_path = '/kaggle/input/tpsoct22-feather-files/test.feather'\nsubmission_path = '/kaggle/input/tabular-playground-series-oct-2022/sample_submission.csv'","metadata":{"execution":{"iopub.status.busy":"2022-10-03T22:17:47.147037Z","iopub.execute_input":"2022-10-03T22:17:47.147913Z","iopub.status.idle":"2022-10-03T22:17:47.166737Z","shell.execute_reply.started":"2022-10-03T22:17:47.147847Z","shell.execute_reply":"2022-10-03T22:17:47.165390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import feather\nimport numpy as np\nimport pandas as pd\nimport os\nimport catboost as cb","metadata":{"execution":{"iopub.status.busy":"2022-10-03T22:17:47.168636Z","iopub.execute_input":"2022-10-03T22:17:47.169401Z","iopub.status.idle":"2022-10-03T22:17:47.179151Z","shell.execute_reply.started":"2022-10-03T22:17:47.169355Z","shell.execute_reply":"2022-10-03T22:17:47.178104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"player_y_cols = [f'p{i}_pos_y' for i in range(6)]\nplayer_x_cols = [f'p{i}_pos_x' for i in range(6)]\nto_keep = ['ball_pos_y', 'ball_vel_y'] + player_y_cols + player_x_cols\ntarget_cols = ['team_A_scoring_within_10sec', 'team_B_scoring_within_10sec']\ntrain_only_cols = ['game_num', 'event_id', 'event_time', 'player_scoring_next', 'team_scoring_next']","metadata":{"execution":{"iopub.status.busy":"2022-10-03T22:17:47.181027Z","iopub.execute_input":"2022-10-03T22:17:47.181753Z","iopub.status.idle":"2022-10-03T22:17:47.195502Z","shell.execute_reply.started":"2022-10-03T22:17:47.181716Z","shell.execute_reply":"2022-10-03T22:17:47.194388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = {\n    'learning_rate': 35e-2, \n    'depth': 6, \n    'l2_leaf_reg': 3, \n    'loss_function': 'MultiLogloss', \n    'eval_metric': 'MultiLogloss', \n    'task_type': 'CPU', \n    'iterations': 200,\n    'od_type': 'Iter', \n    'boosting_type': 'Plain', \n    'bootstrap_type': 'Bernoulli', \n    'allow_const_label': True, \n}","metadata":{"execution":{"iopub.status.busy":"2022-10-03T22:17:47.197746Z","iopub.execute_input":"2022-10-03T22:17:47.198116Z","iopub.status.idle":"2022-10-03T22:17:47.211426Z","shell.execute_reply.started":"2022-10-03T22:17:47.198084Z","shell.execute_reply":"2022-10-03T22:17:47.209960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classifiers = {}","metadata":{"execution":{"iopub.status.busy":"2022-10-03T22:17:47.213454Z","iopub.execute_input":"2022-10-03T22:17:47.214312Z","iopub.status.idle":"2022-10-03T22:17:47.227606Z","shell.execute_reply.started":"2022-10-03T22:17:47.214262Z","shell.execute_reply":"2022-10-03T22:17:47.226279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CatBoostClassifierWrapper(cb.CatBoostClassifier):\n    #def __init__(self, X_val, y_val, *args, **kwargs):\n        #self.X_val = X_val\n        #self.y_val = y_val\n        #super(CatBoostClassifierWrapper, self).__init__(*args, **kwargs)\n\n    def fit(self, X_train, y_train=None, **fit_params):\n        #print(X_train[:3])        \n        #print(X_train.shape, y_train.shape)\n\n        return super().fit(\n            X_train, y_train,\n            #eval_set=(self.X_val, self.y_val),\n            verbose=True, **fit_params)\n    \n    def predict_proba(self, y):\n        return super().predict_proba(y)","metadata":{"execution":{"iopub.status.busy":"2022-10-03T22:17:47.229335Z","iopub.execute_input":"2022-10-03T22:17:47.230216Z","iopub.status.idle":"2022-10-03T22:17:47.244386Z","shell.execute_reply.started":"2022-10-03T22:17:47.230091Z","shell.execute_reply":"2022-10-03T22:17:47.242776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.decomposition import PCA\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\n\nfor index, train_path in train_paths.items():\n    print(f'Processing train slice: {index}')\n    train = pd.read_feather(train_path)\n    \n    train_len = len(train)\n    #print(f'{train.isna().any(axis=1).sum() / train_len:.2f}% missing values')\n    #train = train.dropna()\n\n    X = train.drop(target_cols + train_only_cols, axis=1)\n    \n    #X.team_scoring_next = X.team_scoring_next.map(dict(A=0.0, B=1.0))\n    y = train[target_cols]\n\n    X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.1, random_state=137)\n    #params['X_val'] = X_val\n    #params['y_val'] = y_val \n    \n    clf = Pipeline([\n        ('imputer', SimpleImputer(missing_values=np.nan, strategy='mean')),\n        ('scaler', StandardScaler()),\n        ('pca', PCA(n_components=8)),\n        ('classifier', CatBoostClassifierWrapper(**params))\n    ])\n\n    clf.fit(X_train, y_train)\n    classifiers[index] = clf","metadata":{"execution":{"iopub.status.busy":"2022-10-03T22:18:14.981419Z","iopub.execute_input":"2022-10-03T22:18:14.982057Z","iopub.status.idle":"2022-10-03T22:20:29.921914Z","shell.execute_reply.started":"2022-10-03T22:18:14.982022Z","shell.execute_reply":"2022-10-03T22:20:29.919732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classifiers","metadata":{"execution":{"iopub.status.busy":"2022-10-03T21:51:09.650966Z","iopub.execute_input":"2022-10-03T21:51:09.651316Z","iopub.status.idle":"2022-10-03T21:51:09.660586Z","shell.execute_reply.started":"2022-10-03T21:51:09.651285Z","shell.execute_reply":"2022-10-03T21:51:09.659407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_feather(test_path)","metadata":{"execution":{"iopub.status.busy":"2022-10-03T21:51:09.662393Z","iopub.execute_input":"2022-10-03T21:51:09.663477Z","iopub.status.idle":"2022-10-03T21:51:11.713871Z","shell.execute_reply.started":"2022-10-03T21:51:09.663442Z","shell.execute_reply":"2022-10-03T21:51:11.712954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids = test['id']\ntest.drop('id', inplace=True, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-10-03T21:51:11.717272Z","iopub.execute_input":"2022-10-03T21:51:11.718152Z","iopub.status.idle":"2022-10-03T21:51:11.823625Z","shell.execute_reply.started":"2022-10-03T21:51:11.718105Z","shell.execute_reply":"2022-10-03T21:51:11.822711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-03T21:51:11.825576Z","iopub.execute_input":"2022-10-03T21:51:11.826445Z","iopub.status.idle":"2022-10-03T21:51:11.839376Z","shell.execute_reply.started":"2022-10-03T21:51:11.826399Z","shell.execute_reply":"2022-10-03T21:51:11.838133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = [clf.predict_proba(test) for clf in classifiers.values()]","metadata":{"execution":{"iopub.status.busy":"2022-10-03T21:54:22.443357Z","iopub.execute_input":"2022-10-03T21:54:22.444146Z","iopub.status.idle":"2022-10-03T21:54:24.958916Z","shell.execute_reply.started":"2022-10-03T21:54:22.444101Z","shell.execute_reply":"2022-10-03T21:54:24.957724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = np.sum(np.array(preds), axis=0) / len(classifiers)","metadata":{"execution":{"iopub.status.busy":"2022-10-03T21:54:24.960915Z","iopub.execute_input":"2022-10-03T21:54:24.961291Z","iopub.status.idle":"2022-10-03T21:54:24.978803Z","shell.execute_reply.started":"2022-10-03T21:54:24.961259Z","shell.execute_reply":"2022-10-03T21:54:24.977757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv(submission_path)","metadata":{"execution":{"iopub.status.busy":"2022-10-03T21:54:24.980740Z","iopub.execute_input":"2022-10-03T21:54:24.981179Z","iopub.status.idle":"2022-10-03T21:54:25.180542Z","shell.execute_reply.started":"2022-10-03T21:54:24.981131Z","shell.execute_reply":"2022-10-03T21:54:25.179236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission[['team_A_scoring_within_10sec', 'team_B_scoring_within_10sec']] = preds\nsubmission['id'] = ids","metadata":{"execution":{"iopub.status.busy":"2022-10-03T21:54:25.183493Z","iopub.execute_input":"2022-10-03T21:54:25.184125Z","iopub.status.idle":"2022-10-03T21:54:25.201126Z","shell.execute_reply.started":"2022-10-03T21:54:25.184088Z","shell.execute_reply":"2022-10-03T21:54:25.199967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-03T21:54:25.202671Z","iopub.execute_input":"2022-10-03T21:54:25.203360Z","iopub.status.idle":"2022-10-03T21:54:25.214723Z","shell.execute_reply.started":"2022-10-03T21:54:25.203323Z","shell.execute_reply":"2022-10-03T21:54:25.213345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-10-03T21:54:25.216566Z","iopub.execute_input":"2022-10-03T21:54:25.216970Z","iopub.status.idle":"2022-10-03T21:54:27.997244Z","shell.execute_reply.started":"2022-10-03T21:54:25.216938Z","shell.execute_reply":"2022-10-03T21:54:27.995638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head -10 submission.csv","metadata":{"execution":{"iopub.status.busy":"2022-10-03T21:54:27.999797Z","iopub.execute_input":"2022-10-03T21:54:28.001214Z","iopub.status.idle":"2022-10-03T21:54:29.222828Z","shell.execute_reply.started":"2022-10-03T21:54:28.001162Z","shell.execute_reply":"2022-10-03T21:54:29.221474Z"},"trusted":true},"execution_count":null,"outputs":[]}]}