{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport random\nimport pickle\n\nfrom tqdm import tqdm\n\nimport lightgbm as lgb\nimport xgboost as xgb\nfrom catboost import CatBoostRegressor, Pool\nfrom sklearn.ensemble import RandomForestRegressor\n\nimport optuna\nfrom optuna.samplers import TPESampler\n\nfrom sklearn.model_selection import train_test_split, KFold, GroupKFold, StratifiedKFold\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import mean_squared_error\n\nfrom sklearn.preprocessing import LabelEncoder\n\nimport gc\ngc.enable()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-01T04:54:42.298368Z","iopub.execute_input":"2022-10-01T04:54:42.298918Z","iopub.status.idle":"2022-10-01T04:54:44.143734Z","shell.execute_reply.started":"2022-10-01T04:54:42.298827Z","shell.execute_reply":"2022-10-01T04:54:44.142646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"r_seed = 2022\n\ndef fix_seed(seed):\n    random.seed(seed)\n    np.random.seed(seed)\n    \nfix_seed(r_seed)","metadata":{"execution":{"iopub.status.busy":"2022-10-01T04:54:44.144977Z","iopub.execute_input":"2022-10-01T04:54:44.145323Z","iopub.status.idle":"2022-10-01T04:54:44.151377Z","shell.execute_reply.started":"2022-10-01T04:54:44.145280Z","shell.execute_reply":"2022-10-01T04:54:44.150188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INPUT = '../input/tabular-playground-series-oct-2022/'\n\ndf_train_dtypes = pd.read_csv(INPUT + 'train_dtypes.csv')\ndf_test_dtypes = pd.read_csv(INPUT + 'test_dtypes.csv')\ntrain_dtypes = {k: v for (k, v) in zip(df_train_dtypes.column, df_train_dtypes.dtype)}\ntest_dtypes = {k: v for (k, v) in zip(df_test_dtypes.column, df_test_dtypes.dtype)}\n\ntrain_list = []\nfor i in tqdm(range(1)):\n    train_list.append(pd.read_csv(INPUT + f'train_{i}.csv', dtype = train_dtypes))\n\ndf_test = pd.read_csv(INPUT + 'test.csv', dtype = test_dtypes)\ndf_sample = pd.read_csv(INPUT + 'sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-10-01T04:54:44.152942Z","iopub.execute_input":"2022-10-01T04:54:44.153838Z","iopub.status.idle":"2022-10-01T04:55:09.845256Z","shell.execute_reply.started":"2022-10-01T04:54:44.153792Z","shell.execute_reply":"2022-10-01T04:55:09.843556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_list[0].head()","metadata":{"execution":{"iopub.status.busy":"2022-10-01T04:55:09.848533Z","iopub.execute_input":"2022-10-01T04:55:09.848922Z","iopub.status.idle":"2022-10-01T04:55:09.883609Z","shell.execute_reply.started":"2022-10-01T04:55:09.848887Z","shell.execute_reply":"2022-10-01T04:55:09.882450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_list[0].shape)\nprint(df_test.shape)\nprint(df_sample.shape)","metadata":{"execution":{"iopub.status.busy":"2022-10-01T04:55:09.884886Z","iopub.execute_input":"2022-10-01T04:55:09.885493Z","iopub.status.idle":"2022-10-01T04:55:09.892363Z","shell.execute_reply.started":"2022-10-01T04:55:09.885457Z","shell.execute_reply":"2022-10-01T04:55:09.890114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target = train_list[0]['team_A_scoring_within_10sec']\ntrain = train_list[0].drop(['game_num', 'event_id', 'event_time', 'player_scoring_next', 'team_scoring_next', 'team_A_scoring_within_10sec', 'team_B_scoring_within_10sec'], axis = 1)\ntest = df_test.drop(['id'], axis = 1)\n\nX_train, X_valid, y_train, y_valid = train_test_split(train, target, test_size = 0.2, random_state = r_seed, shuffle=True)\n\nparams = {'objective': 'binary',\n          'seed': r_seed\n         }\ntrain_data = lgb.Dataset(X_train, label = y_train)\nvalid_data = lgb.Dataset(X_valid, label = y_valid, reference = train_data)\n\nmodel = lgb.train({**params},\n                  train_data, \n                  1000,\n                  valid_sets = [train_data, valid_data],\n                  early_stopping_rounds=10,\n                  verbose_eval=1000)\n\npreds = model.predict(test)\n\ndf_sample['team_A_scoring_within_10sec'] = preds\ndf_sample['team_B_scoring_within_10sec'] = 1 - preds\ndf_sample.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-10-01T04:55:09.894419Z","iopub.execute_input":"2022-10-01T04:55:09.894865Z","iopub.status.idle":"2022-10-01T04:59:33.244660Z","shell.execute_reply.started":"2022-10-01T04:55:09.894819Z","shell.execute_reply":"2022-10-01T04:59:33.243559Z"},"trusted":true},"execution_count":null,"outputs":[]}]}