{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-03T18:55:12.432762Z","iopub.execute_input":"2022-11-03T18:55:12.433624Z","iopub.status.idle":"2022-11-03T18:55:12.474690Z","shell.execute_reply.started":"2022-11-03T18:55:12.433520Z","shell.execute_reply":"2022-11-03T18:55:12.472455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import tqdm\n\n# for i in tqdm.tqdm(range(10)):\n#     df = pd.read_csv(f'/kaggle/input/tabular-playground-series-oct-2022/train_{i}.csv').to_parquet(\n#         f'train_{i}.parquet.gzip', compression='gzip')\n\n# 100%|██████████| 10/10 [13:33<00:00, 81.38s/it]","metadata":{"execution":{"iopub.status.busy":"2022-11-03T18:55:13.184855Z","iopub.execute_input":"2022-11-03T18:55:13.185337Z","iopub.status.idle":"2022-11-03T18:55:13.191001Z","shell.execute_reply.started":"2022-11-03T18:55:13.185308Z","shell.execute_reply":"2022-11-03T18:55:13.189621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# read dtypes\ntrain_dtypes = pd.read_csv(\"/kaggle/input/tabular-playground-series-oct-2022/train_dtypes.csv\")\ntest_dtypes = pd.read_csv(\"/kaggle/input/tabular-playground-series-oct-2022/test_dtypes.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-11-03T18:55:13.763058Z","iopub.execute_input":"2022-11-03T18:55:13.763534Z","iopub.status.idle":"2022-11-03T18:55:13.791772Z","shell.execute_reply.started":"2022-11-03T18:55:13.763504Z","shell.execute_reply":"2022-11-03T18:55:13.789647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save dtypes in dict\ncols_dtypes = {k: v for (k, v) in zip(train_dtypes.column, train_dtypes.dtype)}","metadata":{"execution":{"iopub.status.busy":"2022-11-03T18:55:14.388918Z","iopub.execute_input":"2022-11-03T18:55:14.389420Z","iopub.status.idle":"2022-11-03T18:55:14.403930Z","shell.execute_reply.started":"2022-11-03T18:55:14.389389Z","shell.execute_reply":"2022-11-03T18:55:14.402150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# columns only in trainig\n[c for c in train_dtypes['column'] if c not in test_dtypes['column'].values]","metadata":{"execution":{"iopub.status.busy":"2022-11-03T18:55:14.862794Z","iopub.execute_input":"2022-11-03T18:55:14.863867Z","iopub.status.idle":"2022-11-03T18:55:14.875457Z","shell.execute_reply.started":"2022-11-03T18:55:14.863828Z","shell.execute_reply":"2022-11-03T18:55:14.874343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# drop colums that doesn't exist in test_df.  (id == event_id)\nuseless_cols = ['game_num', 'event_id', 'event_time', 'player_scoring_next', 'team_scoring_next']\nuse_cols = [c for c in train_dtypes['column'] if c not in useless_cols]","metadata":{"execution":{"iopub.status.busy":"2022-11-03T18:55:15.534087Z","iopub.execute_input":"2022-11-03T18:55:15.534894Z","iopub.status.idle":"2022-11-03T18:55:15.546971Z","shell.execute_reply.started":"2022-11-03T18:55:15.534845Z","shell.execute_reply":"2022-11-03T18:55:15.544676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Garbage Collection \nimport gc\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-11-03T18:55:15.970467Z","iopub.execute_input":"2022-11-03T18:55:15.971323Z","iopub.status.idle":"2022-11-03T18:55:16.159937Z","shell.execute_reply.started":"2022-11-03T18:55:15.971288Z","shell.execute_reply":"2022-11-03T18:55:16.158057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"`Feather` 🪶 files are 100 times faster, while reading and writing from the disk, as compared to `CSV` files.","metadata":{}},{"cell_type":"code","source":"import tqdm\n\ntrain_dfs = []\nfor i in tqdm.tqdm(range(10)):\n    train_dfs.append(\n        pd.read_csv(f'/kaggle/input/tabular-playground-series-oct-2022/train_{i}.csv',\n                    usecols=use_cols, dtype=cols_dtypes)\n    )\n\npd.concat(train_dfs).reset_index(drop=True).to_feather('train.feather')","metadata":{"execution":{"iopub.status.busy":"2022-11-03T18:55:16.848324Z","iopub.execute_input":"2022-11-03T18:55:16.848840Z","iopub.status.idle":"2022-11-03T19:00:01.311474Z","shell.execute_reply.started":"2022-11-03T18:55:16.848810Z","shell.execute_reply":"2022-11-03T19:00:01.304363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- `Parquet` took 13 minutes\n- `Feather` 🪶 took only 5 minutes, We have a winner 🕺","metadata":{}},{"cell_type":"code","source":"# read test dataframe\ndf = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/test.csv', dtype=cols_dtypes)\ndf.to_feather('test.feather')","metadata":{"execution":{"iopub.status.busy":"2022-11-03T19:00:01.323233Z","iopub.execute_input":"2022-11-03T19:00:01.325850Z","iopub.status.idle":"2022-11-03T19:00:09.798156Z","shell.execute_reply.started":"2022-11-03T19:00:01.325759Z","shell.execute_reply":"2022-11-03T19:00:09.795470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# free space\ndel (df, train_dfs, train_dtypes, cols_dtypes, useless_cols, use_cols)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-11-03T19:00:09.801727Z","iopub.execute_input":"2022-11-03T19:00:09.802291Z","iopub.status.idle":"2022-11-03T19:00:10.091739Z","shell.execute_reply.started":"2022-11-03T19:00:09.802254Z","shell.execute_reply":"2022-11-03T19:00:10.090572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# read all the 10 dfs\ntrain_df = pd.read_feather('train.feather')\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-03T19:00:10.095269Z","iopub.execute_input":"2022-11-03T19:00:10.096013Z","iopub.status.idle":"2022-11-03T19:00:14.747170Z","shell.execute_reply.started":"2022-11-03T19:00:10.095951Z","shell.execute_reply":"2022-11-03T19:00:14.745603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 21 MILLION rows\ntrain_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-11-03T19:00:14.749125Z","iopub.execute_input":"2022-11-03T19:00:14.749531Z","iopub.status.idle":"2022-11-03T19:00:14.759745Z","shell.execute_reply.started":"2022-11-03T19:00:14.749502Z","shell.execute_reply":"2022-11-03T19:00:14.757918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check Null values\ntrain_df.isna().sum()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-11-03T19:00:14.761307Z","iopub.execute_input":"2022-11-03T19:00:14.761718Z","iopub.status.idle":"2022-11-03T19:00:17.351025Z","shell.execute_reply.started":"2022-11-03T19:00:14.761685Z","shell.execute_reply":"2022-11-03T19:00:17.349627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Pre-processing","metadata":{}},{"cell_type":"code","source":"# impute Null values\ntrain_df.fillna(0, inplace=True)\ntrain_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-11-03T19:00:17.352636Z","iopub.execute_input":"2022-11-03T19:00:17.352998Z","iopub.status.idle":"2022-11-03T19:00:19.106315Z","shell.execute_reply.started":"2022-11-03T19:00:17.352968Z","shell.execute_reply":"2022-11-03T19:00:19.105281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-11-03T19:00:19.107910Z","iopub.execute_input":"2022-11-03T19:00:19.108570Z","iopub.status.idle":"2022-11-03T19:00:19.285410Z","shell.execute_reply.started":"2022-11-03T19:00:19.108529Z","shell.execute_reply":"2022-11-03T19:00:19.284398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Split the dataframe","metadata":{}},{"cell_type":"code","source":"# Create a target (A, B) dataframe & remove them from the train_df\ntarget_dt = pd.DataFrame()\ntarget_dt['team_A_scoring_within_10sec'] = train_df.pop('team_A_scoring_within_10sec')\ntarget_dt['team_B_scoring_within_10sec'] = train_df.pop('team_B_scoring_within_10sec')","metadata":{"execution":{"iopub.status.busy":"2022-11-03T19:00:19.287569Z","iopub.execute_input":"2022-11-03T19:00:19.288346Z","iopub.status.idle":"2022-11-03T19:00:20.618915Z","shell.execute_reply.started":"2022-11-03T19:00:19.288311Z","shell.execute_reply":"2022-11-03T19:00:20.617686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Validation set size\n20089957 * 0.005","metadata":{"execution":{"iopub.status.busy":"2022-11-03T19:00:20.622311Z","iopub.execute_input":"2022-11-03T19:00:20.622717Z","iopub.status.idle":"2022-11-03T19:00:20.633450Z","shell.execute_reply.started":"2022-11-03T19:00:20.622679Z","shell.execute_reply":"2022-11-03T19:00:20.631338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_X, val_X, train_y, val_y = train_test_split(train_df, target_dt,\n                                                  test_size=0.005, random_state=10)","metadata":{"execution":{"iopub.status.busy":"2022-11-03T19:00:20.635113Z","iopub.execute_input":"2022-11-03T19:00:20.635553Z","iopub.status.idle":"2022-11-03T19:00:47.120796Z","shell.execute_reply.started":"2022-11-03T19:00:20.635515Z","shell.execute_reply":"2022-11-03T19:00:47.119033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# free space\ndel train_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-11-03T19:00:47.122522Z","iopub.execute_input":"2022-11-03T19:00:47.122961Z","iopub.status.idle":"2022-11-03T19:00:47.326450Z","shell.execute_reply.started":"2022-11-03T19:00:47.122920Z","shell.execute_reply":"2022-11-03T19:00:47.324977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print shapes\ntrain_X.shape, val_X.shape, train_y.shape, val_y.shape","metadata":{"execution":{"iopub.status.busy":"2022-11-03T19:00:47.328653Z","iopub.execute_input":"2022-11-03T19:00:47.329232Z","iopub.status.idle":"2022-11-03T19:00:47.342014Z","shell.execute_reply.started":"2022-11-03T19:00:47.329163Z","shell.execute_reply":"2022-11-03T19:00:47.340200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train a Model","metadata":{}},{"cell_type":"code","source":"from lightgbm import LGBMClassifier\n\nmodel_A = LGBMClassifier(objective=\"binary\", early_stopping_round=100)\nmodel_A.fit(train_X, train_y.iloc[:, 0],\n            eval_set=[(val_X, val_y.iloc[:, 0])],\n            eval_metric=\"binary_logloss\"\n           )\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-11-03T16:22:35.674331Z","iopub.execute_input":"2022-11-03T16:22:35.674726Z","iopub.status.idle":"2022-11-03T16:33:11.315565Z","shell.execute_reply.started":"2022-11-03T16:22:35.674684Z","shell.execute_reply":"2022-11-03T16:33:11.314310Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_B = LGBMClassifier(objective=\"binary\", early_stopping_round=100)\nmodel_B.fit(train_X, train_y.iloc[:, 1],\n            eval_set=[(val_X, val_y.iloc[:, 1])],\n            eval_metric=\"binary_logloss\"\n           )","metadata":{"execution":{"iopub.status.busy":"2022-11-03T16:33:11.317281Z","iopub.execute_input":"2022-11-03T16:33:11.317672Z","iopub.status.idle":"2022-11-03T16:43:35.939656Z","shell.execute_reply.started":"2022-11-03T16:33:11.317630Z","shell.execute_reply":"2022-11-03T16:43:35.938251Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Evaluate the model","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import log_loss\n\nlog_loss(val_y.iloc[:, 0], model_A.predict_proba(val_X))","metadata":{"execution":{"iopub.status.busy":"2022-11-03T16:52:59.610356Z","iopub.execute_input":"2022-11-03T16:52:59.610976Z","iopub.status.idle":"2022-11-03T16:52:59.969360Z","shell.execute_reply.started":"2022-11-03T16:52:59.610936Z","shell.execute_reply":"2022-11-03T16:52:59.967928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_loss(val_y.iloc[:, 1], model_B.predict_proba(val_X))","metadata":{"execution":{"iopub.status.busy":"2022-11-03T16:53:00.995195Z","iopub.execute_input":"2022-11-03T16:53:00.995654Z","iopub.status.idle":"2022-11-03T16:53:01.348599Z","shell.execute_reply.started":"2022-11-03T16:53:00.995617Z","shell.execute_reply":"2022-11-03T16:53:01.347528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-11-03T16:53:06.282530Z","iopub.execute_input":"2022-11-03T16:53:06.283049Z","iopub.status.idle":"2022-11-03T16:53:06.525716Z","shell.execute_reply.started":"2022-11-03T16:53:06.283012Z","shell.execute_reply":"2022-11-03T16:53:06.524080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submit submission","metadata":{}},{"cell_type":"code","source":"# read the feather test dataset\ntest_df = pd.read_feather('test.feather')\n\n# drop ID column\ntest_df.drop(columns=['id'], inplace=True)\n\n# impute Null values\ntest_df.fillna(0, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-11-03T16:53:12.336692Z","iopub.execute_input":"2022-11-03T16:53:12.337113Z","iopub.status.idle":"2022-11-03T16:53:12.868618Z","shell.execute_reply.started":"2022-11-03T16:53:12.337081Z","shell.execute_reply":"2022-11-03T16:53:12.867362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_a_pred = model_A.predict_proba(test_df)\ny_b_pred = model_B.predict_proba(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-11-03T16:53:14.696020Z","iopub.execute_input":"2022-11-03T16:53:14.696457Z","iopub.status.idle":"2022-11-03T16:53:19.050477Z","shell.execute_reply.started":"2022-11-03T16:53:14.696420Z","shell.execute_reply":"2022-11-03T16:53:19.049404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('../input/tabular-playground-series-oct-2022/sample_submission.csv', usecols = ['id'])\nsubmission['team_A_scoring_within_10sec'] = y_a_pred[:,1]\nsubmission['team_B_scoring_within_10sec'] = y_b_pred[:,1]\nsubmission.to_csv('submission.csv', index = False)\n\nprint('Submission saved')\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-03T16:53:19.053090Z","iopub.execute_input":"2022-11-03T16:53:19.054177Z","iopub.status.idle":"2022-11-03T16:53:22.091919Z","shell.execute_reply.started":"2022-11-03T16:53:19.054133Z","shell.execute_reply":"2022-11-03T16:53:22.090755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Resources\n- [TPS-OCT-2022 - Simple TF](https://www.kaggle.com/code/eavelardev/tps-oct-2022-simple-tf/notebook#Data-source)\n- [🚀How to load 21M rows in 1 minute using 2 lines](https://www.kaggle.com/code/donatoriccio/how-to-load-21m-rows-in-1-minute-using-2-lines/notebook)\n- [LightGBM Classifier in Python](https://www.kaggle.com/code/prashant111/lightgbm-classifier-in-python)","metadata":{}}]}