{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### In this notebook I will try to explain you what's going on with the competition metric (LogLoss) and what you can/can't do with it to receive good score\n\n### Preparations (we will use only the `train_0.csv` train file to speedup the experiments)","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport lightgbm\nfrom sklearn.metrics import log_loss\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-23T20:35:33.298173Z","iopub.execute_input":"2022-10-23T20:35:33.299524Z","iopub.status.idle":"2022-10-23T20:35:35.444357Z","shell.execute_reply.started":"2022-10-23T20:35:33.299369Z","shell.execute_reply":"2022-10-23T20:35:35.443233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INPUT_DIR = '../input/tabular-playground-series-oct-2022/'","metadata":{"execution":{"iopub.status.busy":"2022-10-23T20:35:37.746805Z","iopub.execute_input":"2022-10-23T20:35:37.747307Z","iopub.status.idle":"2022-10-23T20:35:37.751879Z","shell.execute_reply.started":"2022-10-23T20:35:37.747268Z","shell.execute_reply":"2022-10-23T20:35:37.750749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dtypes_df = pd.read_csv(f'{INPUT_DIR}/train_dtypes.csv')\ntrain_dtypes = {k: v for (k, v) in zip(train_dtypes_df.column, train_dtypes_df.dtype)}","metadata":{"execution":{"iopub.status.busy":"2022-10-23T20:35:40.894151Z","iopub.execute_input":"2022-10-23T20:35:40.894606Z","iopub.status.idle":"2022-10-23T20:35:40.917868Z","shell.execute_reply.started":"2022-10-23T20:35:40.894574Z","shell.execute_reply":"2022-10-23T20:35:40.916704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv(f'{INPUT_DIR}/train_0.csv', dtype=train_dtypes)\ntrain_data.shape","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-10-23T20:35:45.372965Z","iopub.execute_input":"2022-10-23T20:35:45.373404Z","iopub.status.idle":"2022-10-23T20:36:16.961676Z","shell.execute_reply.started":"2022-10-23T20:35:45.373366Z","shell.execute_reply":"2022-10-23T20:36:16.960420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### All the work further we will do on the `team_A_scoring_within_10sec` target (but it will also work for the second target as well)","metadata":{}},{"cell_type":"code","source":"train_data['team_A_scoring_within_10sec'].mean()","metadata":{"execution":{"iopub.status.busy":"2022-10-23T20:38:45.023798Z","iopub.execute_input":"2022-10-23T20:38:45.024256Z","iopub.status.idle":"2022-10-23T20:38:45.033680Z","shell.execute_reply.started":"2022-10-23T20:38:45.024220Z","shell.execute_reply":"2022-10-23T20:38:45.032733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### So the mean (the prior probability to be equal to 1) for this target is ~0.0583 and if we want to use the constant prediction - this value will be the best one. Let me show you what I'm talking about:","metadata":{}},{"cell_type":"code","source":"y = train_data['team_A_scoring_within_10sec'].values\npreds_1 = np.zeros(len(y)) + np.mean(y)\npreds_2 = np.zeros(len(y)) + 0.05\npreds_3 = np.zeros(len(y)) + 0.06\npreds_4 = np.zeros(len(y)) + 0.1\npreds_5 = np.zeros(len(y)) + 0.01\nprint('LogLoss for mean prediction: {:.5f}'.format(log_loss(y, preds_1)))\nprint('LogLoss for 0.05 prediction: {:.5f}'.format(log_loss(y, preds_2)))\nprint('LogLoss for 0.06 prediction: {:.5f}'.format(log_loss(y, preds_3)))\nprint('LogLoss for 0.1  prediction: {:.5f}'.format(log_loss(y, preds_4)))\nprint('LogLoss for 0.01 prediction: {:.5f}'.format(log_loss(y, preds_5)))","metadata":{"execution":{"iopub.status.busy":"2022-10-23T20:45:06.522796Z","iopub.execute_input":"2022-10-23T20:45:06.525722Z","iopub.status.idle":"2022-10-23T20:45:09.123059Z","shell.execute_reply.started":"2022-10-23T20:45:06.525616Z","shell.execute_reply":"2022-10-23T20:45:09.121759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### As you can see, the mean prediction is really the best one and equal to `0.22229`. So we need to have the predictions near (or exactly equal) our train mean to receive good scores. Now it's time to create our train and validation sets to train the models, but we need them with the equal target mean and non-intersecting `game_num` (according to [this post](https://www.kaggle.com/competitions/tabular-playground-series-oct-2022/discussion/359714)):","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedGroupKFold\ngroups = train_data['game_num'].values\ncv = StratifiedGroupKFold(n_splits=3)\nfor train_ids, test_ids in cv.split(y, y, groups):\n    break\ntr_data = train_data.iloc[train_ids, :]\nte_data = train_data.iloc[test_ids, :]\ntr_data.shape, te_data.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-23T20:55:12.269933Z","iopub.execute_input":"2022-10-23T20:55:12.270432Z","iopub.status.idle":"2022-10-23T20:55:15.334237Z","shell.execute_reply.started":"2022-10-23T20:55:12.270392Z","shell.execute_reply":"2022-10-23T20:55:15.333361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let's check how the data is splitted:","metadata":{}},{"cell_type":"code","source":"tr_data['team_A_scoring_within_10sec'].mean(), te_data['team_A_scoring_within_10sec'].mean()","metadata":{"execution":{"iopub.status.busy":"2022-10-23T20:56:17.813597Z","iopub.execute_input":"2022-10-23T20:56:17.814063Z","iopub.status.idle":"2022-10-23T20:56:17.825196Z","shell.execute_reply.started":"2022-10-23T20:56:17.814026Z","shell.execute_reply":"2022-10-23T20:56:17.823824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set(tr_data['game_num'].values).intersection(set(te_data['game_num'].values))","metadata":{"execution":{"iopub.status.busy":"2022-10-23T20:57:31.220245Z","iopub.execute_input":"2022-10-23T20:57:31.220710Z","iopub.status.idle":"2022-10-23T20:57:31.377190Z","shell.execute_reply.started":"2022-10-23T20:57:31.220675Z","shell.execute_reply":"2022-10-23T20:57:31.375770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### The target mean between train and test is almost equal and the `game_nums` are non-intersected. Let's split the train data into train and validation parts:","metadata":{}},{"cell_type":"code","source":"y = tr_data['team_A_scoring_within_10sec'].values\ngroups = tr_data['game_num'].values\ncv = StratifiedGroupKFold(n_splits=5)\nfor train_ids, valid_ids in cv.split(y, y, groups):\n    break\nva_data = tr_data.iloc[valid_ids, :]\ntr_data = tr_data.iloc[train_ids, :]\n\ntr_data.shape, va_data.shape, te_data.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-23T21:02:55.392923Z","iopub.execute_input":"2022-10-23T21:02:55.393672Z","iopub.status.idle":"2022-10-23T21:02:57.354414Z","shell.execute_reply.started":"2022-10-23T21:02:55.393623Z","shell.execute_reply":"2022-10-23T21:02:57.351706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(tr_data['team_A_scoring_within_10sec'].mean(), \n      va_data['team_A_scoring_within_10sec'].mean(), \n      te_data['team_A_scoring_within_10sec'].mean())","metadata":{"execution":{"iopub.status.busy":"2022-10-23T21:03:35.173730Z","iopub.execute_input":"2022-10-23T21:03:35.174246Z","iopub.status.idle":"2022-10-23T21:03:35.185403Z","shell.execute_reply.started":"2022-10-23T21:03:35.174172Z","shell.execute_reply":"2022-10-23T21:03:35.183853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Mean targets are a little bit different but still almost the same. Now we are ready to train the models:","metadata":{}},{"cell_type":"code","source":"y_train = tr_data['team_A_scoring_within_10sec'].values\ny_valid = va_data['team_A_scoring_within_10sec'].values\ny_test  = te_data['team_A_scoring_within_10sec'].values\nprint(y_train.shape, y_valid.shape, y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T21:08:44.350092Z","iopub.execute_input":"2022-10-23T21:08:44.350551Z","iopub.status.idle":"2022-10-23T21:08:44.357758Z","shell.execute_reply.started":"2022-10-23T21:08:44.350513Z","shell.execute_reply":"2022-10-23T21:08:44.356554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_to_drop = ['game_num', 'event_id', 'event_time', 'player_scoring_next', 'team_scoring_next', \n                'team_A_scoring_within_10sec','team_B_scoring_within_10sec']\ntr_data.drop(cols_to_drop, axis = 1, inplace = True)\nva_data.drop(cols_to_drop, axis = 1, inplace = True)\nte_data.drop(cols_to_drop, axis = 1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T21:12:58.795813Z","iopub.execute_input":"2022-10-23T21:12:58.796292Z","iopub.status.idle":"2022-10-23T21:12:59.030875Z","shell.execute_reply.started":"2022-10-23T21:12:58.796247Z","shell.execute_reply":"2022-10-23T21:12:59.029425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_dist_cols(df):\n    for i in range(6):\n        df[f'dist_p_ball_{i}'] = np.sqrt((df.ball_pos_x - df[f'p{i}_pos_x'])**2 + (df.ball_pos_y - df[f'p{i}_pos_y'])**2 + (df.ball_pos_z - df[f'p{i}_pos_z'])**2)\n    return df\n\ntr_data = add_dist_cols(tr_data)\nva_data = add_dist_cols(va_data)\nte_data = add_dist_cols(te_data)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T21:14:33.460850Z","iopub.execute_input":"2022-10-23T21:14:33.461389Z","iopub.status.idle":"2022-10-23T21:14:33.604108Z","shell.execute_reply.started":"2022-10-23T21:14:33.461345Z","shell.execute_reply":"2022-10-23T21:14:33.602047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-23T21:14:44.599960Z","iopub.execute_input":"2022-10-23T21:14:44.600424Z","iopub.status.idle":"2022-10-23T21:14:44.638080Z","shell.execute_reply.started":"2022-10-23T21:14:44.600385Z","shell.execute_reply":"2022-10-23T21:14:44.637196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MODEL_PARAMS = {    \n    'objective': 'binary',\n    'seed': 42,\n    'num_leaves': 120,\n    'n_estimators': 500,\n    'max_depth': 8,\n    'learning_rate': 0.01,\n    'feature_fraction': 0.75,\n    'subsample': 0.7,\n    'subsample_freq': 8,\n    'n_jobs': -1,\n    'reg_alpha': 1,\n    'reg_lambda': 2,\n    'min_child_samples': 80}\n\nmodel = lightgbm.LGBMClassifier(**MODEL_PARAMS)\n\nmodel.fit(X=tr_data, y=y_train,\n          eval_set=[(va_data, y_valid)],\n          eval_names=['valid'],\n          callbacks=[lightgbm.log_evaluation(period=50, show_stdv=True),\n                     lightgbm.early_stopping(stopping_rounds=50)])","metadata":{"execution":{"iopub.status.busy":"2022-10-23T21:17:08.058015Z","iopub.execute_input":"2022-10-23T21:17:08.058564Z","iopub.status.idle":"2022-10-23T21:19:22.648109Z","shell.execute_reply.started":"2022-10-23T21:17:08.058522Z","shell.execute_reply":"2022-10-23T21:19:22.646866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_pred = model.predict_proba(va_data)[:, 1]\ntest_pred = model.predict_proba(te_data)[:, 1]","metadata":{"execution":{"iopub.status.busy":"2022-10-23T21:21:41.581558Z","iopub.execute_input":"2022-10-23T21:21:41.582653Z","iopub.status.idle":"2022-10-23T21:22:03.607478Z","shell.execute_reply.started":"2022-10-23T21:21:41.582603Z","shell.execute_reply":"2022-10-23T21:22:03.606422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('LogLoss for valid_data: {:.5f}'.format(log_loss(y_valid, valid_pred)))\nprint('LogLoss for  test_data: {:.5f}'.format(log_loss(y_test, test_pred)))","metadata":{"execution":{"iopub.status.busy":"2022-10-23T21:23:02.591503Z","iopub.execute_input":"2022-10-23T21:23:02.591980Z","iopub.status.idle":"2022-10-23T21:23:02.785517Z","shell.execute_reply.started":"2022-10-23T21:23:02.591946Z","shell.execute_reply":"2022-10-23T21:23:02.784092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### The LogLoss for this variant of LGBM model looks fine. Let's check the prediction histograms:","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (20, 10))\nplt.hist(valid_pred, bins = 100, alpha=0.5, label='Valid preds')\nplt.hist(test_pred, bins = 100, alpha=0.5, label='Test preds')\nplt.legend(loc='upper right')\nplt.grid(True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-23T21:27:38.913014Z","iopub.execute_input":"2022-10-23T21:27:38.913481Z","iopub.status.idle":"2022-10-23T21:27:39.574232Z","shell.execute_reply.started":"2022-10-23T21:27:38.913442Z","shell.execute_reply":"2022-10-23T21:27:39.573096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### So the peaks of both predictions are near the prior mean:","metadata":{}},{"cell_type":"code","source":"print('Mean prediction for valid_data: {:.5f}'.format(np.mean(valid_pred)))\nprint('Mean prediction for  test_data: {:.5f}'.format(np.mean(test_pred)))","metadata":{"execution":{"iopub.status.busy":"2022-10-23T21:30:43.361148Z","iopub.execute_input":"2022-10-23T21:30:43.361634Z","iopub.status.idle":"2022-10-23T21:30:43.368582Z","shell.execute_reply.started":"2022-10-23T21:30:43.361594Z","shell.execute_reply":"2022-10-23T21:30:43.367398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Now it's time to change the mean for the train and valid data (using sampling) to see how it will change the situation:","metadata":{}},{"cell_type":"code","source":"# Here we create new DataFrames with target mean equal to 33%\ndef create_sampled_df(data, target):\n    data['target'] = target\n    pos_part = data[data['target'] == 1]\n    neg_part = data[data['target'] == 0]\n    df = pd.concat([pos_part, \n                   neg_part.sample(len(pos_part) * 2)])\n    return df.drop('target', axis = 1), df['target'].values\n\ntr_data_sampled, y_train_sampled = create_sampled_df(tr_data, y_train)\nva_data_sampled, y_valid_sampled = create_sampled_df(va_data, y_valid)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T21:58:06.179361Z","iopub.execute_input":"2022-10-23T21:58:06.179808Z","iopub.status.idle":"2022-10-23T21:58:06.898204Z","shell.execute_reply.started":"2022-10-23T21:58:06.179772Z","shell.execute_reply":"2022-10-23T21:58:06.896819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr_data.drop('target', axis = 1, inplace = True)\nva_data.drop('target', axis = 1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T21:58:09.158251Z","iopub.execute_input":"2022-10-23T21:58:09.158720Z","iopub.status.idle":"2022-10-23T21:58:09.321548Z","shell.execute_reply.started":"2022-10-23T21:58:09.158685Z","shell.execute_reply":"2022-10-23T21:58:09.320296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(np.mean(y_train_sampled), np.mean(y_valid_sampled), np.mean(y_test))","metadata":{"execution":{"iopub.status.busy":"2022-10-23T21:58:10.293046Z","iopub.execute_input":"2022-10-23T21:58:10.293847Z","iopub.status.idle":"2022-10-23T21:58:10.301422Z","shell.execute_reply.started":"2022-10-23T21:58:10.293804Z","shell.execute_reply":"2022-10-23T21:58:10.300119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### So now we have train and valid target rate at 33% and test one is still ~6%. Training LGBM model once again:","metadata":{}},{"cell_type":"code","source":"model = lightgbm.LGBMClassifier(**MODEL_PARAMS)\n\nmodel.fit(X=tr_data_sampled, y=y_train_sampled,\n          eval_set=[(va_data_sampled, y_valid_sampled)],\n          eval_names=['valid'],\n          callbacks=[lightgbm.log_evaluation(period=50, show_stdv=True),\n                     lightgbm.early_stopping(stopping_rounds=50)])","metadata":{"execution":{"iopub.status.busy":"2022-10-23T21:58:19.385430Z","iopub.execute_input":"2022-10-23T21:58:19.385833Z","iopub.status.idle":"2022-10-23T21:58:44.093092Z","shell.execute_reply.started":"2022-10-23T21:58:19.385793Z","shell.execute_reply":"2022-10-23T21:58:44.092052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Below we are predicting for the initial validation and test datasets to check the final LogLoss scores:","metadata":{}},{"cell_type":"code","source":"valid_pred_sampled = model.predict_proba(va_data)[:, 1]\ntest_pred_sampled = model.predict_proba(te_data)[:, 1]","metadata":{"execution":{"iopub.status.busy":"2022-10-23T21:58:49.766148Z","iopub.execute_input":"2022-10-23T21:58:49.766630Z","iopub.status.idle":"2022-10-23T21:58:57.802209Z","shell.execute_reply.started":"2022-10-23T21:58:49.766594Z","shell.execute_reply":"2022-10-23T21:58:57.801170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('LogLoss for valid_data: {:.5f}'.format(log_loss(y_valid, valid_pred_sampled)))\nprint('LogLoss for  test_data: {:.5f}'.format(log_loss(y_test, test_pred_sampled)))","metadata":{"execution":{"iopub.status.busy":"2022-10-23T21:59:10.627070Z","iopub.execute_input":"2022-10-23T21:59:10.627518Z","iopub.status.idle":"2022-10-23T21:59:10.819976Z","shell.execute_reply.started":"2022-10-23T21:59:10.627478Z","shell.execute_reply":"2022-10-23T21:59:10.818563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### The scores become dramatically worse in comparison with the initial ones (and they will be even worse if we increase the sampled dataset target rate). Now we can check how the predictions look like in comparison with previous model:","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (20, 10))\nplt.hist(valid_pred, bins = 100, alpha=0.5, label='Valid preds')\nplt.hist(test_pred, bins = 100, alpha=0.5, label='Test preds')\nplt.hist(valid_pred_sampled, bins = 100, alpha=0.5, label='Valid preds (sampled)')\nplt.hist(test_pred_sampled, bins = 100, alpha=0.5, label='Test preds (sampled)')\nplt.legend(loc='upper right')\nplt.grid(True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-23T21:59:29.087633Z","iopub.execute_input":"2022-10-23T21:59:29.088056Z","iopub.status.idle":"2022-10-23T21:59:30.046558Z","shell.execute_reply.started":"2022-10-23T21:59:29.088023Z","shell.execute_reply":"2022-10-23T21:59:30.045293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Mean prediction for         valid_data: {:.5f}'.format(np.mean(valid_pred)))\nprint('Mean prediction for          test_data: {:.5f}'.format(np.mean(test_pred)))\nprint('Mean prediction for sampled valid_data: {:.5f}'.format(np.mean(valid_pred_sampled)))\nprint('Mean prediction for sampled  test_data: {:.5f}'.format(np.mean(test_pred_sampled)))","metadata":{"execution":{"iopub.status.busy":"2022-10-23T21:59:38.379598Z","iopub.execute_input":"2022-10-23T21:59:38.380720Z","iopub.status.idle":"2022-10-23T21:59:38.389712Z","shell.execute_reply.started":"2022-10-23T21:59:38.380670Z","shell.execute_reply":"2022-10-23T21:59:38.388294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Conclusion: if you are working with LogLoss metric, be careful with you validation and check the target rates to get the best model. Sampling and changing the target rate will kill all your work!!!","metadata":{}}]}