{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-09T10:42:26.005715Z","iopub.execute_input":"2022-10-09T10:42:26.006773Z","iopub.status.idle":"2022-10-09T10:42:26.040438Z","shell.execute_reply.started":"2022-10-09T10:42:26.006643Z","shell.execute_reply":"2022-10-09T10:42:26.039460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Library import","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nimport pandas as pd\nimport lightgbm as lgb\nfrom sklearn import metrics\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import precision_score, recall_score\nfrom sklearn.metrics import confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2022-10-09T11:18:27.397808Z","iopub.execute_input":"2022-10-09T11:18:27.398244Z","iopub.status.idle":"2022-10-09T11:18:27.405399Z","shell.execute_reply.started":"2022-10-09T11:18:27.398209Z","shell.execute_reply":"2022-10-09T11:18:27.404059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Overview\nEach of the data from train0 to 9 was used at 1 second intervals, reducing the data size by a factor of 10.","metadata":{}},{"cell_type":"markdown","source":"# Data Loading","metadata":{}},{"cell_type":"code","source":"dtypes_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_dtypes.csv')\ndtypes = {k: v for (k, v) in zip(dtypes_df.column, dtypes_df.dtype)}\ntrain0_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_0.csv', dtype=dtypes)\ntrain1_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_1.csv', dtype=dtypes)\ntrain2_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_2.csv', dtype=dtypes)\ntrain3_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_3.csv', dtype=dtypes)\ntrain4_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_4.csv', dtype=dtypes)\ntrain5_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_5.csv', dtype=dtypes)\ntrain6_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_6.csv', dtype=dtypes)\ntrain7_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_7.csv', dtype=dtypes)\ntrain8_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_8.csv', dtype=dtypes)\ntrain9_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_9.csv', dtype=dtypes)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T10:47:33.860468Z","iopub.execute_input":"2022-10-09T10:47:33.860924Z","iopub.status.idle":"2022-10-09T10:52:33.130367Z","shell.execute_reply.started":"2022-10-09T10:47:33.860887Z","shell.execute_reply":"2022-10-09T10:52:33.128142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"step = 10 #time interval\ntrain0_selected = train0_df.loc[::step]\ntrain1_selected = train1_df.loc[::step]\ntrain2_selected = train2_df.loc[::step]\ntrain3_selected = train3_df.loc[::step]\ntrain4_selected = train4_df.loc[::step]\ntrain5_selected = train5_df.loc[::step]\ntrain6_selected = train6_df.loc[::step]\ntrain7_selected = train7_df.loc[::step]\ntrain8_selected = train8_df.loc[::step]\ntrain9_selected = train9_df.loc[::step]","metadata":{"execution":{"iopub.status.busy":"2022-10-09T10:53:06.289790Z","iopub.execute_input":"2022-10-09T10:53:06.290269Z","iopub.status.idle":"2022-10-09T10:53:06.301132Z","shell.execute_reply.started":"2022-10-09T10:53:06.290219Z","shell.execute_reply":"2022-10-09T10:53:06.299652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_list =[train0_selected, \n            train1_selected,\n            train2_selected,\n            train3_selected,\n            train4_selected,\n            train5_selected,\n            train6_selected,\n            train7_selected,\n            train8_selected,\n            train9_selected,\n             ]\nname_list = [f'train{i}' for i in range(10)]","metadata":{"execution":{"iopub.status.busy":"2022-10-09T10:53:09.775412Z","iopub.execute_input":"2022-10-09T10:53:09.775842Z","iopub.status.idle":"2022-10-09T10:53:09.781946Z","shell.execute_reply.started":"2022-10-09T10:53:09.775786Z","shell.execute_reply":"2022-10-09T10:53:09.780773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"join_train = pd.DataFrame(columns=list(train0_df.columns) + ['source'])\nfor train, name in zip(train_list, name_list):\n    train['source'] = name\n    join_train = pd.concat([join_train, train])","metadata":{"execution":{"iopub.status.busy":"2022-10-09T10:53:12.732522Z","iopub.execute_input":"2022-10-09T10:53:12.732929Z","iopub.status.idle":"2022-10-09T10:53:17.696658Z","shell.execute_reply.started":"2022-10-09T10:53:12.732888Z","shell.execute_reply":"2022-10-09T10:53:17.695184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"join_train","metadata":{"execution":{"iopub.status.busy":"2022-10-09T10:53:17.699059Z","iopub.execute_input":"2022-10-09T10:53:17.699499Z","iopub.status.idle":"2022-10-09T10:53:18.881988Z","shell.execute_reply.started":"2022-10-09T10:53:17.699454Z","shell.execute_reply":"2022-10-09T10:53:18.880851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"markdown","source":"## Heatmap of correlations","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize =(12,12))\nsns.heatmap(join_train.corr())\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T10:53:28.122342Z","iopub.execute_input":"2022-10-09T10:53:28.122754Z","iopub.status.idle":"2022-10-09T10:53:48.819175Z","shell.execute_reply.started":"2022-10-09T10:53:28.122718Z","shell.execute_reply":"2022-10-09T10:53:48.818146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Compared the ratio of data scoring by A and B each data","metadata":{}},{"cell_type":"code","source":"train_list = [f'train{i}' for i in range(10)]\nfig = plt.figure(figsize=(15,6))\nfig.suptitle('compared the ratio of data scoring by A and B each data', fontsize =16)\nplt.subplots_adjust(wspace=0.4, hspace=0.3)\nfor i, train in enumerate(train_list):\n    temp=join_train[join_train['source']==train]\n    plt.subplot(2, 5, i+1)  \n    x = [len(temp[temp['team_scoring_next']=='A']), \n         len(temp[temp['team_scoring_next']=='B']), \n         len(temp[temp['team_scoring_next'].isnull()])]\n    label = [\"A\", \"B\", \"nan\"]\n    plt.pie(x, labels=label, autopct=\"%1.1f%%\")\n    plt.title(f'{train}')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T10:53:48.820910Z","iopub.execute_input":"2022-10-09T10:53:48.821784Z","iopub.status.idle":"2022-10-09T10:53:53.915716Z","shell.execute_reply.started":"2022-10-09T10:53:48.821748Z","shell.execute_reply":"2022-10-09T10:53:53.914495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Compared the ratio of target data in each train data","metadata":{}},{"cell_type":"code","source":"fig = plt.figure(figsize=(15,6))\nfig.suptitle('compared the ratio of target data in each train data', fontsize =16)\nplt.subplots_adjust(wspace=0.4, hspace=0.3)\nfor i, train in enumerate(train_list):\n    temp=join_train[join_train['source']==train]\n    plt.subplot(2, 5, i+1)  \n    x = [len(temp[temp['team_A_scoring_within_10sec']==1]), \n      len(temp[temp['team_B_scoring_within_10sec']==1]), \n      len(temp)-len(temp[temp['team_A_scoring_within_10sec']==1])-len(temp[temp['team_B_scoring_within_10sec']==1])]\n    label = [\"team_A_scoring_within_10sec\", \"team_B_scoring_within_10sec\", \"other\"]\n    plt.pie(x, labels=label, autopct=\"%1.1f%%\")\n    plt.title(f'{train}')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T10:54:32.168277Z","iopub.execute_input":"2022-10-09T10:54:32.169595Z","iopub.status.idle":"2022-10-09T10:54:36.246090Z","shell.execute_reply.started":"2022-10-09T10:54:32.169538Z","shell.execute_reply":"2022-10-09T10:54:36.244869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All data from train0 to 9 are imbalance data","metadata":{}},{"cell_type":"markdown","source":"# LightGBM baseline A and B","metadata":{}},{"cell_type":"code","source":"test_dtypes_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/test_dtypes.csv')\ntest_dtypes = {k: v for (k, v) in zip(test_dtypes_df.column, test_dtypes_df.dtype)}\ntest = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/test.csv', dtype=test_dtypes)\nsub = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-10-09T11:00:29.039868Z","iopub.execute_input":"2022-10-09T11:00:29.040389Z","iopub.status.idle":"2022-10-09T11:00:39.075183Z","shell.execute_reply.started":"2022-10-09T11:00:29.040350Z","shell.execute_reply":"2022-10-09T11:00:39.074277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"join_train['team_A_scoring_within_10sec']=join_train['team_A_scoring_within_10sec'].astype(int)\njoin_train['team_B_scoring_within_10sec']=join_train['team_B_scoring_within_10sec'].astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T11:18:01.356840Z","iopub.execute_input":"2022-10-09T11:18:01.357974Z","iopub.status.idle":"2022-10-09T11:18:02.262567Z","shell.execute_reply.started":"2022-10-09T11:18:01.357924Z","shell.execute_reply":"2022-10-09T11:18:02.261434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = join_train.drop(['game_num', 'event_id', 'event_time', 'player_scoring_next', 'team_scoring_next', 'team_A_scoring_within_10sec', 'team_B_scoring_within_10sec', 'source'], axis=1)\nya = join_train['team_A_scoring_within_10sec']\nyb = join_train['team_B_scoring_within_10sec']\ntest_temp = test.drop(['id'], axis =1)\n\n#cross validation\nn_splits = 5\ncv = list(StratifiedKFold(n_splits = n_splits, shuffle=True, random_state=0).split(X, ya))\n\nmetrics=[]\nevaluate_df =  pd.DataFrame()\npred_df = pd.DataFrame()\nscore_list = []\n\nfor nfold in np.arange(n_splits):\n    print('-'*30, nfold, '-'*30)\n    evaluate_df =  pd.DataFrame()\n\n    idx_tr, idx_va = cv[nfold][0], cv[nfold][1]\n    x_tr, ya_tr, yb_tr = X.iloc[idx_tr], ya.iloc[idx_tr], yb.iloc[idx_tr]\n    x_va, ya_va, yb_va = X.iloc[idx_va], ya.iloc[idx_va], yb.iloc[idx_va]\n\n    #team A predict\n    modelA = lgb.LGBMClassifier(\n        objective='binary',\n        num_leaves=64,\n        min_child_samples=20,\n        max_depth=7\n    )\n    modelA.fit(x_tr, ya_tr)\n    ya_tr_pred = modelA.predict(x_tr)\n    ya_va_pred = modelA.predict(x_va)\n    ya_va_proba = modelA.predict_proba(x_va)\n    evaluate_df['A_true'] = ya_va\n    evaluate_df[f'A_proba'] = ya_va_proba[:, 1]\n    test_pred_proba = modelA.predict_proba(test_temp)\n    pred_df[f'A_fold{nfold}'] = test_pred_proba[:, 1]\n\n    ac_tr = accuracy_score(ya_tr, ya_tr_pred)\n    ac_va = accuracy_score(ya_va, ya_va_pred)\n    print(f'team_A:accuracy score: training {ac_tr}, validation{ac_va}')\n\n    #team B predict\n    modelB = lgb.LGBMClassifier(\n        objective='binary',\n        num_leaves=64,\n        min_child_samples=20,\n        max_depth=7\n    )\n    modelB.fit(x_tr, yb_tr)\n    yb_tr_pred = modelB.predict(x_tr)\n    yb_va_pred = modelB.predict(x_va)\n    yb_va_proba = modelB.predict_proba(x_va)\n    evaluate_df['B_true'] = yb_va\n    evaluate_df[f'B_proba'] = yb_va_proba[:, 1]\n    test_pred_proba = modelB.predict_proba(test_temp)\n    pred_df[f'B_fold{nfold}'] = test_pred_proba[:, 1]\n\n    ac_tr = accuracy_score(yb_tr, yb_tr_pred)\n    ac_va = accuracy_score(yb_va, yb_va_pred)\n    print(f'team_B:accuracy score: training {ac_tr}, validation{ac_va}')\n\n    #calculate score\n    evaluate_df['A_score'] = evaluate_df['A_true']*np.log(evaluate_df['A_proba']) + (1-evaluate_df['A_true'])*np.log(1-evaluate_df['A_proba'])\n    A_score = -1*evaluate_df[f'A_score'].mean()\n    evaluate_df['B_score'] = evaluate_df['B_true']*np.log(evaluate_df['B_proba']) + (1-evaluate_df['B_true'])*np.log(1-evaluate_df['B_proba'])\n    B_score = -1*evaluate_df[f'B_score'].mean()\n    score = (A_score + B_score)/2\n    score_list.append([f'{nfold}' ,A_score, B_score, score])\n    print(f'A_score : {A_score}\\nB_score : {B_score}\\nscore : {score}')","metadata":{"execution":{"iopub.status.busy":"2022-10-09T11:18:36.420570Z","iopub.execute_input":"2022-10-09T11:18:36.421021Z","iopub.status.idle":"2022-10-09T11:29:22.702284Z","shell.execute_reply.started":"2022-10-09T11:18:36.420982Z","shell.execute_reply":"2022-10-09T11:29:22.700892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score_df = pd.DataFrame(score_list, columns=['fold', 'A_score', 'B_score', 'score'])\nscore_df","metadata":{"execution":{"iopub.status.busy":"2022-10-09T11:36:01.738129Z","iopub.execute_input":"2022-10-09T11:36:01.738630Z","iopub.status.idle":"2022-10-09T11:36:01.754367Z","shell.execute_reply.started":"2022-10-09T11:36:01.738594Z","shell.execute_reply":"2022-10-09T11:36:01.753330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"A_columns = [c for c in pred_df.columns if 'A' in c]\nB_columns = [c for c in pred_df.columns if 'B' in c]\npred_df['A_mean'] = pred_df[A_columns].mean(axis=1)\npred_df['B_mean'] = pred_df[B_columns].mean(axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T11:36:25.711128Z","iopub.execute_input":"2022-10-09T11:36:25.711559Z","iopub.status.idle":"2022-10-09T11:36:25.849909Z","shell.execute_reply.started":"2022-10-09T11:36:25.711527Z","shell.execute_reply":"2022-10-09T11:36:25.848654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"submission =sub.copy()\nsubmission[['team_A_scoring_within_10sec', 'team_B_scoring_within_10sec']] = pred_df[['A_mean', 'B_mean']]","metadata":{"execution":{"iopub.status.busy":"2022-10-09T11:39:23.569044Z","iopub.execute_input":"2022-10-09T11:39:23.570329Z","iopub.status.idle":"2022-10-09T11:39:23.593658Z","shell.execute_reply.started":"2022-10-09T11:39:23.570282Z","shell.execute_reply":"2022-10-09T11:39:23.592198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv(\"/kaggle/working/submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T11:39:32.122463Z","iopub.execute_input":"2022-10-09T11:39:32.122885Z","iopub.status.idle":"2022-10-09T11:39:34.841075Z","shell.execute_reply.started":"2022-10-09T11:39:32.122851Z","shell.execute_reply":"2022-10-09T11:39:34.839999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}