{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Importing Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sb\nimport matplotlib.pyplot as plt\n\nfrom sklearn.model_selection import train_test_split\nfrom tqdm.notebook import tqdm, trange\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom keras import layers\nimport gc\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-10-07T15:34:10.554870Z","iopub.execute_input":"2022-10-07T15:34:10.555626Z","iopub.status.idle":"2022-10-07T15:34:16.080138Z","shell.execute_reply.started":"2022-10-07T15:34:10.555534Z","shell.execute_reply":"2022-10-07T15:34:16.079120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Analysis","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('../input/tabular-playground-series-oct-2022/train_0.csv')\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-07T15:34:16.082060Z","iopub.execute_input":"2022-10-07T15:34:16.082802Z","iopub.status.idle":"2022-10-07T15:34:43.815105Z","shell.execute_reply.started":"2022-10-07T15:34:16.082764Z","shell.execute_reply":"2022-10-07T15:34:43.813986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-07T15:34:43.817050Z","iopub.execute_input":"2022-10-07T15:34:43.817726Z","iopub.status.idle":"2022-10-07T15:34:43.857974Z","shell.execute_reply.started":"2022-10-07T15:34:43.817682Z","shell.execute_reply":"2022-10-07T15:34:43.856693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Even if team_scoring_next is B the team_B_scoring_within_10sec is 0 this is because the event time is -33 that implies that there is still 33 sec in the game to come to an end. And the target column is about last 10sec this is why is contains 0.","metadata":{}},{"cell_type":"markdown","source":"Some of the important points regarding extra columns than the testing data is that:\n* game_num, event_id, event_time these three are not there in the testing data.\n* player_scoring_next or team_scoring_next contain informaton regarding the target variable.******","metadata":{}},{"cell_type":"code","source":"for col in df.columns:\n    k = df[col].isnull().sum()\n    if k > 0:\n        print(col, '->', k)        ","metadata":{"execution":{"iopub.status.busy":"2022-10-07T15:34:43.861437Z","iopub.execute_input":"2022-10-07T15:34:43.862500Z","iopub.status.idle":"2022-10-07T15:34:44.171886Z","shell.execute_reply.started":"2022-10-07T15:34:43.862458Z","shell.execute_reply":"2022-10-07T15:34:44.170840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### We will be removing team_scoring_next column any how and in the remaining columns removing 20000 data points from 2 million rows of data is not going to effect the model's performance much.","metadata":{}},{"cell_type":"code","source":"features = df.loc[:,'ball_pos_x':'boost5_timer']\n\nplt.figure(figsize=(15,15))\nsb.heatmap(features.corr() > 0.8, annot=True, cbar=False)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-07T15:34:44.173405Z","iopub.execute_input":"2022-10-07T15:34:44.173907Z","iopub.status.idle":"2022-10-07T15:35:08.728770Z","shell.execute_reply.started":"2022-10-07T15:34:44.173864Z","shell.execute_reply":"2022-10-07T15:35:08.727765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Here we can see that there are no highly correlated features here.","metadata":{}},{"cell_type":"markdown","source":"# Model Development","metadata":{}},{"cell_type":"code","source":"model = keras.models.Sequential([\n    layers.Dense(256, activation='relu', input_shape=[54]),\n    layers.BatchNormalization(),\n    layers.Dense(256, activation='relu'),\n    layers.BatchNormalization(),\n    layers.Dropout(0.1),\n    layers.Dense(256, activation='relu'),\n    layers.BatchNormalization(),\n    layers.Dropout(0.3),\n    layers.Dense(3, activation='softmax')\n])\n\nmodel.compile(optimizer='adam',\n             loss=tf.keras.losses.CategoricalCrossentropy(from_logits=True),\n             metrics=['AUC', 'accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-10-07T15:36:22.544767Z","iopub.execute_input":"2022-10-07T15:36:22.545641Z","iopub.status.idle":"2022-10-07T15:36:25.158217Z","shell.execute_reply.started":"2022-10-07T15:36:22.545600Z","shell.execute_reply":"2022-10-07T15:36:25.156081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-10-07T15:36:25.160269Z","iopub.execute_input":"2022-10-07T15:36:25.160705Z","iopub.status.idle":"2022-10-07T15:36:25.171146Z","shell.execute_reply.started":"2022-10-07T15:36:25.160669Z","shell.execute_reply":"2022-10-07T15:36:25.169979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"to_remove = ['game_num', 'event_id', 'event_time', 'player_scoring_next', 'team_scoring_next']","metadata":{"execution":{"iopub.status.busy":"2022-10-07T15:36:25.173064Z","iopub.execute_input":"2022-10-07T15:36:25.173793Z","iopub.status.idle":"2022-10-07T15:36:25.191373Z","shell.execute_reply.started":"2022-10-07T15:36:25.173745Z","shell.execute_reply":"2022-10-07T15:36:25.190412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### In the logical XOR part what I have done is created an extra column which will contain 1 if none of the two teams has scored a goal in next 10 sec otherwise it will contain 0.","metadata":{}},{"cell_type":"code","source":"for i in trange(10):\n    path = f'../input/tabular-playground-series-oct-2022/train_{i}.csv'\n    df = pd.read_csv(path)\n    \n#    Data loading and preprocessing \n    df['no_team_scored'] = np.logical_xor(df['team_A_scoring_within_10sec'],\n                                          df['team_B_scoring_within_10sec'])\n    df['no_team_scored'] = (~df['no_team_scored']).astype(int)\n    \n    df.drop(to_remove, axis=1, inplace=True)\n    df.dropna(inplace=True, axis=0)\n    \n    features = df.loc[:,'ball_pos_x':'boost5_timer']\n    target = df.loc[:,'team_A_scoring_within_10sec':'no_team_scored']\n\n    X_train, X_val,\\\n    Y_train, Y_val = train_test_split(features, target,\n                                      test_size = 0.03,\n                                      random_state=22)\n    \n    print(f'Training on dataset number {i+1}.')\n    \n    model.fit(X_train, Y_train,\n              batch_size=64,\n              epochs=1,\n              verbose=1,\n              validation_data=(X_val, Y_val))\n    \n    del df, X_train, X_val, Y_train, Y_val\n    gc.collect()\n    print()","metadata":{"execution":{"iopub.status.busy":"2022-10-07T15:36:27.604514Z","iopub.execute_input":"2022-10-07T15:36:27.604882Z","iopub.status.idle":"2022-10-07T16:12:25.791063Z","shell.execute_reply.started":"2022-10-07T15:36:27.604852Z","shell.execute_reply":"2022-10-07T16:12:25.789902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### From the above training we can say that training on all the data has no effect on the accuracy as the accuracy score or AUC is stick to the 90% and 96% respectively.","metadata":{}},{"cell_type":"markdown","source":"## Predictions\nOne of the important factor here is predicting three classes B's score, A's Score and no ones score to get better results on the leaderboard. Due to this reason only sum of the probabilities for just A and B won't be equal to one.","metadata":{}},{"cell_type":"code","source":"df_test = pd.read_csv('../input/tabular-playground-series-oct-2022/test.csv')\ndf_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-07T16:27:50.352469Z","iopub.execute_input":"2022-10-07T16:27:50.352821Z","iopub.status.idle":"2022-10-07T16:27:54.978475Z","shell.execute_reply.started":"2022-10-07T16:27:50.352792Z","shell.execute_reply":"2022-10-07T16:27:54.977425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.isnull().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2022-10-07T16:27:54.980386Z","iopub.execute_input":"2022-10-07T16:27:54.980847Z","iopub.status.idle":"2022-10-07T16:27:55.055828Z","shell.execute_reply.started":"2022-10-07T16:27:54.980810Z","shell.execute_reply":"2022-10-07T16:27:55.054851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in df_test.columns:\n    if df_test[col].isnull().sum() > 0:\n        temp = df_test[col].mean()\n        df_test[col] = df_test[col].fillna(temp)\n\ndf_test.isnull().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2022-10-07T16:23:38.542983Z","iopub.execute_input":"2022-10-07T16:23:38.543564Z","iopub.status.idle":"2022-10-07T16:23:38.909646Z","shell.execute_reply.started":"2022-10-07T16:23:38.543528Z","shell.execute_reply":"2022-10-07T16:23:38.908579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-07T16:23:40.892381Z","iopub.execute_input":"2022-10-07T16:23:40.892735Z","iopub.status.idle":"2022-10-07T16:23:40.918692Z","shell.execute_reply.started":"2022-10-07T16:23:40.892706Z","shell.execute_reply":"2022-10-07T16:23:40.917767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = df_test.loc[:,'ball_pos_x':'boost5_timer']\npreds = model.predict(test)","metadata":{"execution":{"iopub.status.busy":"2022-10-07T16:23:42.822635Z","iopub.execute_input":"2022-10-07T16:23:42.823049Z","iopub.status.idle":"2022-10-07T16:24:06.260865Z","shell.execute_reply.started":"2022-10-07T16:23:42.823015Z","shell.execute_reply":"2022-10-07T16:24:06.259868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-07T16:24:06.262986Z","iopub.execute_input":"2022-10-07T16:24:06.263336Z","iopub.status.idle":"2022-10-07T16:24:06.269961Z","shell.execute_reply.started":"2022-10-07T16:24:06.263303Z","shell.execute_reply":"2022-10-07T16:24:06.269071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss = pd.read_csv('../input/tabular-playground-series-oct-2022/sample_submission.csv')\nss['team_A_scoring_within_10sec'] = preds[:,0]\nss['team_B_scoring_within_10sec'] = preds[:,1]\nss.to_csv('Submission.csv', index=False)\nss.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-07T16:24:06.271459Z","iopub.execute_input":"2022-10-07T16:24:06.272105Z","iopub.status.idle":"2022-10-07T16:24:08.023973Z","shell.execute_reply.started":"2022-10-07T16:24:06.272071Z","shell.execute_reply":"2022-10-07T16:24:08.022869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### This will be our baseline score as we have done nothing but fit and predict on the training and testing data respectively. Let's see what feature engineering or different techniques could help us to get better results than this.","metadata":{}}]}