{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# TPS OCT 2022","metadata":{}},{"cell_type":"markdown","source":"**The goal of the competition is to predict -- from a given snapshot in the game -- for each team, the probability that they will score within the next 10 seconds of game time.**","metadata":{}},{"cell_type":"markdown","source":"**If you like my work, please, leave an upvote: it will be really appreciated and it will motivate me in offering more content to the Kaggle community ! :)**","metadata":{}},{"cell_type":"markdown","source":"# Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator\nfrom matplotlib.colors import ListedColormap\nimport seaborn as sns\nfrom scipy.special import softmax\nfrom cycler import cycler\nfrom IPython.display import display\nimport datetime\nimport joblib\nimport gc\nfrom pathlib import Path\nfrom fastai.tabular.all import *\nimport fastai.losses as loss\n\nfrom sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.feature_selection import mutual_info_classif\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.linear_model import Ridge\nfrom plotly.offline import plot, iplot, init_notebook_mode\nimport plotly.graph_objs as go\ninit_notebook_mode(connected=True)\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \npd.set_option('display.max_columns', None)   ","metadata":{"execution":{"iopub.status.busy":"2022-10-19T02:35:13.731395Z","iopub.execute_input":"2022-10-19T02:35:13.731947Z","iopub.status.idle":"2022-10-19T02:35:13.751853Z","shell.execute_reply.started":"2022-10-19T02:35:13.731922Z","shell.execute_reply":"2022-10-19T02:35:13.750444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Let's start by reading the data and looking at the first few rows:**","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/tabular-playground-series-oct-2022/train_0.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-10-19T02:35:13.753515Z","iopub.execute_input":"2022-10-19T02:35:13.753890Z","iopub.status.idle":"2022-10-19T02:35:24.035679Z","shell.execute_reply.started":"2022-10-19T02:35:13.753857Z","shell.execute_reply":"2022-10-19T02:35:24.034354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print()\nprint('df')\ndisplay(df.head())","metadata":{"execution":{"iopub.status.busy":"2022-10-19T02:35:24.036990Z","iopub.execute_input":"2022-10-19T02:35:24.037311Z","iopub.status.idle":"2022-10-19T02:35:24.097361Z","shell.execute_reply.started":"2022-10-19T02:35:24.037282Z","shell.execute_reply":"2022-10-19T02:35:24.096216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-10-19T02:35:24.099164Z","iopub.execute_input":"2022-10-19T02:35:24.099886Z","iopub.status.idle":"2022-10-19T02:35:24.112974Z","shell.execute_reply.started":"2022-10-19T02:35:24.099840Z","shell.execute_reply":"2022-10-19T02:35:24.112017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import missingno as msno\n%matplotlib inline\nmsno.matrix(df.sample(250))","metadata":{"execution":{"iopub.status.busy":"2022-10-19T02:35:24.114700Z","iopub.execute_input":"2022-10-19T02:35:24.115649Z","iopub.status.idle":"2022-10-19T02:35:24.514974Z","shell.execute_reply.started":"2022-10-19T02:35:24.115617Z","shell.execute_reply":"2022-10-19T02:35:24.513796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-10-19T02:35:24.516264Z","iopub.execute_input":"2022-10-19T02:35:24.516543Z","iopub.status.idle":"2022-10-19T02:35:27.805691Z","shell.execute_reply.started":"2022-10-19T02:35:24.516519Z","shell.execute_reply":"2022-10-19T02:35:27.804315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr = df.corr()\nsns.set(rc = {\"figure.figsize\": (14, 10)})\n\nsns.heatmap(corr, xticklabels = corr.columns, yticklabels = corr.columns, cmap = \"YlGnBu\")","metadata":{"execution":{"iopub.status.busy":"2022-10-19T02:35:27.806982Z","iopub.execute_input":"2022-10-19T02:35:27.807348Z","iopub.status.idle":"2022-10-19T02:35:43.771309Z","shell.execute_reply.started":"2022-10-19T02:35:27.807317Z","shell.execute_reply":"2022-10-19T02:35:43.770051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"float_features = [f for f in df.columns if df[f].dtype == 'float64']\n\n# Training histograms\nfig, axs = plt.subplots(4, 4, figsize=(16, 16))\nfor f, ax in zip(float_features, axs.ravel()):\n    ax.hist(df[f], density=True, bins=100)\n    ax.set_title(f'Train {f}, std={df[f].std():.1f}')\nplt.suptitle('Histograms of the float features', y=0.93, fontsize=20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-19T02:35:43.772653Z","iopub.execute_input":"2022-10-19T02:35:43.773401Z","iopub.status.idle":"2022-10-19T02:35:48.903855Z","shell.execute_reply.started":"2022-10-19T02:35:43.773367Z","shell.execute_reply":"2022-10-19T02:35:48.903094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"f, axes = plt.subplots(nrows = 1, ncols = 2, figsize = (25, 10))\nlabels = [f\"Games with {i+1} events\" for i in range(10)]\ndf.groupby(['game_num'])['event_id'].nunique().value_counts().plot(kind=\"pie\", ax=axes[0], labels=labels)\ndf.groupby(['game_num'])['event_id'].nunique().value_counts().plot(kind=\"bar\", ax=axes[1])","metadata":{"execution":{"iopub.status.busy":"2022-10-19T02:35:48.907819Z","iopub.execute_input":"2022-10-19T02:35:48.908477Z","iopub.status.idle":"2022-10-19T02:35:49.355422Z","shell.execute_reply.started":"2022-10-19T02:35:48.908451Z","shell.execute_reply":"2022-10-19T02:35:49.354483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['event_time_diff'] = abs(df['event_time'].shift(1).fillna(method='ffill')-df['event_time'])\naverage_event_time_diff_by_game = df.groupby(['game_num'])['event_time_diff'].mean()\naverage_event_time_diff_by_game.plot(kind=\"hist\") ","metadata":{"execution":{"iopub.status.busy":"2022-10-19T02:35:49.356629Z","iopub.execute_input":"2022-10-19T02:35:49.357609Z","iopub.status.idle":"2022-10-19T02:35:49.646074Z","shell.execute_reply.started":"2022-10-19T02:35:49.357571Z","shell.execute_reply":"2022-10-19T02:35:49.644776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, (ax1, ax2) = plt.subplots(1, 2, figsize = (15, 7))\nlabels_a = [f\"{p:.2f}%\" for p in df['team_A_scoring_within_10sec'].value_counts()/df['team_A_scoring_within_10sec'].value_counts().sum()*100]\nlabels_b = [f\"{p:.2f}%\" for p in df['team_B_scoring_within_10sec'].value_counts()/df['team_A_scoring_within_10sec'].value_counts().sum()*100]\ndf['team_A_scoring_within_10sec'].value_counts().plot(kind='pie', ax=ax1, labels=labels_a, startangle=-140)\ndf['team_B_scoring_within_10sec'].value_counts().plot(kind='pie', ax=ax2, labels=labels_b, startangle=-140)","metadata":{"execution":{"iopub.status.busy":"2022-10-19T02:35:49.647959Z","iopub.execute_input":"2022-10-19T02:35:49.648359Z","iopub.status.idle":"2022-10-19T02:35:49.940015Z","shell.execute_reply.started":"2022-10-19T02:35:49.648325Z","shell.execute_reply":"2022-10-19T02:35:49.939217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15, 7))\nsns.histplot(data=df, x='event_time')\nplt.title('Time before event (s)')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-19T02:35:49.941353Z","iopub.execute_input":"2022-10-19T02:35:49.941917Z","iopub.status.idle":"2022-10-19T02:35:51.197550Z","shell.execute_reply.started":"2022-10-19T02:35:49.941884Z","shell.execute_reply":"2022-10-19T02:35:51.196753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# features","metadata":{}},{"cell_type":"code","source":"features = [\n    'ball_pos_x', 'ball_pos_y','ball_pos_z', 'ball_vel_x', 'ball_vel_y', 'ball_vel_z', \n    'p0_pos_x', 'p0_pos_y', 'p0_pos_z', 'p0_vel_x', 'p0_vel_y', 'p0_vel_z', 'p0_boost', 'p0_na',\n    'p1_pos_x', 'p1_pos_y', 'p1_pos_z', 'p1_vel_x', 'p1_vel_y', 'p1_vel_z', 'p1_boost', 'p1_na',\n    'p2_pos_x', 'p2_pos_y', 'p2_pos_z', 'p2_vel_x', 'p2_vel_y', 'p2_vel_z', 'p2_boost', 'p2_na',\n    'p3_pos_x', 'p3_pos_y', 'p3_pos_z', 'p3_vel_x', 'p3_vel_y', 'p3_vel_z', 'p3_boost', 'p3_na',\n    'p4_pos_x', 'p4_pos_y', 'p4_pos_z', 'p4_vel_x', 'p4_vel_y', 'p4_vel_z', 'p4_boost', 'p4_na',\n    'p5_pos_x', 'p5_pos_y', 'p5_pos_z', 'p5_vel_x', 'p5_vel_y', 'p5_vel_z', 'p5_boost', 'p5_na',\n    'boost0_timer', 'boost1_timer', \n    'boost2_timer', 'boost3_timer',\n    'boost4_timer', 'boost5_timer']\n\nfeatures_x_pos = [pos for pos, feature in enumerate(features) if feature.endswith('_x')]\nfeatures_y_pos = [pos for pos, feature in enumerate(features) if feature.endswith('_y')]\n\ntargets = [\n    'team_A_scoring_within_10sec',\n    'team_B_scoring_within_10sec']","metadata":{"execution":{"iopub.status.busy":"2022-10-19T02:35:51.198801Z","iopub.execute_input":"2022-10-19T02:35:51.200013Z","iopub.status.idle":"2022-10-19T02:35:51.208288Z","shell.execute_reply.started":"2022-10-19T02:35:51.199947Z","shell.execute_reply":"2022-10-19T02:35:51.206738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nDEBUG = False\ninput_path = Path('../input/fast-loading-high-compression-with-feather/feather_data')\n\ndef fe(x):\n    # indicators for respawns...\n    x['p0_na'] = x['p0_pos_x'].isna().astype('int8')\n    x['p1_na'] = x['p1_pos_x'].isna().astype('int8')\n    x['p2_na'] = x['p2_pos_x'].isna().astype('int8')\n    x['p3_na'] = x['p3_pos_x'].isna().astype('int8')\n    x['p4_na'] = x['p4_pos_x'].isna().astype('int8')\n    x['p5_na'] = x['p5_pos_x'].isna().astype('int8')\n    for feature in features:\n        if feature.endswith('_na'):\n            continue\n        # this is just scaling the features to something reasonable\n        # it might make sense to apply a transformation to the z-dimension.\n        if feature.endswith('_x'):\n            x[feature] = (x[feature] / 82).fillna(0).astype('float16')\n        if feature.endswith('_y'):\n            x[feature] = (x[feature] / 120).fillna(0).astype('float16')\n        if feature.endswith('_z'):\n            x[feature] = (x[feature] / 40).fillna(0).astype('float16')\n        if feature.endswith('_boost'):\n            x[feature] = (x[feature] / 100).fillna(0).astype('float16')\n        if feature.endswith('_timer'):\n            x[feature] = (-x[feature] / 100).astype('float16')\n    return x\n\ndef read_train():\n    dfs = []\n    for i in range(10):\n        dfs.append(fe(pd.read_feather(input_path / f'train_{i}_compressed.ftr')))\n    result = pd.concat(dfs)\n    if DEBUG:\n        result = result.sample(frac=0.05)\n    return result\n\ndef read_test():\n    return fe(pd.read_feather(input_path / 'test_compressed.ftr'))\n\ndf_train = read_train()\ngc.collect()\n\nprint(f'Train Rows = {len(df_train):,}  ' \n      f'Memory Usage = {df_train.memory_usage(deep=True).sum() / (1024 * 1024):4.1f} Mb'\n     '\\n')","metadata":{"execution":{"iopub.status.busy":"2022-10-19T02:35:51.210041Z","iopub.execute_input":"2022-10-19T02:35:51.210544Z","iopub.status.idle":"2022-10-19T02:36:12.295018Z","shell.execute_reply.started":"2022-10-19T02:35:51.210509Z","shell.execute_reply":"2022-10-19T02:36:12.293659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1 = pd.read_csv(\"../input/tps-2022-10-fastai-with-multistart-and-tta/model_fastai_multistart_tta.csv\") # 0.19443","metadata":{"execution":{"iopub.status.busy":"2022-10-19T02:36:12.296259Z","iopub.execute_input":"2022-10-19T02:36:12.297007Z","iopub.status.idle":"2022-10-19T02:36:12.492878Z","shell.execute_reply.started":"2022-10-19T02:36:12.296980Z","shell.execute_reply":"2022-10-19T02:36:12.491753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = [0.19443, 0.19674, 0.19778, 0.19895]","metadata":{"execution":{"iopub.status.busy":"2022-10-19T02:36:12.494104Z","iopub.execute_input":"2022-10-19T02:36:12.494407Z","iopub.status.idle":"2022-10-19T02:36:12.499033Z","shell.execute_reply.started":"2022-10-19T02:36:12.494381Z","shell.execute_reply":"2022-10-19T02:36:12.497933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(1 - scores[0]) / (len(scores) - sum(scores))","metadata":{"execution":{"iopub.status.busy":"2022-10-19T02:36:12.500318Z","iopub.execute_input":"2022-10-19T02:36:12.500601Z","iopub.status.idle":"2022-10-19T02:36:12.514002Z","shell.execute_reply.started":"2022-10-19T02:36:12.500571Z","shell.execute_reply":"2022-10-19T02:36:12.512977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#result_df = pd.DataFrame()\n#result_df[\"id\"] = df1.id\n#result_df[\"team_A_scoring_within_10sec\"] = df1.team_A_scoring_within_10sec * ((1 - scores[0]) / (len(scores) - sum(scores))) + df2.team_A_scoring_within_10sec * ((1 - scores[1]) / (len(scores) - sum(scores))) + df3.team_A_scoring_within_10sec * ((1 - scores[2]) / (len(scores) - sum(scores))) + df4.team_A_scoring_within_10sec * ((1 - scores[3]) / (len(scores) - sum(scores)))\n#result_df[\"team_B_scoring_within_10sec\"] = df1.team_B_scoring_within_10sec * ((1 - scores[0]) / (len(scores) - sum(scores))) + df2.team_B_scoring_within_10sec * ((1 - scores[1]) / (len(scores) - sum(scores))) + df3.team_B_scoring_within_10sec * ((1 - scores[2]) / (len(scores) - sum(scores))) + df4.team_B_scoring_within_10sec * ((1 - scores[3]) / (len(scores) - sum(scores)))","metadata":{"execution":{"iopub.status.busy":"2022-10-19T02:36:12.515387Z","iopub.execute_input":"2022-10-19T02:36:12.515681Z","iopub.status.idle":"2022-10-19T02:36:12.523383Z","shell.execute_reply.started":"2022-10-19T02:36:12.515658Z","shell.execute_reply":"2022-10-19T02:36:12.522243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-19T02:36:12.524757Z","iopub.execute_input":"2022-10-19T02:36:12.525906Z","iopub.status.idle":"2022-10-19T02:36:12.542256Z","shell.execute_reply.started":"2022-10-19T02:36:12.525860Z","shell.execute_reply":"2022-10-19T02:36:12.541487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"df1.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-10-19T02:36:12.543535Z","iopub.execute_input":"2022-10-19T02:36:12.544338Z","iopub.status.idle":"2022-10-19T02:36:14.391318Z","shell.execute_reply.started":"2022-10-19T02:36:12.544303Z","shell.execute_reply":"2022-10-19T02:36:14.389845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Thank you for reading!","metadata":{}},{"cell_type":"markdown","source":"# Please let me know if you have any questions and I look forward to any suggestions 🙂","metadata":{}}]}