{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle'): #/input\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-14T00:28:16.083017Z","iopub.execute_input":"2022-10-14T00:28:16.083438Z","iopub.status.idle":"2022-10-14T00:28:16.091956Z","shell.execute_reply.started":"2022-10-14T00:28:16.083406Z","shell.execute_reply":"2022-10-14T00:28:16.091066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dtypes_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_dtypes.csv')\ndtypes = {k: v for (k, v) in zip(dtypes_df.column, dtypes_df.dtype)}\ntrain0_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_0.csv', dtype=dtypes)\ntrain0_df","metadata":{"execution":{"iopub.status.busy":"2022-10-14T00:28:16.105463Z","iopub.execute_input":"2022-10-14T00:28:16.105965Z","iopub.status.idle":"2022-10-14T00:28:45.503799Z","shell.execute_reply.started":"2022-10-14T00:28:16.105920Z","shell.execute_reply":"2022-10-14T00:28:45.502561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_df = train0_df\n\n#for i in range(1,10):\n#    temp_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_{0}.csv'.format(i), dtype=dtypes)\n#    train_df = pd.concat([train_df,temp_df])\n#train_df","metadata":{"execution":{"iopub.status.busy":"2022-10-14T00:28:45.506026Z","iopub.execute_input":"2022-10-14T00:28:45.506731Z","iopub.status.idle":"2022-10-14T00:28:45.511668Z","shell.execute_reply.started":"2022-10-14T00:28:45.506685Z","shell.execute_reply":"2022-10-14T00:28:45.510345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train0_df[200:230]","metadata":{"execution":{"iopub.status.busy":"2022-10-14T00:28:45.513011Z","iopub.execute_input":"2022-10-14T00:28:45.513322Z","iopub.status.idle":"2022-10-14T00:28:45.559075Z","shell.execute_reply.started":"2022-10-14T00:28:45.513293Z","shell.execute_reply":"2022-10-14T00:28:45.557987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def no_team_scoring(team_A,team_B):\n    if team_A == 0 and team_B == 0:\n        return 1\n    else:\n        return 0\n    \ntrain0_df['no_team_scoring_within_10sec'] = train0_df.apply(lambda x: no_team_scoring(x['team_A_scoring_within_10sec'],x['team_B_scoring_within_10sec']), axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-10-14T00:28:45.561775Z","iopub.execute_input":"2022-10-14T00:28:45.562165Z","iopub.status.idle":"2022-10-14T00:29:26.031330Z","shell.execute_reply.started":"2022-10-14T00:28:45.562119Z","shell.execute_reply":"2022-10-14T00:29:26.030457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train0_df[200:230]","metadata":{"execution":{"iopub.status.busy":"2022-10-14T00:29:26.032870Z","iopub.execute_input":"2022-10-14T00:29:26.033382Z","iopub.status.idle":"2022-10-14T00:29:26.074210Z","shell.execute_reply.started":"2022-10-14T00:29:26.033351Z","shell.execute_reply":"2022-10-14T00:29:26.073015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train0_df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-10-14T00:29:26.075769Z","iopub.execute_input":"2022-10-14T00:29:26.076149Z","iopub.status.idle":"2022-10-14T00:29:32.663052Z","shell.execute_reply.started":"2022-10-14T00:29:26.076108Z","shell.execute_reply":"2022-10-14T00:29:32.661896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train0_df['team_A_scoring_within_10sec'].mean() + train0_df['team_B_scoring_within_10sec'].mean() + train0_df['no_team_scoring_within_10sec'].mean()","metadata":{"execution":{"iopub.status.busy":"2022-10-14T00:29:32.664232Z","iopub.execute_input":"2022-10-14T00:29:32.664794Z","iopub.status.idle":"2022-10-14T00:29:32.679619Z","shell.execute_reply.started":"2022-10-14T00:29:32.664763Z","shell.execute_reply":"2022-10-14T00:29:32.678420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train0_df.columns","metadata":{"execution":{"iopub.status.busy":"2022-10-14T00:29:32.681163Z","iopub.execute_input":"2022-10-14T00:29:32.681909Z","iopub.status.idle":"2022-10-14T00:29:32.689626Z","shell.execute_reply.started":"2022-10-14T00:29:32.681853Z","shell.execute_reply":"2022-10-14T00:29:32.688545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train0_df['team_scoring_next'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-10-14T00:29:32.691246Z","iopub.execute_input":"2022-10-14T00:29:32.692709Z","iopub.status.idle":"2022-10-14T00:29:32.783272Z","shell.execute_reply.started":"2022-10-14T00:29:32.692658Z","shell.execute_reply":"2022-10-14T00:29:32.782269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define train and test sets","metadata":{}},{"cell_type":"code","source":"target = ['team_A_scoring_within_10sec', 'team_B_scoring_within_10sec', 'no_team_scoring_within_10sec']\n#target = ['team_A_scoring_within_10sec']\ny = train0_df[target]\ny.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-14T00:29:32.788567Z","iopub.execute_input":"2022-10-14T00:29:32.789769Z","iopub.status.idle":"2022-10-14T00:29:32.809018Z","shell.execute_reply.started":"2022-10-14T00:29:32.789734Z","shell.execute_reply":"2022-10-14T00:29:32.808285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = train0_df.columns[3:-5]","metadata":{"execution":{"iopub.status.busy":"2022-10-14T00:29:32.809984Z","iopub.execute_input":"2022-10-14T00:29:32.810700Z","iopub.status.idle":"2022-10-14T00:29:32.820799Z","shell.execute_reply.started":"2022-10-14T00:29:32.810671Z","shell.execute_reply":"2022-10-14T00:29:32.820038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features","metadata":{"execution":{"iopub.status.busy":"2022-10-14T00:29:32.821742Z","iopub.execute_input":"2022-10-14T00:29:32.822729Z","iopub.status.idle":"2022-10-14T00:29:32.834753Z","shell.execute_reply.started":"2022-10-14T00:29:32.822697Z","shell.execute_reply":"2022-10-14T00:29:32.834037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#features2 = ['ball_pos_x', 'ball_pos_y', 'ball_pos_z', 'ball_vel_x', 'ball_vel_y','ball_vel_z']\nX = train0_df[features]\nX.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-14T00:29:32.835703Z","iopub.execute_input":"2022-10-14T00:29:32.836558Z","iopub.status.idle":"2022-10-14T00:29:33.028138Z","shell.execute_reply.started":"2022-10-14T00:29:32.836527Z","shell.execute_reply":"2022-10-14T00:29:33.027089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.isnull().any()","metadata":{"execution":{"iopub.status.busy":"2022-10-14T00:29:33.029445Z","iopub.execute_input":"2022-10-14T00:29:33.029760Z","iopub.status.idle":"2022-10-14T00:29:33.175418Z","shell.execute_reply.started":"2022-10-14T00:29:33.029734Z","shell.execute_reply":"2022-10-14T00:29:33.174476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.fillna(0, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-10-14T00:29:33.177530Z","iopub.execute_input":"2022-10-14T00:29:33.178472Z","iopub.status.idle":"2022-10-14T00:29:33.408185Z","shell.execute_reply.started":"2022-10-14T00:29:33.178438Z","shell.execute_reply":"2022-10-14T00:29:33.407287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.isnull().any()","metadata":{"execution":{"iopub.status.busy":"2022-10-14T00:29:33.409248Z","iopub.execute_input":"2022-10-14T00:29:33.409552Z","iopub.status.idle":"2022-10-14T00:29:33.557913Z","shell.execute_reply.started":"2022-10-14T00:29:33.409523Z","shell.execute_reply":"2022-10-14T00:29:33.556833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n# Split into validation and training data\ntrain_X, val_X, train_y, val_y = train_test_split(X, y, random_state=1, test_size=0.1)","metadata":{"execution":{"iopub.status.busy":"2022-10-14T00:29:33.559222Z","iopub.execute_input":"2022-10-14T00:29:33.559539Z","iopub.status.idle":"2022-10-14T00:29:35.318124Z","shell.execute_reply.started":"2022-10-14T00:29:33.559511Z","shell.execute_reply":"2022-10-14T00:29:35.317152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print (\"number of training examples = \" + str(train_X.shape[0]))\nprint (\"number of test examples = \" + str(val_X.shape[0]))\nprint (\"train_X shape: \" + str(train_X.shape))\nprint (\"train_y shape: \" + str(train_y.shape))\nprint (\"val_X shape: \" + str(val_X.shape))\nprint (\"val_y shape: \" + str(val_y.shape))","metadata":{"execution":{"iopub.status.busy":"2022-10-14T00:29:35.319272Z","iopub.execute_input":"2022-10-14T00:29:35.319581Z","iopub.status.idle":"2022-10-14T00:29:35.326707Z","shell.execute_reply.started":"2022-10-14T00:29:35.319553Z","shell.execute_reply":"2022-10-14T00:29:35.325619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define sequential model","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow.keras.layers as tfl","metadata":{"execution":{"iopub.status.busy":"2022-10-14T00:29:35.328162Z","iopub.execute_input":"2022-10-14T00:29:35.328594Z","iopub.status.idle":"2022-10-14T00:29:41.070919Z","shell.execute_reply.started":"2022-10-14T00:29:35.328554Z","shell.execute_reply":"2022-10-14T00:29:41.069962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def simple_model():\n    \"\"\"\n    Implements the forward propagation for the binary? classification model:\n    DENSE -> RELU -> DENSE -> SIGMOID\n    \n    Arguments:\n    None\n\n    Returns:\n    model -- TF Keras model (object containing the information for the entire training process) \n    \"\"\"\n    model = tf.keras.Sequential([\n        tfl.Dense(16, input_shape=(train_X.shape[1],)),\n        tfl.Dense(3, activation='sigmoid')\n        ])\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-10-14T00:29:41.072325Z","iopub.execute_input":"2022-10-14T00:29:41.072930Z","iopub.status.idle":"2022-10-14T00:29:41.078374Z","shell.execute_reply.started":"2022-10-14T00:29:41.072893Z","shell.execute_reply":"2022-10-14T00:29:41.077218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 1st way: use defined function\n#model = simple_model()","metadata":{"execution":{"iopub.status.busy":"2022-10-14T00:29:41.079941Z","iopub.execute_input":"2022-10-14T00:29:41.081133Z","iopub.status.idle":"2022-10-14T00:29:41.104543Z","shell.execute_reply.started":"2022-10-14T00:29:41.081090Z","shell.execute_reply":"2022-10-14T00:29:41.103315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 2nd way: use model add\nmodel = tf.keras.Sequential()\nmodel.add(tfl.Input(shape=(train_X.shape[1],)))\nmodel.add(tfl.Dense(256, activation='relu'))\n#model.add(tfl.Dense(256, input_shape=(train_X.shape[1],)))\nmodel.add(tfl.Dense(256, activation='relu'))\nmodel.add(tfl.Dense(256, activation='relu'))\nmodel.add(tfl.Dense(256, activation='relu'))\nmodel.add(tfl.Dense(256, activation='relu'))\nmodel.add(tfl.Dense(256, activation='relu'))\nmodel.add(tfl.Dense(256, activation='relu'))\nmodel.add(tfl.Dense(256, activation='relu'))\nmodel.add(tfl.Dense(256, activation='relu'))\nmodel.add(tfl.Dense(256, activation='relu'))\nmodel.add(tfl.Dense(256, activation='relu'))\nmodel.add(tfl.Dense(256, activation='relu'))\nmodel.add(tfl.Dense(256, activation='relu'))\nmodel.add(tfl.Dense(3, activation='sigmoid'))","metadata":{"execution":{"iopub.status.busy":"2022-10-14T00:30:50.077715Z","iopub.execute_input":"2022-10-14T00:30:50.078708Z","iopub.status.idle":"2022-10-14T00:30:50.204798Z","shell.execute_reply.started":"2022-10-14T00:30:50.078668Z","shell.execute_reply":"2022-10-14T00:30:50.203547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer='adam',\n                   loss='binary_crossentropy',\n                   metrics=['accuracy'])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-10-14T00:30:51.744263Z","iopub.execute_input":"2022-10-14T00:30:51.744667Z","iopub.status.idle":"2022-10-14T00:30:51.757662Z","shell.execute_reply.started":"2022-10-14T00:30:51.744636Z","shell.execute_reply":"2022-10-14T00:30:51.756273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train model","metadata":{}},{"cell_type":"code","source":"history = model.fit(x=train_X, y=train_y,\n                    epochs=30, batch_size=16,\n                    validation_data=(val_X, val_y))","metadata":{"execution":{"iopub.status.busy":"2022-10-14T00:30:55.308862Z","iopub.execute_input":"2022-10-14T00:30:55.309790Z","iopub.status.idle":"2022-10-14T00:31:37.621938Z","shell.execute_reply.started":"2022-10-14T00:30:55.309739Z","shell.execute_reply":"2022-10-14T00:31:37.620339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('/kaggle/working/trained_model_large')\nmodel.save_weights('/kaggle/working/trained_weights_large')","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:39:40.258622Z","iopub.execute_input":"2022-10-13T14:39:40.259111Z","iopub.status.idle":"2022-10-13T14:39:41.087234Z","shell.execute_reply.started":"2022-10-13T14:39:40.259078Z","shell.execute_reply":"2022-10-13T14:39:41.086154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model.fit(train_X, train_y, epochs=5, batch_size=16)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.evaluate(val_X, val_y)","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:41:32.168675Z","iopub.execute_input":"2022-10-13T14:41:32.170579Z","iopub.status.idle":"2022-10-13T14:41:41.097472Z","shell.execute_reply.started":"2022-10-13T14:41:32.170494Z","shell.execute_reply":"2022-10-13T14:41:41.096226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The history.history[\"loss\"] entry is a dictionary with as many values as epochs that the\n# model was trained on. \ndf_loss_acc = pd.DataFrame(history.history)\ndf_loss= df_loss_acc[['loss','val_loss']]\ndf_loss.rename(columns={'loss':'train','val_loss':'validation'},inplace=True)\ndf_acc= df_loss_acc[['accuracy','val_accuracy']]\ndf_acc.rename(columns={'accuracy':'train','val_accuracy':'validation'},inplace=True)\ndf_loss.plot(title='Model loss',figsize=(12,8)).set(xlabel='Epoch',ylabel='Loss')\ndf_acc.plot(title='Model Accuracy',figsize=(12,8)).set(xlabel='Epoch',ylabel='Accuracy')","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:39:39.616993Z","iopub.execute_input":"2022-10-13T14:39:39.617987Z","iopub.status.idle":"2022-10-13T14:39:40.257159Z","shell.execute_reply.started":"2022-10-13T14:39:39.617945Z","shell.execute_reply":"2022-10-13T14:39:40.255819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict over test set","metadata":{}},{"cell_type":"code","source":"dtypes_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/test_dtypes.csv')\ndtypes = {k: v for (k, v) in zip(dtypes_df.column, dtypes_df.dtype)}\ntest_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/test.csv', dtype=dtypes)\ntest_df","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:43:07.638563Z","iopub.execute_input":"2022-10-13T14:43:07.639056Z","iopub.status.idle":"2022-10-13T14:43:19.092971Z","shell.execute_reply.started":"2022-10-13T14:43:07.639017Z","shell.execute_reply":"2022-10-13T14:43:19.091848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/sample_submission.csv', dtype=dtypes)\nsample_sub_df","metadata":{"execution":{"iopub.status.busy":"2022-10-11T15:20:15.289688Z","iopub.execute_input":"2022-10-11T15:20:15.290044Z","iopub.status.idle":"2022-10-11T15:20:15.401255Z","shell.execute_reply.started":"2022-10-11T15:20:15.290013Z","shell.execute_reply":"2022-10-11T15:20:15.400125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = test_df[features]\nX_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:43:19.095038Z","iopub.execute_input":"2022-10-13T14:43:19.095395Z","iopub.status.idle":"2022-10-13T14:43:19.142860Z","shell.execute_reply.started":"2022-10-13T14:43:19.095347Z","shell.execute_reply":"2022-10-13T14:43:19.141737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test = model.predict(X_test)\ny_test","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:43:19.144321Z","iopub.execute_input":"2022-10-13T14:43:19.145503Z","iopub.status.idle":"2022-10-13T14:43:38.652627Z","shell.execute_reply.started":"2022-10-13T14:43:19.145462Z","shell.execute_reply":"2022-10-13T14:43:38.650846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:43:38.656031Z","iopub.execute_input":"2022-10-13T14:43:38.656570Z","iopub.status.idle":"2022-10-13T14:43:38.666070Z","shell.execute_reply.started":"2022-10-13T14:43:38.656520Z","shell.execute_reply":"2022-10-13T14:43:38.664889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_sub_df = pd.DataFrame({'Id': test_df.id,\n                          'team_A_scoring_within_10sec': y_test[:, 0],\n                          'team_B_scoring_within_10sec': y_test[:, 1]})\nmy_sub_df","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:43:38.668239Z","iopub.execute_input":"2022-10-13T14:43:38.669178Z","iopub.status.idle":"2022-10-13T14:43:38.694542Z","shell.execute_reply.started":"2022-10-13T14:43:38.669123Z","shell.execute_reply":"2022-10-13T14:43:38.693407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_sub_df.fillna(0, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:43:38.695890Z","iopub.execute_input":"2022-10-13T14:43:38.696338Z","iopub.status.idle":"2022-10-13T14:43:38.707222Z","shell.execute_reply.started":"2022-10-13T14:43:38.696304Z","shell.execute_reply":"2022-10-13T14:43:38.705753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_sub_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-10-13T14:43:38.708519Z","iopub.execute_input":"2022-10-13T14:43:38.709111Z","iopub.status.idle":"2022-10-13T14:43:40.860421Z","shell.execute_reply.started":"2022-10-13T14:43:38.709069Z","shell.execute_reply":"2022-10-13T14:43:40.858790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}