{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 🌵 A Simple Keras Model...\nI will build a simple neuronal network with minimun amount of data to use the model for feature engineering development and improvement in performance...\n\n### Notebook Ideas...\n* Start simple and add complexity...\n* Keep well documented code...\n* Complete a simple end to end model...\n* Submit predictions and continue building improvements...","metadata":{}},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"## 1.0 Loading Libraries...","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-30T02:51:33.371482Z","iopub.execute_input":"2022-10-30T02:51:33.372283Z","iopub.status.idle":"2022-10-30T02:51:33.400104Z","shell.execute_reply.started":"2022-10-30T02:51:33.372184Z","shell.execute_reply":"2022-10-30T02:51:33.399282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading other important libraries\nimport matplotlib.pyplot as plt # Import visualization library\nimport seaborn as sns # Import visualization library","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:51:33.405998Z","iopub.execute_input":"2022-10-30T02:51:33.406353Z","iopub.status.idle":"2022-10-30T02:51:33.916379Z","shell.execute_reply.started":"2022-10-30T02:51:33.406320Z","shell.execute_reply":"2022-10-30T02:51:33.915459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"## 2.0 Notebook Configurations...","metadata":{}},{"cell_type":"code","source":"%%time\n# I like to disable my Notebook Warnings.\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:51:33.917959Z","iopub.execute_input":"2022-10-30T02:51:33.918330Z","iopub.status.idle":"2022-10-30T02:51:33.927672Z","shell.execute_reply.started":"2022-10-30T02:51:33.918294Z","shell.execute_reply":"2022-10-30T02:51:33.923710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Notebook Configuration.\n\n# Amount of data we want to load into the model from Pandas.\nDATA_ROWS = None\n\n# Memory and replicability\nDATA_PCT = 0.35 # Load only 50% of the dataset to avoid memory issues...\nSEED = 777 # This will be the seed utilized across the notebook\n\n# Dataframe, the amount of rows and cols to visualize.\nNROWS = 20\nNCOLS = 15\n\n# Main data location base path.\nBASE_PATH = '...'\n\n# Model development parameters\nTEST_PCT = 0.20\nDATA_PCT = 0.20","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:51:33.931334Z","iopub.execute_input":"2022-10-30T02:51:33.931638Z","iopub.status.idle":"2022-10-30T02:51:33.939106Z","shell.execute_reply.started":"2022-10-30T02:51:33.931611Z","shell.execute_reply":"2022-10-30T02:51:33.937982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Configure notebook display settings to only use 2 decimal places, tables look nicer and compressed.\npd.options.display.float_format = '{:,.3f}'.format\npd.set_option('display.max_columns', NCOLS) \npd.set_option('display.max_rows', NROWS)","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:51:33.940715Z","iopub.execute_input":"2022-10-30T02:51:33.941248Z","iopub.status.idle":"2022-10-30T02:51:33.948562Z","shell.execute_reply.started":"2022-10-30T02:51:33.941196Z","shell.execute_reply":"2022-10-30T02:51:33.947624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"## 3.0 Loading the Datasets","metadata":{}},{"cell_type":"code","source":"%%time\n# Load the Test Datset utilizing the provided dtype arrays for memory eficiency\n\ndtypes_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/test_dtypes.csv')\ndtypes = {k: v for (k, v) in zip(dtypes_df.column, dtypes_df.dtype)}","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:51:33.950130Z","iopub.execute_input":"2022-10-30T02:51:33.950924Z","iopub.status.idle":"2022-10-30T02:51:33.972649Z","shell.execute_reply.started":"2022-10-30T02:51:33.950889Z","shell.execute_reply":"2022-10-30T02:51:33.971559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nTRN_PATH = '/kaggle/input/tabular-playground-series-oct-2022/train_1.csv'\ntrn_data = pd.read_csv(TRN_PATH, dtype = dtypes)","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:51:33.974098Z","iopub.execute_input":"2022-10-30T02:51:33.974742Z","iopub.status.idle":"2022-10-30T02:51:59.845179Z","shell.execute_reply.started":"2022-10-30T02:51:33.974707Z","shell.execute_reply":"2022-10-30T02:51:59.844008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nTST_PATH = '/kaggle/input/tabular-playground-series-oct-2022/test.csv'\ntst_data = pd.read_csv(TST_PATH, dtype = dtypes)","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:51:59.846988Z","iopub.execute_input":"2022-10-30T02:51:59.847403Z","iopub.status.idle":"2022-10-30T02:52:08.081100Z","shell.execute_reply.started":"2022-10-30T02:51:59.847361Z","shell.execute_reply":"2022-10-30T02:52:08.080028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Create\nSUB_PATH = '/kaggle/input/tabular-playground-series-oct-2022/sample_submission.csv'\nsubmission = pd.read_csv(SUB_PATH)","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:52:08.082773Z","iopub.execute_input":"2022-10-30T02:52:08.083452Z","iopub.status.idle":"2022-10-30T02:52:08.250622Z","shell.execute_reply.started":"2022-10-30T02:52:08.083410Z","shell.execute_reply":"2022-10-30T02:52:08.249454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"## 4.0 Exploring the Datasets...","metadata":{}},{"cell_type":"code","source":"%%time\n# Display some basic dataset information, the most important is the memory usage...\n\ntrn_data.info(verbose=False)","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:52:08.252469Z","iopub.execute_input":"2022-10-30T02:52:08.253187Z","iopub.status.idle":"2022-10-30T02:52:08.274965Z","shell.execute_reply.started":"2022-10-30T02:52:08.253143Z","shell.execute_reply":"2022-10-30T02:52:08.273547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Display information of the first 5 rows in the datset...\n\ntrn_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:52:08.276561Z","iopub.execute_input":"2022-10-30T02:52:08.276936Z","iopub.status.idle":"2022-10-30T02:52:08.307301Z","shell.execute_reply.started":"2022-10-30T02:52:08.276899Z","shell.execute_reply":"2022-10-30T02:52:08.306375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Display advanced statistics of the dataset...\n\ntrn_data.describe() ","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:52:08.308981Z","iopub.execute_input":"2022-10-30T02:52:08.309384Z","iopub.status.idle":"2022-10-30T02:52:13.560246Z","shell.execute_reply.started":"2022-10-30T02:52:08.309350Z","shell.execute_reply":"2022-10-30T02:52:13.558647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Display the number of NaNs in each column of the dataset...\n\ntrn_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:52:13.564770Z","iopub.execute_input":"2022-10-30T02:52:13.565068Z","iopub.status.idle":"2022-10-30T02:52:13.860011Z","shell.execute_reply.started":"2022-10-30T02:52:13.565040Z","shell.execute_reply":"2022-10-30T02:52:13.858815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"## 5.0 Feature Engineering...\nHello, at this point I don't have any feature engineering built in the Notebbok, first I like to try how the feature provided will perform... ","metadata":{}},{"cell_type":"code","source":"# ...\n# ...","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:52:13.861642Z","iopub.execute_input":"2022-10-30T02:52:13.862237Z","iopub.status.idle":"2022-10-30T02:52:13.866905Z","shell.execute_reply.started":"2022-10-30T02:52:13.862198Z","shell.execute_reply":"2022-10-30T02:52:13.865726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"## 6.0 Preparing the Datasets for the Model...","metadata":{}},{"cell_type":"code","source":"%%time\n# The type of ML model I'm currenthly using can deal with NaNs\n\nDEFAULT_VALUE = -1000\ntrn_data = trn_data.fillna(value = DEFAULT_VALUE)\ntst_data = tst_data.fillna(value = DEFAULT_VALUE)","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:52:13.868288Z","iopub.execute_input":"2022-10-30T02:52:13.869374Z","iopub.status.idle":"2022-10-30T02:52:14.392473Z","shell.execute_reply.started":"2022-10-30T02:52:13.869336Z","shell.execute_reply":"2022-10-30T02:52:14.391411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# ...\n\n# List of featuare that needs to be avoided...\nskip = ['game_num', \n        'event_id', \n        'event_time', \n        'player_scoring_next', \n        'team_scoring_next', \n        'team_A_scoring_within_10sec',\n        'team_B_scoring_within_10sec'\n       ]\n\n# Using a list comprehension we generate a list of features to train the model...\nfeatures = [feat for feat in trn_data.columns if feat not in skip]\n\nlabel_a = 'team_A_scoring_within_10sec'\nlabel_b = 'team_B_scoring_within_10sec'","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:52:14.394288Z","iopub.execute_input":"2022-10-30T02:52:14.394917Z","iopub.status.idle":"2022-10-30T02:52:14.402387Z","shell.execute_reply.started":"2022-10-30T02:52:14.394873Z","shell.execute_reply":"2022-10-30T02:52:14.401203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Scale the train and test datasets to improve the learning capabilities of the model.\n# Modified to be during the training stage.\n\n# scaler = StandardScaler()\n# trn_data[features] = scaler.fit_transform(trn_data[features])","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:52:14.403667Z","iopub.execute_input":"2022-10-30T02:52:14.403929Z","iopub.status.idle":"2022-10-30T02:52:14.413440Z","shell.execute_reply.started":"2022-10-30T02:52:14.403905Z","shell.execute_reply":"2022-10-30T02:52:14.412191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"## 7.0 Baseline Modeling, Using a Keras Classifier...","metadata":{}},{"cell_type":"markdown","source":"### 7.1 Importing Tenforflow Libraries","metadata":{}},{"cell_type":"code","source":"%%time\nimport tensorflow as tf\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, LearningRateScheduler, EarlyStopping\nfrom tensorflow.keras.layers import Dense, Input, InputLayer, Add, BatchNormalization, Dropout, Concatenate","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:52:14.414568Z","iopub.execute_input":"2022-10-30T02:52:14.414845Z","iopub.status.idle":"2022-10-30T02:52:19.231737Z","shell.execute_reply.started":"2022-10-30T02:52:14.414819Z","shell.execute_reply":"2022-10-30T02:52:19.230688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom sklearn.preprocessing import StandardScaler, RobustScaler, MinMaxScaler\nfrom sklearn.model_selection import KFold, StratifiedKFold, GroupKFold\nfrom sklearn.metrics import roc_auc_score, roc_curve, accuracy_score","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:52:19.234318Z","iopub.execute_input":"2022-10-30T02:52:19.235205Z","iopub.status.idle":"2022-10-30T02:52:19.334334Z","shell.execute_reply.started":"2022-10-30T02:52:19.235163Z","shell.execute_reply":"2022-10-30T02:52:19.333248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nimport datetime\nimport random\nimport math","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:52:19.335662Z","iopub.execute_input":"2022-10-30T02:52:19.336313Z","iopub.status.idle":"2022-10-30T02:52:19.342022Z","shell.execute_reply.started":"2022-10-30T02:52:19.336271Z","shell.execute_reply":"2022-10-30T02:52:19.340966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 7.2 Defining Model Architecture","metadata":{}},{"cell_type":"code","source":"%%time\ndef nn_model_one():\n    \n    '''\n    Function to define the Neuronal Network architecture...\n    '''\n    \n    L2 = 65e-6\n    dropout_value = 0.05\n    activation_func = 'swish'\n    inputs = Input(shape = (len(features)))\n    \n    x0 = Dense(256, kernel_regularizer = tf.keras.regularizers.l2(L2), activation = activation_func)(inputs)\n    x0 = BatchNormalization()(x0)\n    x0 = Dropout(dropout_value)(x0)\n    \n    x1 = Dense(256, kernel_regularizer = tf.keras.regularizers.l2(L2), activation = activation_func)(x0)\n    x1 = BatchNormalization()(x1)\n    x1 = Dropout(dropout_value)(x1)\n    \n    x1 = Dense(64,  kernel_regularizer = tf.keras.regularizers.l2(L2), activation = activation_func)(x1)\n    x1 = Concatenate()([x1, x0])\n    x1 = BatchNormalization()(x1)\n    x1 = Dropout(dropout_value)(x1)\n    \n    x1 = Dense(32, kernel_regularizer = tf.keras.regularizers.l2(L2), activation = activation_func)(x1)\n    x1 = BatchNormalization()(x1)\n    x1 = Dropout(dropout_value)(x1)\n    \n    x1 = Dense(1,  \n               #kernel_regularizer = tf.keras.regularizers.l2(4e-4), \n               activation = 'sigmoid')(x1)\n    \n    model = Model(inputs, x1)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:52:19.343679Z","iopub.execute_input":"2022-10-30T02:52:19.344395Z","iopub.status.idle":"2022-10-30T02:52:19.356267Z","shell.execute_reply.started":"2022-10-30T02:52:19.344359Z","shell.execute_reply":"2022-10-30T02:52:19.355211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 7.3 Visualizing Model Architecture","metadata":{}},{"cell_type":"code","source":"%%time\narchitecture = nn_model_one()\narchitecture.summary()","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:52:19.357708Z","iopub.execute_input":"2022-10-30T02:52:19.358111Z","iopub.status.idle":"2022-10-30T02:52:22.146750Z","shell.execute_reply.started":"2022-10-30T02:52:19.358076Z","shell.execute_reply":"2022-10-30T02:52:22.145745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 7.4 Defining Model Parameters","metadata":{}},{"cell_type":"code","source":"%%time\n# Defining model parameters...\nBATCH_SIZE         = 256\n\nEPOCHS             = 5\nEPOCHS_COSINEDECAY = 5\n\nDIAGRAMS           = True\nUSE_PLATEAU        = True\nINFERENCE          = False\nVERBOSE            = 1 \n\nTARGET             = label_a","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:52:22.148283Z","iopub.execute_input":"2022-10-30T02:52:22.149184Z","iopub.status.idle":"2022-10-30T02:52:22.156459Z","shell.execute_reply.started":"2022-10-30T02:52:22.149142Z","shell.execute_reply":"2022-10-30T02:52:22.155296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 7.5 Defining Model Train Function","metadata":{}},{"cell_type":"code","source":"%%time\n# Defining model training function...\n\ndef fit_model(X_train, y_train, X_val, y_val, run = 0):\n    '''\n    This function train an NN model...\n    '''\n    \n    lr_start = 0.2 # Initial value for learning rate...\n\n    start_time = datetime.datetime.now()\n    scaler = StandardScaler()\n    \n    X_train[features] = scaler.fit_transform(X_train[features])\n\n    epochs = EPOCHS    \n    lr = ReduceLROnPlateau(monitor = 'val_loss', factor = 0.1, patience = 8, verbose = VERBOSE)\n    es = EarlyStopping(monitor = 'val_loss',patience = 16, verbose = 1, mode = 'min', restore_best_weights = True)\n    tm = tf.keras.callbacks.TerminateOnNaN()\n    callbacks = [lr, es, tm]\n    \n    # Cosine Learning Rate Decay\n    if USE_PLATEAU == False:\n        epochs = EPOCHS_COSINEDECAY\n        lr_end = 0.0002\n\n        def cosine_decay(epoch):\n            if epochs > 1:\n                w = (1 + math.cos(epoch / (epochs - 1) * math.pi)) / 2\n            else:\n                w = 1\n            return w * lr_start + (1 - w) * lr_end\n        \n        lr = LearningRateScheduler(cosine_decay, verbose = 0)\n        callbacks = [lr, tm]\n        \n    model = nn_model_one()\n    \n    optimizer_func = tf.keras.optimizers.Adam(learning_rate = lr_start)\n    loss_func = tf.keras.losses.BinaryCrossentropy()\n    model.compile(optimizer = optimizer_func, loss = loss_func)\n    \n    X_val[features] = scaler.transform(X_val[features])\n    validation_data = (X_val, y_val)\n    \n    history = model.fit(X_train, \n                        y_train, \n                        validation_data = validation_data, \n                        epochs          = epochs,\n                        verbose         = VERBOSE,\n                        batch_size      = BATCH_SIZE,\n                        shuffle         = True,\n                        callbacks       = callbacks\n                       )\n    \n    history_list.append(history.history)\n    print(f'Training loss:{history_list[-1][\"loss\"][-1]:.3f}')\n    callbacks, es, lr, tm, history = None, None, None, None, None\n    \n    \n    y_val_pred_proba = model.predict(X_val, batch_size = BATCH_SIZE, verbose = VERBOSE)\n    y_val_pred = [1 if x > 0.5 else 0 for x in y_val_pred_proba]\n    \n    acc_score = accuracy_score(y_val, y_val_pred)\n    auc_score = roc_auc_score(y_val, y_val_pred_proba)\n    \n    print(f'Fold {run}.{fold} | {str(datetime.datetime.now() - start_time)[-12:-7]}'\n          f' | ACC: {acc_score:.5f} | AUC: {auc_score:.5f}')\n    \n    auc_score_list.append(auc_score)\n    \n    tst_data_scaled = tst_data.copy(deep = True)\n    tst_data_scaled[features] = scaler.transform(tst_data_scaled[features])\n    tst_pred = model.predict(tst_data_scaled[features])\n    predictions.append(tst_pred)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:52:22.158424Z","iopub.execute_input":"2022-10-30T02:52:22.159157Z","iopub.status.idle":"2022-10-30T02:52:22.173689Z","shell.execute_reply.started":"2022-10-30T02:52:22.159118Z","shell.execute_reply":"2022-10-30T02:52:22.172595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 7.6 Defining a Cross Validation Function & Training the Model","metadata":{}},{"cell_type":"code","source":"%%time\n# Create empty lists to store NN information...\nTARGET = label_a\n\nhistory_list = []\nauc_score_list = []\npredictions  = []\n\n# Define kfolds for training purposes...\n#kf = KFold(n_splits = 5)\nkf = StratifiedKFold(n_splits = 5)\n\n#kf = GroupKFold(n_splits = 5)\nfor fold, (trn_idx, val_idx) in enumerate(kf.split(trn_data, trn_data[TARGET])):\n    x_train, x_val = trn_data.iloc[trn_idx][features], trn_data.iloc[val_idx][features]\n    y_train, y_val = trn_data.iloc[trn_idx][TARGET], trn_data.iloc[val_idx][TARGET]\n    \n    fit_model(x_train, y_train, x_val, y_val)\n    \nprint(f'OOF AUC: {np.mean(auc_score_list):.5f}')","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:52:22.176030Z","iopub.execute_input":"2022-10-30T02:52:22.176613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['team_A_scoring_within_10sec'] = np.mean(predictions, axis = 0)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Create empty lists to store NN information...\nTARGET = label_b\n\nhistory_list = []\nauc_score_list = []\npredictions  = []\n\n# Define kfolds for training purposes...\n#kf = KFold(n_splits = 5)\nkf = StratifiedKFold(n_splits = 5)\n\n#kf = GroupKFold(n_splits = 5)\nfor fold, (trn_idx, val_idx) in enumerate(kf.split(trn_data, trn_data[TARGET])):\n    x_train, x_val = trn_data.iloc[trn_idx][features], trn_data.iloc[val_idx][features]\n    y_train, y_val = trn_data.iloc[trn_idx][TARGET], trn_data.iloc[val_idx][TARGET]\n    \n    fit_model(x_train, y_train, x_val, y_val)\n    \nprint(f'OOF AUC: {np.mean(auc_score_list):.5f}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['team_B_scoring_within_10sec'] = np.mean(predictions, axis = 0)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"## 8.0 Creating Submission Files.","metadata":{}},{"cell_type":"code","source":"%%time\n# Creates a kaggle submission file...\nsubmission.to_csv('submission_10292022.csv', index = False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}