{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Importing Libraries and Loading datasets","metadata":{}},{"cell_type":"code","source":"import os\nimport random\nimport numpy as np\nimport pandas as pd\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nimport tensorflow as tf\ntf.config.threading.set_intra_op_parallelism_threads(6)\ntf.config.threading.set_inter_op_parallelism_threads(2)\n\nfrom tensorflow import keras\nfrom tensorflow.keras import layers, callbacks\n\n!git clone https://github.com/analokmaus/kuma_utils.git\nimport sys; sys.path.append(\"kuma_utils/\")\nfrom kuma_utils.preprocessing.imputer import LGBMImputer\n\nfrom sklearn.preprocessing import LabelEncoder\n\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.feature_selection import mutual_info_classif\n\nfrom sklearn.linear_model import LogisticRegression","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-14T04:12:52.993432Z","iopub.execute_input":"2022-08-14T04:12:52.994715Z","iopub.status.idle":"2022-08-14T04:12:54.159748Z","shell.execute_reply.started":"2022-08-14T04:12:52.994660Z","shell.execute_reply":"2022-08-14T04:12:54.158083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"../input/tabular-playground-series-aug-2022/train.csv\", index_col='id')\ntest = pd.read_csv(\"../input/tabular-playground-series-aug-2022/test.csv\", index_col='id')\nsub = pd.read_csv(\"../input/tabular-playground-series-aug-2022/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:12:54.162684Z","iopub.execute_input":"2022-08-14T04:12:54.163112Z","iopub.status.idle":"2022-08-14T04:12:54.403293Z","shell.execute_reply.started":"2022-08-14T04:12:54.163073Z","shell.execute_reply":"2022-08-14T04:12:54.402028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Explore Data","metadata":{}},{"cell_type":"code","source":"train.head()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-14T04:12:54.404750Z","iopub.execute_input":"2022-08-14T04:12:54.405194Z","iopub.status.idle":"2022-08-14T04:12:54.440023Z","shell.execute_reply.started":"2022-08-14T04:12:54.405155Z","shell.execute_reply":"2022-08-14T04:12:54.439032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-14T04:12:54.443282Z","iopub.execute_input":"2022-08-14T04:12:54.443845Z","iopub.status.idle":"2022-08-14T04:12:54.561391Z","shell.execute_reply.started":"2022-08-14T04:12:54.443790Z","shell.execute_reply":"2022-08-14T04:12:54.560470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Columns: \\n{0}\".format(list(train.columns)))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-14T04:12:54.562397Z","iopub.execute_input":"2022-08-14T04:12:54.562714Z","iopub.status.idle":"2022-08-14T04:12:54.568164Z","shell.execute_reply.started":"2022-08-14T04:12:54.562686Z","shell.execute_reply":"2022-08-14T04:12:54.567181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Basic Data Check","metadata":{}},{"cell_type":"code","source":"print('Train data shape:', train.shape)\nprint('Test data shape:', test.shape)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-14T04:12:54.569151Z","iopub.execute_input":"2022-08-14T04:12:54.569467Z","iopub.status.idle":"2022-08-14T04:12:54.582834Z","shell.execute_reply.started":"2022-08-14T04:12:54.569439Z","shell.execute_reply":"2022-08-14T04:12:54.581535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Missing values","metadata":{}},{"cell_type":"code","source":"missing_values_train = train.isna().sum().sum()\nprint('Missing values in train data: {0}'.format(missing_values_train[missing_values_train > 0]))\n\nmissing_values_test = test.isna().sum().sum()\nprint('Missing values in test data: {0}'.format(missing_values_test[missing_values_test > 0]))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-14T04:12:54.585099Z","iopub.execute_input":"2022-08-14T04:12:54.586323Z","iopub.status.idle":"2022-08-14T04:12:54.609266Z","shell.execute_reply.started":"2022-08-14T04:12:54.586270Z","shell.execute_reply":"2022-08-14T04:12:54.608158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Duplicates","metadata":{}},{"cell_type":"code","source":"duplicates_train = train.duplicated().sum()\nprint('Duplicates in train data: {0}'.format(duplicates_train))\n\nduplicates_test = test.duplicated().sum()\nprint('Duplicates in test data: {0}'.format(duplicates_test))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-14T04:12:54.611234Z","iopub.execute_input":"2022-08-14T04:12:54.611782Z","iopub.status.idle":"2022-08-14T04:12:54.698217Z","shell.execute_reply.started":"2022-08-14T04:12:54.611717Z","shell.execute_reply":"2022-08-14T04:12:54.696982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Features","metadata":{}},{"cell_type":"code","source":"train_data = train.drop('failure', axis=1).copy()\ntest_data = test.copy()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:12:54.699833Z","iopub.execute_input":"2022-08-14T04:12:54.700202Z","iopub.status.idle":"2022-08-14T04:12:54.715295Z","shell.execute_reply.started":"2022-08-14T04:12:54.700169Z","shell.execute_reply":"2022-08-14T04:12:54.714004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Numerical Features","metadata":{}},{"cell_type":"code","source":"numerical_cols = train_data.select_dtypes(np.number).columns.values.tolist()\ntrain_data[numerical_cols].head()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-14T04:12:54.718852Z","iopub.execute_input":"2022-08-14T04:12:54.719614Z","iopub.status.idle":"2022-08-14T04:12:54.755397Z","shell.execute_reply.started":"2022-08-14T04:12:54.719515Z","shell.execute_reply":"2022-08-14T04:12:54.754012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"There are {} numerical columns.\".format(len(numerical_cols)))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-14T04:12:54.757600Z","iopub.execute_input":"2022-08-14T04:12:54.758257Z","iopub.status.idle":"2022-08-14T04:12:54.764049Z","shell.execute_reply.started":"2022-08-14T04:12:54.758220Z","shell.execute_reply":"2022-08-14T04:12:54.762904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Categorical Features","metadata":{}},{"cell_type":"code","source":"categorical_cols = [x for x in train_data.columns.values if (x not in numerical_cols)]\ntrain_data[categorical_cols].head()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-14T04:12:54.766128Z","iopub.execute_input":"2022-08-14T04:12:54.766610Z","iopub.status.idle":"2022-08-14T04:12:54.785509Z","shell.execute_reply.started":"2022-08-14T04:12:54.766575Z","shell.execute_reply":"2022-08-14T04:12:54.784397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"There are {} categorical columns.\".format(len(categorical_cols)))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-14T04:12:54.787701Z","iopub.execute_input":"2022-08-14T04:12:54.788472Z","iopub.status.idle":"2022-08-14T04:12:54.795340Z","shell.execute_reply.started":"2022-08-14T04:12:54.788421Z","shell.execute_reply":"2022-08-14T04:12:54.793996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Target Distribution","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nplt.title('Target distribution')\nax = sns.countplot(x=train['failure'], data=train)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-14T04:12:54.796798Z","iopub.execute_input":"2022-08-14T04:12:54.797536Z","iopub.status.idle":"2022-08-14T04:12:55.023904Z","shell.execute_reply.started":"2022-08-14T04:12:54.797499Z","shell.execute_reply":"2022-08-14T04:12:55.022973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"markdown","source":"## Use missing values as a feature\n\nCredits to https://www.kaggle.com/competitions/tabular-playground-series-aug-2022/discussion/342319586078","metadata":{}},{"cell_type":"code","source":"def use_missing_values(column):\n    new_column = column + '_missing'\n    train_data[new_column] = train_data[column].isna() * 1\n    test_data[new_column] = test_data[column].isna() * 1\nfor column in numerical_cols:\n    use_missing_values(column)\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:12:55.025302Z","iopub.execute_input":"2022-08-14T04:12:55.026007Z","iopub.status.idle":"2022-08-14T04:12:55.090281Z","shell.execute_reply.started":"2022-08-14T04:12:55.025970Z","shell.execute_reply":"2022-08-14T04:12:55.089045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fill Missing Values","metadata":{}},{"cell_type":"code","source":"imputer = LGBMImputer(n_iter=50)\nimputer.fit(train_data[numerical_cols].append(test_data[numerical_cols]))\ntrain_data[numerical_cols] = imputer.transform(train_data[numerical_cols])\ntest_data[numerical_cols] = imputer.transform(test_data[numerical_cols])","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-14T04:12:55.092038Z","iopub.execute_input":"2022-08-14T04:12:55.092495Z","iopub.status.idle":"2022-08-14T04:13:02.379370Z","shell.execute_reply.started":"2022-08-14T04:12:55.092463Z","shell.execute_reply":"2022-08-14T04:13:02.378074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-14T04:13:02.383788Z","iopub.execute_input":"2022-08-14T04:13:02.384914Z","iopub.status.idle":"2022-08-14T04:13:02.412084Z","shell.execute_reply.started":"2022-08-14T04:13:02.384863Z","shell.execute_reply":"2022-08-14T04:13:02.410741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_values_train = train_data.isna().any().sum()\nprint('Missing values in train data: {0}'.format(missing_values_train[missing_values_train > 0]))\n\nmissing_values_test = test_data.isna().any().sum()\nprint('Missing values in test data: {0}'.format(missing_values_test[missing_values_test > 0]))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-14T04:13:02.414103Z","iopub.execute_input":"2022-08-14T04:13:02.415003Z","iopub.status.idle":"2022-08-14T04:13:02.436851Z","shell.execute_reply.started":"2022-08-14T04:13:02.414948Z","shell.execute_reply":"2022-08-14T04:13:02.435401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Engineering\n\nCredits to https://www.kaggle.com/code/themikejones/tps-aug-22-votingclassifier/notebook?scriptVersionId=102761965#4.1-Combine-features-and-create-new","metadata":{}},{"cell_type":"code","source":"def feature_engineering(data, numerical_cols):\n    data['attribute_2*3'] = data['attribute_2'] * data['attribute_3']\n    numerical_cols = numerical_cols + ['attribute_2*3']\n    \n    meas_gr1_cols = [f\"measurement_{i:d}\" for i in list(range(3, 5)) + list(range(9, 18))]\n    data['meas_gr1_avg'] = np.mean(data[meas_gr1_cols], axis=1)\n    numerical_cols = numerical_cols + ['meas_gr1_avg']\n    data['meas_gr1_std'] = np.std(data[meas_gr1_cols], axis=1)\n    numerical_cols = numerical_cols + ['meas_gr1_std']\n\n    meas_gr2_cols = [f\"measurement_{i:d}\" for i in list(range(5, 9))]\n    data['meas_gr2_avg'] = np.mean(data[meas_gr2_cols], axis=1)\n    numerical_cols = numerical_cols + ['meas_gr2_avg']\n\n    data['meas17/meas_gr2_avg'] = data['measurement_17'] / data['meas_gr2_avg']\n    numerical_cols = numerical_cols + ['meas17/meas_gr2_avg']\n    return numerical_cols\nfeature_engineering(train_data, numerical_cols)\nnumerical_cols = feature_engineering(test_data, numerical_cols)\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:13:02.438918Z","iopub.execute_input":"2022-08-14T04:13:02.439445Z","iopub.status.idle":"2022-08-14T04:13:02.522967Z","shell.execute_reply.started":"2022-08-14T04:13:02.439400Z","shell.execute_reply":"2022-08-14T04:13:02.521522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Encoding","metadata":{}},{"cell_type":"code","source":"for column in categorical_cols:\n    label_encoder = LabelEncoder()\n    label_encoder.fit(train_data[column].append(test_data[column]))\n    train_data[column] = label_encoder.transform(train_data[column])\n    test_data[column] = label_encoder.transform(test_data[column])\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:13:02.526965Z","iopub.execute_input":"2022-08-14T04:13:02.528101Z","iopub.status.idle":"2022-08-14T04:13:02.598305Z","shell.execute_reply.started":"2022-08-14T04:13:02.528055Z","shell.execute_reply":"2022-08-14T04:13:02.597291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature selection","metadata":{}},{"cell_type":"code","source":"X = train_data.copy()\ny = train.failure.copy()\ntest_X = test_data.copy()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:13:30.168177Z","iopub.execute_input":"2022-08-14T04:13:30.169242Z","iopub.status.idle":"2022-08-14T04:13:30.179073Z","shell.execute_reply.started":"2022-08-14T04:13:30.169195Z","shell.execute_reply":"2022-08-14T04:13:30.177942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_mi_scores(mi_scores, X, y):\n    mi_scores = pd.Series(mi_scores, name=\"MI Scores\")\n    mi_scores = mi_scores.sort_values(ascending=False)\n    return mi_scores\n\ndef plot_mi_scores(scores, X):\n    scores = scores.sort_values(ascending=True)\n    width = np.arange(len(scores))\n    ticks = X.columns[scores.index]\n    plt.barh(width, scores)\n    plt.yticks(width, ticks)\n    plt.title(\"Mutual Information Scores\")\n    return ticks","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:13:31.036761Z","iopub.execute_input":"2022-08-14T04:13:31.037941Z","iopub.status.idle":"2022-08-14T04:13:31.044768Z","shell.execute_reply.started":"2022-08-14T04:13:31.037880Z","shell.execute_reply":"2022-08-14T04:13:31.043496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mi_scores = mutual_info_classif(X, y, random_state=1)\nmi_scores_classif = make_mi_scores(mi_scores, X, y)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:13:32.631702Z","iopub.execute_input":"2022-08-14T04:13:32.632131Z","iopub.status.idle":"2022-08-14T04:13:38.930152Z","shell.execute_reply.started":"2022-08-14T04:13:32.632094Z","shell.execute_reply":"2022-08-14T04:13:38.928987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(dpi=100, figsize=(12, 8))\ncolumns = plot_mi_scores(mi_scores_classif[mi_scores_classif > 1e-3], X)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:13:38.932096Z","iopub.execute_input":"2022-08-14T04:13:38.932459Z","iopub.status.idle":"2022-08-14T04:13:39.284530Z","shell.execute_reply.started":"2022-08-14T04:13:38.932427Z","shell.execute_reply":"2022-08-14T04:13:39.283449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modelling","metadata":{}},{"cell_type":"code","source":"N_SPLITS = 5\n\nmy_seed = 1\ndef seedAll(seed):\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n    random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\nseedAll(my_seed)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:14:38.457159Z","iopub.execute_input":"2022-08-14T04:14:38.458587Z","iopub.status.idle":"2022-08-14T04:14:38.465464Z","shell.execute_reply.started":"2022-08-14T04:14:38.458533Z","shell.execute_reply":"2022-08-14T04:14:38.463846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_data[columns].copy()\ny = train.failure.copy()\ntest_X = test_data[columns].copy()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:14:43.645811Z","iopub.execute_input":"2022-08-14T04:14:43.646263Z","iopub.status.idle":"2022-08-14T04:14:43.657906Z","shell.execute_reply.started":"2022-08-14T04:14:43.646227Z","shell.execute_reply":"2022-08-14T04:14:43.656768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Logistic Regression","metadata":{}},{"cell_type":"code","source":"scores = []\ntest_predictions = []\ncv = StratifiedKFold(n_splits=N_SPLITS, random_state=my_seed, shuffle=True)\nfor fold, (train_idx, test_idx) in enumerate(cv.split(X, y)):\n    train_X, val_X = X.iloc[train_idx], X.iloc[test_idx]\n    train_y, val_y = y.iloc[train_idx], y.iloc[test_idx]\n    \n    model = LogisticRegression(C=0.0001, penalty='l2', solver='newton-cg')\n    model.fit(train_X, train_y)\n    \n    predictions = model.predict_proba(val_X)[:, 1]\n    score = roc_auc_score(val_y, predictions)\n    scores.append(score)\n    print(f\"Fold {fold + 1} \\t\\t AUC: {score}\")\n    \n    test_predictions.append(model.predict_proba(test_X)[:, 1])\nprint('Overall AUC: ', np.mean(scores))","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:14:44.823253Z","iopub.execute_input":"2022-08-14T04:14:44.824068Z","iopub.status.idle":"2022-08-14T04:14:48.170961Z","shell.execute_reply.started":"2022-08-14T04:14:44.824027Z","shell.execute_reply":"2022-08-14T04:14:48.169273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub['failure'] = np.mean(test_predictions, axis=0)\nsub.to_csv('submission_lr.csv', index=False)\nsub","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:13:15.204053Z","iopub.execute_input":"2022-08-14T04:13:15.205265Z","iopub.status.idle":"2022-08-14T04:13:15.341100Z","shell.execute_reply.started":"2022-08-14T04:13:15.205192Z","shell.execute_reply":"2022-08-14T04:13:15.340069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Neural Network","metadata":{}},{"cell_type":"code","source":"EPOCHS = 200\nBATCH_SIZE = 512\nACTIVATION = 'swish'\n\ndef load_model():\n    early_stopping = callbacks.EarlyStopping(\n        monitor=\"val_loss\",     # Quantity to be monitored\n        patience=20,                # How many epochs to wait before stopping\n        restore_best_weights=True)\n    \n    reduce_lr = callbacks.ReduceLROnPlateau(\n        monitor='val_loss', \n        factor=0.5,                # Factor by which the learning rate will be reduced\n        patience=5)                # Number of epochs with no improvement\n    \n    model = keras.Sequential([\n        layers.Dense(108, activation=ACTIVATION, input_shape=[X.shape[1]]),      \n        layers.Dense(64, activation=ACTIVATION), \n        layers.Dense(32, activation=ACTIVATION),\n        layers.Dense(1, activation='sigmoid')\n    ])\n\n    model.compile(\n        optimizer='adam',\n        loss='binary_crossentropy',\n        metrics=['AUC'])\n    \n    return model, [early_stopping, reduce_lr]","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:13:15.342391Z","iopub.execute_input":"2022-08-14T04:13:15.342752Z","iopub.status.idle":"2022-08-14T04:13:15.351388Z","shell.execute_reply.started":"2022-08-14T04:13:15.342720Z","shell.execute_reply":"2022-08-14T04:13:15.350476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training","metadata":{}},{"cell_type":"code","source":"scores = []\nf_scores = []\ntest_predictions = []\ncv = StratifiedKFold(n_splits=N_SPLITS, random_state=my_seed, shuffle=True)\nfor fold, (train_idx, test_idx) in enumerate(cv.split(X, y)):\n    train_X, val_X = X.iloc[train_idx], X.iloc[test_idx]\n    train_y, val_y = y.iloc[train_idx], y.iloc[test_idx]\n\n    model, CALLBACKS = load_model()\n    history = model.fit(\n        train_X, train_y,\n        validation_data=(val_X, val_y),\n        batch_size=BATCH_SIZE,\n        epochs=EPOCHS,\n        callbacks=CALLBACKS,        # Put your callbacks in a list\n        verbose=0)                  # Turn off training log\n\n    predictions = model.predict(val_X)\n    score = roc_auc_score(val_y, predictions)\n    scores.append(score)\n    print(f\"Fold {fold + 1} \\t\\t AUC: {score}\")\n\n    test_predictions.append(model.predict(test_X))\n\n    # Saving history to plot at the end\n    hist = pd.DataFrame(history.history)\n    hist['folds'] = fold + 1\n    f_scores = hist if fold == 0 else pd.concat([f_scores, hist], axis=0)\nprint('Overall AUC: ', np.mean(scores))","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:13:15.352856Z","iopub.execute_input":"2022-08-14T04:13:15.353520Z","iopub.status.idle":"2022-08-14T04:13:27.464114Z","shell.execute_reply.started":"2022-08-14T04:13:15.353486Z","shell.execute_reply":"2022-08-14T04:13:27.462253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Outcomes","metadata":{}},{"cell_type":"code","source":"for fold in range(f_scores['folds'].nunique()):\n    fold = fold + 1\n    history_f = f_scores[f_scores['folds'] == fold]\n\n    fig, ax = plt.subplots(1, 2, tight_layout=True, figsize=(14,4))\n    fig.suptitle('Fold : ' + str(fold), fontsize=14)\n        \n    plt.subplot(1,2,1)\n    plt.plot(history_f.loc[:, ['loss', 'val_loss']], label= ['loss', 'val_loss'])\n    plt.legend(fontsize=15)\n    plt.grid()\n    \n    plt.subplot(1,2,2)\n    plt.plot(history_f.loc[:, ['auc', 'val_auc']],label= ['auc', 'val_auc'])\n    plt.legend(fontsize=15)\n    plt.grid()\n    \n    print(\"Validation Loss: {:0.4f}\".format(history_f['val_loss'].min()));","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:13:27.465244Z","iopub.status.idle":"2022-08-14T04:13:27.465668Z","shell.execute_reply.started":"2022-08-14T04:13:27.465469Z","shell.execute_reply":"2022-08-14T04:13:27.465488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub['failure'] = np.mean(test_predictions, axis=0)\nsub.to_csv('submission_nn.csv', index=False)\nsub","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:13:27.467087Z","iopub.status.idle":"2022-08-14T04:13:27.467480Z","shell.execute_reply.started":"2022-08-14T04:13:27.467293Z","shell.execute_reply":"2022-08-14T04:13:27.467311Z"},"trusted":true},"execution_count":null,"outputs":[]}]}