{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.layers import Dense, Dropout\nfrom keras.optimizers import adam\nfrom keras.callbacks import Callback\nfrom keras.layers.advanced_activations import LeakyReLU\nfrom sklearn.metrics import roc_auc_score\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport timeit\nimport sys\nfrom termcolor import colored as coloured\nfrom sklearn.datasets import make_blobs","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9a16d59130dad7ca8de02b738e3312451a6bb6bf"},"cell_type":"code","source":"class RocCallback(Callback):\n    def __init__(self, training_data, test_data, max_epochs):\n        self.x = training_data[0]\n        self.y = training_data[1]\n        self.x_test = test_data[0]\n        self.y_test = test_data[1]\n        self.max_roc_auc = 0  # For use in finding max ROC AUC and saving the model\n        global train_roc_auc_glob\n        global test_roc_auc_glob\n        train_roc_auc_glob = []\n        test_roc_auc_glob = []\n        self.epoch_count = 0\n        self.best_epoch_classvar = 0\n        self.max_epochs = max_epochs\n\n    def on_train_begin(self, logs={}):\n        # Get time the training started for use in calculating time left\n        self.fitting_time_init = timeit.default_timer()\n        return\n\n    def on_train_end(self, logs={}):\n        return\n\n    def on_epoch_begin(self, epoch, logs={}):\n        return\n\n    def on_epoch_end(self, epoch, logs={}):\n        self.epoch_count = self.epoch_count + 1\n        # Get start time of current epoch for use in calcaulting time left\n        temp_time_since_fitting_began = timeit.default_timer() - self.fitting_time_init\n        # How long per epoch so far, multiplied but amount left will give a time left\n        temp_time_left_until_end = (temp_time_since_fitting_began/self.epoch_count) * (self.max_epochs-self.epoch_count)\n        percent_done = int(self.epoch_count/self.max_epochs*100)\n\n        # Calculate metrics at this epoch in training\n        y_pred = self.model.predict(self.x)\n        roc = roc_auc_score(self.y, y_pred)\n        train_roc_auc_glob.append(roc)\n        y_pred_test = self.model.predict(self.x_test)\n        roc_test = roc_auc_score(self.y_test, y_pred_test)\n        test_roc_auc_glob.append(roc_test)\n\n        # Updating output:\n        # Format string to pad 0s so progress output doesn't move\n        sys.stdout.write('\\rTrain ROC AUC: {0} || Val ROC AUC: {1} - Best Val ROCAUC: {2} at Epoch {3}'.format(\n            coloured('{:.4f}'.format(np.round(roc, 4)), 'green', attrs=['bold']),\n            coloured('{:.4f}'.format(round(roc_test, 4)), 'green', attrs=['bold']),\n            coloured('{:.4f}'.format(np.round(self.max_roc_auc, 4)), 'cyan', attrs=['bold']),\n            coloured(str(self.best_epoch_classvar), 'cyan', attrs=['bold'])))\n\n        sys.stdout.write(coloured('          {0:2d}% Done ({1}/{2})'.format(int(percent_done), int(self.epoch_count),\n                                                                         int(self.max_epochs)),\n                                  'magenta', attrs=['bold']))\n        # If more than an hour, don't display seconds\n        if temp_time_left_until_end > 3600:\n            sys.stdout.write(coloured('          {0:.0f}h {1:2.0f}m left'.format(np.floor(temp_time_left_until_end / 3600),\n                                                                                 temp_time_left_until_end % 60),\n                                      'yellow', attrs=['bold']))\n\n        # If less than a minute, don't display minutes, just seconds\n        elif temp_time_left_until_end < 60:\n            sys.stdout.write(coloured('             {0:2.0f}s left'.format(temp_time_left_until_end),\n                                       'yellow', attrs=['bold']))\n        # Else display minutes and seconds\n        else:\n            sys.stdout.write(coloured('          {0:2.0f}m {1:2.0f}s left'.format(np.floor(temp_time_left_until_end/60),\n                                                                                temp_time_left_until_end % 60),\n                                     'yellow', attrs=['bold']))\n\n        sys.stdout.flush()\n\n        # Save weights and epoch number of model with best test ROC AUC\n\n        if roc_test > self.max_roc_auc:  # Save weights of model with best test ROC AUC\n            self.max_roc_auc = roc_test\n            self.best_epoch_classvar = epoch + 1\n            global best_epoch\n            best_epoch = epoch + 1\n        return\n\n    def on_batch_begin(self, batch, logs={}):\n        return\n\n    def on_batch_end(self, batch, logs={}):\n        return","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"epochs = 1000\ntrain_prop = 0.8\nsamples = 250","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"02c901707755517fc716cdd7ab208a9d1b9ad09d"},"cell_type":"code","source":"# Make data\nx, y = make_blobs(n_samples=samples, n_features=2, centers=2, cluster_std=6, random_state = 123)\n\n# Plot\ndf = pd.DataFrame(dict(x=x[:, 0], y=x[:, 1], label=y))\n\ngroups = df.groupby('label')\n\nfig, ax = plt.subplots()\nax.margins(0.05)  # Optional, just adds 5% padding to the autoscaling\nfor name, group in groups:\n    ax.plot(group.x, group.y, marker='o', linestyle='', ms=6, label=name)\nax.legend()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"428b259603bd0c3b561642510dd3f5110cee0d2d"},"cell_type":"code","source":"# Split into test and train and minmax scale:\ntrain_samples = int(round(x.shape[0] * train_prop, 0))\nsequence = np.linspace(0, x.shape[0] - 1, num=x.shape[0], dtype=int)\nnp.random.shuffle(sequence)\ntrain_ind = sequence[0:train_samples]\nval_ind = sequence[train_samples:]\n\nx_train = df.iloc[train_ind, [0, 1]]\ny_train = df.iloc[train_ind, 2]\n\nx_val = df.iloc[val_ind, [0, 1]]\ny_val = df.iloc[val_ind, 2]\n\nx_train = (x_train-x_train.min())/(x_train.max()-x_train.min())\nx_val = (x_val-x_val.min())/(x_val.max()-x_val.min())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"377abe8d2dfdd2fdb08de1c29edac47fa92c2312"},"cell_type":"code","source":"# Construct layers then compile\nmodel = Sequential()\nmodel.add(Dense(20, input_dim=x_train.shape[1]))\nmodel.add(LeakyReLU())\nmodel.add(Dropout(0.1))\n\nmodel.add(Dense(10))\nmodel.add(LeakyReLU())\nmodel.add(Dropout(0.1))\n\n\nmodel.add(Dense(1, activation='sigmoid'))\n\nmodel.compile(loss='binary_crossentropy', optimizer=adam(lr=0.0001), metrics=['accuracy'])\n\nmodel_hist = model.fit(x_train, y_train, validation_data=(x_val, y_val), epochs=epochs, batch_size=32,\n                       verbose=0, callbacks=[RocCallback(training_data=(x_train, y_train),\n                                                                test_data=(x_val, y_val),\n                                                                max_epochs=epochs)])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7a8a3d21c32419756d3b7c5a98a1a876fc232393"},"cell_type":"code","source":"# PLOT - Accuracy and loss combined\n# Only plot first 10% of data to avoid axes being stretched\nf, axarr = plt.subplots(2, sharex=True)\naxarr[0].plot(model_hist.history['acc'], label=\"Train\", linewidth=0.75)\naxarr[0].plot(model_hist.history['val_acc'], label=\"Test\", linewidth=0.75)\n\naxarr[0].legend(loc='best')\naxarr[0].set_title('Model Accuracy')\naxarr[1].set_title('Model Loss')\naxarr[0].set_ylabel(\"Accuracy\")\naxarr[1].set_xlabel(\"Epoch\")\naxarr[1].set_ylabel(\"Loss\")\n\naxarr[1].plot(model_hist.history['loss'], label=\"Train\", linewidth=0.75)\naxarr[1].plot(model_hist.history['val_loss'], label=\"Test\", linewidth=0.75)\naxarr[1].legend(loc='best')\naxarr[0].grid(True)\naxarr[1].grid(True)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d1610f5fa3ead8c0c2cc8f837ba1b5a57664c392"},"cell_type":"code","source":"# PLOT - Accuracy and loss combined - MA\n# Only plot where MA is calculated correctly\n\n# MA, has int(epochs/100)+1 as cannot convolve with N = 0\n# Only use last 90% of data as in other plots\nMA_Var = int(epochs / 30) + 1\nacc_ma = np.convolve(np.asarray(model_hist.history['acc']),\n                     np.ones((MA_Var,)) / MA_Var,\n                     mode='valid')\nval_acc_ma = np.convolve(np.asarray(model_hist.history['val_acc']),\n                         np.ones((MA_Var,)) / MA_Var,\n                         mode='valid')\nloss_ma = np.convolve(np.asarray(model_hist.history['loss']),\n                      np.ones((MA_Var,)) / MA_Var,\n                      mode='valid')\nval_loss_ma = np.convolve(np.asarray(model_hist.history['val_loss']),\n                          np.ones((MA_Var,)) / MA_Var, mode='valid')\n\n\nf, axarr = plt.subplots(2, sharex=True)\naxarr[0].plot(list(range(int(MA_Var), epochs+1)),\n              acc_ma, label=\"Train\", linewidth=0.75)\naxarr[0].plot(list(range(int(MA_Var), epochs+1)),\n              val_acc_ma, label=\"Test\", linewidth=0.75)\n\naxarr[0].legend(loc='best')\naxarr[0].set_title('Model Accuracy - Moving Average')\naxarr[0].set_ylabel(\"Accuracy\")\naxarr[1].set_xlabel(\"Epoch\")\naxarr[1].set_ylabel(\"Loss\")\naxarr[1].set_title('Model Loss - Moving Average')\naxarr[1].plot(list(range(int(MA_Var), epochs+1)),\n              loss_ma, label=\"Train\", linewidth=0.75)\naxarr[1].plot(list(range(int(MA_Var), epochs+1)),\n              val_loss_ma, label=\"Test\", linewidth=0.75)\naxarr[1].legend(loc='best')\naxarr[0].grid(True)\naxarr[1].grid(True)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1cbf8328f20bde677afff2ddedb1c16dda5a7f04"},"cell_type":"code","source":"\nf, ax = plt.subplots()\nax.plot(train_roc_auc_glob, label=\"Train\", linewidth=0.75)\nax.plot(test_roc_auc_glob, label=\"Test\", linewidth=0.75)\nbest_iter = np.argmax(test_roc_auc_glob)\nplt.axvline(x=best_iter, ls='--', color='black', label='Best Test Value', linewidth=0.75)\nax.set_xlabel(\"Epoch\")\nax.set_ylabel(\"Area under ROC\")\nax.legend(loc='best')\nax.grid(True)\n# ax.plot([0, 1], [0, 1], transform=ax.transAxes,ls=\"--\",c=\"0.3\") - Plots diagonal line\nax.set_title(\"ROC AUC by Epoch\")\nplt.show()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}