{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nDATADIR = \"../input/LANL-Earthquake-Prediction\"\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"df = pd.read_csv(os.path.join(DATADIR, 'train.csv'),\n                 dtype={'acoustic_data': np.int16, 'time_to_failure': np.float64})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import linear_signal_py as linear_signal\n\nclass SignalFeatures(linear_signal.SignalFeatureGenerator):\n    SEQUENCE_LENGHT = 150_000\n\n    S_MEAN = 4\n    S_STD = 10\n\n    def __init__(self, normalize=True):\n        self.normalize = normalize\n\n    def shape(self):\n        return (self.SEQUENCE_LENGHT, 1)\n\n    def generate(self, df: pd.DataFrame, predict=False):\n        \"\"\" The performance of this function when vectorized is 10x of\n        an iterative loop.\n        \"\"\"\n        X = df['acoustic_data'].values[:, np.newaxis]\n        if self.normalize:\n            X = (X - self.S_MEAN) / self.S_STD\n        if predict:\n            return X\n        y = df['time_to_failure'].iloc[df.shape[0] - 1]\n        return X, np.array([y])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class StaticSplit(object):\n    def __init__(self, n_blocks, test_slice):\n        self.n_blocks = n_blocks\n        indices = set(range(n_blocks))\n        for i in test_slice:\n            indices.remove(i)\n        self.train_slice = list(indices)\n        self.test_slice = test_slice\n    def train(self):\n        return self.train_slice\n    def test(self):\n        return self.test_slice\n\nkFolds = [\n    StaticSplit(25, [0, 1])\n]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"SEGMENT_SIZE = 150_000\nSTRIDES = 4_000\n\ndef get_generators(spliter):\n    ds_train = linear_signal.LinearDatasetAccessor(df, spliter.n_blocks, spliter.train())\n    ds_eval = linear_signal.LinearDatasetAccessor(df, spliter.n_blocks, spliter.test())\n\n    gen_train = linear_signal.LinearSignalGenerator(\n        ds_train, SEGMENT_SIZE, SignalFeatures(), strides=STRIDES, batch_size=256)\n    gen_eval = linear_signal.LinearSignalGenerator(\n        ds_eval, SEGMENT_SIZE, SignalFeatures(), strides=16_000)\n\n    return gen_train, gen_eval","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.keras.models import load_model\nmodel = load_model('../input/signal-convolution/model-signal-conv.h5')\nmodel.load_weights('../input/signal-convolution/signal-conv.0.ckpt.hdf5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"gen_train, gen_eval = get_generators(kFolds[0])\nmodel.evaluate_generator(gen_eval)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import shap\n\nbackground = gen_train[0][0]\ne = shap.DeepExplainer(model, background)\ntest_values = gen_eval[0][0][:4]\ny_true = gen_eval[0][1][:4]\ny_pred = model.predict(test_values)\nshap_values = e.shap_values(test_values)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"SHAP calculates the important of the features (in this case the value of the signal at each specific time).\n\nThe graphs bellow show the raw signal from a validation set example followed by the feature importance, as computed by shap.\n\nIn this case we can see that the convolutional model is being able to detect spikes in the input signal. However, it is interesting to note that not all spikes are being taken into account."},{"metadata":{"trusted":true},"cell_type":"code","source":"%matplotlib inline\nimport matplotlib.pyplot as plt\n\nfig, axes = plt.subplots(8, 1, figsize=(20, 20), sharex=True)\nfor i, ax in enumerate(axes):\n    ex = i // 2\n    if i % 2 == 0:\n        ax.set_title('y_true: {0}'.format(y_true[ex]))\n        ax.set(ylim=(-15.0, 15.0))\n        ax.plot(test_values[ex], c='b')\n    else:\n        s = shap_values[0][ex]\n        neg = np.ma.masked_where(s < 0.0, s)\n        pos = np.ma.masked_where(s > 0.0, s)\n        ax.set(ylim=(-0.05, 0.05))\n        ax.set_title('y_pred: {0}'.format(y_pred[ex]))\n        x = np.arange(150_000)\n        ax.plot(x, neg, x, pos)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Show the activation of the convolutional layers for specific validation set examples."},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.keras.models import Model\nlayer_outputs = [layer.output for layer in model.layers[1:]]\n\nviz_model = Model(inputs = model.input, outputs = layer_outputs)\n\nbackground_outs = viz_model.predict(background)\nouts = viz_model.predict(test_values)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def show_conv_layers(outs, ex, background_outs=None):\n    width = np.max(np.array([outs[i*2].shape[-1] for i in range(4)]))\n    fig = plt.figure(figsize=(20, 10))\n    gs = fig.add_gridspec(4, width, wspace=0.0)\n    for i in range(4):\n        ax_left = None\n        index = i * 2\n        x = outs[index][ex]\n        if background_outs is not None:\n            b = background_outs[index]\n            x -= b.mean()\n            x /= b.std()\n        else:\n            x -= x.mean()\n            x /= x.std()\n\n        for j in range(outs[index].shape[-1]):\n            step = width // outs[index].shape[-1]\n            s = slice(j * step, (j + 1) * step)\n            g = gs[i, s]\n            ax = fig.add_subplot(g, sharey=ax_left)\n            if j == 0:\n                ax_left = ax\n            else:\n                plt.setp(ax.get_yticklabels(), visible=False)\n            ax.plot(x[:,j])\n\n    plt.show()","execution_count":1,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"show_conv_layers(outs, 0, background_outs)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"show_conv_layers(outs, 1, background_outs)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}