{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport os\nfrom tqdm import tqdm\n\n# Fix seeds\nfrom numpy.random import seed\nseed(639)\nfrom tensorflow import set_random_seed\nset_random_seed(5944)\n\n# Import\nfloat_data = pd.read_csv(\"../input/train.csv\", nrows=300000, dtype={\"acoustic_data\": np.float32, \"time_to_failure\": np.float32}).values\n\n# Helper function for the data generator. Extracts mean, standard deviation, and quantiles per time step.\n# Can easily be extended. Expects a two dimensional array.\ndef extract_features(z):\n     return np.c_[z.mean(axis=1), \n                  z.min(axis=1),\n                  z.max(axis=1),\n                  z.std(axis=1)]\n\n# For a given ending position \"last_index\", we split the last 150'000 values \n# of \"x\" into 150 pieces of length 1000 each. So n_steps * step_length should equal 150'000.\n# From each piece, a set features are extracted. This results in a feature matrix \n# of dimension (150 time steps x features).  \ndef create_X_Y(x, last_index=None, n_steps=150, step_length=1000):\n    if last_index == None:\n        last_index=len(x)\n       \n    assert last_index - n_steps * step_length >= 0\n\n    # Reshaping and without approximate standardization with mean 5 and std 3.\n    chunk = x[(last_index - n_steps * step_length):last_index]\n    X = (chunk.T[0].reshape(n_steps, step_length) - 4.5 ) / 10.7\n    Y = chunk.T[1].min()\n    \n    # Extracts features of sequences of full length 1000, of the last 100 values and finally also \n    # of the last 10 observations. \n    return np.c_[extract_features(X),\n                 extract_features(X[:, -step_length // 10:]),\n                 extract_features(X[:, -step_length // 100:])],Y\ndef create_X(x, last_index=None, n_steps=150, step_length=1000):\n    if last_index == None:\n        last_index=len(x)\n       \n    assert last_index - n_steps * step_length >= 0\n\n    # Reshaping and without approximate standardization with mean 5 and std 3.\n    chunk = x[(last_index - n_steps * step_length):last_index]\n    X = (chunk.reshape(n_steps, step_length) - 4.5 ) / 10.7\n    \n    # Extracts features of sequences of full length 1000, of the last 100 values and finally also \n    # of the last 10 observations. \n    return np.c_[extract_features(X),\n                 extract_features(X[:, -step_length // 10:]),\n                 extract_features(X[:, -step_length // 100:])]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Query \"create_X\" to figure out the number of features\nfeatures,targets = create_X_Y(float_data[0:150000])\nn_features = features.shape[1]\nprint(\"Our RNN is based on %i features\"% n_features)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"create_X_Y returns X,Y: <br>\nX 150 * 12 matrix<br>\nY 150 * 1 matrix"},{"metadata":{"trusted":true},"cell_type":"code","source":"# define create_dataset\n# train = create_dataset ...\n# validation = create_dataset ...\n# test = create_dataset ...","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n_train = 5000\nfloat_data = pd.read_csv(\"../input/train.csv\", nrows=n_train*150000, dtype={\"acoustic_data\": np.float32, \"time_to_failure\": np.float32}).values\n#float_data = pd.read_csv(\"../input/train.csv\", dtype={\"acoustic_data\": np.float32, \"time_to_failure\": np.float32})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# len(float_data)/150000\n# print(float_data[:5])\na = float_data[:,0]\nprint(np.mean(a))\nprint(np.std(a))\nlen(float_data)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# The generator endlessly selects \"batch_size\" ending positions of sub-time series. For each ending position,\n# the \"time_to_failure\" serves as target, while the features are created by the function \"create_X\".\ndef generator(data, min_index=0, max_index=None, batch_size=16, n_steps=150, step_length=1000):\n    if max_index is None:\n        max_index = len(data) - 1\n     \n    while True:\n        # Pick indices of ending positions\n        rows = np.random.randint(min_index + n_steps * step_length, max_index, size=batch_size)\n         \n        # Initialize feature matrices and targets\n        samples = np.zeros((batch_size, n_steps, n_features))\n        targets = np.zeros(batch_size, )\n        \n        for j, row in enumerate(rows):\n            samples[j],targets[j] = create_X_Y(data, last_index=row, n_steps=n_steps, step_length=step_length)\n        yield samples, targets\n\nbatch_size = 64\n\n# Position of second (of 16) earthquake. Used to have a clean split\n# between train and validation\nsecond_earthquake = 50085877\n#float_data[second_earthquake, 1]\n\ntv_bd = 3800 * 150000\n\n# steps = 300\n# step_length = 150000/steps\n# Initialize generators\ntrain_gen = generator(float_data, max_index = 4000 * 150000, batch_size=batch_size,n_steps=300,step_length=500) # Use this for better score\n# train_gen = generator(float_data, batch_size=batch_size, min_index=second_earthquake + 1)\nvalid_gen = generator(float_data, min_index= 3800 * 150000,batch_size=batch_size,n_steps=300, step_length=500)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Define model\nfrom keras import regularizers\nfrom keras.models import Sequential\nfrom keras.layers import Dense, CuDNNGRU,CuDNNLSTM,Bidirectional,Activation,Dropout\nfrom keras.optimizers import adam,RMSprop\nfrom keras.callbacks import ModelCheckpoint\n\ncb = [ModelCheckpoint(\"model.hdf5\", save_best_only=True, period=3)]\n\nmodel = Sequential()\nmodel.add(CuDNNLSTM(64,return_sequences=True,input_shape=(None, n_features)))\nmodel.add(CuDNNGRU(48))\nmodel.add(Dense(16,activation=\"relu\"))\nmodel.add(Dense(1))\n#model.add(Dense(16,activation=\"tanh\",kernel_regularizer=regularizers.l2(0.05)))\n#model.add(CuDNNGRU(48, input_shape=(None, n_features)))\n\n\nmodel.summary()\n\n# Compile and fit model\nmodel.compile(optimizer=RMSprop(lr=0.0005), loss=\"mae\")\n\nhistory = model.fit_generator(train_gen,\n                              steps_per_epoch=1000,\n                              epochs=100,\n                              verbose=2,\n                              callbacks=cb,\n                              validation_data=valid_gen,\n                              validation_steps=200)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Visualize accuracies\nimport matplotlib.pyplot as plt\n\ndef perf_plot(history, what = 'loss'):\n    x = history.history[what]\n    val_x = history.history['val_' + what]\n    epochs = np.asarray(history.epoch) + 1\n    \n    plt.plot(epochs, x, 'bo', label = \"Training \" + what)\n    plt.plot(epochs, val_x, 'b', label = \"Validation \" + what)\n    plt.title(\"Training and validation \" + what)\n    plt.xlabel(\"Epochs\")\n    plt.legend()\n    plt.show()\n    return None\n\nperf_plot(history)\n\n# Load submission file\nsubmission = pd.read_csv('../input/sample_submission.csv', index_col='seg_id', dtype={\"time_to_failure\": np.float32})\n\n# Load each test data, create the feature matrix, get numeric prediction\nfor i, seg_id in enumerate(tqdm(submission.index)):\n  #  print(i)\n    seg = pd.read_csv('../input/test/' + seg_id + '.csv')\n    \n    x = seg['acoustic_data'].values\n    submission.time_to_failure[i] = model.predict(np.expand_dims(create_X(x,n_steps=300, step_length=500), 0))\n\nsubmission.head()\n\n# Save\nsubmission.to_csv('submission.csv')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}