{"cells":[{"metadata":{"_uuid":"80f28a6efc7c4b5298c077a70c7d539b2a8dbebd"},"cell_type":"raw","source":""},{"metadata":{"_uuid":"d82e0e6a27ed41c56b3fa21ab01bc18a51c4262a"},"cell_type":"markdown","source":"  # Loading packages"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\nimport seaborn as sns\nsns.set()\n\nfrom IPython.display import HTML\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":false},"cell_type":"markdown","source":"# Data "},{"metadata":{"trusted":true,"_uuid":"a54c08442e5f9f59067da2fa0a44ed1c8040fd99"},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\", nrows=10000000)\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"140f4ac58f14a8cd8b2a6223d95dffbed9fcad56"},"cell_type":"markdown","source":"# Column Rename \nWe can see two columns: Acoustic data and time_to_failure. The further is the seismic singal and the latter corresponds to the time until the laboratory earthquake takes place.\n"},{"metadata":{"trusted":true,"_uuid":"df9bf33a8cb5d8f748726e86c2aa0f64550a5ed6"},"cell_type":"code","source":"train.rename({\"acoustic_data\": \"signal\", \"time_to_failure\": \"quaketime\"}, axis=\"columns\", inplace=True)\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c4c1a5a9ae0f2b0a87f444715799ad67c890f32e"},"cell_type":"markdown","source":"# See earthquack time , cross check"},{"metadata":{"trusted":true,"_uuid":"2856ab2e0974205348eb84fbcbefdd60b3d08ee5"},"cell_type":"code","source":"for n in range(5):\n    print(train.quaketime.values[n])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"decbcb48fb44f376165977cf0d80c36b68e59f31"},"cell_type":"markdown","source":"# Let's Plot this data"},{"metadata":{"trusted":true,"_uuid":"b4ce02ce66bce7948d9716492fe5ac91a8f73edb"},"cell_type":"code","source":"fig, ax = plt.subplots(2,1, figsize=(20,12))\nax[0].plot(train.index.values, train.quaketime.values)\nax[0].set_title(\"Quaketime of 10 Mio rows\")\nax[0].set_xlabel(\"Index\")\nax[0].set_ylabel(\"Quaketime in ms\");\nax[1].plot(train.index.values, train.signal.values, c=\"green\")\nax[1].set_title(\"Signal of 10 Mio rows\")\nax[1].set_xlabel(\"Index\")\nax[1].set_ylabel(\"Acoustic Signal\");","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"875a824edfc4ac825a82bc67dbc43f653d400d1a"},"cell_type":"markdown","source":"# How does the earthquck changes "},{"metadata":{"trusted":true,"_uuid":"4accdef56f51b17c02b89e2b119b4c3e7b0341a2"},"cell_type":"code","source":"fig, ax = plt.subplots(3,1,figsize=(20,18))\nax[0].plot(train.index.values[0:50000], train.quaketime.values[0:50000], c=\"Red\")\nax[0].set_xlabel(\"Index\")\nax[0].set_ylabel(\"Time to quake\")\nax[0].set_title(\"How does the second quaketime pattern look like?\")\nax[1].plot(train.index.values[0:49999], np.diff(train.quaketime.values[0:50000]))\nax[1].set_xlabel(\"Index\")\nax[1].set_ylabel(\"Difference between quaketimes\")\nax[1].set_title(\"Are the jumps always the same?\")\nax[2].plot(train.index.values[0:4000], train.quaketime.values[0:4000])\nax[2].set_xlabel(\"Index from 0 to 4000\")\nax[2].set_ylabel(\"Quaketime\")\nax[2].set_title(\"How does the quaketime changes within the first block?\");\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ac359bedc50d5a70985e589ae43010f786a9a6b8"},"cell_type":"markdown","source":"# Test Data"},{"metadata":{"trusted":true,"_uuid":"a7813afb6c9fea0b27402384cc8df78d5f744882"},"cell_type":"code","source":"import os \nfrom os import listdir\ntest_path = \"../input/test/\"\ntest_files = listdir(\"../input/test\")\nprint(test_files[0:5])\n\nlen(test_files)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"19ae93bcaaf8c6133b7f0e5295fcde42773dbae5"},"cell_type":"markdown","source":"# Check with sample submission file"},{"metadata":{"trusted":true,"_uuid":"84c63af66f02613b34c2c9a94b8d3d3e4bf99029"},"cell_type":"code","source":"sample_submission = pd.read_csv(\"../input/sample_submission.csv\")\nprint (sample_submission.head())\nlen(sample_submission.seg_id.values)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a644848e31fc3c9697810e63bfbdd55a6daf0dad"},"cell_type":"markdown","source":"# Check signals of test data"},{"metadata":{"trusted":true,"_uuid":"4d2840a1662b99595b8034c048df679fbc1fc23b"},"cell_type":"code","source":"fig, ax = plt.subplots(4,1, figsize=(20,25))\n\nfor n in range(4):\n    seg = pd.read_csv(test_path  + test_files[n])\n    ax[n].plot(seg.acoustic_data.values, c=\"mediumseagreen\")\n    ax[n].set_xlabel(\"Index\")\n    ax[n].set_ylabel(\"Signal\")\n    ax[n].set_ylim([-300, 300])\n    ax[n].set_title(\"Test {}\".format(test_files[n]));\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b64f3bbd95b9255f03ce75ab275801d98f967c7e"},"cell_type":"markdown","source":"# EDA "},{"metadata":{"trusted":true,"_uuid":"816ad827584c8779b2fde0efa886339d6cb4f97f"},"cell_type":"code","source":"train.describe()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ffcfae295c6c59a0f4655bbc6b580b6d0fcf8d50"},"cell_type":"markdown","source":"# Signal Distribution "},{"metadata":{"trusted":true,"_uuid":"ffd6abc668f66e7033a9018c33774522a98af2cb"},"cell_type":"code","source":"fig, ax = plt.subplots(1,2, figsize=(20,5))\nsns.distplot(train.signal.values, ax=ax[0], color=\"Red\", bins=100, kde=False)\nax[0].set_xlabel(\"Signal\")\nax[0].set_ylabel(\"Density\")\nax[0].set_title(\"Signal distribution\")\n\nlow = train.signal.mean() - 3 * train.signal.std()\nhigh = train.signal.mean() + 3 * train.signal.std() \nsns.distplot(train.loc[(train.signal >= low) & (train.signal <= high), \"signal\"].values,\n             ax=ax[1],\n             color=\"Orange\",\n             bins=150, kde=False)\nax[1].set_xlabel(\"Signal\")\nax[1].set_ylabel(\"Density\")\nax[1].set_title(\"Signal distribution without peaks\");","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8ffaf6377e1211048187b9bdcaf7eea2e243d0ae"},"cell_type":"markdown","source":"# Counting Stepsize"},{"metadata":{"trusted":true,"_uuid":"9762bc30fd33e5faa45c46b02b23494f16cd42da"},"cell_type":"code","source":"stepsize = np.diff(train.quaketime)\ntrain = train.drop(train.index[len(train)-1])\ntrain[\"stepsize\"] = stepsize\ntrain.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"55d744f06abf9d4178fdc3b0f2182a3a02967210"},"cell_type":"code","source":"train.stepsize = train.stepsize.apply(lambda l: np.round(l, 10))\nstepsize_counts = train.stepsize.value_counts()\nstepsize_counts","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"27f1e2d02478bad369089a033d978051a7acac34"},"cell_type":"markdown","source":"# Rolling sizes and window features"},{"metadata":{"trusted":true,"_uuid":"7d5b1d998c1aeb5dea977d1d421063b9e14233e4"},"cell_type":"code","source":"window_sizes = [10, 50, 100, 1000]\nfor window in window_sizes:\n    train[\"rolling_mean_\" + str(window)] = train.signal.rolling(window=window).mean()\n    train[\"rolling_std_\" + str(window)] = train.signal.rolling(window=window).std()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"faf992c8d873fa8ef3ad079a5b43a46c23cfa247"},"cell_type":"code","source":"fig, ax = plt.subplots(len(window_sizes),1,figsize=(20,6*len(window_sizes)))\n\nn = 0\nfor col in train.columns.values:\n    if \"rolling_\" in col:\n        if \"mean\" in col:\n            mean_df = train.iloc[4435000:4445000][col]\n            ax[n].plot(mean_df, label=col, color=\"mediumseagreen\")\n        if \"std\" in col:\n            std = train.iloc[4435000:4445000][col].values\n            ax[n].fill_between(mean_df.index.values,\n                               mean_df.values-std, mean_df.values+std,\n                               facecolor='lightgreen',\n                               alpha = 0.5, label=col)\n            ax[n].legend()\n            n+=1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"91cfcf298adf036f296403e09f6f530f46fef71e"},"cell_type":"code","source":"float_data = pd.read_csv(\"../input/train.csv\",dtype={\"acoustic_data\": np.float32, \"time_to_failure\": np.float32})\nfloat_data = float_data.values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d7226ac138c7bb8425bc72833202c0025b3a2a86"},"cell_type":"code","source":"#Helper Function\ndef extract_features(z):\n     return np.c_[z.mean(axis=1), \n                  np.median(np.abs(z), axis=1),\n                  z.std(axis=1), \n                  z.max(axis=1),\n                  z.min(axis=1)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2df25c010d0de1543fe168171768631522a3e50f"},"cell_type":"code","source":"def create_X(x, last_index=None, n_steps=150, step_length=1000):\n    if last_index == None:\n        last_index=len(x)\n    assert last_index - n_steps * step_length >= 0\n    temp = (x[(last_index - n_steps * step_length):last_index].reshape(n_steps, -1) - 5 ) / 3\n    return np.c_[extract_features(temp),\n                 extract_features(temp[:, -step_length // 10:]),\n                 extract_features(temp[:, -step_length // 100:]),\n                 temp[:, -1:]]\n        ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fcf5d17a1fa9e62a81587828130a44678bec00d6"},"cell_type":"code","source":"n_features = 16\ndef generator(data, min_index=0, max_index=None, batch_size=16, n_steps=150, step_length=1000):\n    if max_index is None:\n        max_index = len(data) - 1\n    while True:\n        # Pick indices of ending positions\n        rows = np.random.randint(min_index + n_steps * step_length, max_index, size=batch_size)\n         \n        # Initialize feature matrices and targets\n        samples = np.zeros((batch_size, n_steps, n_features))\n        targets = np.zeros(batch_size, )\n        \n        for j, row in enumerate(rows):\n            samples[j] = create_X(data[:, 0], last_index=row, n_steps=n_steps, step_length=step_length)\n            targets[j] = data[row, 1]\n        yield samples, targets","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"db81502b2223a0d6e03fb63d7a153898249a52bd"},"cell_type":"code","source":"batch_size = 32\n\ntrain_gen = generator(float_data, batch_size=batch_size)\nvalid_gen = generator(float_data, batch_size=batch_size)\n\n# Define model\nfrom keras.models import Sequential\nfrom keras.layers import Dense, CuDNNGRU\nfrom keras.optimizers import adam\nfrom keras.callbacks import ModelCheckpoint\n\ncb = [ModelCheckpoint(\"model.hdf5\", monitor='val_loss', save_weights_only=False, period=3)]\n\nmodel = Sequential()\nmodel.add(CuDNNGRU(48, input_shape=(None, n_features)))\nmodel.add(Dense(10, activation='relu'))\nmodel.add(Dense(1))\n\nmodel.summary()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"909c27819abc5881694bb0a354cb3a214f4f8315"},"cell_type":"code","source":"model.compile(optimizer=adam(lr=0.0005), loss=\"mae\")\n\nhistory = model.fit_generator(train_gen,\n                              steps_per_epoch=1000,#n_train // batch_size,\n                              epochs=30,\n                              verbose=0,\n                              callbacks=cb,\n                              validation_data=valid_gen,\n                              validation_steps=100)#n_valid // batch_size)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"54521983bbcb82f323e16cd93081f06499ade93c"},"cell_type":"code","source":"import matplotlib.pyplot as plt\n\ndef perf_plot(history, what = 'loss'):\n    x = history.history[what]\n    val_x = history.history['val_' + what]\n    epochs = np.asarray(history.epoch) + 1\n    \n    plt.plot(epochs, x, 'bo', label = \"Training \" + what)\n    plt.plot(epochs, val_x, 'b', label = \"Validation \" + what)\n    plt.title(\"Training and validation \" + what)\n    plt.xlabel(\"Epochs\")\n    plt.legend()\n    plt.show()\n    return None\n\nperf_plot(history)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}