{"cells":[{"metadata":{"_uuid":"93e36bbb6826818b439ae57db1f438168c381646"},"cell_type":"markdown","source":"## Through this competition I am trying to learn on : \n\nGiven seismic signals, predict the time until the onset of laboratory earthquakes.\n* The training data is a single sequence of signal\n* In contrast the test data is called segments\n* For each seg_id, predict it's  time until the  earthquake"},{"metadata":{"_uuid":"d3457484c9d6ba5136f23749b1cd94142df23467"},"cell_type":"markdown","source":"## The P (python) Packages :D"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"## pd and np to begin with\nimport pandas as pd\nimport numpy as np \n\n## old plots\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\n## my sea horse seaborn\nimport seaborn as sns\nsns.set()\n\n### HTML hmm\nfrom IPython.display import HTML\n\n### Check the files\nfrom os import listdir\nprint(listdir(\"../input\"))\n\n## supress those annyoing warnings\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\", nrows=10000000)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3ef9c7a76fd2e71009d10de64616f580573b961d"},"cell_type":"code","source":"train.head(2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8ff5ca307d2fb0ee9eb8199feb23d13bfab62d6a"},"cell_type":"code","source":"train.rename({\"acoustic_data\" : \"sig\",\"time_to_failure\" : \"qtime\"}, axis = \"columns\", inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8922228579b246fb69f4a7b311731bcc276acb7b"},"cell_type":"code","source":"### lets see the values in such sensitive data\nprint(range(3))\nfor n in range(3):\n    print(train.qtime.values[n])","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true,"_uuid":"9a593f59bf16cdd00d40f0492e90a004598022fa"},"cell_type":"code","source":"fig, ax = plt.subplots(2,1, figsize=(20,12))\n\nax[0].plot(train.index.values, train.sig.values, c=\"blue\")\nax[0].set_title(\"Sig of 10 M rows\")\nax[0].set_xlabel(\"Index\")\nax[0].set_ylabel(\"Signal\");\n\nax[1].plot(train.index.values, train.qtime.values, c=\"green\")\nax[1].set_title(\"Qtime of 10 M rows\")\nax[1].set_xlabel(\"Index\")\nax[1].set_ylabel(\"Qtime in ms\");","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true,"_uuid":"33104fb28f462352eb6da660484088a1d21e24ad"},"cell_type":"code","source":"fig, ax = plt.subplots(3,1,figsize=(20,18))\nax[0].plot(train.index.values[0:50000], train.qtime.values[0:50000], c=\"Red\")\nax[0].set_xlabel(\"Index\")\nax[0].set_ylabel(\"Time to quake\")\nax[0].set_title(\"How does the second quaketime pattern look like?\")\nax[1].plot(train.index.values[0:49999], np.diff(train.qtime.values[0:50000]))\nax[1].set_xlabel(\"Index\")\nax[1].set_ylabel(\"Difference between quaketimes\")\nax[1].set_title(\"Are the jumps always the same?\")\nax[2].plot(train.index.values[0:4000], train.qtime.values[0:4000])\nax[2].set_xlabel(\"Index from 0 to 4000\")\nax[2].set_ylabel(\"Quaketime\")\nax[2].set_title(\"How does the quaketime changes within the first block?\");","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f5f067e8975e93384f3e029c0e2fd11290ab1864"},"cell_type":"code","source":"test_path = \"../input/test/\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bdebb1e826a0b554feec3c05b26d7a29fa86e34a"},"cell_type":"code","source":"test_files = listdir(\"../input/test\")\nsample_submission = pd.read_csv(\"../input/sample_submission.csv\")\n","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true,"_uuid":"0fec497da40633de95566b5a0c300ba552bea45d"},"cell_type":"code","source":"fig, ax = plt.subplots(4,1, figsize=(20,25))\n\nfor n in range(4):\n    seg = pd.read_csv(test_path  + test_files[n])\n    ax[n].plot(seg.acoustic_data.values, c=\"Red\")\n    ax[n].set_xlabel(\"Index\")\n    ax[n].set_ylabel(\"Signal\")\n    ax[n].set_ylim([-300, 300])\n    ax[n].set_title(\"Test {}\".format(test_files[n]));","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true,"_uuid":"38973e6d701a80b5f88242c0d89e78c4408ebf57"},"cell_type":"code","source":"fig, ax = plt.subplots(1,2, figsize=(20,5))\nsns.distplot(train.sig.values, ax=ax[0], color=\"Green\", bins=100, kde=False)\nax[0].set_xlabel(\"Signal\")\nax[0].set_ylabel(\"Density\")\nax[0].set_title(\"Signal distribution\")\n\nlow = train.sig.mean() - 3 * train.sig.std()\nhigh = train.sig.mean() + 3 * train.sig.std() \nsns.distplot(train.loc[(train.sig >= low) & (train.sig <= high), \"sig\"].values,\n             ax=ax[1],\n             color=\"Red\",\n             bins=150, kde=False)\nax[1].set_xlabel(\"Signal\")\nax[1].set_ylabel(\"Density\")\nax[1].set_title(\"Signal distribution without peaks\");","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":false,"trusted":true,"_uuid":"34177c0b9415954f1ec609be0697389777d1ce6d"},"cell_type":"code","source":"stepsize = np.diff(train.qtime)\ntrain = train.drop(train.index[len(train)-1])\ntrain[\"stepsize\"] = stepsize\ntrain.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"deddf4f8c5f02057b7070ef0cf4f98e4759f0aa9"},"cell_type":"code","source":"train.stepsize = train.stepsize.apply(lambda l: np.round(l, 10))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1d1278916c8e5d07a0af08c3987a7cb299f4ec50"},"cell_type":"code","source":"stepsize_counts = train.stepsize.value_counts()\nstepsize_counts","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1538154e8e3c3c0729e247a6e739e6ae4da6e3bc"},"cell_type":"markdown","source":"**BE HOLD THE TIME  SERIES SPLIT**"},{"metadata":{"trusted":true,"_uuid":"f1e541fb7ce6c00eb449f12a917b74020bd89369"},"cell_type":"code","source":"from sklearn.model_selection import TimeSeriesSplit\n\ncv = TimeSeriesSplit(n_splits=5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"066ee08ef873bfccbe743f7d701e63f5d5f256e2"},"cell_type":"code","source":"### Rolling Window Approach \nwindow_sizes = [10, 50, 100, 1000]\nfor window in window_sizes:\n    train[\"rolling_mean_\" + str(window)] = train.sig.rolling(window=window).mean()\n    train[\"rolling_std_\" + str(window)] = train.sig.rolling(window=window).std()","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true,"_uuid":"1cc2e8b6affafe38880b16d3409f667faf6e79d4"},"cell_type":"code","source":"fig, ax = plt.subplots(len(window_sizes),1,figsize=(20,6*len(window_sizes)))\n\nn = 0\nfor col in train.columns.values:\n    if \"rolling_\" in col:\n        if \"mean\" in col:\n            mean_df = train.iloc[4435000:4445000][col]\n            ax[n].plot(mean_df, label=col, color=\"Green\")\n        if \"std\" in col:\n            std = train.iloc[4435000:4445000][col].values\n            ax[n].fill_between(mean_df.index.values,\n                               mean_df.values-std, mean_df.values+std,\n                               facecolor='Orange',\n                               alpha = 0.5, label=col)\n            ax[n].legend()\n            n+=1\n","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}