{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-05T06:28:04.470269Z","iopub.execute_input":"2023-03-05T06:28:04.470590Z","iopub.status.idle":"2023-03-05T06:28:04.477157Z","shell.execute_reply.started":"2023-03-05T06:28:04.470511Z","shell.execute_reply":"2023-03-05T06:28:04.476162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Using the same notebook with features and other process to try the GRU model on the dataset.\n","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom tqdm import tqdm\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import mean_absolute_error","metadata":{"execution":{"iopub.status.busy":"2023-03-05T06:28:19.367247Z","iopub.execute_input":"2023-03-05T06:28:19.367637Z","iopub.status.idle":"2023-03-05T06:28:19.372751Z","shell.execute_reply.started":"2023-03-05T06:28:19.367576Z","shell.execute_reply":"2023-03-05T06:28:19.371501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/train.csv', dtype={'acoustic_data': np.int16, 'time_to_failure': np.float64})","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.status.busy":"2023-03-05T06:34:09.592621Z","iopub.execute_input":"2023-03-05T06:34:09.592962Z","iopub.status.idle":"2023-03-05T06:37:09.980146Z","shell.execute_reply.started":"2023-03-05T06:34:09.592909Z","shell.execute_reply":"2023-03-05T06:37:09.979561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-05T06:37:09.981066Z","iopub.execute_input":"2023-03-05T06:37:09.981368Z","iopub.status.idle":"2023-03-05T06:37:10.014774Z","shell.execute_reply.started":"2023-03-05T06:37:09.981325Z","shell.execute_reply":"2023-03-05T06:37:10.014038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pandas doesn't show us all the decimals\npd.options.display.precision = 15","metadata":{"execution":{"iopub.status.busy":"2023-03-05T06:37:10.015810Z","iopub.execute_input":"2023-03-05T06:37:10.016159Z","iopub.status.idle":"2023-03-05T06:37:10.020209Z","shell.execute_reply.started":"2023-03-05T06:37:10.016108Z","shell.execute_reply":"2023-03-05T06:37:10.018846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# much better!\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-05T06:37:14.578270Z","iopub.execute_input":"2023-03-05T06:37:14.578581Z","iopub.status.idle":"2023-03-05T06:37:14.591465Z","shell.execute_reply.started":"2023-03-05T06:37:14.578520Z","shell.execute_reply":"2023-03-05T06:37:14.590236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a training file with simple derived features\n\nrows = 150_000\nsegments = int(np.floor(train.shape[0] / rows))\n\nX_train = pd.DataFrame(index=range(segments), dtype=np.float64,\n                       columns=['ave', 'std', 'max', 'min'])\ny_train = pd.DataFrame(index=range(segments), dtype=np.float64,\n                       columns=['time_to_failure'])\n\nfor segment in tqdm(range(segments)):\n    seg = train.iloc[segment*rows:segment*rows+rows]\n    x = seg['acoustic_data'].values\n    y = seg['time_to_failure'].values[-1]\n    \n    y_train.loc[segment, 'time_to_failure'] = y\n    \n    X_train.loc[segment, 'ave'] = x.mean()\n    X_train.loc[segment, 'std'] = x.std()\n    X_train.loc[segment, 'max'] = x.max()\n    X_train.loc[segment, 'min'] = x.min()","metadata":{"execution":{"iopub.status.busy":"2023-03-05T06:37:27.504626Z","iopub.execute_input":"2023-03-05T06:37:27.505017Z","iopub.status.idle":"2023-03-05T06:37:35.039480Z","shell.execute_reply.started":"2023-03-05T06:37:27.504941Z","shell.execute_reply":"2023-03-05T06:37:35.038274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-05T06:37:35.040508Z","iopub.execute_input":"2023-03-05T06:37:35.040775Z","iopub.status.idle":"2023-03-05T06:37:35.056658Z","shell.execute_reply.started":"2023-03-05T06:37:35.040727Z","shell.execute_reply":"2023-03-05T06:37:35.056007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-05T06:39:17.322090Z","iopub.execute_input":"2023-03-05T06:39:17.322428Z","iopub.status.idle":"2023-03-05T06:39:17.328964Z","shell.execute_reply.started":"2023-03-05T06:39:17.322376Z","shell.execute_reply":"2023-03-05T06:39:17.327753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaler = StandardScaler()\nscaler.fit(X_train)\nX_train_scaled = scaler.transform(X_train)","metadata":{"execution":{"iopub.status.busy":"2023-03-05T06:39:19.680604Z","iopub.execute_input":"2023-03-05T06:39:19.681259Z","iopub.status.idle":"2023-03-05T06:39:19.686753Z","shell.execute_reply.started":"2023-03-05T06:39:19.681200Z","shell.execute_reply":"2023-03-05T06:39:19.686185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_scaled.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-05T06:39:21.881060Z","iopub.execute_input":"2023-03-05T06:39:21.881694Z","iopub.status.idle":"2023-03-05T06:39:21.887352Z","shell.execute_reply.started":"2023-03-05T06:39:21.881634Z","shell.execute_reply":"2023-03-05T06:39:21.886240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### GRU","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nfrom keras.models import Sequential\nfrom keras.layers import Dense\nfrom keras.layers import GRU","metadata":{"execution":{"iopub.status.busy":"2023-03-05T06:39:24.300197Z","iopub.execute_input":"2023-03-05T06:39:24.300608Z","iopub.status.idle":"2023-03-05T06:39:27.155442Z","shell.execute_reply.started":"2023-03-05T06:39:24.300514Z","shell.execute_reply":"2023-03-05T06:39:27.153829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_scaled = X_train_scaled.reshape(X_train_scaled.shape[0],X_train_scaled.shape[1],1)","metadata":{"execution":{"iopub.status.busy":"2023-03-05T06:39:27.159727Z","iopub.execute_input":"2023-03-05T06:39:27.160044Z","iopub.status.idle":"2023-03-05T06:39:27.165267Z","shell.execute_reply.started":"2023-03-05T06:39:27.160001Z","shell.execute_reply":"2023-03-05T06:39:27.163718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_scaled.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-05T06:50:06.738210Z","iopub.execute_input":"2023-03-05T06:50:06.738581Z","iopub.status.idle":"2023-03-05T06:50:06.745181Z","shell.execute_reply.started":"2023-03-05T06:50:06.738503Z","shell.execute_reply":"2023-03-05T06:50:06.743932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model=Sequential()\nmodel.add(GRU(128,return_sequences=True,input_shape=(X_train_scaled.shape[1], X_train_scaled.shape[2])))\nmodel.add(GRU(32,return_sequences=True))\nmodel.add(GRU(32,return_sequences=True))\nmodel.add(GRU(32))\nmodel.add(Dense(1))\n\nmodel.compile(optimizer='RMSprop', loss='mae')\nhistory = model.fit(X_train_scaled, \n                    y_train, \n                    epochs=15,\n                    batch_size=64,\n                    verbose=0)\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-03-05T06:39:27.166453Z","iopub.execute_input":"2023-03-05T06:39:27.166754Z","iopub.status.idle":"2023-03-05T06:39:47.817575Z","shell.execute_reply.started":"2023-03-05T06:39:27.166701Z","shell.execute_reply":"2023-03-05T06:39:47.816843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate model\nfrom sklearn.metrics import mean_absolute_error\n    \ny_val= model.predict(X_train_scaled)\nmae = mean_absolute_error(y_train, y_val)\nprint('%.5f' % mae)","metadata":{"execution":{"iopub.status.busy":"2023-03-05T06:39:51.777993Z","iopub.execute_input":"2023-03-05T06:39:51.778301Z","iopub.status.idle":"2023-03-05T06:39:52.384200Z","shell.execute_reply.started":"2023-03-05T06:39:51.778257Z","shell.execute_reply":"2023-03-05T06:39:52.383287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('../input/sample_submission.csv', index_col='seg_id')","metadata":{"execution":{"iopub.status.busy":"2023-03-05T06:39:55.036377Z","iopub.execute_input":"2023-03-05T06:39:55.036661Z","iopub.status.idle":"2023-03-05T06:39:55.056043Z","shell.execute_reply.started":"2023-03-05T06:39:55.036624Z","shell.execute_reply":"2023-03-05T06:39:55.054329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = pd.DataFrame(columns=X_train.columns, dtype=np.float64, index=submission.index)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-05T06:39:55.383321Z","iopub.execute_input":"2023-03-05T06:39:55.383641Z","iopub.status.idle":"2023-03-05T06:39:55.390701Z","shell.execute_reply.started":"2023-03-05T06:39:55.383584Z","shell.execute_reply":"2023-03-05T06:39:55.389518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for seg_id in X_test.index:\n    seg = pd.read_csv('../input/test/' + seg_id + '.csv')\n    \n    x = seg['acoustic_data'].values\n    \n    X_test.loc[seg_id, 'ave'] = x.mean()\n    X_test.loc[seg_id, 'std'] = x.std()\n    X_test.loc[seg_id, 'max'] = x.max()\n    X_test.loc[seg_id, 'min'] = x.min()","metadata":{"execution":{"iopub.status.busy":"2023-03-05T06:39:55.718551Z","iopub.execute_input":"2023-03-05T06:39:55.718890Z","iopub.status.idle":"2023-03-05T06:41:10.866436Z","shell.execute_reply.started":"2023-03-05T06:39:55.718822Z","shell.execute_reply":"2023-03-05T06:41:10.864911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-05T06:41:10.868786Z","iopub.execute_input":"2023-03-05T06:41:10.869192Z","iopub.status.idle":"2023-03-05T06:41:10.886984Z","shell.execute_reply.started":"2023-03-05T06:41:10.869109Z","shell.execute_reply":"2023-03-05T06:41:10.886206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-05T06:41:10.888095Z","iopub.execute_input":"2023-03-05T06:41:10.888441Z","iopub.status.idle":"2023-03-05T06:41:10.902694Z","shell.execute_reply.started":"2023-03-05T06:41:10.888390Z","shell.execute_reply":"2023-03-05T06:41:10.901232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test_scaled = scaler.transform(X_test)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-05T06:50:28.881783Z","iopub.execute_input":"2023-03-05T06:50:28.882114Z","iopub.status.idle":"2023-03-05T06:50:28.887941Z","shell.execute_reply.started":"2023-03-05T06:50:28.882058Z","shell.execute_reply":"2023-03-05T06:50:28.886613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = X_test_scaled.reshape(X_test_scaled.shape[0],X_test_scaled.shape[1],1)","metadata":{"execution":{"iopub.status.busy":"2023-03-05T06:50:43.249661Z","iopub.execute_input":"2023-03-05T06:50:43.249956Z","iopub.status.idle":"2023-03-05T06:50:43.254152Z","shell.execute_reply.started":"2023-03-05T06:50:43.249915Z","shell.execute_reply":"2023-03-05T06:50:43.253298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-05T06:50:46.151476Z","iopub.execute_input":"2023-03-05T06:50:46.152151Z","iopub.status.idle":"2023-03-05T06:50:46.156377Z","shell.execute_reply.started":"2023-03-05T06:50:46.152094Z","shell.execute_reply":"2023-03-05T06:50:46.155857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['time_to_failure'] = model.predict(X_test)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-05T06:50:49.013910Z","iopub.execute_input":"2023-03-05T06:50:49.014317Z","iopub.status.idle":"2023-03-05T06:50:49.264806Z","shell.execute_reply.started":"2023-03-05T06:50:49.014276Z","shell.execute_reply":"2023-03-05T06:50:49.263915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-05T06:50:55.557421Z","iopub.execute_input":"2023-03-05T06:50:55.557776Z","iopub.status.idle":"2023-03-05T06:50:55.569940Z","shell.execute_reply.started":"2023-03-05T06:50:55.557706Z","shell.execute_reply":"2023-03-05T06:50:55.569029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-03-05T06:50:59.350785Z","iopub.execute_input":"2023-03-05T06:50:59.351075Z","iopub.status.idle":"2023-03-05T06:50:59.361955Z","shell.execute_reply.started":"2023-03-05T06:50:59.351035Z","shell.execute_reply":"2023-03-05T06:50:59.360608Z"},"trusted":true},"execution_count":null,"outputs":[]}]}