{"cells":[{"metadata":{"_uuid":"cbca22b129c737f401ca4ed5d9e5ec6374d70222"},"cell_type":"markdown","source":"Here, Am I and has beginner, I had take many other code referrence and try study  it. \nThen created the model with data featuring. \n\nHere, Shown evey library and used alphabet as symbol for each."},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np \nimport pandas as pd \n\n%matplotlib inline\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm \nfrom xgboost import XGBRegressor as xgbr\nfrom sklearn.pipeline import make_pipeline as mpp\nfrom sklearn.preprocessing import Imputer as imp\nfrom sklearn.ensemble import AdaBoostRegressor as adr\nfrom sklearn.model_selection import cross_val_score as cvs\nfrom sklearn.linear_model import Lasso as lss\nfrom sklearn.linear_model import Ridge as rdg\nfrom sklearn.linear_model import ElasticNet as ecn\nfrom sklearn.metrics import accuracy_score as aus\nfrom sklearn.linear_model import LinearRegression as lmr \nfrom sklearn.metrics import mean_absolute_error as mae\nfrom sklearn.preprocessing import StandardScaler as scl\nfrom sklearn.model_selection import ShuffleSplit as sst\n\nimport time \nimport math","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"%%time\ntrain = pd.read_csv('../input/train.csv', dtype={'acoustic_data': np.int16, 'time_to_failure': np.float32})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b71026536de48d5c5770aadc0e23da0f698b4b42"},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"07584cc7e1914231710750c6aeaeae5d0a358bcd"},"cell_type":"code","source":"train.shape","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"375d64bd62dfc6df89f3888e2c885e4fa7149794"},"cell_type":"markdown","source":"Looking the data flow with graph."},{"metadata":{"trusted":true,"_uuid":"41ea6112f0ce958d0c58eea15e318a0c478e0e00","scrolled":false},"cell_type":"code","source":"tsx = train['acoustic_data'].values[::100]\ntsy = train['time_to_failure'].values[::100]\n\nfig, ax = plt.subplots(figsize = (16, 8))\nplt.title('Tester')\nplt.plot(tsx, color = 'r')\nax.set_ylabel('Data')\nplt.legend(['acoustic_data'])\naxinv = ax.twinx()\n\nplt.plot(tsy, color = 'b')\naxinv.set_ylabel('time_to_failure')\nplt.legend(['time_to_failure'], loc = (0.875, 0.9))\nplt.grid(False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bd8d24cd1696e56dde898b773e22d9bb8e00cbd6"},"cell_type":"markdown","source":"Allocating the numbers  of rows to roll in the loop and segments also. To collect the data which is preprosecced one. Creating two extrenal function ensued the training set.\n"},{"metadata":{"trusted":true,"_uuid":"c9020702f808ae665491680003ddb5a292811e40"},"cell_type":"code","source":"rows = 150_000\nsgt = int(np.floor(train.shape[0] / rows))\nrjp = 75000\nnjp = int(5)\n\ndef data_feature(ar, abs_values = False):\n    \n    idx = np.array(range(len(ar)))\n    \n    if abs_values:\n        ar = np.abs(ar)\n    lr = lmr()\n    lr.fit(idx.reshape(-1, 1), ar)\n    return lr.coef_[0]\n\ndef unknow_func(x, lsa, lla):\n    \n    sta = np.cumsum(x ** 2)\n    sta = np.require(sta, dtype = np.float)\n    lta = sta.copy()\n    \n    sta[lsa:] = sta[lsa:] - sta[:-lsa]\n    sta /= lsa\n    lta[lla:] = lta[lla:] - lta[:-lla]\n    lta /= lla\n         \n    sta[:lla - 1] = 0\n    \n    dty = np.finfo(0.0).tiny\n    idx = lta < dty\n    lta[idx] = dty\n    \n    return sta / lta","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5c32b6e87b12b6f7e4b235ae315f1d64b22eeaeb"},"cell_type":"code","source":"xtr = pd.DataFrame(index = range(sgt), dtype = np.float64)\nytr = pd.DataFrame(index = range(sgt), dtype = np.float64, columns = ['time_to_failure'])\n     \nmn = train['acoustic_data'].mean()\nsd = train['acoustic_data'].std()\nmx = train['acoustic_data'].max()\nmi = train['acoustic_data'].min()\ntt = np.abs(train['acoustic_data']).sum() \n\nfor sgt in tqdm(range(sgt)) :\n    sg = train.iloc[sgt * rows + njp * rjp : sgt * rows + rows + njp * rjp]\n    \n    X = pd.Series(sg['acoustic_data'].values)\n    y = train['time_to_failure'].values[-1]\n    \n    xtr.loc[sgt + njp * sgt, 'part1'] = X.mean()\n    xtr.loc[sgt + njp * sgt, 'part2'] = X.std()\n    xtr.loc[sgt + njp * sgt, 'part3'] = X.max()\n    xtr.loc[sgt + njp * sgt, 'part4'] = X.min()\n    xtr.loc[sgt + njp * sgt, 'part5'] = X.kurtosis()\n    xtr.loc[sgt + njp * sgt, 'part6'] = X.skew()\n    xtr.loc[sgt + njp * sgt, 'part7.0'] = X.quantile()\n    xtr.loc[sgt + njp * sgt, 'part7.1'] = np.count_nonzero(X < np.quantile(X,0.05))\n    xtr.loc[sgt + njp * sgt, 'part7.2'] = np.count_nonzero(X < np.quantile(X,0.010))\n    xtr.loc[sgt + njp * sgt, 'part7.3'] = np.count_nonzero(X > np.quantile(X,0.015))\n    xtr.loc[sgt + njp * sgt, 'part7.4'] = np.count_nonzero(X > np.quantile(X,0.020))\n    xtr.loc[sgt + njp * sgt, 'part7.5'] = np.count_nonzero(X < np.quantile(X,0.025))\n    xtr.loc[sgt + njp * sgt, 'part7.6'] = np.count_nonzero(X < np.quantile(X,0.030))\n    xtr.loc[sgt + njp * sgt, 'part7.7 '] = np.count_nonzero(X > np.quantile(X,0.035))\n    xtr.loc[sgt + njp * sgt, 'part7.8'] = np.count_nonzero(X > np.quantile(X,0.040))\n    xtr.loc[sgt + njp * sgt, 'part7.9'] = np.count_nonzero(X < np.quantile(X,0.045))\n    xtr.loc[sgt + njp * sgt, 'part7.10'] = np.count_nonzero(X < np.quantile(X,0.050))\n    xtr.loc[sgt + njp * sgt, 'part8'] = data_feature(X)\n    xtr.loc[sgt + njp * sgt, 'part9'] = data_feature(X, abs_values = True)\n    xtr.loc[sgt + njp * sgt, 'part10.0'] = unknow_func(X, 500, 10000).mean()\n    xtr.loc[sgt + njp * sgt, 'part10.1'] = unknow_func(X, 625, 25000).mean()\n    xtr.loc[sgt + njp * sgt, 'part11'] = np.abs(X).max()\n    xtr.loc[sgt + njp * sgt, 'part12'] = np.abs(X).min()\n    xtr.loc[sgt + njp * sgt, 'part13'] = X.mean() - X.std()\n    xtr.loc[sgt + njp * sgt, 'part14'] = X.max() - X.min()\n    xtr.loc[sgt + njp * sgt, 'part15'] = np.abs(X).max() - np.abs(X).min()\n\n    for win in [25, 225, 3375] :\n        xrllsd = X.rolling(win).std().dropna().values \n        xrllmn = X.rolling(win).std().dropna().values\n        \n        xtr.loc[sgt + njp * sgt, 'WindowsPartition1.1' + str(win)] = xrllsd.mean()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition1.2' + str(win)] = xrllsd.std()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition1.3' + str(win)] = xrllsd.max()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition1.4' + str(win)] = xrllsd.min()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition1.5' + str(win)] = np.quantile(xrllsd, 0.0125)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition1.6' + str(win)] = np.quantile(xrllsd, 0.0750)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition1.7' + str(win)] = np.quantile(xrllsd, 0.125)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition1.8' + str(win)] = np.quantile(xrllsd, 0.105)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition1.9' + str(win)] = np.mean(np.diff(xrllsd))\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition1.10' + str(win)] = np.mean(np.nonzero((np.diff(xrllsd) / xrllsd[:-1]))[0])\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition1.11' + str(win)] = np.abs(xrllsd).max()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition1.12' + str(win)] = xrllsd.mean() - xrllsd.std()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition1.13' + str(win)] = xrllsd.max()  - xrllsd.min()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition1.14' + str(win)] = np.quantile(xrllsd, 0.0250) - np.quantile(xrllsd, 0.0125)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition1.15' + str(win)] = np.quantile(xrllsd, 0.0900) - np.quantile(xrllsd, 0.0750)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition1.16' + str(win)] = np.quantile(xrllsd, 0.250) - np.quantile(xrllsd, 0.125)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition1.17' + str(win)] = np.quantile(xrllsd, 0.210) - np.quantile(xrllsd, 0.105)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition1.18' + str(win)] = np.quantile(xrllsd, 0.2575) \n        xtr.loc[sgt + njp * sgt, 'WindowsPartition1.19' + str(win)] = np.mean(np.nonzero((np.diff(xrllsd) / xrllsd[:-1]))[0]) - np.abs(xrllsd).max()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition1.20' + str(win)] = np.mean(np.nonzero((np.diff(xrllsd) / xrllsd[:-1]))[0]) + np.abs(xrllsd).max()\n        \n        xtr.loc[sgt + njp * sgt, 'WindowsPartition2.1' + str(win)] = xrllmn.mean()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition2.2' + str(win)] = xrllmn.std()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition2.3' + str(win)] = xrllmn.max()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition2.4' + str(win)] = xrllmn.min()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition2.5' + str(win)] = np.quantile(xrllmn, 0.0125)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition2.6' + str(win)] = np.quantile(xrllmn, 0.0750)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition2.7' + str(win)] = np.quantile(xrllmn, 0.125)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition2.8' + str(win)] = np.quantile(xrllmn, 0.105)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition2.9' + str(win)] = np.mean(np.diff(xrllsd))\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition2.10' + str(win)] = np.mean(np.nonzero((np.diff(xrllmn) / xrllmn[:-1]))[0])\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition2.11' + str(win)] = np.abs(xrllmn).max()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition2.12' + str(win)] = xrllmn.mean() - xrllmn.std()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition2.13' + str(win)] = xrllmn.max()  - xrllmn.min()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition2.14' + str(win)] = np.quantile(xrllmn, 0.0250) - np.quantile(xrllmn, 0.0125)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition2.15' + str(win)] = np.quantile(xrllmn, 0.0900) - np.quantile(xrllmn, 0.0750)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition2.16' + str(win)] = np.quantile(xrllmn, 0.250) - np.quantile(xrllmn, 0.125)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition2.17' + str(win)] =  np.quantile(xrllmn, 0.210) - np.quantile(xrllmn, 0.105)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition2.18' + str(win)] = np.quantile(xrllmn, 0.2575)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition2.19' + str(win)] = np.mean(np.nonzero((np.diff(xrllmn) / xrllmn[:-1]))[0]) - np.abs(xrllmn).max()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition2.20' + str(win)] = np.mean(np.nonzero((np.diff(xrllmn) / xrllmn[:-1]))[0]) + np.abs(xrllmn).max()\n        \n    for win in [50, 375, 6225] :\n        xrllsd = X.rolling(win).std().dropna().values \n        xrllmn = X.rolling(win).std().dropna().values\n        \n        xtr.loc[sgt + njp * sgt, 'WindowsPartition3.1' + str(win)] = xrllsd.mean()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition3.2' + str(win)] = xrllsd.std()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition3.3' + str(win)] = xrllsd.max()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition3.4' + str(win)] = xrllsd.min()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition3.5' + str(win)] = np.quantile(xrllsd, 0.0125)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition3.6' + str(win)] = np.quantile(xrllsd, 0.0750)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition3.7' + str(win)] = np.quantile(xrllsd, 0.125)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition3.8' + str(win)] = np.quantile(xrllsd, 0.105)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition3.9' + str(win)] = np.mean(np.diff(xrllsd))\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition3.10' + str(win)] = np.mean(np.nonzero((np.diff(xrllsd) / xrllsd[:-1]))[0])\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition3.11' + str(win)] = np.abs(xrllsd).max()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition3.12' + str(win)] = xrllsd.mean() - xrllsd.std()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition3.13' + str(win)] = xrllsd.max()  - xrllsd.min()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition3.14' + str(win)] = np.quantile(xrllsd, 0.0250) - np.quantile(xrllsd, 0.0125)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition3.15' + str(win)] = np.quantile(xrllsd, 0.0900) - np.quantile(xrllsd, 0.0750)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition3.16' + str(win)] = np.quantile(xrllsd, 0.250) - np.quantile(xrllsd, 0.125)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition3.17' + str(win)] = np.quantile(xrllsd, 0.210) - np.quantile(xrllsd, 0.105)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition3.18' + str(win)] = np.quantile(xrllsd, 0.2575) \n        xtr.loc[sgt + njp * sgt, 'WindowsPartition3.19' + str(win)] = np.mean(np.nonzero((np.diff(xrllsd) / xrllsd[:-1]))[0]) - np.abs(xrllsd).max()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition3.20' + str(win)] = np.mean(np.nonzero((np.diff(xrllsd) / xrllsd[:-1]))[0]) + np.abs(xrllsd).max()\n        \n        xtr.loc[sgt + njp * sgt, 'WindowsPartition4.1' + str(win)] = xrllmn.mean()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition4.2' + str(win)] = xrllmn.std()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition4.3' + str(win)] = xrllmn.max()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition4.4' + str(win)] = xrllmn.min()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition4.5' + str(win)] = np.quantile(xrllmn, 0.0125)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition4.6' + str(win)] = np.quantile(xrllmn, 0.0750)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition4.7' + str(win)] = np.quantile(xrllmn, 0.125)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition4.8' + str(win)] = np.quantile(xrllmn, 0.105)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition4.9' + str(win)] = np.mean(np.diff(xrllsd))\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition4.10' + str(win)] = np.mean(np.nonzero((np.diff(xrllmn) / xrllmn[:-1]))[0])\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition4.11' + str(win)] = np.abs(xrllmn).max()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition4.12' + str(win)] = xrllmn.mean() - xrllmn.std()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition4.13' + str(win)] = xrllmn.max()  - xrllmn.min()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition4.14' + str(win)] = np.quantile(xrllmn, 0.0250) - np.quantile(xrllmn, 0.0125)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition4.15' + str(win)] = np.quantile(xrllmn, 0.0900) - np.quantile(xrllmn, 0.0750)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition4.16' + str(win)] = np.quantile(xrllmn, 0.250) - np.quantile(xrllmn, 0.125)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition4.17' + str(win)] = np.quantile(xrllmn, 0.210) - np.quantile(xrllmn, 0.105)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition4.18' + str(win)] = np.quantile(xrllmn, 0.2575)\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition4.19' + str(win)] = np.mean(np.nonzero((np.diff(xrllmn) / xrllmn[:-1]))[0]) - np.abs(xrllmn).max()\n        xtr.loc[sgt + njp * sgt, 'WindowsPartition4.20' + str(win)] = np.mean(np.nonzero((np.diff(xrllmn) / xrllmn[:-1]))[0]) + np.abs(xrllmn).max()\n            \n    ytr.loc[sgt + njp * sgt, 'time_to_failure'] = y\n    pass","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1455401fac53800db7df92c3fe9eccc1e2ba45ec"},"cell_type":"markdown","source":"Scaling the training set for the prediction and load the data for the model also."},{"metadata":{"trusted":true,"_uuid":"5f62bdf61818e7aaed831bc8c8b1f95dddd922c0"},"cell_type":"code","source":"sc = scl()\nsc.fit(xtr)\n\nsxtr = pd.DataFrame(sc.transform(xtr), columns = xtr.columns)\n\nytr.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b313e1afa44404badc11ddd308a4938ad7279292"},"cell_type":"markdown","source":"Models which are deployed for the foretell. And there score on the based on mean squared error which is going to directly added the average to the test data. "},{"metadata":{"trusted":true,"_uuid":"625b7af6032bfbe211ba686964b45aca79704f1b","scrolled":true},"cell_type":"code","source":"lm = lmr()\nlm.fit(sxtr, ytr)\n\nlmpr = lm.predict(sxtr)\ns = mae(ytr, lmpr)\nprint(s)\n\nxgbm = xgbr()\nxgbm.fit(sxtr, ytr)\n\nxgbpr = xgbm.predict(sxtr)\nxgbs = mae(ytr, xgbpr)\nprint(xgbs)\n\nlaso = lss(alpha = 0.1)\nlaso.fit(sxtr, ytr)\n\n\nlspr = laso.predict(sxtr)\nlsss = mae(ytr, lspr)\nprint(lsss)\n\nrdgm = rdg()\nrdgm.fit(sxtr, ytr)\n\nrdgr = rdgm.predict(sxtr)\nrdgs = mae(ytr, rdgr)\nprint(rdgs)\n\necnm = ecn()\necnm.fit(sxtr, ytr)\n\necnpr = ecnm.predict(sxtr)\necnms = mae(ytr, ecnpr)\nprint(ecnms)\n\nadrm = adr()\nadrm.fit(sxtr, ytr)\n\nadrpr = adrm.predict(sxtr)\nadrms = mae(ytr, adrpr)\nprint(adrms)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d58a9c2eb9ca91efa678808d2bb5a97aed4c505d"},"cell_type":"markdown","source":"Submission file."},{"metadata":{"trusted":true,"_uuid":"c536a148cc5d1b09832f6f2c52b5d966a3dd1f90"},"cell_type":"code","source":"def sigmoid(z):\n    \n    return 1 / (1 + np.exp(-z))\n\na = np.array(xtr)\nb = sigmoid(a)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"663e89bfed0a3bc53278e0eefce13668d5262401"},"cell_type":"code","source":"def cost(theta, e, d, lr) :\n    theta = np.matrix(theta) \n    e = np.matrix(e)\n    d = np.matrix(d)\n        \n    fst = np.multiply(-e, np.log(sigmoid(d * theta.T)))\n    scd = np.multiply((1 - e), np.log(1 - sigmoid(d * theta.T)))\n    reg = (lr / 2 * len(e)) * np.sum(np.power(theta[:, 1:theta.shape[1]], 2))\n    \n    return np.sum(fst - scd) / (len(e)) + reg \n\ne = train['acoustic_data'].values[-1]\nf = train['time_to_failure'].values[-1]\n\nef = cost(0.13741, e, f, 0.25)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d37bdb2243e8e6795231c48529a7d86140e76fee"},"cell_type":"code","source":"sub = pd.read_csv('../input/sample_submission.csv', index_col='seg_id')\n\nxtst = pd.DataFrame(columns = xtr.columns, dtype = np.float64, index = sub.index)\n\nfor i, seg_id in enumerate(tqdm(xtst.index)) :\n    seg = pd.read_csv('../input/test/' + seg_id + '.csv')\n    \n    X = seg['acoustic_data']\n    \n    xtst.loc[seg_id, 'part1'] = X.mean()\n    xtst.loc[seg_id, 'part2'] = X.std()\n    xtst.loc[seg_id, 'part3'] = X.max()\n    xtst.loc[seg_id, 'part4'] = X.min()\n    xtst.loc[seg_id, 'part5'] = X.kurtosis()\n    xtst.loc[seg_id, 'part6'] = X.skew()\n    xtst.loc[seg_id, 'part7.0'] = X.quantile()\n    xtst.loc[seg_id, 'part7.1'] = np.count_nonzero(X < np.quantile(X,0.05))\n    xtst.loc[seg_id, 'part7.2'] = np.count_nonzero(X < np.quantile(X,0.010))\n    xtst.loc[seg_id, 'part7.3'] = np.count_nonzero(X > np.quantile(X,0.015))\n    xtst.loc[seg_id, 'part7.4'] = np.count_nonzero(X > np.quantile(X,0.020))\n    xtst.loc[seg_id, 'part7.5'] = np.count_nonzero(X < np.quantile(X,0.025))\n    xtst.loc[seg_id, 'part7.6'] = np.count_nonzero(X < np.quantile(X,0.030))\n    xtst.loc[seg_id, 'part7.7 '] = np.count_nonzero(X > np.quantile(X,0.035))\n    xtst.loc[seg_id, 'part7.8'] = np.count_nonzero(X > np.quantile(X,0.040))\n    xtst.loc[seg_id, 'part7.9'] = np.count_nonzero(X < np.quantile(X,0.045))\n    xtst.loc[seg_id, 'part7.10'] = np.count_nonzero(X < np.quantile(X,0.050))\n    xtst.loc[seg_id, 'part8'] = data_feature(X)\n    xtst.loc[seg_id, 'part9'] = data_feature(X, abs_values = True)\n    xtst.loc[seg_id, 'part10.0'] = unknow_func(X, 500, 10000).mean()\n    xtst.loc[seg_id, 'part10.1'] = unknow_func(X, 625, 25000).mean()\n    xtst.loc[seg_id, 'part11'] = np.abs(X).max()\n    xtst.loc[seg_id, 'part12'] = np.abs(X).min()\n    xtst.loc[seg_id, 'part13'] = X.mean() - X.std()\n    xtst.loc[seg_id, 'part14'] = X.max() - X.min()\n    xtst.loc[seg_id, 'part15'] = np.abs(X).max() - np.abs(X).min()\n    \n    for win in [25, 225, 3375] :\n        xrllsd = X.rolling(win).std().dropna().values \n        xrllmn = X.rolling(win).std().dropna().values\n        \n        xtst.loc[seg_id, 'WindowsPartition1.1' + str(win)] = xrllsd.mean()\n        xtst.loc[seg_id, 'WindowsPartition1.2' + str(win)] = xrllsd.std()\n        xtst.loc[seg_id, 'WindowsPartition1.3' + str(win)] = xrllsd.max()\n        xtst.loc[seg_id, 'WindowsPartition1.4' + str(win)] = xrllsd.min()\n        xtst.loc[seg_id, 'WindowsPartition1.5' + str(win)] = np.quantile(xrllsd, 0.0125)\n        xtst.loc[seg_id, 'WindowsPartition1.6' + str(win)] = np.quantile(xrllsd, 0.0750)\n        xtst.loc[seg_id, 'WindowsPartition1.7' + str(win)] = np.quantile(xrllsd, 0.125)\n        xtst.loc[seg_id, 'WindowsPartition1.8' + str(win)] = np.quantile(xrllsd, 0.105)\n        xtst.loc[seg_id, 'WindowsPartition1.9' + str(win)] = np.mean(np.diff(xrllsd))\n        xtst.loc[seg_id, 'WindowsPartition1.10' + str(win)] = np.mean(np.nonzero((np.diff(xrllsd) / xrllsd[:-1]))[0])\n        xtst.loc[seg_id, 'WindowsPartition1.11' + str(win)] = np.abs(xrllsd).max()\n        xtst.loc[seg_id, 'WindowsPartition1.12' + str(win)] = xrllsd.mean() - xrllsd.std()\n        xtst.loc[seg_id, 'WindowsPartition1.13' + str(win)] = xrllsd.max()  - xrllsd.min()\n        xtst.loc[seg_id, 'WindowsPartition1.14' + str(win)] = np.quantile(xrllsd, 0.0250) - np.quantile(xrllsd, 0.0125)\n        xtst.loc[seg_id, 'WindowsPartition1.15' + str(win)] = np.quantile(xrllsd, 0.0900) - np.quantile(xrllsd, 0.0750)\n        xtst.loc[seg_id, 'WindowsPartition1.16' + str(win)] = np.quantile(xrllsd, 0.250) - np.quantile(xrllsd, 0.125)\n        xtst.loc[seg_id, 'WindowsPartition1.17' + str(win)] = np.quantile(xrllsd, 0.210) - np.quantile(xrllsd, 0.105)\n        xtst.loc[seg_id, 'WindowsPartition1.18' + str(win)] = np.quantile(xrllsd, 0.2575) \n        xtst.loc[seg_id, 'WindowsPartition1.19' + str(win)] = np.mean(np.nonzero((np.diff(xrllsd) / xrllsd[:-1]))[0]) - np.abs(xrllsd).max()\n        xtst.loc[seg_id, 'WindowsPartition1.20' + str(win)] = np.mean(np.nonzero((np.diff(xrllsd) / xrllsd[:-1]))[0]) + np.abs(xrllsd).max()\n        \n        \n        xtst.loc[seg_id, 'WindowsPartition2.1' + str(win)] = xrllmn.mean()\n        xtst.loc[seg_id, 'WindowsPartition2.2' + str(win)] = xrllmn.std()\n        xtst.loc[seg_id, 'WindowsPartition2.3' + str(win)] = xrllmn.max()\n        xtst.loc[seg_id, 'WindowsPartition2.4' + str(win)] = xrllmn.min()\n        xtst.loc[seg_id, 'WindowsPartition2.5' + str(win)] = np.quantile(xrllmn, 0.0125)\n        xtst.loc[seg_id, 'WindowsPartition2.6' + str(win)] = np.quantile(xrllmn, 0.0750)\n        xtst.loc[seg_id, 'WindowsPartition2.7' + str(win)] = np.quantile(xrllmn, 0.125)\n        xtst.loc[seg_id, 'WindowsPartition2.8' + str(win)] = np.quantile(xrllmn, 0.105)\n        xtst.loc[seg_id, 'WindowsPartition2.9' + str(win)] = np.mean(np.diff(xrllsd))\n        xtst.loc[seg_id, 'WindowsPartition2.10' + str(win)] = np.mean(np.nonzero((np.diff(xrllmn) / xrllmn[:-1]))[0])\n        xtst.loc[seg_id, 'WindowsPartition2.11' + str(win)] = np.abs(xrllmn).max()        \n        xtst.loc[seg_id, 'WindowsPartition2.12' + str(win)] = xrllmn.mean() - xrllmn.std()\n        xtst.loc[seg_id, 'WindowsPartition2.13' + str(win)] = xrllmn.max()  - xrllmn.min()\n        xtst.loc[seg_id, 'WindowsPartition2.14' + str(win)] = np.quantile(xrllmn, 0.0250) - np.quantile(xrllmn, 0.0125)\n        xtst.loc[seg_id, 'WindowsPartition2.15' + str(win)] = np.quantile(xrllmn, 0.0900) - np.quantile(xrllmn, 0.0750)\n        xtst.loc[seg_id, 'WindowsPartition2.16' + str(win)] = np.quantile(xrllmn, 0.250) - np.quantile(xrllmn, 0.125)\n        xtst.loc[seg_id, 'WindowsPartition2.17' + str(win)] = np.quantile(xrllmn, 0.210) - np.quantile(xrllmn, 0.105)\n        xtst.loc[seg_id, 'WindowsPartition2.18' + str(win)] = np.quantile(xrllmn, 0.2575)\n        xtst.loc[seg_id, 'WindowsPartition2.19' + str(win)] = np.mean(np.nonzero((np.diff(xrllmn) / xrllmn[:-1]))[0]) - np.abs(xrllmn).max()\n        xtst.loc[seg_id, 'WindowsPartition2.20' + str(win)] = np.mean(np.nonzero((np.diff(xrllmn) / xrllmn[:-1]))[0]) + np.abs(xrllmn).max()\n        \n    for win in [50, 375, 6225] :\n        xrllsd = X.rolling(win).std().dropna().values \n        xrllmn = X.rolling(win).std().dropna().values\n        \n        xtst.loc[seg_id, 'WindowsPartition3.1' + str(win)] = xrllsd.mean()\n        xtst.loc[seg_id, 'WindowsPartition3.2' + str(win)] = xrllsd.std()\n        xtst.loc[seg_id, 'WindowsPartition3.3' + str(win)] = xrllsd.max()\n        xtst.loc[seg_id, 'WindowsPartition3.4' + str(win)] = xrllsd.min()\n        xtst.loc[seg_id, 'WindowsPartition3.5' + str(win)] = np.quantile(xrllsd, 0.0125)\n        xtst.loc[seg_id, 'WindowsPartition3.6' + str(win)] = np.quantile(xrllsd, 0.0750)\n        xtst.loc[seg_id, 'WindowsPartition3.7' + str(win)] = np.quantile(xrllsd, 0.125)\n        xtst.loc[seg_id, 'WindowsPartition3.8' + str(win)] = np.quantile(xrllsd, 0.105)\n        xtst.loc[seg_id, 'WindowsPartition3.9' + str(win)] = np.mean(np.diff(xrllsd))\n        xtst.loc[seg_id, 'WindowsPartition3.10' + str(win)] = np.mean(np.nonzero((np.diff(xrllsd) / xrllsd[:-1]))[0])\n        xtst.loc[seg_id, 'WindowsPartition3.11' + str(win)] = np.abs(xrllsd).max()\n        xtst.loc[seg_id, 'WindowsPartition3.12' + str(win)] = xrllsd.mean() - xrllsd.std()\n        xtst.loc[seg_id, 'WindowsPartition3.13' + str(win)] = xrllsd.max()  - xrllsd.min()\n        xtst.loc[seg_id, 'WindowsPartition3.14' + str(win)] = np.quantile(xrllsd, 0.0250) - np.quantile(xrllsd, 0.0125)\n        xtst.loc[seg_id, 'WindowsPartition3.15' + str(win)] = np.quantile(xrllsd, 0.0900) - np.quantile(xrllsd, 0.0750)\n        xtst.loc[seg_id, 'WindowsPartition3.16' + str(win)] = np.quantile(xrllsd, 0.250) - np.quantile(xrllsd, 0.125)\n        xtst.loc[seg_id, 'WindowsPartition3.17' + str(win)] = np.quantile(xrllsd, 0.210) - np.quantile(xrllsd, 0.105)\n        xtst.loc[seg_id, 'WindowsPartition3.18' + str(win)] = np.quantile(xrllsd, 0.2575) \n        xtst.loc[seg_id, 'WindowsPartition3.19' + str(win)] = np.mean(np.nonzero((np.diff(xrllsd) / xrllsd[:-1]))[0]) - np.abs(xrllsd).max()\n        xtst.loc[seg_id, 'WindowsPartition3.20' + str(win)] = np.mean(np.nonzero((np.diff(xrllsd) / xrllsd[:-1]))[0]) + np.abs(xrllsd).max()\n        \n        xtst.loc[seg_id, 'WindowsPartition4.1' + str(win)] = xrllmn.mean()\n        xtst.loc[seg_id, 'WindowsPartition4.2' + str(win)] = xrllmn.std()\n        xtst.loc[seg_id, 'WindowsPartition4.3' + str(win)] = xrllmn.max()\n        xtst.loc[seg_id, 'WindowsPartition4.4' + str(win)] = xrllmn.min()\n        xtst.loc[seg_id, 'WindowsPartition4.5' + str(win)] = np.quantile(xrllmn, 0.0125)\n        xtst.loc[seg_id, 'WindowsPartition4.6' + str(win)] = np.quantile(xrllmn, 0.0750)\n        xtst.loc[seg_id, 'WindowsPartition4.7' + str(win)] = np.quantile(xrllmn, 0.125)\n        xtst.loc[seg_id, 'WindowsPartition4.8' + str(win)] = np.quantile(xrllmn, 0.105)\n        xtst.loc[seg_id, 'WindowsPartition4.9' + str(win)] = np.mean(np.diff(xrllsd))\n        xtst.loc[seg_id, 'WindowsPartition4.10' + str(win)] = np.mean(np.nonzero((np.diff(xrllmn) / xrllmn[:-1]))[0])\n        xtst.loc[seg_id, 'WindowsPartition4.11' + str(win)] = np.abs(xrllmn).max()\n        xtst.loc[seg_id, 'WindowsPartition4.12' + str(win)] = xrllmn.mean() - xrllmn.std()\n        xtst.loc[seg_id, 'WindowsPartition4.13' + str(win)] = xrllmn.max()  - xrllmn.min()\n        xtst.loc[seg_id, 'WindowsPartition4.14' + str(win)] = np.quantile(xrllmn, 0.0250) - np.quantile(xrllmn, 0.0125)\n        xtst.loc[seg_id, 'WindowsPartition4.15' + str(win)] = np.quantile(xrllmn, 0.0900) - np.quantile(xrllmn, 0.0750)\n        xtst.loc[seg_id, 'WindowsPartition4.16' + str(win)] = np.quantile(xrllmn, 0.250) - np.quantile(xrllmn, 0.125)\n        xtst.loc[seg_id, 'WindowsPartition4.17' + str(win)] =  np.quantile(xrllmn, 0.210) - np.quantile(xrllmn, 0.105)\n        xtst.loc[seg_id, 'WindowsPartition4.18' + str(win)] = np.quantile(xrllmn, 0.2575)\n        xtst.loc[seg_id, 'WindowsPartition4.19' + str(win)] = np.mean(np.nonzero((np.diff(xrllmn) / xrllmn[:-1]))[0]) - np.abs(xrllmn).max()\n        xtst.loc[seg_id, 'WindowsPartition4.20' + str(win)] = np.mean(np.nonzero((np.diff(xrllmn) / xrllmn[:-1]))[0]) + np.abs(xrllmn).max()\n            \n    pass","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ce4023f1119c31de869fad6d5bde86335a70e7c4"},"cell_type":"code","source":"sc.fit(xtst)\nxtsc = pd.DataFrame(sc.transform(xtst), columns = xtst.columns)\n\nmp1 = mpp(imp(), xgbr())\nmp2 = mpp(imp(), rdg())\nmp3 = mpp(imp(), adr())\nmp4 = mpp(imp(), lss())\nmp1.fit(sxtr, ytr)\nmp2.fit(sxtr, ytr)\nmp3.fit(sxtr, ytr)\nmp4.fit(sxtr, ytr)\n\nmppr1 = mp1.predict(xtsc)\nmppr2 = mp2.predict(xtsc)\nmppr3 = mp3.predict(xtsc)\nmppr4 = mp4.predict(xtsc)\n\nab = (lm.predict(xtsc) + ecnm.predict(xtsc))  / 4\ncd = (mppr1 + mppr2 + mppr3 + mppr4) / 16 \n\nzz = ((ab + cd) / 2) + ef","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"18b49d3b025311e6b9cd948bddc8c89c6e91bdcf"},"cell_type":"code","source":"sub['time_to_failure'] = zz \n\nprint(sub)\nsub.to_csv('sub.csv', index = True)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}