{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Denoising the signal by Singular Value Decomposition."},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport warnings\nfrom tqdm import tqdm_notebook\nimport matplotlib.pyplot as plt\n% matplotlib inline\nwarnings.filterwarnings('ignore')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Define SVD functions"},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_data(data, row, col):\n    return np.asarray(data).reshape((row, col))\n\ndef svd_decompose(data):\n    U, sigma, V = np.linalg.svd(data)\n    sigma_norm = sigma / np.sum(sigma)\n    #print('sigma is: ', sigma_norm)\n    #print('cumulative sigma is', np.cumsum(sigma_norm))\n    return U, sigma, V\n\ndef svd_recompose(data, K):\n    U, sigma, V = svd_decompose(data)\n    sig = np.zeros((K, K))\n    for i in range(K):\n        sig[i, i] = sigma[i]\n    sig = np.mat(sig)\n    recon = U[:, :K] * sig * V[:K, :]\n    return recon.getA().reshape(-1)\n\ndef svd_plot(before, after):\n    fig = plt.figure().add_subplot(111)\n    fig.plot(np.asarray(before), 'b*', label='Original data')\n    fig.plot(after, 'r-', label='Processed data')\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Compare signal before and after SVD"},{"metadata":{"trusted":true},"cell_type":"code","source":"df=pd.read_csv('../input/test/seg_0a0fbb.csv')\nsvd_df=(svd_recompose(load_data(df, 300, 500), 5))\nsvd_plot(df, svd_df)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"\nFirst Difference"},{"metadata":{"trusted":true},"cell_type":"code","source":"diff_df=abs(df).diff().fillna(0).values\nsvd_diff_df=(svd_recompose(load_data(diff_df, 300, 500), 5))\nsvd_plot(diff_df, svd_diff_df)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Second Difference"},{"metadata":{"trusted":true},"cell_type":"code","source":"diff_df=pd.Series(diff_df.flatten())\ndiff_df2=abs(diff_df).diff().fillna(0).values\nsvd_diff_df2=(svd_recompose(load_data(diff_df2, 300, 500), 5))\nsvd_plot(diff_df2, svd_diff_df2)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Pearson Correlation Coefficience "},{"metadata":{"trusted":true},"cell_type":"code","source":"def create_features(df, segment, seg_id, filter=True, window_range=[10]):\n    x = pd.Series(segment['acoustic_data'].values)\n    if filter == True:\n        x = svd_recompose(load_data(x, 300, 500), 80)\n        x = pd.Series(x)\n    for windows in window_range:\n        X_roll_kurtosis = x.rolling(windows).kurt().dropna().values\n        df.loc[seg_id, 'q025_roll_kurt_' + str(windows)] = np.quantile(X_roll_kurtosis, 0.025)\n        df.loc[seg_id, 'q050_roll_kurt_' + str(windows)] = np.quantile(X_roll_kurtosis, 0.05)\n        df.loc[seg_id, 'q100_roll_kurt_' + str(windows)] = np.quantile(X_roll_kurtosis, 0.1)\n        df.loc[seg_id, 'q400_roll_kurt_' + str(windows)] = np.quantile(X_roll_kurtosis, 0.4)\n        \ntrain = pd.read_csv('../input/train.csv', dtype={'acoustic_data': np.int16, 'time_to_failure': np.float32})  #\nrows = 150000\nsegments = int(np.floor(train.shape[0] / rows))\nX_tr = pd.DataFrame(index=range(segments), dtype=np.float64)\ny_tr = pd.DataFrame(index=range(segments), dtype=np.float64, columns=['time_to_failure'])\n\nfor segment in tqdm_notebook(range(segments)):\n    seg = train.iloc[segment * rows:segment * rows + rows]\n    y = seg['time_to_failure'].values[-1]\n    y_tr.loc[segment, 'time_to_failure'] = y\n    create_features(df=X_tr, segment=seg, seg_id=segment, filter=False)\ntr = pd.concat([X_tr, y_tr], axis=1)\nprint((np.abs(tr.corrwith(tr['time_to_failure'])).sort_values(ascending=False)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for segment in tqdm_notebook(range(segments)):\n    seg = train.iloc[segment * rows:segment * rows + rows]\n    y = seg['time_to_failure'].values[-1]\n    y_tr.loc[segment, 'time_to_failure'] = y\n    create_features(df=X_tr, segment=seg, seg_id=segment, filter=True)\ntr = pd.concat([X_tr, y_tr], axis=1)\nprint((np.abs(tr.corrwith(tr['time_to_failure'])).sort_values(ascending=False)))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"It seems SVD filters improve the performance of roll_kurtosis features."},{"metadata":{},"cell_type":"markdown","source":""}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}