{"cells":[{"metadata":{"_uuid":"390fe13a8427f1ea4a6ed89bc34e1ce175f12c18"},"cell_type":"markdown","source":"# This Script has few functions for generating features from the signal data we have for each signal. Most of the Help is taken from tsfresh package."},{"metadata":{"trusted":false,"_uuid":"00f72fada5b2ef6382c62dab22df3d704b49cbf5"},"cell_type":"code","source":"import pandas as pd, numpy as np\nimport scipy.stats as ss, statsmodels,pywt\nfrom scipy import signal","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"42c7438a46b1b1af5506648074ede4d93877aa53"},"cell_type":"code","source":"param1 =[{'coeff': 0, 'attr': 'abs'},{'coeff': 1, 'attr': 'abs'},{'coeff': 2, 'attr': 'abs'},{'coeff': 3, 'attr': 'abs'},\n       {'coeff': 4, 'attr': 'abs'},{'coeff': 5, 'attr': 'abs'},{'coeff': 6, 'attr': 'abs'},{'coeff': 7, 'attr': 'abs'},\n       {'coeff': 8, 'attr': 'abs'},{'coeff': 9, 'attr': 'abs'},{'coeff': 10, 'attr': 'abs'},{'coeff': 11, 'attr': 'abs'},\n        {'coeff': 12, 'attr': 'abs'},{'coeff': 13, 'attr': 'abs'},{'coeff': 14, 'attr': 'abs'},{'coeff': 15, 'attr': 'abs'}]\nparam2 = [{\"aggtype\": \"centroid\"},{\"aggtype\": 'variance'},{'aggtype':'skew'},{'aggtype':'kurtosis'}]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"30ab007f793b9183ede34b00bd78882694e1e5bb"},"cell_type":"code","source":"# x is the each individual signal\ndef width_char(x):\n    indices = signal.find_peaks(x)[0]\n    character = signal.peak_widths(x,indices)\n    return character[0],character[1]\n\ndef prominences(x):\n    indices = signal.find_peaks(x)[0]\n    Prominence = signal.peak_prominences(x,indices)\n    Prominence = Prominence[0]\n    return Prominence.mean(),Prominence.max(),Prominence.min(),Prominence.std(),Prominence.var()\n\ndef abs_energy(x):\n    if not isinstance(x, (np.ndarray, pd.Series)):\n        x = np.asarray(x)\n    return np.dot(x, x)\n\ndef complexity(x,normalize=True):\n    if not isinstance(x, (np.ndarray, pd.Series)):\n        x = np.asarray(x)\n    if normalize:\n        s = np.std(x)\n        if s!=0:\n            x = (x - np.mean(x))/s\n        else:\n            return 0.0\n    x = np.diff(x)\n    return np.sqrt(np.dot(x, x))\n\ndef _roll(a, shift):\n    if not isinstance(a, np.ndarray):\n        a = np.asarray(a)\n    idx = shift % len(a)\n    return np.concatenate([a[-idx:], a[:-idx]])\n\ndef non_linearity(x,lag):\n    if not isinstance(x, (np.ndarray, pd.Series)):\n        x = np.asarray(x)\n    n = x.size\n    if 2 * lag >= n:\n        return 0\n    else:\n        return np.mean((_roll(x, 2 * -lag) * _roll(x, -lag) * x)[0:(n - 2 * lag)])\n    \ndef binned_entropy(x):\n    if not isinstance(x, (np.ndarray, pd.Series)):\n        x = np.asarray(x)\n    hist, bin_edges = np.histogram(x, bins=4)\n    probs = hist / x.size\n    return - np.sum(p * np.math.log(p) for p in probs if p != 0)\n\ndef range_count(x, min, max):\n    return np.sum((x >= min) & (x < max))\n\ndef sum_of_reoccurring_data_points(x):\n    unique, counts = np.unique(x, return_counts=True)\n    counts[counts < 2] = 0\n    return np.sum(counts * unique)\n\ndef first_location_of_minimum(x):\n    if not isinstance(x, (np.ndarray, pd.Series)):\n        x = np.asarray(x)\n    return np.argmin(x) / len(x) if len(x) > 0 else np.NaN\n\ndef last_location_of_minimum(x):\n    x = np.asarray(x)\n    return 1.0 - np.argmin(x[::-1]) / len(x) if len(x) > 0 else np.NaN\n\ndef last_location_of_maximum(x):\n    x = np.asarray(x)\n    return 1.0 - np.argmax(x[::-1]) / len(x) if len(x) > 0 else np.NaN\n\ndef first_location_of_maximum(x):\n    if not isinstance(x, (np.ndarray, pd.Series)):\n        x = np.asarray(x)\n    return np.argmax(x) / len(x) if len(x) > 0 else np.NaN\n\ndef energy_wavelet(x):\n    decomposition = pywt.wavedec(x,'db4')[0]\n    return decomposition,np.sqrt(np.sum(np.array(decomposition) ** 2)) / len(decomposition)\n\ndef energy_detailed_coeff(x):\n    cA,cD = pywt.dwt(x,'db4')\n    return np.sqrt(np.sum(np.array(cD) ** 2)) / len(cD)\n\n# for fft_aggregated , use param2\ndef fft_aggregated(x, param):\n    \"\"\"\n    Returns the spectral centroid (mean), variance, skew, and kurtosis of the absolute fourier transform spectrum.\n\n    :param x: the time series to calculate the feature of\n    :type x: pandas.Series\n    :param param: contains dictionaries {\"aggtype\": s} where s str and in [\"centroid\", \"variance\",\n        \"skew\", \"kurtosis\"]\n    :type param: list\n    :return: the different feature values\n    :return type: pandas.Series\n    \"\"\"\n\n    assert set([config[\"aggtype\"] for config in param]) <= set([\"centroid\", \"variance\", \"skew\", \"kurtosis\"]), \\\n        'Attribute must be \"centroid\", \"variance\", \"skew\", \"kurtosis\"'\n\n\n    def get_moment(y, moment):\n        \"\"\"\n        Returns the (non centered) moment of the distribution y:\n        E[y**moment] = \\sum_i[index(y_i)^moment * y_i] / \\sum_i[y_i]\n        \n        :param y: the discrete distribution from which one wants to calculate the moment \n        :type y: pandas.Series or np.array\n        :param moment: the moment one wants to calcalate (choose 1,2,3, ... )\n        :type moment: int\n        :return: the moment requested\n        :return type: float\n        \"\"\"\n        return y.dot(np.arange(len(y))**moment) / y.sum()\n\n    def get_centroid(y):\n        \"\"\"\n        :param y: the discrete distribution from which one wants to calculate the centroid \n        :type y: pandas.Series or np.array\n        :return: the centroid of distribution y (aka distribution mean, first moment)\n        :return type: float \n        \"\"\"\n        return get_moment(y, 1)\n\n    def get_variance(y):\n        \"\"\"\n        :param y: the discrete distribution from which one wants to calculate the variance \n        :type y: pandas.Series or np.array\n        :return: the variance of distribution y\n        :return type: float \n        \"\"\"\n        return get_moment(y, 2) - get_centroid(y) ** 2\n\n    def get_skew(y):\n        \"\"\"\n        Calculates the skew as the third standardized moment.\n        Ref: https://en.wikipedia.org/wiki/Skewness#Definition\n        \n        :param y: the discrete distribution from which one wants to calculate the skew \n        :type y: pandas.Series or np.array\n        :return: the skew of distribution y\n        :return type: float \n        \"\"\"\n\n        variance = get_variance(y)\n        # In the limit of a dirac delta, skew should be 0 and variance 0.  However, in the discrete limit,\n        # the skew blows up as variance --> 0, hence return nan when variance is smaller than a resolution of 0.5:\n        if variance < 0.5:\n            return np.nan\n        else:\n            return (\n                get_moment(y, 3) - 3*get_centroid(y)*variance - get_centroid(y)**3\n            ) / get_variance(y)**(1.5)\n\n    def get_kurtosis(y):\n        \"\"\"\n        Calculates the kurtosis as the fourth standardized moment.\n        Ref: https://en.wikipedia.org/wiki/Kurtosis#Pearson_moments\n        \n        :param y: the discrete distribution from which one wants to calculate the kurtosis \n        :type y: pandas.Series or np.array\n        :return: the kurtosis of distribution y\n        :return type: float \n        \"\"\"\n\n        variance = get_variance(y)\n        # In the limit of a dirac delta, kurtosis should be 3 and variance 0.  However, in the discrete limit,\n        # the kurtosis blows up as variance --> 0, hence return nan when variance is smaller than a resolution of 0.5:\n        if variance < 0.5:\n            return np.nan\n        else:\n            return (\n                get_moment(y, 4) - 4*get_centroid(y)*get_moment(y, 3)\n                + 6*get_moment(y, 2)*get_centroid(y)**2 - 3*get_centroid(y)\n            ) / get_variance(y)**2\n\n    calculation = dict(\n        centroid=get_centroid,\n        variance=get_variance,\n        skew=get_skew,\n        kurtosis=get_kurtosis\n    )\n\n    fft_abs = np.abs(np.fft.rfft(x))\n\n    res = [calculation[config[\"aggtype\"]](fft_abs) for config in param]\n    index = ['aggtype_\"{}\"'.format(config[\"aggtype\"]) for config in param]\n    return zip(index, res)\n\n# for fft_coefficient use param1\ndef fft_coefficient(x, param):\n\n    assert min([config[\"coeff\"] for config in param]) >= 0, \"Coefficients must be positive or zero.\"\n    assert set([config[\"attr\"] for config in param]) <= set([\"imag\", \"real\", \"abs\", \"angle\"]), \\\n        'Attribute must be \"real\", \"imag\", \"angle\" or \"abs\"'\n\n    fft = np.fft.rfft(x)\n\n    def complex_agg(x, agg):\n        if agg == \"real\":\n            return x.real\n        elif agg == \"imag\":\n            return x.imag\n        elif agg == \"abs\":\n            return np.abs(x)\n        elif agg == \"angle\":\n            return np.angle(x, deg=True)\n\n    res = [complex_agg(fft[config[\"coeff\"]], config[\"attr\"]) if config[\"coeff\"] < len(fft)\n           else np.NaN for config in param]\n    index = ['coeff_{}__attr_\"{}\"'.format(config[\"coeff\"], config[\"attr\"]) for config in param]\n    return zip(index, res)\n","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.8"}},"nbformat":4,"nbformat_minor":1}