{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":11000,"databundleVersionId":875412,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-26T11:14:54.521356Z","iopub.execute_input":"2024-11-26T11:14:54.521938Z","iopub.status.idle":"2024-11-26T11:14:54.525800Z","shell.execute_reply.started":"2024-11-26T11:14:54.521909Z","shell.execute_reply":"2024-11-26T11:14:54.524745Z"},"_kg_hide-output":true,"trusted":true,"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom scipy.signal import find_peaks\nfrom scipy.signal import spectrogram\nfrom scipy.stats import skew, kurtosis\nfrom sklearn.model_selection import train_test_split, KFold\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.preprocessing import StandardScaler\nfrom lightgbm import LGBMRegressor\nfrom tqdm import tqdm\nfrom statsmodels.tsa.stattools import adfuller\nfrom scipy.signal import spectrogram\nimport librosa\nimport librosa.display\nfrom statsmodels.graphics.tsaplots import plot_acf, plot_pacf","metadata":{"execution":{"iopub.status.busy":"2024-11-30T10:37:24.294937Z","iopub.execute_input":"2024-11-30T10:37:24.295224Z","iopub.status.idle":"2024-11-30T10:37:34.204811Z","shell.execute_reply.started":"2024-11-30T10:37:24.295192Z","shell.execute_reply":"2024-11-30T10:37:34.203989Z"},"trusted":true,"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Introduction\n\nThis work replicates methodology from [this notebook](https://www.kaggle.com/code/gpreda/lanl-earthquake-data-exploration-and-baseline) with minor modifications: this work does not normalise X and y before applying lightGBM. This work used fewer features (15) instead of the original 154, which may lead to a reduction in performance.","metadata":{}},{"cell_type":"code","source":"chunk_size = 10_000_000  # 10 mill to read at a time\ndata_chunks = []\n\nfor chunk in pd.read_csv(\n    \"/kaggle/input/LANL-Earthquake-Prediction/train.csv\",\n    chunksize=chunk_size,\n):\n    data_chunks.append(chunk)\n\ntrain_data = pd.concat(data_chunks, ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2024-11-30T10:37:34.206299Z","iopub.execute_input":"2024-11-30T10:37:34.206811Z","iopub.status.idle":"2024-11-30T10:41:48.961470Z","shell.execute_reply.started":"2024-11-30T10:37:34.206781Z","shell.execute_reply":"2024-11-30T10:41:48.960743Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Shape: \", train_data.shape)\ndf_60mil = train_data.head(60000000) # about 10% of total data\npd.set_option(\"display.precision\", 15)\n#df_60mil.to_csv('df_3mil.csv', index=False)\ntrain_data.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T10:41:48.962428Z","iopub.execute_input":"2024-11-30T10:41:48.962667Z","iopub.status.idle":"2024-11-30T10:41:48.991543Z","shell.execute_reply.started":"2024-11-30T10:41:48.962645Z","shell.execute_reply":"2024-11-30T10:41:48.990538Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Exploratory Data Analysis\n\nFirst we plot the first 10% of data (60 million entries), and the first 2000 entries. Then we visualise frequency density by conducting Fourier Analysis and Short-Time Fourier Transform. Then we plot ACF and PACF to determine the time series data is not AR or MA process since the value doesn't fall under the confidence band (using first 50000 entries.)","metadata":{}},{"cell_type":"code","source":"def plot_time_series(df):\n\n    fig, ax1 = plt.subplots(figsize=(10, 6))\n\n    ax1.plot(df.index, df['acoustic_data'], color='blue', label='Acoustic Data')\n    ax1.set_xlabel('Index')\n    ax1.set_ylabel('Acoustic Data', color='blue')\n    ax1.tick_params(axis='y', labelcolor='blue')\n    \n    ax2 = ax1.twinx()\n    ax2.plot(df.index, df['time_to_failure'], color='red', label='Time to Failure')\n    ax2.set_ylabel('Time to Failure (s)', color='red')\n    ax2.tick_params(axis='y', labelcolor='red')\n\n    plt.title(\"Acoustic Data and Time to Failure\")\n    ax1.legend(loc='upper left')\n    ax2.legend(loc='upper right')\n\n    plt.tight_layout()\n    plt.show()\n\nplot_time_series(df_60mil) #~10% of data, 60 million points\n","metadata":{"execution":{"iopub.status.busy":"2024-11-24T11:43:51.236390Z","iopub.execute_input":"2024-11-24T11:43:51.237461Z","iopub.status.idle":"2024-11-24T11:44:17.839221Z","shell.execute_reply.started":"2024-11-24T11:43:51.237421Z","shell.execute_reply":"2024-11-24T11:44:17.838349Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_time_series(df_60mil[:2000]) # first 2000 points","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T11:47:58.764466Z","iopub.execute_input":"2024-11-24T11:47:58.765427Z","iopub.status.idle":"2024-11-24T11:47:59.267545Z","shell.execute_reply.started":"2024-11-24T11:47:58.765387Z","shell.execute_reply":"2024-11-24T11:47:59.266605Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def visualize_fourier_transform(df, column_name='acoustic_data', sampling_rate=1):\n    \"\"\"\n    Calculate and visualize the Fourier Transform of a signal in a DataFrame.\n    \n    Parameters:\n        df (pd.DataFrame): DataFrame containing the signal data.\n        column_name (str): Name of the column to perform the Fourier Transform on.\n        sampling_rate (float): Sampling rate of the signal (in Hz). Default is 1.\n    \"\"\"\n    \n    signal = df[column_name].values\n    \n    # Perform Fourier Transform\n    fft_result = np.fft.fft(signal)\n    fft_magnitude = np.abs(fft_result)  # Magnitude of the FFT\n    fft_frequencies = np.fft.fftfreq(len(signal), d=1/sampling_rate)  # Frequencies\n\n    # Only take the positive frequencies for visualization\n    positive_freqs = fft_frequencies[:len(fft_frequencies)//2]\n    positive_magnitude = fft_magnitude[:len(fft_magnitude)//2]\n    \n    plt.figure(figsize=(12, 6))\n    plt.plot(positive_freqs, positive_magnitude, color='blue', label='FFT Magnitude')\n    plt.title('Fourier Transform of Acoustic Data')\n    plt.xlabel('Frequency (Hz)')\n    plt.ylabel('Magnitude')\n    plt.grid(True)\n    plt.legend()\n    plt.xlim(0, 0.2)\n    plt.tight_layout()\n    plt.show()\n    \n\nvisualize_fourier_transform(df_60mil[:50000], column_name='acoustic_data', sampling_rate=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T14:50:01.682360Z","iopub.execute_input":"2024-11-26T14:50:01.683440Z","iopub.status.idle":"2024-11-26T14:50:02.093608Z","shell.execute_reply.started":"2024-11-26T14:50:01.683389Z","shell.execute_reply":"2024-11-26T14:50:02.092620Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"chunk_size = 50000\nchunks = [df_60mil.iloc[i:i + chunk_size] for i in range(0, len(df_60mil), chunk_size)]\nprint(f\"Number of chunks: {len(chunks)}\")\nprint(f\"Shape of the first chunk: {chunks[0].shape}\")\n\nFRAME_SIZE = 1024 # params for STFT, window size\nHOP_SIZE=256 # shift size\n\n# Convert the Series to a NumPy array\namp_data = chunks[0]['acoustic_data'].to_numpy()\n# Ensure the data is of type float32\namp_data = amp_data.astype(np.float32)\n\nS_scale = librosa.stft(amp_data, n_fft = FRAME_SIZE, hop_length = HOP_SIZE)\n# type(S_scale[0][0]) -> numpy.complex64 we need to convert complex num to real num\nY_scale = np.abs(S_scale) ** 2\n# type(Y_scale[0][0]) -> numpy.float32, real num","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T14:40:17.483815Z","iopub.execute_input":"2024-11-26T14:40:17.484808Z","iopub.status.idle":"2024-11-26T14:40:26.522984Z","shell.execute_reply.started":"2024-11-26T14:40:17.484768Z","shell.execute_reply":"2024-11-26T14:40:26.522211Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_spectrogram(Y, sr, hop_length, y_axis = \"log\"): # sr - sampling rate: 909090909.09 Hz\n    plt.figure(figsize = (17,10))\n    librosa.display.specshow(Y,\n                             sr = sr,\n                             hop_length = hop_length,\n                             x_axis = \"time\",\n                             y_axis = y_axis)\n    plt.colorbar(format = \"%+2.f\")\n    plt.title(\"Spectrogram for first 60 million data\")\n    plt.xlabel(\"Time\")\n    plt.show()\nY_log_scale = librosa.power_to_db(Y_scale)\nsr =909090909.09 # sampling frequency, 1/sampling time, which is about 1.1*10^-9 from 'time_to_failure'\nplot_spectrogram(Y_log_scale, sr=sr, hop_length=HOP_SIZE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T14:42:16.362453Z","iopub.execute_input":"2024-11-26T14:42:16.363320Z","iopub.status.idle":"2024-11-26T14:42:16.948426Z","shell.execute_reply.started":"2024-11-26T14:42:16.363283Z","shell.execute_reply":"2024-11-26T14:42:16.947406Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"test whether amplitude (`acoustic_data`) is stationary. conclusion: yes there's strong evidence the time series is stationary.","metadata":{}},{"cell_type":"code","source":"result = adfuller(df_60mil['acoustic_data'][:50000])\nprint(f\"ADF Statistic: {result[0]}\")\nprint(f\"p-value: {result[1]}\")\nfor key, value in result[4].items():\n    print(f\"Critical Value ({key}): {value}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T14:31:26.262966Z","iopub.execute_input":"2024-11-26T14:31:26.263366Z","iopub.status.idle":"2024-11-26T14:31:37.800720Z","shell.execute_reply.started":"2024-11-26T14:31:26.263333Z","shell.execute_reply":"2024-11-26T14:31:37.798360Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"series = df_60mil['acoustic_data'][:50000]\n\n# Plot the ACF\nplt.figure(figsize=(12, 6))\nplt.subplot(2, 1, 1)  # First subplot\nplot_acf(series, lags=50, alpha=0.05, ax=plt.gca())  # ACF plot with 95% confidence interval\nplt.ylim(-0.2, 1.2)  # Adjust the Y-axis to make error bars visible\nplt.title('Autocorrelation Function (ACF)')\n\n# Plot the PACF\nplt.subplot(2, 1, 2)  # Second subplot\nplot_pacf(series, lags=50, alpha=0.05, ax=plt.gca())  # PACF plot with 95% confidence interval\nplt.ylim(-0.2, 1.2)\nplt.title('Partial Autocorrelation Function (PACF)')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T15:26:40.014946Z","iopub.execute_input":"2024-11-26T15:26:40.015364Z","iopub.status.idle":"2024-11-26T15:26:41.399815Z","shell.execute_reply.started":"2024-11-26T15:26:40.015328Z","shell.execute_reply":"2024-11-26T15:26:41.398820Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Engineering & Model training\n\nExtract 15 statistical metrics from amplitude data `acoustic_data`, per 150,000 data points, this is our feature matrix X. Use the last `time_to_failure` as y. Then apply lightGBM to train the processed data.","metadata":{}},{"cell_type":"markdown","source":"Below, we divide the data into segments, each containing 150,000 data points, and calculate statistical metrics for each segment of amplitude data.","metadata":{}},{"cell_type":"code","source":"rows = 150000\nsegments = int(np.floor(train_data.shape[0] / rows))\nprint(\"Number of segments: \", segments)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T12:12:57.120690Z","iopub.execute_input":"2024-11-26T12:12:57.120960Z","iopub.status.idle":"2024-11-26T12:12:57.130687Z","shell.execute_reply.started":"2024-11-26T12:12:57.120913Z","shell.execute_reply":"2024-11-26T12:12:57.129495Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"below are some functions created to be implemented for feature creation.","metadata":{}},{"cell_type":"code","source":"def higuchi_fd(time_series, k_max=10):\n    \"\"\"\n    Calculate Higuchi's Fractal Dimension for a 1D time series.\n    :param time_series: 1D array of time series data\n    :param k_max: Maximum segment size (scale)\n    :return: Fractal Dimension (float)\n    \"\"\"\n    n = len(time_series)\n    L = []\n\n    for k in range(1, k_max + 1):\n        Lk = []\n        for m in range(k):\n            # Calculate the \"length\" of the curve\n            idx = np.arange(m, n, k)  # Indices at interval k\n            Lm = np.sum(np.abs(np.diff(time_series[idx]))) * (n - 1) / (len(idx) * k)\n            Lk.append(Lm)\n        L.append(np.mean(Lk))\n    \n    # Fit a line to log-log scale to find slope (fractal dimension)\n    log_k = np.log(range(1, k_max + 1))\n    log_L = np.log(L)\n    return -np.polyfit(log_k, log_L, 1)[0]\n    \ndef hurst_exponent(ts, max_lag=20):\n    \"\"\"\n    Calculate the Hurst Exponent of a time series using R/S analysis.\n    :param ts: 1D numpy array of the time series.\n    :param max_lag: Maximum lag for rescaled range calculation.\n    :return: Estimated Hurst Exponent (float).\n    \"\"\"\n    lags = range(2, max_lag)\n    rs_values = []\n    for lag in lags:\n        lagged_diffs = np.diff(ts[:-(lag - 1)])\n        r = np.max(np.cumsum(lagged_diffs)) - np.min(np.cumsum(lagged_diffs))  # Range\n        s = np.std(lagged_diffs)  # Standard deviation\n        rs_values.append(r / s)\n\n    # Fit a line to log-log plot\n    log_lags = np.log(lags)\n    log_rs = np.log(rs_values)\n    hurst, _ = np.polyfit(log_lags, log_rs, 1)  # Slope is the Hurst Exponent\n    return hurst\n    \ndef calculate_rise_time(xc):\n    \"\"\"\n    Calculate the Rise Time of a signal.\n    :param xc: 1D numpy array of the signal.\n    :param sampling_rate: The sampling rate of the signal (default is 1 if time steps are uniform).\n    :return: Rise Time (float).\n    \"\"\"\n    peak_index = np.argmax(xc)  # Index of the peak amplitude\n    #start_index = 0  # Assuming we start measuring rise time from the first value\n    #rise_time = (peak_index - start_index)\n    return peak_index","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T12:12:57.131861Z","iopub.execute_input":"2024-11-26T12:12:57.132135Z","iopub.status.idle":"2024-11-26T12:12:57.143809Z","shell.execute_reply.started":"2024-11-26T12:12:57.132106Z","shell.execute_reply":"2024-11-26T12:12:57.142887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_features(seg_id, seg):\n    X = pd.DataFrame()\n    xc = np.array(seg['acoustic_data'].values)  # convert to NumPy array for runtime\n    X['mean'] = [xc.mean()]\n    X['std'] = [xc.std()]\n    X['min'] = [xc.min()]\n    X['max'] = [xc.max()]\n    X['skewness'] = skew(xc)\n    X['kurtosis'] = kurtosis(xc)\n    X['range'] = X['max']-X['min']\n    X['mad'] = [np.median(np.abs(xc - np.median(xc)))]  # Median Absolute Deviation (MAD)\n    X['rms'] = [np.sqrt(np.mean(xc**2))] # root mean sqaure (RMS)\n    X['energy'] = [np.sum(xc**2)] # energy of the signal\n    X['zcr'] = np.sum(xc[:-1] * xc[1:] < 0) # Zero-Crossing Rate (ZCR), the number of times the signal crosses the zero line\n    X['fractal_dimension'] = [higuchi_fd(xc)] # calculate the \"statistical self-similarity\" of the time series, min =1, max = 2\n    X['hurst_exponent'] = [hurst_exponent(xc)]  # time series behaviour analysis, \"persistency\"\n    peak_amplitude = np.max(np.abs(xc))\n    X['crest_factor'] = [peak_amplitude / X['rms'].iloc[0]]  # Crest Factor, largest amp / avg. amp\n    X['rise_time'] = [calculate_rise_time(xc)]  # time taken to go from start value to peak amp value\n\n    return X\n#create_features(1, df_60mil[:2000]) #test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T12:14:02.728971Z","iopub.execute_input":"2024-11-26T12:14:02.729674Z","iopub.status.idle":"2024-11-26T12:14:02.736528Z","shell.execute_reply.started":"2024-11-26T12:14:02.729638Z","shell.execute_reply":"2024-11-26T12:14:02.735729Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_X = pd.DataFrame()\ntrain_y = pd.DataFrame(columns=['time_to_failure'])\n\n# Iterate over all segments with a progress bar\nfor i in tqdm(range(segments), desc=\"Processing Segments\"):\n    seg = train_data.iloc[i * rows:(i + 1) * rows]\n    features = create_features(i, seg)  # Generate features\n    train_X = pd.concat([train_X, features], ignore_index=True)  # Append features\n    train_y.loc[i, 'time_to_failure'] = seg['time_to_failure'].values[-1]  # Assign target\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T12:14:08.968176Z","iopub.execute_input":"2024-11-26T12:14:08.968867Z","iopub.status.idle":"2024-11-26T12:17:42.271419Z","shell.execute_reply.started":"2024-11-26T12:14:08.968835Z","shell.execute_reply":"2024-11-26T12:17:42.270299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_X.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T12:26:01.367131Z","iopub.execute_input":"2024-11-26T12:26:01.367547Z","iopub.status.idle":"2024-11-26T12:26:01.383067Z","shell.execute_reply.started":"2024-11-26T12:26:01.367512Z","shell.execute_reply":"2024-11-26T12:26:01.382198Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"train lightGBM with 5-fold CV, based on our extracted X and y (named `scaled_train_X` and `train_y`) from original time series data.","metadata":{}},{"cell_type":"code","source":"# Convert time_to_failure to numeric\ntrain_y = train_y.astype(float)\n# Split data into training and validation sets, 5-fold CV\nX_train, X_val, y_train, y_val = train_test_split(train_X, train_y, test_size=0.2, random_state=42)\n\n# Define LightGBM parameters\nparams = {\n    'objective': 'regression',\n    'metric': 'mae',\n    'learning_rate': 0.01,\n    'num_leaves': 31,\n    'max_depth': -1,\n    'feature_fraction': 0.9,\n    'bagging_fraction': 0.8,\n    'bagging_freq': 5,\n    'verbose': -1  # Suppress output\n}\n\nmodel = LGBMRegressor(**params, n_estimators=20000, n_jobs=-1)\nmodel.fit(\n    X_train, y_train,\n    eval_set=[(X_train, y_train), (X_val, y_val)],\n    eval_metric='mae'\n)\n\ny_pred = model.predict(X_val)\n\n# Evaluate the model\nmae = mean_absolute_error(y_val, y_pred)\nprint(f\"Validation MAE: {mae:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T12:54:23.528252Z","iopub.execute_input":"2024-11-26T12:54:23.528690Z","iopub.status.idle":"2024-11-26T12:54:46.492074Z","shell.execute_reply.started":"2024-11-26T12:54:23.528655Z","shell.execute_reply":"2024-11-26T12:54:46.490379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_importance = pd.DataFrame({\n    'Feature': X_train.columns,  # Feature names\n    'Importance': model.feature_importances_  # Feature importance scores\n})\n\n# Sort features by importance\nfeature_importance = feature_importance.sort_values(by='Importance', ascending=False)\n\n# Plot the feature importance\nplt.figure(figsize=(10, 8))\nplt.barh(feature_importance['Feature'], feature_importance['Importance'])\nplt.xlabel('Importance Score')\nplt.ylabel('Features')\nplt.title('Feature Importance')\nplt.gca().invert_yaxis()  # Reverse the order for better readability\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T13:03:02.345278Z","iopub.execute_input":"2024-11-26T13:03:02.345714Z","iopub.status.idle":"2024-11-26T13:03:02.603180Z","shell.execute_reply.started":"2024-11-26T13:03:02.345679Z","shell.execute_reply":"2024-11-26T13:03:02.602218Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Testing model performance\n\nA lot of space for imporvement: model predicted 6.99, actual 1.47 (unit seconds.)","metadata":{}},{"cell_type":"markdown","source":"now test our model on real time series data `acoustic_data` and see how well it perform to predict target variable `time_to_failure`.","metadata":{}},{"cell_type":"code","source":"# Extract the first 10000 rows of the time series data and wrap it in a DataFrame\nsegment2000 = df_60mil[['acoustic_data']].iloc[:10000]  # Ensure it's a DataFrame\nnew_features = create_features(0, segment2000)\nnew_features = new_features[X_train.columns]\ny_pred = model.predict(new_features)\nprint(f\"Predicted target value (y): {y_pred[0]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T13:55:06.673245Z","iopub.execute_input":"2024-11-26T13:55:06.673600Z","iopub.status.idle":"2024-11-26T13:55:06.956561Z","shell.execute_reply.started":"2024-11-26T13:55:06.673569Z","shell.execute_reply":"2024-11-26T13:55:06.954495Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Actual value (y): {df_60mil['time_to_failure'][9999]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T13:55:23.362321Z","iopub.execute_input":"2024-11-26T13:55:23.362790Z","iopub.status.idle":"2024-11-26T13:55:23.368335Z","shell.execute_reply.started":"2024-11-26T13:55:23.362736Z","shell.execute_reply":"2024-11-26T13:55:23.367210Z"}},"outputs":[],"execution_count":null}]}