{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":11000,"databundleVersionId":875412,"sourceType":"competition"},{"sourceId":7700987,"sourceType":"datasetVersion","datasetId":4495313},{"sourceId":7754938,"sourceType":"datasetVersion","datasetId":4534508},{"sourceId":7831146,"sourceType":"datasetVersion","datasetId":4546322}],"dockerImageVersionId":30646,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport pylab as plt\nimport scipy as sp\nfrom tqdm import tqdm\nfrom scipy import fftpack\nimport pandas as pd\nimport seaborn as sns\n%matplotlib inline\n\nimport sklearn as sk\nfrom sklearn import tree\nfrom sklearn.metrics import accuracy_score\nfrom sklearn import datasets, model_selection\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler\nfrom sklearn import preprocessing\nfrom sklearn.metrics import mean_squared_error\nfrom scipy.signal import find_peaks\n\nfrom tqdm import tqdm\nimport warnings","metadata":{"execution":{"iopub.status.busy":"2024-03-14T03:31:08.168153Z","iopub.execute_input":"2024-03-14T03:31:08.169156Z","iopub.status.idle":"2024-03-14T03:31:09.495610Z","shell.execute_reply.started":"2024-03-14T03:31:08.169111Z","shell.execute_reply":"2024-03-14T03:31:09.494485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.DataFrame()\ndata_list = []  # List to store intermediate DataFrames\n\nfor i in tqdm(range(0,9)):\n    \n    # Read\n    if i < 5:\n        file_name = f'/kaggle/input/lanl-earthquake/train_{i}.csv'\n    else:\n        file_name = f'/kaggle/input/train-data/train_{i}.csv'\n    train = pd.read_csv(file_name)\n    train.columns = ['acoustic_data', 'time_to_failure']\n    data_i = train[::1000]\n    \n    data_list.append(data_i) \n    del train  ","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:46:27.495031Z","iopub.execute_input":"2024-03-14T01:46:27.495525Z","iopub.status.idle":"2024-03-14T01:50:42.205221Z","shell.execute_reply.started":"2024-03-14T01:46:27.495488Z","shell.execute_reply":"2024-03-14T01:50:42.202798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.concat(data_list, ignore_index=True)\nplt.figure(figsize=(10, 6))\nplt.plot(data['acoustic_data']/50,label = 'acoustic signal (AE) ')\nplt.plot(data['time_to_failure'], label = 'time to failure (TTF)')\nplt.ylabel('TTF(s) and AE/50')\nplt.xlabel('data points')\nplt.title('LANL Earthquake Data')\nplt.xlim(0,110000)\nplt.ylim(-25,25)\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T02:06:39.130015Z","iopub.execute_input":"2024-03-14T02:06:39.130666Z","iopub.status.idle":"2024-03-14T02:06:41.731034Z","shell.execute_reply.started":"2024-03-14T02:06:39.130617Z","shell.execute_reply":"2024-03-14T02:06:41.729754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calculateFeatures(data):\n    features = {}\n    \n    # Percentile    \n    for n in [75,90,91,92,93,94,95,96,97,98,99,99.9]:\n        features[f'q_{n}'] = np.percentile(data, n) - np.percentile(data, 100-n)\n    features['max'] = data.max() - data.min() \n    \n    # Statistical Moments\n    features['mean'] = data.mean()\n    features['std'] = data.std()\n    features['m2'] = sp.stats.moment(data, moment = 2)\n    features['m3'] = sp.stats.moment(data, moment = 3)\n    features['m4'] = sp.stats.moment(data, moment = 4)\n    features['m5'] = sp.stats.moment(data, moment = 5)\n    features['m6'] = sp.stats.moment(data, moment = 6)\n    features['m7'] = sp.stats.moment(data, moment = 7)\n    features['var'] = data.std()**2\n    \n    # Roll\n    for roll_length in [10,100,1000,10000]:\n        roll_data = data.rolling(roll_length).std().dropna()\n        roll_data = roll_data.values\n        features[f'roll_{roll_length}_min'] = roll_data.min()\n        features[f'roll_{roll_length}_max'] = roll_data.max()\n        features[f'roll_{roll_length}_mean'] = roll_data.mean()\n        features[f'roll_{roll_length}_std'] = roll_data.std()\n        features[f'roll_{roll_length}_m2'] = sp.stats.moment(roll_data, moment = 2)\n        for n in [75,90,91,92,93,94,95,96,97,98,99,99.9]:\n            features[f'roll_{roll_length}_q{n}'] = np.percentile(roll_data, n)\n            features[f'roll_{roll_length}_q{100-n}'] = np.percentile(roll_data, 100-n)\n    \n    #FFT\n    data_centered = data - data.mean()\n    data_centered = data_centered.values\n    data_fft = fftpack.fft(data_centered)\n    frequency = fftpack.fftfreq(len(data), d=1/4000000)\n    density = np.abs(data_fft)**2\n    frequency_pos = frequency[1:len(data)//2]\n    density_pos = density[1:len(data)//2]\n    for i in range(30):\n        features[f'power_{i}'] = density_pos[2500*i:2500*(i+1)].mean()\n    features['fft_peak_freq'] = frequency_pos[np.argmax(density_pos)]\n    features['fft_max'] = density_pos.max()\n    \n    #Threshold\n    for n in [10,20,30,40,50,60,70,80,90,100,500,1000]:\n        features[f'threshold_{n}'] = sum(data >= n)/150000\n    \n    #peaks\n    for n in [50,75,100,125,150,175,200]:\n        height_threshold = n\n        peaks, _ = find_peaks(data, height=height_threshold)\n        features[f'peak>{n}'] = len(peaks)\n    \n    peaks, _ = find_peaks(data, height=200)\n    peak_mask = np.full(data.shape, False)\n    peak_mask[peaks] = True\n    no_peak_mask = ~peak_mask\n    no_peak_data = data[no_peak_mask]\n    features['std_nopeak'] = np.std(no_peak_data)\n    \n    return features","metadata":{"execution":{"iopub.status.busy":"2024-03-13T04:49:09.223096Z","iopub.execute_input":"2024-03-13T04:49:09.223625Z","iopub.status.idle":"2024-03-13T04:49:09.251671Z","shell.execute_reply.started":"2024-03-13T04:49:09.223584Z","shell.execute_reply":"2024-03-13T04:49:09.249615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Define window_width and window_total here (if not already defined)\n\nwindow_width = 150000\ndata = pd.DataFrame()\ndata_list = []  # List to store intermediate DataFrames\n\nfor i in range(0,9):\n    \n    # Read\n    if i < 5:\n        file_name = f'/kaggle/input/lanl-earthquake/train_{i}.csv'\n    else:\n        file_name = f'/kaggle/input/train-data/train_{i}.csv'\n    train = pd.read_csv(file_name)\n    train.columns = ['acoustic_data', 'time_to_failure']\n    window_total = int(train.shape[0]/window_width)-1\n    \n    # Generate feature set\n    data_X = pd.DataFrame(index=range(window_total), dtype=np.float16)\n    data_Y = pd.DataFrame(index=range(window_total), dtype=np.float16)\n    for j in tqdm(range(window_total)):  \n        window_j = train.iloc[j*window_width:(j+1)*window_width]\n        data_Y.loc[j, 'TTF'] = window_j['time_to_failure'].mean()\n        features = calculateFeatures(window_j['acoustic_data'])\n        for feature_name, feature_value in features.items():\n            data_X.loc[j, feature_name] = feature_value\n    \n    data_i = pd.concat([data_X, data_Y], axis=1)\n    data_list.append(data_i) \n    del train  ","metadata":{"execution":{"iopub.status.busy":"2024-03-13T05:29:11.808654Z","iopub.execute_input":"2024-03-13T05:29:11.810088Z","iopub.status.idle":"2024-03-13T06:11:11.972528Z","shell.execute_reply.started":"2024-03-13T05:29:11.810032Z","shell.execute_reply":"2024-03-13T06:11:11.970928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.concat(data_list, ignore_index=True)\ndata.to_csv('my_dataframe.csv', index=False)\n\ndf = data\ncolumns_with_na = df.columns[df.isna().any()]\ndf_cleaned = df.drop(columns=columns_with_na)\nprint(\"Dropped columns:\", columns_with_na.tolist())","metadata":{"execution":{"iopub.status.busy":"2024-03-13T06:15:11.794355Z","iopub.execute_input":"2024-03-13T06:15:11.794852Z","iopub.status.idle":"2024-03-13T06:15:13.651587Z","shell.execute_reply.started":"2024-03-13T06:15:11.794818Z","shell.execute_reply":"2024-03-13T06:15:13.650427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_cleaned = df_cleaned.drop(columns=['std','var'])\ndf_cleaned.to_csv('/kaggle/working/my_output2.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-03-13T06:15:32.275563Z","iopub.execute_input":"2024-03-13T06:15:32.275989Z","iopub.status.idle":"2024-03-13T06:15:34.078165Z","shell.execute_reply.started":"2024-03-13T06:15:32.275958Z","shell.execute_reply":"2024-03-13T06:15:34.077164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_cleaned = pd.read_csv('/kaggle/input/feature-set/my_output2 (1).csv')\n\nnum_columns = df_cleaned.to_numpy().shape[1] - 1\nprint(num_columns)\ndf_cleaned.head()\n\nX = df_cleaned.drop('TTF', axis=1)\nY = df_cleaned['TTF']","metadata":{"execution":{"iopub.status.busy":"2024-03-14T03:31:22.143554Z","iopub.execute_input":"2024-03-14T03:31:22.144592Z","iopub.status.idle":"2024-03-14T03:31:22.703681Z","shell.execute_reply.started":"2024-03-14T03:31:22.144532Z","shell.execute_reply":"2024-03-14T03:31:22.702067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#X = X[['roll_100_q10', 'roll_100_q25', 'roll_100_q8', 'roll_100_q9','roll_100_q6','roll_100_q75','roll_10_q3','roll_100_q7','roll_1000_q10','roll_1000_q25','mean','roll_1000_q5','power_0','power_1','power_6','power_23','roll_1000_q9','roll_100_q4','power_24','power_22']]\nx_train, x_test, y_train, y_test = model_selection.train_test_split(X, Y, test_size=0.2)\nx_train = x_train.sort_index()\nx_test = x_test.sort_index()\ny_train = y_train.sort_index()\ny_test = y_test.sort_index()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T03:31:25.079799Z","iopub.execute_input":"2024-03-14T03:31:25.080517Z","iopub.status.idle":"2024-03-14T03:31:25.131991Z","shell.execute_reply.started":"2024-03-14T03:31:25.080461Z","shell.execute_reply":"2024-03-14T03:31:25.130794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"abs_error_list = []\nmin_split_list = [2,3,5,7,9]  # Example values for max_samples as a fraction of the size of the dataset\nmin_leaf_list = [1,2,4,6,8]  # Example values for max_features\ndepth_range = range(6, 16)\n\n# Loop over all combinations of max_samples, max_features, and max_depth\nfor min_split in min_split_list:\n    for min_leaf in min_leaf_list:\n        for depth in tqdm(depth_range):\n            clf = RandomForestRegressor(\n                max_depth=depth,\n                criterion='squared_error',\n                min_samples_split=min_split,\n                min_samples_leaf=min_leaf,\n                n_jobs = -1,\n                random_state=42\n            )\n            clf.fit(x_train, y_train)\n            pred_test = clf.predict(x_test)\n            pred_train = clf.predict(x_train)\n            abs_error_test = mean_absolute_error(y_test, pred_test)\n            abs_error_train = mean_absolute_error(y_train, pred_train)\n            \n            # Append the results including the hyperparameters used\n            abs_error_list.append({\n                'max_depth': depth,\n                'min_split': min_split,\n                'min_leaf': min_leaf,\n                'mae_test': abs_error_test,\n                'mae_train': abs_error_train                \n            })","metadata":{"execution":{"iopub.status.busy":"2024-03-14T04:46:40.107466Z","iopub.execute_input":"2024-03-14T04:46:40.107944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results_df = pd.DataFrame(abs_error_list)\nbest_hyperparameters = results_df.loc[results_df['mae_test'].idxmin()]\nbest_hyperparameters","metadata":{"execution":{"iopub.status.busy":"2024-03-14T04:45:15.333484Z","iopub.execute_input":"2024-03-14T04:45:15.337628Z","iopub.status.idle":"2024-03-14T04:45:15.354280Z","shell.execute_reply.started":"2024-03-14T04:45:15.337558Z","shell.execute_reply":"2024-03-14T04:45:15.352951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best = results_df[results_df[max_feature] = ","metadata":{"execution":{"iopub.status.busy":"2024-03-14T04:14:19.577768Z","iopub.execute_input":"2024-03-14T04:14:19.578340Z","iopub.status.idle":"2024-03-14T04:14:19.598725Z","shell.execute_reply.started":"2024-03-14T04:14:19.578297Z","shell.execute_reply":"2024-03-14T04:14:19.597519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.scatter(depth,abs_error_list)\n#plt.scatter(depth,sqr_error_list) ","metadata":{"execution":{"iopub.status.busy":"2024-03-13T08:41:46.109218Z","iopub.execute_input":"2024-03-13T08:41:46.109688Z","iopub.status.idle":"2024-03-13T08:41:46.467089Z","shell.execute_reply.started":"2024-03-13T08:41:46.109643Z","shell.execute_reply":"2024-03-13T08:41:46.465764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import r2_score\n\nclf = sk.ensemble.RandomForestRegressor(criterion='squared_error',n_estimators = 200, max_depth=10)\nclf.fit(x_train,y_train)\npred=clf.predict(x_test)\n\nplt.figure(figsize=(30, 6))\nplt.plot(x_test.index,y_test)\nplt.plot(x_test.index,pred)\nplt.show()\n\nprint(f' mae: {mean_absolute_error(y_test,pred)}')\nprint(f' mse: {mean_squared_error(y_test,pred)}')\nprint(f'R²: {r2_score(y_test, pred)}')","metadata":{"execution":{"iopub.status.busy":"2024-03-14T03:33:16.754251Z","iopub.execute_input":"2024-03-14T03:33:16.754880Z","iopub.status.idle":"2024-03-14T03:34:23.977138Z","shell.execute_reply.started":"2024-03-14T03:33:16.754828Z","shell.execute_reply":"2024-03-14T03:34:23.975698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred=clf.predict(x_train)\nprint(mean_absolute_error(y_train,pred))\nplt.figure(figsize=(30, 6))\nplt.plot(x_train.index,y_train)\nplt.plot(x_train.index,pred)\nplt.show()\n\nprint(f' mae: {mean_absolute_error(y_train,pred)}')\nprint(f' mse: {mean_squared_error(y_train,pred)}')\nprint(f'R²: {r2_score(y_train, pred)}')","metadata":{"execution":{"iopub.status.busy":"2024-03-14T03:40:20.722182Z","iopub.execute_input":"2024-03-14T03:40:20.722665Z","iopub.status.idle":"2024-03-14T03:40:21.163269Z","shell.execute_reply.started":"2024-03-14T03:40:20.722630Z","shell.execute_reply":"2024-03-14T03:40:21.162104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport seaborn as sns\n\n# Get feature importances and sort them\nfeature_names = X.columns\ncolors = sns.color_palette(\"husl\", len(indices))\n\nimportances = clf.feature_importances_\nindices = np.argsort(importances)[::-1]\ntop_n = 20\nsorted_top_indices = indices[:top_n]\n\nplt.figure(figsize=(15, 10))\nplt.title('Feature Importances')\nplt.barh(range(top_n), importances[sorted_top_indices], align='center', color = colors)\n\nplt.yticks(range(top_n), [feature_names[i] for i in sorted_top_indices], fontsize=8, rotation=0)\nplt.ylim(top_n, -1)\n\nplt.xticks([0.01,0.05,0.1])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-13T08:45:26.416042Z","iopub.execute_input":"2024-03-13T08:45:26.416620Z","iopub.status.idle":"2024-03-13T08:45:26.851408Z","shell.execute_reply.started":"2024-03-13T08:45:26.416563Z","shell.execute_reply":"2024-03-13T08:45:26.849752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nimport warnings\nwarnings.filterwarnings('ignore')\n\nwindow_width = 150000\ndata = pd.DataFrame()\ndata_list = []  # List to store intermediate DataFrames\n\n# Assuming 'test_path' is the directory containing your CSV files\ntest_path = '/kaggle/input/LANL-Earthquake-Prediction/test/*.csv'\nfile_paths = glob.glob(test_path)\n\n# Loop over the file paths\nfor file_path in tqdm(file_paths):\n    \n    train = pd.read_csv(file_path)\n    train.columns = ['acoustic_data']\n    window_total = 1\n\n    data_X = pd.DataFrame(index=range(window_total), dtype=np.float16)\n\n    for j in range(window_total):   \n        window_j = train.iloc[j*window_width:(j+1)*window_width]\n        features = calculateFeatures(window_j['acoustic_data'])\n        for feature_name, feature_value in features.items():\n            data_X.loc[j, feature_name] = feature_value\n            \n    data_list.append(data_X) \n    del train  ","metadata":{"execution":{"iopub.status.busy":"2024-03-13T06:29:06.424785Z","iopub.execute_input":"2024-03-13T06:29:06.425367Z","iopub.status.idle":"2024-03-13T06:57:48.171270Z","shell.execute_reply.started":"2024-03-13T06:29:06.425330Z","shell.execute_reply":"2024-03-13T06:57:48.169606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.concat(data_list, ignore_index=True)\ndata.to_csv('valid.csv', index=False)\n\ndf = data\ndf_cleaned = df\n\ndf_cleaned = df_cleaned.drop(columns=['std','var'])\n#df_cleaned = pd.read_csv('/kaggle/input/feature-set/valid.csv')\ndf_cleaned.to_csv('/kaggle/working/valid.csv', index=False)\n\nX = df_cleaned","metadata":{"execution":{"iopub.status.busy":"2024-03-13T06:59:12.645822Z","iopub.execute_input":"2024-03-13T06:59:12.647914Z","iopub.status.idle":"2024-03-13T06:59:23.809573Z","shell.execute_reply.started":"2024-03-13T06:59:12.647838Z","shell.execute_reply":"2024-03-13T06:59:23.808146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_cleaned = pd.read_csv('/kaggle/input/feature-set/valid (2).csv')\n#df_cleaned.to_csv('/kaggle/working/valid.csv', index=False)\nX = df_cleaned\nX = X[['roll_100_q10', 'roll_100_q25', 'roll_100_q8', 'roll_100_q9','roll_100_q6','roll_100_q75','roll_10_q3','roll_100_q7','roll_1000_q10','roll_1000_q25','mean','roll_1000_q5','power_0','power_1','power_6','power_23','roll_1000_q9','roll_100_q4','power_24','power_22']]","metadata":{"execution":{"iopub.status.busy":"2024-03-13T08:46:24.905447Z","iopub.execute_input":"2024-03-13T08:46:24.905974Z","iopub.status.idle":"2024-03-13T08:46:25.081181Z","shell.execute_reply.started":"2024-03-13T08:46:24.905940Z","shell.execute_reply":"2024-03-13T08:46:25.079786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(X))","metadata":{"execution":{"iopub.status.busy":"2024-03-13T08:35:02.201128Z","iopub.execute_input":"2024-03-13T08:35:02.201604Z","iopub.status.idle":"2024-03-13T08:35:02.209260Z","shell.execute_reply.started":"2024-03-13T08:35:02.201571Z","shell.execute_reply":"2024-03-13T08:35:02.207737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred=clf.predict(X)","metadata":{"execution":{"iopub.status.busy":"2024-03-13T08:46:27.057170Z","iopub.execute_input":"2024-03-13T08:46:27.057614Z","iopub.status.idle":"2024-03-13T08:46:27.114715Z","shell.execute_reply.started":"2024-03-13T08:46:27.057582Z","shell.execute_reply":"2024-03-13T08:46:27.113307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(30, 5))\nplt.plot(pred)\nplt.xlim([0,600])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-13T08:46:28.849062Z","iopub.execute_input":"2024-03-13T08:46:28.849516Z","iopub.status.idle":"2024-03-13T08:46:29.257909Z","shell.execute_reply.started":"2024-03-13T08:46:28.849462Z","shell.execute_reply":"2024-03-13T08:46:29.256592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport os\n\ntest_path = '/kaggle/input/LANL-Earthquake-Prediction/test'\nsegment_ids = [fname.replace('.csv', '') for fname in os.listdir(test_path) if fname.endswith('.csv')]\nsubmission = pd.DataFrame({'seg_id': segment_ids, 'time_to_failure': pred})\nsubmission.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-03-13T08:46:30.638156Z","iopub.execute_input":"2024-03-13T08:46:30.638675Z","iopub.status.idle":"2024-03-13T08:46:30.663009Z","shell.execute_reply.started":"2024-03-13T08:46:30.638636Z","shell.execute_reply":"2024-03-13T08:46:30.661825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}