{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Step 1 - Installing dependencies\n\nStep 2 - Importing dataset\n\nStep 3 - Exploratory data analysis\n\nStep 4 - Feature engineering (statistical features added)\n\nStep 5 - Implement \"Catboost\" model\n\nStep 6 - Implement Support Vector Machine + Radial Basis Function model","metadata":{}},{"cell_type":"markdown","source":"Step 1: installing Dependencies","metadata":{}},{"cell_type":"code","source":"#data preprocessing\nimport pandas as pd\n#math operations\nimport numpy as np\n#machine learning\nfrom catboost import CatBoostRegressor, Pool\n#data scaling\nfrom sklearn.preprocessing import StandardScaler\n#hyperparameter optimization\nfrom sklearn.model_selection import GridSearchCV\n#support vector machine model\nfrom sklearn.svm import NuSVR, SVR\n#kernel ridge model\nfrom sklearn.kernel_ridge import KernelRidge\n#data visualization\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-04-11T19:43:00.117398Z","iopub.execute_input":"2023-04-11T19:43:00.118204Z","iopub.status.idle":"2023-04-11T19:43:01.590167Z","shell.execute_reply.started":"2023-04-11T19:43:00.118165Z","shell.execute_reply":"2023-04-11T19:43:01.589028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Step 2: importing Dataset","metadata":{}},{"cell_type":"code","source":"#Extract training data into a dataframe for further manipulation\ntrain = pd.read_csv('/kaggle/input/LANL-Earthquake-Prediction/train.csv', nrows=6000000, dtype={'acoustic_data': np.int16, 'time_to_failure': np.float64})","metadata":{"execution":{"iopub.status.busy":"2023-04-11T19:43:29.689461Z","iopub.execute_input":"2023-04-11T19:43:29.690438Z","iopub.status.idle":"2023-04-11T19:43:31.542516Z","shell.execute_reply.started":"2023-04-11T19:43:29.690399Z","shell.execute_reply":"2023-04-11T19:43:31.541499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print first 10 entries\ntrain.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-04-11T19:44:24.993606Z","iopub.execute_input":"2023-04-11T19:44:24.994183Z","iopub.status.idle":"2023-04-11T19:44:25.014828Z","shell.execute_reply.started":"2023-04-11T19:44:24.994145Z","shell.execute_reply":"2023-04-11T19:44:25.013678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Step 3: Exploratory Data Analysis","metadata":{}},{"cell_type":"code","source":"#visualize 1% of samples data, first 100 datapoints\ntrain_ad_sample_df = train['acoustic_data'].values[::100]\ntrain_ttf_sample_df = train['time_to_failure'].values[::100]\n\n#function for plotting based on both features\ndef plot_acc_ttf_data(train_ad_sample_df, train_ttf_sample_df, title=\"Acoustic data and time to failure: 1% sampled data\"):\n    fig, ax1 = plt.subplots(figsize=(12, 8))\n    plt.title(title)\n    plt.plot(train_ad_sample_df, color='r')\n    ax1.set_ylabel('acoustic data', color='r')\n    plt.legend(['acoustic data'], loc=(0.01, 0.95))\n    ax2 = ax1.twinx()\n    plt.plot(train_ttf_sample_df, color='b')\n    ax2.set_ylabel('time to failure', color='b')\n    plt.legend(['time to failure'], loc=(0.01, 0.9))\n    plt.grid(True)\n\nplot_acc_ttf_data(train_ad_sample_df, train_ttf_sample_df)\ndel train_ad_sample_df\ndel train_ttf_sample_df","metadata":{"execution":{"iopub.status.busy":"2023-04-11T19:45:29.584456Z","iopub.execute_input":"2023-04-11T19:45:29.584827Z","iopub.status.idle":"2023-04-11T19:45:30.059716Z","shell.execute_reply.started":"2023-04-11T19:45:29.584795Z","shell.execute_reply":"2023-04-11T19:45:30.058750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Step 4: Feature Engineering","metadata":{}},{"cell_type":"code","source":"# Step 4 - Feature Engineering and signifiance of these statistical features\n\n#lets create a function to generate some statistical features based on the training data\ndef gen_features(X):\n    strain = []\n    strain.append(X.mean())\n    strain.append(X.std())\n    strain.append(X.min())\n    strain.append(X.max())\n    strain.append(X.kurtosis())\n    strain.append(X.skew())\n    strain.append(np.quantile(X,0.01))\n    strain.append(np.quantile(X,0.05))\n    strain.append(np.quantile(X,0.95))\n    strain.append(np.quantile(X,0.99))\n    strain.append(np.abs(X).max())\n    strain.append(np.abs(X).mean())\n    strain.append(np.abs(X).std())\n    return pd.Series(strain)","metadata":{"execution":{"iopub.status.busy":"2023-04-11T19:47:42.634344Z","iopub.execute_input":"2023-04-11T19:47:42.635175Z","iopub.status.idle":"2023-04-11T19:47:42.642894Z","shell.execute_reply.started":"2023-04-11T19:47:42.635134Z","shell.execute_reply":"2023-04-11T19:47:42.641817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/LANL-Earthquake-Prediction/train.csv', iterator=True, chunksize=150_000, dtype={'acoustic_data': np.int16, 'time_to_failure': np.float64})\n\nX_train = pd.DataFrame()\ny_train = pd.Series()\nfor df in train:\n    ch = gen_features(df['acoustic_data'])\n    X_train = X_train.append(ch, ignore_index=True)\n    y_train = y_train.append(pd.Series(df['time_to_failure'].values[-1]))","metadata":{"execution":{"iopub.status.busy":"2023-04-11T19:48:30.904245Z","iopub.execute_input":"2023-04-11T19:48:30.904640Z","iopub.status.idle":"2023-04-11T19:52:22.188907Z","shell.execute_reply.started":"2023-04-11T19:48:30.904605Z","shell.execute_reply":"2023-04-11T19:52:22.187851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.describe()","metadata":{"execution":{"iopub.status.busy":"2023-04-11T19:52:22.190759Z","iopub.execute_input":"2023-04-11T19:52:22.191091Z","iopub.status.idle":"2023-04-11T19:52:22.240606Z","shell.execute_reply.started":"2023-04-11T19:52:22.191063Z","shell.execute_reply":"2023-04-11T19:52:22.239516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Step 5 - Implement Catboost Model\n\nGradient boosting on decision trees is a form of machine learning that works by progressively training more complex models to maximize the accuracy of predictions.\n\nIt's particularly useful for predictive models that analyze ordered (continuous) data and categorical data.\n\nIt's one of the most efficient ways to build ensemble models. The combination of gradient boosting with decision trees provides state-of-the-art results in many applications with structured data.\n\nOn the first iteration, the algorithm learns the first tree to reduce the training error, shown on left-hand image in figure 1.\n\nThis model usually has a significant error; it’s not a good idea to build very big trees in boosting since they overfit the data.\n\nThe right-hand image in figure 1 shows the second iteration, in which the algorithm learns one more tree to reduce the error made by the first tree.\n\nThe algorithm repeats this procedure until it builds a decent quality model\n","metadata":{}},{"cell_type":"markdown","source":"Each step of Gradient Boosting combines two steps:\n\nStep 1 - Computing gradients of the loss function we want to optimize for each input object\n\nStep 2 - Learning the decision tree which predicts gradients of the loss function\n\nELI5 Time\n\nStep 1 - We first model data with simple models and analyze data for errors.\n\nStep 2 - These errors signify data points that are difficult to fit by a simple model.\n\nStep 3 - Then for later models, we particularly focus on those hard to fit data to get them right.\n\nStep 4 - In the end, we combine all the predictors by giving some weights to each predictor.\n","metadata":{}},{"cell_type":"code","source":"#Model #1 - Catboost\n\ntrain_pool = Pool(X_train, y_train)\nm = CatBoostRegressor(iterations=10000, loss_function='MAE', boosting_type='Ordered')\nm.fit(X_train, y_train, silent=True)\nm.best_score_","metadata":{"execution":{"iopub.status.busy":"2023-04-11T20:38:28.588392Z","iopub.execute_input":"2023-04-11T20:38:28.588874Z","iopub.status.idle":"2023-04-11T20:40:30.384148Z","shell.execute_reply.started":"2023-04-11T20:38:28.588831Z","shell.execute_reply":"2023-04-11T20:40:30.383008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import RandomizedSearchCV\n\nparams = {\n    'iterations': [1000, 5000, 10000],\n    'learning_rate': [0.01, 0.05, 0.1],\n    'depth': [4, 6, 8],\n    'loss_function': ['MAE', 'RMSE'],\n}\n\nrandom_search = RandomizedSearchCV(estimator=CatBoostRegressor(silent=True),\n                                   param_distributions=params,\n                                   n_iter=10,\n                                   cv=5,\n                                   n_jobs=-1)\nrandom_search.fit(X_train, y_train)\n\nbest_params = random_search.best_params_\nbest_score = random_search.best_score_\n\nprint(\"Best Parameters: \", best_params)\nprint(\"Best Score: \", best_score)","metadata":{"execution":{"iopub.status.busy":"2023-04-11T21:09:12.669467Z","iopub.execute_input":"2023-04-11T21:09:12.670433Z","iopub.status.idle":"2023-04-11T21:39:51.644816Z","shell.execute_reply.started":"2023-04-11T21:09:12.670396Z","shell.execute_reply":"2023-04-11T21:39:51.642695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This code performs hyperparameter tuning using RandomizedSearchCV with CatBoostRegressor as the estimator. It searches for the best combination of hyperparameters from the specified parameter distributions (params). The code fits the model with different hyperparameter values (n_iter=10) using 5-fold cross-validation (cv=5) and runs the search in parallel (n_jobs=-1) to speed up the process.\n\nThe output shows the best hyperparameters found by the RandomizedSearchCV (best_params) and the corresponding best score (best_score). In this case, the best hyperparameters found are:\n\n    'loss_function': 'RMSE'\n    'learning_rate': 0.01\n    'iterations': 1000\n    'depth': 4\n\nThe best score (best_score) is a performance metric (e.g., R^2 score, accuracy) obtained using the best hyperparameters during cross-validation. It indicates the performance of the CatBoostRegressor model with the best hyperparameters found by the RandomizedSearchCV. In this case, the best score is 0.39426905699313197, which is the highest score achieved among the hyperparameter combinations tried during the search.","metadata":{}},{"cell_type":"markdown","source":"Need to learn a nonlinear decision boundary? Grab a Kernel\n\nA very simple and intuitive way of thinking about kernels (at least for SVMs) is a similarity function.\n\nGiven two objects, the kernel outputs some similarity score. The objects can be anything starting from two integers, two real valued vectors, trees whatever provided that the kernel function knows how to compare them.\n\nThe arguably simplest example is the linear kernel, also called dot-product. Given two vectors, the similarity is the length of the projection of one vector on another.\n\nAnother interesting kernel examples is Gaussian kernel. Given two vectors, the similarity will diminish with the radius of σ. The distance between two objects is \"reweighted\" by this radius parameter.\n\nThe success of learning with kernels (again, at least for SVMs), very strongly depends on the choice of kernel. You can see a kernel as a compact representation of the knowledge about your classification problem. It is very often problem specific.\n","metadata":{}},{"cell_type":"code","source":"#Model #2 - Support Vector Machine w/ RBF + Grid Search\n\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.svm import NuSVR, SVR\n\n\nscaler = StandardScaler()\nscaler.fit(X_train)\nX_train_scaled = scaler.transform(X_train)\n\nparameters = [{'gamma': [0.001, 0.005, 0.01, 0.02, 0.05, 0.1],\n               'C': [0.1, 0.2, 0.25, 0.5, 1, 1.5, 2]}]\n               #'nu': [0.75, 0.8, 0.85, 0.9, 0.95, 0.97]}]\n\nreg1 = GridSearchCV(SVR(kernel='rbf', tol=0.01), parameters, cv=5, scoring='neg_mean_absolute_error')\nreg1.fit(X_train_scaled, y_train.values.flatten())\ny_pred1 = reg1.predict(X_train_scaled)\n\nprint(\"Best CV score: {:.4f}\".format(reg1.best_score_))\nprint(reg1.best_params_)","metadata":{"execution":{"iopub.status.busy":"2023-04-11T20:40:35.262982Z","iopub.execute_input":"2023-04-11T20:40:35.263594Z","iopub.status.idle":"2023-04-11T20:43:46.559693Z","shell.execute_reply.started":"2023-04-11T20:40:35.263555Z","shell.execute_reply":"2023-04-11T20:43:46.558552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}