{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nplt.style.use('seaborn-white')\n%matplotlib inline\n\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nimport lightgbm as lgbm\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom math import isnan\n\nfile_list = []\nfile_list_train = []\nfile_list_test = []\n\n\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        file_list.append(os.path.join(dirname, filename))\n        \nPATH = '/kaggle/input/predict-volcanic-eruptions-ingv-oe/'\n\nfor dirname, _, filenames in os.walk('/kaggle/input/predict-volcanic-eruptions-ingv-oe/train'):\n    for filename in filenames:\n        file_list_train.append(os.path.join(dirname, filename))\n        \nfor dirname, _, filenames in os.walk('/kaggle/input/predict-volcanic-eruptions-ingv-oe/test'):\n    for filename in filenames:\n        file_list_test.append(os.path.join(dirname, filename))\n  \n\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-12-17T13:24:57.264447Z","iopub.execute_input":"2021-12-17T13:24:57.26509Z","iopub.status.idle":"2021-12-17T13:25:06.635049Z","shell.execute_reply.started":"2021-12-17T13:24:57.26505Z","shell.execute_reply":"2021-12-17T13:25:06.634094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(file_list[2])\n\n\nprint(pd.read_csv(file_list[0]))\nprint(pd.read_csv(file_list[2]).isna().sum())","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:25:06.636956Z","iopub.execute_input":"2021-12-17T13:25:06.637248Z","iopub.status.idle":"2021-12-17T13:25:06.798961Z","shell.execute_reply.started":"2021-12-17T13:25:06.637216Z","shell.execute_reply":"2021-12-17T13:25:06.79836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(file_list_train[0])\nprint(pd.read_csv(file_list_train[0]))","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:25:06.799828Z","iopub.execute_input":"2021-12-17T13:25:06.800571Z","iopub.status.idle":"2021-12-17T13:25:06.934048Z","shell.execute_reply.started":"2021-12-17T13:25:06.80053Z","shell.execute_reply":"2021-12-17T13:25:06.933165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(file_list_train))","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:25:06.935783Z","iopub.execute_input":"2021-12-17T13:25:06.936016Z","iopub.status.idle":"2021-12-17T13:25:06.941082Z","shell.execute_reply.started":"2021-12-17T13:25:06.93598Z","shell.execute_reply":"2021-12-17T13:25:06.940182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(file_list_test))","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:25:06.942364Z","iopub.execute_input":"2021-12-17T13:25:06.942724Z","iopub.status.idle":"2021-12-17T13:25:06.952536Z","shell.execute_reply.started":"2021-12-17T13:25:06.942695Z","shell.execute_reply":"2021-12-17T13:25:06.951731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files_test = [file.split('/')[-1].split('.')[-2] for file in file_list_test]\nfiles_train = [file.split('/')[-1].split('.')[-2] for file in file_list_train]","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:25:06.953722Z","iopub.execute_input":"2021-12-17T13:25:06.953952Z","iopub.status.idle":"2021-12-17T13:25:06.969292Z","shell.execute_reply.started":"2021-12-17T13:25:06.953925Z","shell.execute_reply":"2021-12-17T13:25:06.968564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(files_train[0:10])\nprint(files_test[0:10])","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:25:06.970301Z","iopub.execute_input":"2021-12-17T13:25:06.970855Z","iopub.status.idle":"2021-12-17T13:25:06.979154Z","shell.execute_reply.started":"2021-12-17T13:25:06.970808Z","shell.execute_reply":"2021-12-17T13:25:06.978373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_set = set(files_test)\ntrain_set = set(files_train)\ninter = test_set.intersection(train_set)\n\nprint(inter)","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:25:06.98102Z","iopub.execute_input":"2021-12-17T13:25:06.981626Z","iopub.status.idle":"2021-12-17T13:25:06.991043Z","shell.execute_reply.started":"2021-12-17T13:25:06.981584Z","shell.execute_reply":"2021-12-17T13:25:06.990163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(PATH+'train.csv')","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:25:06.992177Z","iopub.execute_input":"2021-12-17T13:25:06.992513Z","iopub.status.idle":"2021-12-17T13:25:07.01153Z","shell.execute_reply.started":"2021-12-17T13:25:06.992479Z","shell.execute_reply":"2021-12-17T13:25:07.01073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.distplot(train['time_to_eruption'],\n            hist=True,\n            kde=True,\n            bins=100,\n            color='blue',\n            hist_kws={'edgecolor':'black'})","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:25:07.014148Z","iopub.execute_input":"2021-12-17T13:25:07.014918Z","iopub.status.idle":"2021-12-17T13:25:07.638858Z","shell.execute_reply.started":"2021-12-17T13:25:07.014877Z","shell.execute_reply":"2021-12-17T13:25:07.637655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['time_to_eruption'].describe()","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:25:07.640253Z","iopub.execute_input":"2021-12-17T13:25:07.64059Z","iopub.status.idle":"2021-12-17T13:25:07.65277Z","shell.execute_reply.started":"2021-12-17T13:25:07.640548Z","shell.execute_reply":"2021-12-17T13:25:07.651946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_segment_id = pd.read_csv(PATH+'test/473253715.csv')\n\ndf_segment_id.plot(figsize=(20,20),\n                  subplots=True,\n                  layout=(10,1),\n                  rot=0,\n                  lw=1,\n                  title='sergemnt id #473253715')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:25:07.654041Z","iopub.execute_input":"2021-12-17T13:25:07.654262Z","iopub.status.idle":"2021-12-17T13:25:10.416514Z","shell.execute_reply.started":"2021-12-17T13:25:07.654235Z","shell.execute_reply":"2021-12-17T13:25:10.415885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_segment_id.columns","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:25:10.417635Z","iopub.execute_input":"2021-12-17T13:25:10.417944Z","iopub.status.idle":"2021-12-17T13:25:10.423861Z","shell.execute_reply.started":"2021-12-17T13:25:10.417916Z","shell.execute_reply":"2021-12-17T13:25:10.42302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nstats = dict()\n\nstats[\"sensor_1\"] =  0\nstats[\"sensor_2\"] =  0\nstats[\"sensor_3\"] =  0\nstats[\"sensor_4\"] =  0\nstats[\"sensor_5\"] =  0\nstats[\"sensor_6\"] =  0\nstats[\"sensor_7\"] =  0\nstats[\"sensor_8\"] =  0\nstats[\"sensor_9\"] =  0\nstats[\"sensor_10\"] =  0\n\n\nfor i in file_list_test:\n    df_test_stat = pd.read_csv(i)\n    if df_test_stat[\"sensor_1\"].max() > 0:\n        stats[\"sensor_1\"] += 1\n    if df_test_stat[\"sensor_2\"].max() > 0:\n        stats[\"sensor_2\"] += 1\n    if df_test_stat[\"sensor_3\"].max() > 0:\n        stats[\"sensor_3\"] += 1\n    if df_test_stat[\"sensor_4\"].max() > 0:\n        stats[\"sensor_4\"] += 1\n    if df_test_stat[\"sensor_5\"].max() > 0:\n        stats[\"sensor_5\"] += 1\n    if df_test_stat[\"sensor_6\"].max() > 0:\n        stats[\"sensor_6\"] += 1\n    if df_test_stat[\"sensor_7\"].max() > 0:\n        stats[\"sensor_7\"] += 1\n    if df_test_stat[\"sensor_8\"].max() > 0:\n        stats[\"sensor_8\"] += 1\n    if df_test_stat[\"sensor_9\"].max() > 0:\n        stats[\"sensor_9\"] += 1\n    if df_test_stat[\"sensor_10\"].max() > 0:\n        stats[\"sensor_10\"] += 1\n\n        \nprint(stats)  \n\n","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:25:10.425077Z","iopub.execute_input":"2021-12-17T13:25:10.42539Z","iopub.status.idle":"2021-12-17T13:34:21.695185Z","shell.execute_reply.started":"2021-12-17T13:25:10.425363Z","shell.execute_reply":"2021-12-17T13:34:21.6937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure()\nplt.bar(stats.keys(), stats.values())","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:34:21.697757Z","iopub.execute_input":"2021-12-17T13:34:21.698132Z","iopub.status.idle":"2021-12-17T13:34:21.939391Z","shell.execute_reply.started":"2021-12-17T13:34:21.698091Z","shell.execute_reply":"2021-12-17T13:34:21.938345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train.sort_values('time_to_eruption', axis=0, ascending=True).iloc[[0,-1],:])\n\nsegment_id_min = 601524801\nsegment_id_max = 1923243961\n\ndf_segment_id_min = pd.read_csv(PATH+'train/'+str(segment_id_min)+'.csv')\n\ndf_segment_id_max = pd.read_csv(PATH+'train/'+str(segment_id_max)+'.csv')","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:34:21.940911Z","iopub.execute_input":"2021-12-17T13:34:21.941145Z","iopub.status.idle":"2021-12-17T13:34:22.172277Z","shell.execute_reply.started":"2021-12-17T13:34:21.941113Z","shell.execute_reply":"2021-12-17T13:34:22.171406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_segment_id_min.plot(figsize=(20,20),\n                  subplots=True,\n                  layout=(10,1),\n                  rot=0,\n                  lw=1,\n                  title='segment id min')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:34:22.175362Z","iopub.execute_input":"2021-12-17T13:34:22.175659Z","iopub.status.idle":"2021-12-17T13:34:24.707754Z","shell.execute_reply.started":"2021-12-17T13:34:22.175627Z","shell.execute_reply":"2021-12-17T13:34:24.706893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_segment_id_max.plot(figsize=(20,20),\n                  subplots=True,\n                  layout=(10,1),\n                  rot=0,\n                  lw=1,\n                  title='segmanet id max')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:34:24.708997Z","iopub.execute_input":"2021-12-17T13:34:24.71023Z","iopub.status.idle":"2021-12-17T13:34:27.660976Z","shell.execute_reply.started":"2021-12-17T13:34:24.710186Z","shell.execute_reply":"2021-12-17T13:34:27.660373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_features(signal, ts, sensor_id):\n    X = pd.DataFrame()\n    f = np.fft.fft(signal)\n    f_real = np.real(f)\n    X.loc[ts, f'{sensor_id}_sum'] = signal.sum()\n    X.loc[ts, f'{sensor_id}_mean'] = signal.mean()\n    X.loc[ts, f'{sensor_id}_std'] = signal.std()\n    X.loc[ts, f'{sensor_id}_var'] = signal.var()\n    X.loc[ts, f'{sensor_id}_max'] = signal.max()\n    X.loc[ts, f'{sensor_id}_min'] = signal.min()\n    X.loc[ts, f'{sensor_id}_skew'] = signal.skew()\n    X.loc[ts, f'{sensor_id}_mad'] = signal.mad()\n    X.loc[ts, f'{sensor_id}_kurtosis'] = signal.kurtosis()\n    X.loc[ts, f'{sensor_id}_quantile99'] = np.quantile(signal, 0.99)\n    X.loc[ts, f'{sensor_id}_quantile95'] = np.quantile(signal, 0.95)\n    X.loc[ts, f'{sensor_id}_quantile85'] = np.quantile(signal, 0.85)\n    X.loc[ts, f'{sensor_id}_quantile75'] = np.quantile(signal, 0.75)\n    X.loc[ts, f'{sensor_id}_quantile55'] = np.quantile(signal, 0.55)\n    X.loc[ts, f'{sensor_id}_quantile45'] = np.quantile(signal, 0.45)\n    X.loc[ts, f'{sensor_id}_quantile25'] = np.quantile(signal, 0.25)\n    X.loc[ts, f'{sensor_id}_quantile15'] = np.quantile(signal, 0.15)\n    X.loc[ts, f'{sensor_id}_quantile05'] = np.quantile(signal, 0.05)\n    X.loc[ts, f'{sensor_id}_quantile01'] = np.quantile(signal, 0.01)\n    X.loc[ts, f'{sensor_id}_fft_real_mean'] = f_real.mean()\n    X.loc[ts, f'{sensor_id}_fft_real_std'] = f_real.std()\n    X.loc[ts, f'{sensor_id}_fft_real_max'] = f_real.max()\n    X.loc[ts, f'{sensor_id}_fft_real_min'] = f_real.min()\n    \n    return X","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:34:27.662193Z","iopub.execute_input":"2021-12-17T13:34:27.662973Z","iopub.status.idle":"2021-12-17T13:34:27.676842Z","shell.execute_reply.started":"2021-12-17T13:34:27.662936Z","shell.execute_reply":"2021-12-17T13:34:27.675926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain_set = list()\nseg = 0\n\nfor seg, segment_id in enumerate(train.segment_id):\n    signals = pd.read_csv(PATH+'train/'+str(segment_id)+'.csv')\n    train_row = []\n    \n    if seg%200 == 0:\n        print('Processing segment_id={}'.format(seg))\n        \n    for sensor in range(0, 10):\n        sensor_id = f'sensor_{sensor+1}'\n        train_row.append(build_features(signals[sensor_id].fillna(0), segment_id, sensor_id))\n        \n    train_row = pd.concat(train_row, axis=1)\n    train_set.append(train_row)\n    seg+=1\n    \ntrain_set = pd.concat(train_set)","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:34:27.678382Z","iopub.execute_input":"2021-12-17T13:34:27.678623Z","iopub.status.idle":"2021-12-17T14:13:48.379257Z","shell.execute_reply.started":"2021-12-17T13:34:27.678594Z","shell.execute_reply":"2021-12-17T14:13:48.378358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_set = train_set.reset_index()\ntrain_set = train_set.rename(columns={'index':  'segment_id'})\n\ntrain_set = pd.merge(train_set, train, on='segment_id')","metadata":{"execution":{"iopub.status.busy":"2021-12-17T14:13:48.380942Z","iopub.execute_input":"2021-12-17T14:13:48.381261Z","iopub.status.idle":"2021-12-17T14:13:48.42294Z","shell.execute_reply.started":"2021-12-17T14:13:48.381219Z","shell.execute_reply":"2021-12-17T14:13:48.422328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_set.head(3))","metadata":{"execution":{"iopub.status.busy":"2021-12-17T14:13:48.424016Z","iopub.execute_input":"2021-12-17T14:13:48.424741Z","iopub.status.idle":"2021-12-17T14:13:48.439935Z","shell.execute_reply.started":"2021-12-17T14:13:48.424691Z","shell.execute_reply":"2021-12-17T14:13:48.438835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_files = []\nfor dirname, _, filenames in os.walk(PATH+'test/'):\n    for filename in filenames:\n        test_files.append(filename[:-4])\n        \ntest = pd.DataFrame(test_files, columns=['segment_id'])","metadata":{"execution":{"iopub.status.busy":"2021-12-17T14:13:48.441471Z","iopub.execute_input":"2021-12-17T14:13:48.442561Z","iopub.status.idle":"2021-12-17T14:13:49.682163Z","shell.execute_reply.started":"2021-12-17T14:13:48.442514Z","shell.execute_reply":"2021-12-17T14:13:49.681571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_set = list()\nseg = 0\n\nfor seg, segment_id in enumerate(test.segment_id):\n    signals = pd.read_csv(PATH+'test/'+str(segment_id)+'.csv')\n    test_row = []\n    \n    if seg%200 == 0:\n        print('Processing segment_id={}'.format(seg))\n        \n    for sensor in range(0, 10):\n        sensor_id = f'sensor_{sensor+1}'\n        test_row.append(build_features(signals[sensor_id].fillna(0), segment_id, sensor_id))\n        \n    test_row = pd.concat(test_row, axis=1)\n    test_set.append(test_row)\n    seg+=1\n    \ntest_set = pd.concat(test_set)","metadata":{"execution":{"iopub.status.busy":"2021-12-17T14:13:49.683734Z","iopub.execute_input":"2021-12-17T14:13:49.684028Z","iopub.status.idle":"2021-12-17T14:53:45.606963Z","shell.execute_reply.started":"2021-12-17T14:13:49.68399Z","shell.execute_reply":"2021-12-17T14:53:45.605958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_set = test_set.reset_index()\ntest_set = test_set.rename(columns={'index':  'segment_id'})\n\ntest_set = pd.merge(test_set, test, on='segment_id')","metadata":{"execution":{"iopub.status.busy":"2021-12-17T14:53:45.608714Z","iopub.execute_input":"2021-12-17T14:53:45.609028Z","iopub.status.idle":"2021-12-17T14:53:45.650997Z","shell.execute_reply.started":"2021-12-17T14:53:45.608986Z","shell.execute_reply":"2021-12-17T14:53:45.649986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test_set.head(3))","metadata":{"execution":{"iopub.status.busy":"2021-12-17T14:53:45.652769Z","iopub.execute_input":"2021-12-17T14:53:45.653301Z","iopub.status.idle":"2021-12-17T14:53:45.672447Z","shell.execute_reply.started":"2021-12-17T14:53:45.653253Z","shell.execute_reply":"2021-12-17T14:53:45.67152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_set.drop(['segment_id', 'time_to_eruption'], axis=1)\ny = train_set['time_to_eruption']\n\nX_train, X_valid, y_train, y_valid = train_test_split(X, y, \n                                                      test_size=0.2,\n                                                      random_state=42)","metadata":{"execution":{"iopub.status.busy":"2021-12-17T14:53:45.674301Z","iopub.execute_input":"2021-12-17T14:53:45.674878Z","iopub.status.idle":"2021-12-17T14:53:45.704933Z","shell.execute_reply.started":"2021-12-17T14:53:45.674831Z","shell.execute_reply":"2021-12-17T14:53:45.704071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X_train.head(3))\nprint('np.shape(X_train) = ', np.shape(X_train))","metadata":{"execution":{"iopub.status.busy":"2021-12-17T14:53:45.709872Z","iopub.execute_input":"2021-12-17T14:53:45.710538Z","iopub.status.idle":"2021-12-17T14:53:45.727808Z","shell.execute_reply.started":"2021-12-17T14:53:45.710497Z","shell.execute_reply":"2021-12-17T14:53:45.726946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_train.head(3))\nprint('np.shape(y_train) = ', np.shape(y_train))","metadata":{"execution":{"iopub.status.busy":"2021-12-17T14:53:45.729259Z","iopub.execute_input":"2021-12-17T14:53:45.729519Z","iopub.status.idle":"2021-12-17T14:53:45.738664Z","shell.execute_reply.started":"2021-12-17T14:53:45.729489Z","shell.execute_reply":"2021-12-17T14:53:45.73781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\n\nmodel = RandomForestRegressor(max_depth=20, random_state=0)\nmodel.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2021-12-17T14:53:45.739925Z","iopub.execute_input":"2021-12-17T14:53:45.740612Z","iopub.status.idle":"2021-12-17T14:54:38.835454Z","shell.execute_reply.started":"2021-12-17T14:53:45.740567Z","shell.execute_reply":"2021-12-17T14:54:38.834372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(X_valid)","metadata":{"execution":{"iopub.status.busy":"2021-12-17T14:54:38.837095Z","iopub.execute_input":"2021-12-17T14:54:38.837372Z","iopub.status.idle":"2021-12-17T14:54:38.890888Z","shell.execute_reply.started":"2021-12-17T14:54:38.837329Z","shell.execute_reply":"2021-12-17T14:54:38.89005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import r2_score\n\nconf_mat = r2_score(y_valid, y_pred)\nprint(conf_mat)","metadata":{"execution":{"iopub.status.busy":"2021-12-17T14:54:38.892247Z","iopub.execute_input":"2021-12-17T14:54:38.892532Z","iopub.status.idle":"2021-12-17T14:54:38.898601Z","shell.execute_reply.started":"2021-12-17T14:54:38.892501Z","shell.execute_reply":"2021-12-17T14:54:38.897582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error\n\nmse = mean_squared_error(y_valid, y_pred)\nfig = plt.figure()\nmulreg = fig.add_subplot(1, 1, 1)\nmulreg.scatter(y_valid, y_pred, color='r')\nmulreg.set_title('Nonlinear Regression') ","metadata":{"execution":{"iopub.status.busy":"2021-12-17T14:54:38.899965Z","iopub.execute_input":"2021-12-17T14:54:38.900188Z","iopub.status.idle":"2021-12-17T14:54:39.138358Z","shell.execute_reply.started":"2021-12-17T14:54:38.90016Z","shell.execute_reply":"2021-12-17T14:54:39.137452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction = model.predict(test_set.drop(columns=['segment_id']))","metadata":{"execution":{"iopub.status.busy":"2021-12-17T14:54:39.13978Z","iopub.execute_input":"2021-12-17T14:54:39.140102Z","iopub.status.idle":"2021-12-17T14:54:39.277149Z","shell.execute_reply.started":"2021-12-17T14:54:39.140061Z","shell.execute_reply":"2021-12-17T14:54:39.276506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame()  \nsubmission['segment_id'] = test_set['segment_id']\nsubmission['time_to_eruption'] = prediction\nsubmission.to_csv('submission.csv', header=True, index=False)","metadata":{"execution":{"iopub.status.busy":"2021-12-17T14:54:39.278428Z","iopub.execute_input":"2021-12-17T14:54:39.279161Z","iopub.status.idle":"2021-12-17T14:54:39.308217Z","shell.execute_reply.started":"2021-12-17T14:54:39.279121Z","shell.execute_reply":"2021-12-17T14:54:39.307568Z"},"trusted":true},"execution_count":null,"outputs":[]}]}