{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import gc\nimport math\nimport numpy as np \nimport pandas as pd\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import r2_score\nfrom sklearn.ensemble import RandomForestRegressor\n\nfrom matplotlib import pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-11-29T08:10:04.080305Z","iopub.execute_input":"2022-11-29T08:10:04.081256Z","iopub.status.idle":"2022-11-29T08:10:05.656014Z","shell.execute_reply.started":"2022-11-29T08:10:04.081151Z","shell.execute_reply":"2022-11-29T08:10:05.654577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Розуміння задачі","metadata":{}},{"cell_type":"code","source":"INPUT_PATH = \"/kaggle/input/predict-volcanic-eruptions-ingv-oe\"\nINPUT_PATH_TRAIN = INPUT_PATH + \"/train/\"\nINPUT_PATH_TEST = INPUT_PATH + \"/test/\"\n\ndf_train = pd.read_csv(INPUT_PATH + \"/train.csv\")\ndf_samplesub = pd.read_csv(INPUT_PATH + \"/sample_submission.csv\")\n\nprint(\"train.csv:\\n\", df_train)\nprint(\"\\nsample_submission.csv:\\n\", df_samplesub)","metadata":{"execution":{"iopub.status.busy":"2022-11-29T08:10:07.812149Z","iopub.execute_input":"2022-11-29T08:10:07.812723Z","iopub.status.idle":"2022-11-29T08:10:07.861202Z","shell.execute_reply.started":"2022-11-29T08:10:07.812635Z","shell.execute_reply":"2022-11-29T08:10:07.859891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_filenames, test_filenames = [], []\n\nfor seg_id in df_train[\"segment_id\"]:\n    train_filenames.append(INPUT_PATH_TRAIN + str(seg_id) + \".csv\")\n\nfor seg_id in df_samplesub[\"segment_id\"]:\n    test_filenames.append(INPUT_PATH_TEST + str(seg_id) + \".csv\")\n    \nprint(\"Train:\", len(train_filenames), \"files\")\nprint(\"Test:\", len(test_filenames), \"files\")","metadata":{"execution":{"iopub.status.busy":"2022-11-29T08:10:13.277527Z","iopub.execute_input":"2022-11-29T08:10:13.277979Z","iopub.status.idle":"2022-11-29T08:10:13.297166Z","shell.execute_reply.started":"2022-11-29T08:10:13.277941Z","shell.execute_reply":"2022-11-29T08:10:13.295435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Дослідження даних","metadata":{}},{"cell_type":"code","source":"pd.read_csv(train_filenames[0])","metadata":{"execution":{"iopub.status.busy":"2022-11-29T08:10:18.645482Z","iopub.execute_input":"2022-11-29T08:10:18.645905Z","iopub.status.idle":"2022-11-29T08:10:18.823811Z","shell.execute_reply.started":"2022-11-29T08:10:18.645867Z","shell.execute_reply":"2022-11-29T08:10:18.822448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv(test_filenames[0])","metadata":{"execution":{"iopub.status.busy":"2022-11-29T08:10:22.487346Z","iopub.execute_input":"2022-11-29T08:10:22.487813Z","iopub.status.idle":"2022-11-29T08:10:22.651108Z","shell.execute_reply.started":"2022-11-29T08:10:22.487772Z","shell.execute_reply":"2022-11-29T08:10:22.650134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_ordered = df_train.sort_values(\"time_to_eruption\")\ndf_train_ordered","metadata":{"execution":{"iopub.status.busy":"2022-11-29T08:10:25.969429Z","iopub.execute_input":"2022-11-29T08:10:25.969888Z","iopub.status.idle":"2022-11-29T08:10:25.988907Z","shell.execute_reply.started":"2022-11-29T08:10:25.969840Z","shell.execute_reply":"2022-11-29T08:10:25.987066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"segment_ids_ordered = list(df_train_ordered[\"segment_id\"])\nmin_tte_id, max_tte_id = segment_ids_ordered[0], segment_ids_ordered[-1]\nprint(min_tte_id)\nprint(max_tte_id)","metadata":{"execution":{"iopub.status.busy":"2022-11-29T08:10:31.428110Z","iopub.execute_input":"2022-11-29T08:10:31.428499Z","iopub.status.idle":"2022-11-29T08:10:31.435408Z","shell.execute_reply.started":"2022-11-29T08:10:31.428467Z","shell.execute_reply":"2022-11-29T08:10:31.434524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_min_tte = pd.read_csv(INPUT_PATH_TRAIN + str(min_tte_id) + \".csv\")\ndf_max_tte = pd.read_csv(INPUT_PATH_TRAIN + str(max_tte_id) + \".csv\")\n\npalette = sns.color_palette(\"bright\", n_colors=11)\n\nfig = plt.figure(figsize=(18, 30))\n\nfor df, df_name, col_i in [\n    (df_min_tte, \"Min Time To Eruption\", -1), \n    (df_max_tte, \"Max Time to Eruption\", 0)\n]:\n    for sensor_i in [1, 2, 3, 4, 5, 6, 7, 8, 9, 10]:\n        ax = plt.subplot(10, 2, sensor_i*2 + col_i)\n        if sensor_i == 1:\n            ax.set_title(df_name)\n        df[f\"sensor_{sensor_i}\"].plot(ax=ax, color=palette[sensor_i])\n        ax.legend(loc=\"upper right\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-29T08:10:35.312613Z","iopub.execute_input":"2022-11-29T08:10:35.313899Z","iopub.status.idle":"2022-11-29T08:10:39.410945Z","shell.execute_reply.started":"2022-11-29T08:10:35.313852Z","shell.execute_reply":"2022-11-29T08:10:39.409313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ax = plt.subplot(1, 1, 1)\nsns.histplot(df_train[\"time_to_eruption\"])\nax.set_title(\"Distribution of records in timeline\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-29T08:18:35.128026Z","iopub.execute_input":"2022-11-29T08:18:35.128423Z","iopub.status.idle":"2022-11-29T08:18:35.392176Z","shell.execute_reply.started":"2022-11-29T08:18:35.128389Z","shell.execute_reply":"2022-11-29T08:18:35.390711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_check_dfs = [ pd.read_csv(filename, nrows=2) for filename in train_filenames ]\n\ntrain_nancounts = { column: 0 for column in train_check_dfs[0] }\n\nfor df in train_check_dfs:\n    for column in df:\n        if np.isnan(df[column][0]): \n            train_nancounts[column] += 1\n\nprint(f\"{len(train_check_dfs)} files loaded\")\n            \nfig = plt.figure(figsize=(10, 7))\nax = plt.subplot(1, 1, 1)\nsns.barplot(x=list(train_nancounts.keys()), y=list(train_nancounts.values()))\nax.set_title(\"Amount of NaN from each sensor [ train ]\")\nplt.show()\n\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-11-29T08:18:41.529625Z","iopub.execute_input":"2022-11-29T08:18:41.530039Z","iopub.status.idle":"2022-11-29T08:19:52.844039Z","shell.execute_reply.started":"2022-11-29T08:18:41.530008Z","shell.execute_reply":"2022-11-29T08:19:52.842943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_check_dfs = [ pd.read_csv(filename, nrows=2) for filename in test_filenames ]\n\ntest_nancounts = { column: 0 for column in test_check_dfs[0] }\n\nfor df in test_check_dfs:\n    for column in df:\n        if np.isnan(df[column][0]): \n            test_nancounts[column] += 1\n\nprint(f\"{len(test_check_dfs)} files loaded\")\n            \nfig = plt.figure(figsize=(10, 7))\nax = plt.subplot(1, 1, 1)\nsns.barplot(x=list(test_nancounts.keys()), y=list(test_nancounts.values()))\nax.set_title(\"Amount of NaN from each sensor [ test ]\")\nplt.show()\n\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-11-29T08:20:33.217304Z","iopub.execute_input":"2022-11-29T08:20:33.217727Z","iopub.status.idle":"2022-11-29T08:21:56.190741Z","shell.execute_reply.started":"2022-11-29T08:20:33.217693Z","shell.execute_reply":"2022-11-29T08:21:56.189434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Перетворення даних для навчання","metadata":{}},{"cell_type":"code","source":"extractors = [\n    lambda x: x.min(),\n    lambda x: x.max(),\n    lambda x: x.mean(),\n    lambda x: x.std(),\n    lambda x: x.var(),\n    lambda x: x.skew(),\n    lambda x: np.quantile(x, 0.01),\n    lambda x: np.quantile(x, 0.05),\n    lambda x: np.quantile(x, 0.10),\n    lambda x: np.quantile(x, 0.30),\n    lambda x: np.quantile(x, 0.70),\n    lambda x: np.quantile(x, 0.90),\n    lambda x: np.quantile(x, 0.95),\n    lambda x: np.quantile(x, 0.99)\n]\n\ndef convert_one_signal(signal):\n    if not signal.count():\n        return [0]*len(extractors)\n    return [ f(signal) for f in extractors ]\n\ndef convert_one_record(record):\n    result = []\n    for sensor_id in record: \n         for t in convert_one_signal(record[sensor_id]):\n            result.append(t)\n    return result\n\ndef convert_files(file_list):\n    converted = [ convert_one_record(pd.read_csv(filename)) for filename in file_list ]\n    _ = gc.collect()\n    return np.nan_to_num(np.array(converted))","metadata":{"execution":{"iopub.status.busy":"2022-11-29T08:24:44.082639Z","iopub.execute_input":"2022-11-29T08:24:44.083057Z","iopub.status.idle":"2022-11-29T08:24:44.093973Z","shell.execute_reply.started":"2022-11-29T08:24:44.083022Z","shell.execute_reply":"2022-11-29T08:24:44.092719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_converted = convert_files(train_filenames)\n\nprint(train_converted.shape)","metadata":{"execution":{"iopub.status.busy":"2022-11-29T08:24:48.823388Z","iopub.execute_input":"2022-11-29T08:24:48.824187Z","iopub.status.idle":"2022-11-29T08:25:08.313364Z","shell.execute_reply.started":"2022-11-29T08:24:48.824148Z","shell.execute_reply":"2022-11-29T08:25:08.311983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_converted\nY = np.array(df_train[\"time_to_eruption\"])\n\nx_train, x_eval, y_train, y_eval = train_test_split(X, Y, test_size = 0.3)\n\nprint(\"Training:\", x_train.shape[0], \"records\")\nprint(\"Evaluation:\", x_eval.shape[0], \"records\")","metadata":{"execution":{"iopub.status.busy":"2022-11-28T21:40:18.938691Z","iopub.execute_input":"2022-11-28T21:40:18.939117Z","iopub.status.idle":"2022-11-28T21:40:18.948583Z","shell.execute_reply.started":"2022-11-28T21:40:18.939082Z","shell.execute_reply":"2022-11-28T21:40:18.947171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Моделювання","metadata":{}},{"cell_type":"code","source":"model = RandomForestRegressor(max_depth=19)\nmodel.fit(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-11-28T21:43:02.212447Z","iopub.execute_input":"2022-11-28T21:43:02.212928Z","iopub.status.idle":"2022-11-28T21:43:02.492788Z","shell.execute_reply.started":"2022-11-28T21:43:02.212891Z","shell.execute_reply":"2022-11-28T21:43:02.491527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Оцінка точності моделі","metadata":{}},{"cell_type":"code","source":"y_pred = model.predict(x_eval)\nprint(\"- M. sq. error :\", mean_squared_error(y_eval, y_pred))\nprint(\"- Std. dev.    :\", math.sqrt(mean_squared_error(y_eval, y_pred)))\nprint(\"- R^2 score    :\", r2_score(y_eval, y_pred), \"\\n\")\n\nfig = plt.figure(figsize=(9, 9))\nax = plt.subplot(1, 1, 1)\nplt.scatter(y_eval, y_pred, color=\"orange\")\nax.set_xlabel(\"Real Time To Eruption\")\nax.set_ylabel(\"Predicted Time To Eruption\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-28T21:48:26.325876Z","iopub.execute_input":"2022-11-28T21:48:26.326272Z","iopub.status.idle":"2022-11-28T21:48:26.504158Z","shell.execute_reply.started":"2022-11-28T21:48:26.326242Z","shell.execute_reply":"2022-11-28T21:48:26.502793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Впровадження моделі","metadata":{}},{"cell_type":"code","source":"X_test_converted = convert_files(test_filenames)\ndf_samplesub[\"time_to_eruption\"] = model.predict(X_test_converted)\ndf_samplesub","metadata":{"execution":{"iopub.status.busy":"2022-11-28T21:47:30.098021Z","iopub.execute_input":"2022-11-28T21:47:30.099174Z","iopub.status.idle":"2022-11-28T21:47:35.391462Z","shell.execute_reply.started":"2022-11-28T21:47:30.099129Z","shell.execute_reply":"2022-11-28T21:47:35.389796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_samplesub.to_csv(\"submission.csv\", header=True, index=False)\nprint(\"done\")","metadata":{},"execution_count":null,"outputs":[]}]}