{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## 1 Розуміння задачі","metadata":{"_uuid":"74925434-c8be-4ed0-a745-a33e042eb7c2","_cell_guid":"c23c9896-739b-4187-be15-3fdf1798ff16","trusted":true}},{"cell_type":"code","source":"import os\nimport math\nimport numpy as np \nimport pandas as pd\n\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\n\nINPUT_PATH = \"/kaggle/input/predict-volcanic-eruptions-ingv-oe\"\nINPUT_PATH_TRAIN = INPUT_PATH + \"/train/\"\nINPUT_PATH_TEST = INPUT_PATH + \"/test/\"\n\ndf_train = pd.read_csv(INPUT_PATH + \"/train.csv\")\ndf_samplesub = pd.read_csv(INPUT_PATH + \"/sample_submission.csv\")\n\n# Get lists of train/test file names\ntrain_filenames = [ INPUT_PATH_TRAIN + str(sid) + \".csv\" for sid in df_train[\"segment_id\"] ]\ntest_filenames = [ INPUT_PATH_TEST + str(sid) + \".csv\" for sid in df_samplesub[\"segment_id\"] ]\n\nprint(\"Ready\\n\")","metadata":{"_uuid":"94bd0e38-fe03-4878-93d9-988c8a026014","_cell_guid":"3294a31d-b718-4689-ac82-d1c808f7f70a","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-11-28T07:09:52.144934Z","iopub.execute_input":"2022-11-28T07:09:52.145441Z","iopub.status.idle":"2022-11-28T07:09:53.342884Z","shell.execute_reply.started":"2022-11-28T07:09:52.145313Z","shell.execute_reply":"2022-11-28T07:09:53.341596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2 Розуміння даних","metadata":{}},{"cell_type":"code","source":"print(\"train.csv:\\n\", df_train)\nprint(\"\\nsample_submission.csv:\\n\", df_samplesub)","metadata":{"execution":{"iopub.status.busy":"2022-11-28T07:10:53.704502Z","iopub.execute_input":"2022-11-28T07:10:53.705325Z","iopub.status.idle":"2022-11-28T07:10:53.715639Z","shell.execute_reply.started":"2022-11-28T07:10:53.705266Z","shell.execute_reply":"2022-11-28T07:10:53.714771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print how much files and a file content example for train and test set\nprint(\"Training files:\", len(train_filenames), \"\\n\")\nprint(\"Training file content example: \\n\", pd.read_csv(train_filenames[0]))\nprint(\"\\n------------------------------------\\n\")\nprint(\"Test files:\", len(test_filenames), \"\\n\")\nprint(\"Test file content example: \\n\", pd.read_csv(test_filenames[0]))","metadata":{"execution":{"iopub.status.busy":"2022-11-28T07:11:15.286656Z","iopub.execute_input":"2022-11-28T07:11:15.287046Z","iopub.status.idle":"2022-11-28T07:11:15.591437Z","shell.execute_reply.started":"2022-11-28T07:11:15.287015Z","shell.execute_reply":"2022-11-28T07:11:15.590146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for intersections of train and test\nset(df_train[\"segment_id\"]).intersection(\n    set(df_samplesub[\"segment_id\"])\n)","metadata":{"execution":{"iopub.status.busy":"2022-11-28T07:11:25.004182Z","iopub.execute_input":"2022-11-28T07:11:25.004582Z","iopub.status.idle":"2022-11-28T07:11:25.019239Z","shell.execute_reply.started":"2022-11-28T07:11:25.004550Z","shell.execute_reply":"2022-11-28T07:11:25.017725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize distribution of records in timeline\nsns.distplot(df_train[\"time_to_eruption\"], hist=True, hist_kws={\"edgecolor\": \"darkblue\"})\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-28T07:11:31.944515Z","iopub.execute_input":"2022-11-28T07:11:31.944936Z","iopub.status.idle":"2022-11-28T07:11:32.232633Z","shell.execute_reply.started":"2022-11-28T07:11:31.944901Z","shell.execute_reply":"2022-11-28T07:11:32.230363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize one of record files\ndf = pd.read_csv(train_filenames[127])\nprint(df.info())\n\nfig, axes = plt.subplots(nrows=10, ncols=1, figsize=(18, 40))\nfor i in range(10):\n    df[\"sensor_\"+str(i+1)].plot(ax=axes[i], color=sns.color_palette(\"bright\", n_colors=10)[i])\n    axes[i].legend(loc=\"upper right\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-28T07:14:14.176941Z","iopub.execute_input":"2022-11-28T07:14:14.177331Z","iopub.status.idle":"2022-11-28T07:14:16.686889Z","shell.execute_reply.started":"2022-11-28T07:14:14.177300Z","shell.execute_reply":"2022-11-28T07:14:16.685572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3 Перетворення даних для навчання","metadata":{}},{"cell_type":"code","source":"QUANTILES = [0.01, 0.05, 0.10, 0.25, 0.4, 0.6, 0.75, 0.90, 0.95, 0.99]\n\ndef build_one_row(sample):\n    result = []\n    for sensor in sample: \n        col = sample[sensor]\n        if col.count():\n            result.append(col.mean())\n            result.append(col.std())\n            result.append(col.var())\n            result.append(col.max())\n            result.append(col.mean())\n            result.append(col.skew())\n            for qu in QUANTILES: \n                result.append(np.quantile(col, qu))\n        else:\n            for i in range(16):\n                result.append(0)\n    return result\n\ndef build_all(filenames):\n    count, total = 0, len(filenames)\n    nancounts = False\n    print(\"Processing\\n\")\n    result = []\n    for filename in filenames:\n        sample = pd.read_csv(filename)\n        if not nancounts:\n            nancounts = { sensor: 0 for sensor in sample }\n        for sensor in sample:\n            if not sample[sensor].count(): \n                nancounts[sensor]+=1\n        result.append(build_one_row(sample))\n        if count % 10 == 0: \n            print(f\"\\r[{'='*(30*count//total)}{' '*(30*(total-count)//total)}] {count} of {total} done     \", end=\"\")\n        count += 1\n    print(f\"\\r[{'='*30}] {count} of {total} done     \", end=\"\")\n    return result, nancounts\n","metadata":{"execution":{"iopub.status.busy":"2022-11-27T18:57:48.451547Z","iopub.execute_input":"2022-11-27T18:57:48.451962Z","iopub.status.idle":"2022-11-27T18:58:11.633737Z","shell.execute_reply.started":"2022-11-27T18:57:48.451931Z","shell.execute_reply":"2022-11-27T18:58:11.632282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_files_processed, train_nancounts = build_all(train_filenames)\nX = np.nan_to_num(np.array(train_files_processed))\nY = np.array(df_train[\"time_to_eruption\"])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(12,7))\nax = plt.subplot(1, 1, 1)\nsns.barplot(x=list(train_nancounts.keys()), y=list(train_nancounts.values()))\nax.set_title(\"Amount of missing data from each sensor\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-27T18:59:30.043854Z","iopub.execute_input":"2022-11-27T18:59:30.044425Z","iopub.status.idle":"2022-11-27T18:59:30.302846Z","shell.execute_reply.started":"2022-11-27T18:59:30.044380Z","shell.execute_reply":"2022-11-27T18:59:30.301538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"X shape:\", X.shape)\nprint(\"Y shape:\", Y.shape)","metadata":{"execution":{"iopub.status.busy":"2022-11-27T19:03:25.766150Z","iopub.execute_input":"2022-11-27T19:03:25.767763Z","iopub.status.idle":"2022-11-27T19:03:25.776437Z","shell.execute_reply.started":"2022-11-27T19:03:25.767686Z","shell.execute_reply":"2022-11-27T19:03:25.774950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nx_train, x_test, y_train, y_test = train_test_split(X, Y, test_size=0.2)\n\nprint(\"Train set:\", x_train.shape[0], \"entries\")\nprint(\"Test set:\", x_test.shape[0], \"entries\")","metadata":{"execution":{"iopub.status.busy":"2022-11-27T16:48:12.531863Z","iopub.execute_input":"2022-11-27T16:48:12.532443Z","iopub.status.idle":"2022-11-27T16:48:12.559303Z","shell.execute_reply.started":"2022-11-27T16:48:12.532397Z","shell.execute_reply":"2022-11-27T16:48:12.557126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4 Моделювання, оцінка точності\n\n### sklearn.linear_model.LinearRegression","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import r2_score\n\nmodel = LinearRegression()\nmodel.fit(x_train, y_train)\ny_pred = model.predict(x_test)\nprint(\"- M. sq. error :\", mean_squared_error(y_test, y_pred))\nprint(\"- Std. dev.    :\", math.sqrt(mean_squared_error(y_test, y_pred)))\nprint(\"- R^2 score    :\", r2_score(y_test, y_pred), \"\\n\")\n\nfig = plt.figure(figsize=(12, 9))\nax = plt.subplot(1, 1, 1)\nplt.scatter(y_test, y_pred)\nax.set_xlabel(\"real TTE\")\nax.set_ylabel(\"predicted TTE\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-27T17:02:43.146346Z","iopub.execute_input":"2022-11-27T17:02:43.147741Z","iopub.status.idle":"2022-11-27T17:02:43.553100Z","shell.execute_reply.started":"2022-11-27T17:02:43.147682Z","shell.execute_reply":"2022-11-27T17:02:43.552282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### sklearn.neural_network.MLPRegressor","metadata":{}},{"cell_type":"code","source":"from sklearn.neural_network import MLPRegressor\n\nmodel = MLPRegressor(hidden_layer_sizes=(256, 64, 16), max_iter=4000, activation=\"relu\", \n                     alpha=0.0005, solver=\"adam\", verbose=0, random_state=None, tol=0.0001)\n\nmodel.fit(x_train, y_train)\ny_pred = model.predict(x_test)\nprint(\"- M. sq. error :\", mean_squared_error(y_test, y_pred))\nprint(\"- Std. dev.    :\", math.sqrt(mean_squared_error(y_test, y_pred)))\nprint(\"- R^2 score    :\", r2_score(y_test, y_pred), \"\\n\")\n\nfig = plt.figure(figsize=(12, 9))\nax = plt.subplot(1, 1, 1)\nplt.scatter(y_test, y_pred)\nax.set_xlabel(\"real TTE\")\nax.set_ylabel(\"predicted TTE\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-27T17:10:46.520694Z","iopub.execute_input":"2022-11-27T17:10:46.521255Z","iopub.status.idle":"2022-11-27T17:11:48.638308Z","shell.execute_reply.started":"2022-11-27T17:10:46.521192Z","shell.execute_reply":"2022-11-27T17:11:48.637052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### sklearn.ensemble.RandomForestRegressor","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\n\nmodel = RandomForestRegressor(max_depth=19)\nmodel.fit(x_train, y_train)\ny_pred = model.predict(x_test)\nprint(\"- M. sq. error :\", mean_squared_error(y_test, y_pred))\nprint(\"- Std. dev.    :\", math.sqrt(mean_squared_error(y_test, y_pred)))\nprint(\"- R^2 score    :\", r2_score(y_test, y_pred), \"\\n\")\n\nfig = plt.figure(figsize=(12, 9))\nax = plt.subplot(1, 1, 1)\nplt.scatter(y_test, y_pred)\nax.set_xlabel(\"real TTE\")\nax.set_ylabel(\"predicted TTE\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-27T17:20:05.933763Z","iopub.execute_input":"2022-11-27T17:20:05.934681Z","iopub.status.idle":"2022-11-27T17:20:39.730036Z","shell.execute_reply.started":"2022-11-27T17:20:05.934637Z","shell.execute_reply":"2022-11-27T17:20:39.728661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5 Впровадження моделі\n\nВ production буде використовуватися RandomForestRegressor з точністю r2_score = 0.78","metadata":{}},{"cell_type":"code","source":"prod_files_processed, prod_nancounts = build_all(test_filenames)\nx_prod = np.nan_to_num(np.array(prod_files_processed))","metadata":{"execution":{"iopub.status.busy":"2022-11-27T19:02:04.818292Z","iopub.execute_input":"2022-11-27T19:02:04.818947Z","iopub.status.idle":"2022-11-27T19:02:39.788391Z","shell.execute_reply.started":"2022-11-27T19:02:04.818901Z","shell.execute_reply":"2022-11-27T19:02:39.786978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(12,7))\nax = plt.subplot(1, 1, 1)\nsns.barplot(x=list(prod_nancounts.keys()), y=list(prod_nancounts.values()))\nax.set_title(\"Amount of missing data from each sensor\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-27T19:02:42.081075Z","iopub.execute_input":"2022-11-27T19:02:42.081618Z","iopub.status.idle":"2022-11-27T19:02:42.368761Z","shell.execute_reply.started":"2022-11-27T19:02:42.081579Z","shell.execute_reply":"2022-11-27T19:02:42.367349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_samplesub[\"time_to_eruption\"] = model.predict(x_prod)\ndf_samplesub","metadata":{"execution":{"iopub.status.busy":"2022-11-27T17:44:04.884758Z","iopub.execute_input":"2022-11-27T17:44:04.885308Z","iopub.status.idle":"2022-11-27T17:44:05.008306Z","shell.execute_reply.started":"2022-11-27T17:44:04.885266Z","shell.execute_reply":"2022-11-27T17:44:05.007131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_samplesub.to_csv('submission.csv', header=True, index=False)\nprint(\"Export - done!\")","metadata":{"execution":{"iopub.status.busy":"2022-11-27T17:45:43.615908Z","iopub.execute_input":"2022-11-27T17:45:43.616393Z","iopub.status.idle":"2022-11-27T17:45:43.640811Z","shell.execute_reply.started":"2022-11-27T17:45:43.616358Z","shell.execute_reply":"2022-11-27T17:45:43.639474Z"},"trusted":true},"execution_count":null,"outputs":[]}]}