{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**1. Розуміння задачі**","metadata":{}},{"cell_type":"code","source":"import os\nimport math\nimport numpy as np \nimport pandas as pd\n\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\n\nINPUT_PATH = \"/kaggle/input/predict-volcanic-eruptions-ingv-oe\"\nINPUT_PATH_TRAIN = INPUT_PATH + \"/train/\"\nINPUT_PATH_TEST = INPUT_PATH + \"/test/\"\n\ndf_train = pd.read_csv(INPUT_PATH + \"/train.csv\")\ndf_samplesub = pd.read_csv(INPUT_PATH + \"/sample_submission.csv\")\n\n# Get lists of train/test file names\ntrain_filenames = [ INPUT_PATH_TRAIN + str(sid) + \".csv\" for sid in df_train[\"segment_id\"] ]\ntest_filenames = [ INPUT_PATH_TEST + str(sid) + \".csv\" for sid in df_samplesub[\"segment_id\"] ]\n\nprint(\"Ready\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-12-22T20:22:25.067110Z","iopub.execute_input":"2022-12-22T20:22:25.068101Z","iopub.status.idle":"2022-12-22T20:22:25.091812Z","shell.execute_reply.started":"2022-12-22T20:22:25.068052Z","shell.execute_reply":"2022-12-22T20:22:25.090816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**2. Розуміння даних**","metadata":{}},{"cell_type":"code","source":"print(\"train.csv:\\n\", df_train)\nprint(\"\\nsample_submission.csv:\\n\", df_samplesub)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T20:22:25.093563Z","iopub.execute_input":"2022-12-22T20:22:25.094559Z","iopub.status.idle":"2022-12-22T20:22:25.106755Z","shell.execute_reply.started":"2022-12-22T20:22:25.094522Z","shell.execute_reply":"2022-12-22T20:22:25.105290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Training files:\", len(train_filenames), \"\\n\")\nprint(\"Training file content example: \\n\", pd.read_csv(train_filenames[0]))\nprint(\"\\n------------------------------------\\n\")\nprint(\"Test files:\", len(test_filenames), \"\\n\")\nprint(\"Test file content example: \\n\", pd.read_csv(test_filenames[0]))","metadata":{"execution":{"iopub.status.busy":"2022-12-22T20:22:25.108411Z","iopub.execute_input":"2022-12-22T20:22:25.108925Z","iopub.status.idle":"2022-12-22T20:22:25.286231Z","shell.execute_reply.started":"2022-12-22T20:22:25.108876Z","shell.execute_reply":"2022-12-22T20:22:25.284586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set(df_train[\"segment_id\"]).intersection(\n    set(df_samplesub[\"segment_id\"])\n)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T20:22:25.288268Z","iopub.execute_input":"2022-12-22T20:22:25.288820Z","iopub.status.idle":"2022-12-22T20:22:25.300002Z","shell.execute_reply.started":"2022-12-22T20:22:25.288749Z","shell.execute_reply":"2022-12-22T20:22:25.298310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.distplot(df_train[\"time_to_eruption\"], hist=True, hist_kws={\"edgecolor\": \"darkblue\"})\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-22T20:22:25.303074Z","iopub.execute_input":"2022-12-22T20:22:25.304218Z","iopub.status.idle":"2022-12-22T20:22:25.527694Z","shell.execute_reply.started":"2022-12-22T20:22:25.304151Z","shell.execute_reply":"2022-12-22T20:22:25.526071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(train_filenames[127])\nprint(df.info())\n\nfig, axes = plt.subplots(nrows=10, ncols=1, figsize=(18, 40))\nfor i in range(10):\n    df[\"sensor_\"+str(i+1)].plot(ax=axes[i], color=sns.color_palette(\"bright\", n_colors=10)[i])\n    axes[i].legend(loc=\"upper right\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-22T20:22:25.529641Z","iopub.execute_input":"2022-12-22T20:22:25.531078Z","iopub.status.idle":"2022-12-22T20:22:27.610728Z","shell.execute_reply.started":"2022-12-22T20:22:25.531020Z","shell.execute_reply":"2022-12-22T20:22:27.609618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" **3. Перетворення даних для навчання**","metadata":{}},{"cell_type":"code","source":"QUANTILES = [0.01, 0.05, 0.10, 0.25, 0.4, 0.6, 0.75, 0.90, 0.95, 0.99]\ndef build_one_row(sample):\n    result = []\n    for sensor in sample: \n        col = sample[sensor]\n        if col.count():\n            result.append(col.mean())\n            result.append(col.std())\n            result.append(col.var())\n            result.append(col.max())\n            result.append(col.mean())\n            result.append(col.skew())\n            for qu in QUANTILES: \n                result.append(np.quantile(col, qu))\n        else:\n            for i in range(16):\n                result.append(0)\n    return result\n\ndef build_all(filenames):\n    count, total = 0, len(filenames)\n    nancounts = False\n    print(\"Processing\\n\")\n    result = []\n    for filename in filenames:\n        sample = pd.read_csv(filename)\n        if not nancounts:\n            nancounts = { sensor: 0 for sensor in sample }\n        for sensor in sample:\n            if not sample[sensor].count(): \n                nancounts[sensor]+=1\n        result.append(build_one_row(sample))\n        if count % 10 == 0: \n            print(f\"\\r[{'='*(30*count//total)}{' '*(30*(total-count)//total)}] {count} of {total} done     \", end=\"\")\n        count += 1\n    print(f\"\\r[{'='*30}] {count} of {total} done     \", end=\"\")\n    return result, nancounts","metadata":{"execution":{"iopub.status.busy":"2022-12-22T20:22:27.612315Z","iopub.execute_input":"2022-12-22T20:22:27.613685Z","iopub.status.idle":"2022-12-22T20:22:27.627849Z","shell.execute_reply.started":"2022-12-22T20:22:27.613638Z","shell.execute_reply":"2022-12-22T20:22:27.626574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_files_processed, train_nancounts = build_all(train_filenames)\nX = np.nan_to_num(np.array(train_files_processed))\nY = np.array(df_train[\"time_to_eruption\"])","metadata":{"execution":{"iopub.status.busy":"2022-12-22T20:22:27.630013Z","iopub.execute_input":"2022-12-22T20:22:27.630474Z","iopub.status.idle":"2022-12-22T20:38:04.002129Z","shell.execute_reply.started":"2022-12-22T20:22:27.630426Z","shell.execute_reply":"2022-12-22T20:38:04.000033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(12,7))\nax = plt.subplot(1, 1, 1)\nsns.barplot(x=list(train_nancounts.keys()), y=list(train_nancounts.values()))\nax.set_title(\"Amount of missing data from each sensor\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-22T20:38:04.007189Z","iopub.execute_input":"2022-12-22T20:38:04.008237Z","iopub.status.idle":"2022-12-22T20:38:04.333649Z","shell.execute_reply.started":"2022-12-22T20:38:04.008012Z","shell.execute_reply":"2022-12-22T20:38:04.332057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"X shape:\", X.shape)\nprint(\"Y shape:\", Y.shape)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T20:38:04.335369Z","iopub.execute_input":"2022-12-22T20:38:04.335776Z","iopub.status.idle":"2022-12-22T20:38:04.342826Z","shell.execute_reply.started":"2022-12-22T20:38:04.335737Z","shell.execute_reply":"2022-12-22T20:38:04.341351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nx_train, x_test, y_train, y_test = train_test_split(X, Y, test_size=0.2)\n\nprint(\"Train set:\", x_train.shape[0], \"entries\")\nprint(\"Test set:\", x_test.shape[0], \"entries\")","metadata":{"execution":{"iopub.status.busy":"2022-12-22T20:38:04.344498Z","iopub.execute_input":"2022-12-22T20:38:04.345589Z","iopub.status.idle":"2022-12-22T20:38:04.557550Z","shell.execute_reply.started":"2022-12-22T20:38:04.345547Z","shell.execute_reply":"2022-12-22T20:38:04.556118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**4. Моделювання, оцінка точності**","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import r2_score\n\nmodel = LinearRegression()\nmodel.fit(x_train, y_train)\ny_pred = model.predict(x_test)\nprint(\"- M. sq. error :\", mean_squared_error(y_test, y_pred))\nprint(\"- Std. dev.    :\", math.sqrt(mean_squared_error(y_test, y_pred)))\nprint(\"- R^2 score    :\", r2_score(y_test, y_pred), \"\\n\")\n\nfig = plt.figure(figsize=(12, 9))\nax = plt.subplot(1, 1, 1)\nplt.scatter(y_test, y_pred)\nax.set_xlabel(\"real TTE\")\nax.set_ylabel(\"predicted TTE\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-22T20:38:04.559252Z","iopub.execute_input":"2022-12-22T20:38:04.559661Z","iopub.status.idle":"2022-12-22T20:38:05.018256Z","shell.execute_reply.started":"2022-12-22T20:38:04.559624Z","shell.execute_reply":"2022-12-22T20:38:05.016903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.neural_network import MLPRegressor\n\nmodel = MLPRegressor(hidden_layer_sizes=(256, 64, 16), max_iter=4000, activation=\"relu\", \n                     alpha=0.0005, solver=\"adam\", verbose=0, random_state=None, tol=0.0001)\n\nmodel.fit(x_train, y_train)\ny_pred = model.predict(x_test)\nprint(\"- M. sq. error :\", mean_squared_error(y_test, y_pred))\nprint(\"- Std. dev.    :\", math.sqrt(mean_squared_error(y_test, y_pred)))\nprint(\"- R^2 score    :\", r2_score(y_test, y_pred), \"\\n\")\n\nfig = plt.figure(figsize=(12, 9))\nax = plt.subplot(1, 1, 1)\nplt.scatter(y_test, y_pred)\nax.set_xlabel(\"real TTE\")\nax.set_ylabel(\"predicted TTE\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-22T20:38:05.020365Z","iopub.execute_input":"2022-12-22T20:38:05.021234Z","iopub.status.idle":"2022-12-22T20:38:31.527050Z","shell.execute_reply.started":"2022-12-22T20:38:05.021179Z","shell.execute_reply":"2022-12-22T20:38:31.524964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\n\nmodel = RandomForestRegressor(max_depth=19)\nmodel.fit(x_train, y_train)\ny_pred = model.predict(x_test)\nprint(\"- M. sq. error :\", mean_squared_error(y_test, y_pred))\nprint(\"- Std. dev.    :\", math.sqrt(mean_squared_error(y_test, y_pred)))\nprint(\"- R^2 score    :\", r2_score(y_test, y_pred), \"\\n\")\n\nfig = plt.figure(figsize=(12, 9))\nax = plt.subplot(1, 1, 1)\nplt.scatter(y_test, y_pred)\nax.set_xlabel(\"real TTE\")\nax.set_ylabel(\"predicted TTE\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-22T20:38:31.532958Z","iopub.execute_input":"2022-12-22T20:38:31.533336Z","iopub.status.idle":"2022-12-22T20:39:05.851475Z","shell.execute_reply.started":"2022-12-22T20:38:31.533304Z","shell.execute_reply":"2022-12-22T20:39:05.849994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**5. Впровадження моделі**","metadata":{}},{"cell_type":"code","source":"prod_files_processed, prod_nancounts = build_all(test_filenames)\nx_prod = np.nan_to_num(np.array(prod_files_processed))","metadata":{"execution":{"iopub.status.busy":"2022-12-22T20:39:05.853309Z","iopub.execute_input":"2022-12-22T20:39:05.853699Z","iopub.status.idle":"2022-12-22T20:54:51.869964Z","shell.execute_reply.started":"2022-12-22T20:39:05.853665Z","shell.execute_reply":"2022-12-22T20:54:51.868231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(12,7))\nax = plt.subplot(1, 1, 1)\nsns.barplot(x=list(prod_nancounts.keys()), y=list(prod_nancounts.values()))\nax.set_title(\"Amount of missing data from each sensor\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-22T20:54:51.872367Z","iopub.execute_input":"2022-12-22T20:54:51.873806Z","iopub.status.idle":"2022-12-22T20:54:52.183252Z","shell.execute_reply.started":"2022-12-22T20:54:51.873713Z","shell.execute_reply":"2022-12-22T20:54:52.181753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_samplesub[\"time_to_eruption\"] = model.predict(x_prod)\ndf_samplesub","metadata":{"execution":{"iopub.status.busy":"2022-12-22T20:54:52.185225Z","iopub.execute_input":"2022-12-22T20:54:52.185661Z","iopub.status.idle":"2022-12-22T20:54:52.336238Z","shell.execute_reply.started":"2022-12-22T20:54:52.185624Z","shell.execute_reply":"2022-12-22T20:54:52.334657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_samplesub.to_csv('submission.csv', header=True, index=False)\nprint(\"Export - done!\")","metadata":{"execution":{"iopub.status.busy":"2022-12-22T20:54:52.338145Z","iopub.execute_input":"2022-12-22T20:54:52.338712Z","iopub.status.idle":"2022-12-22T20:54:52.362234Z","shell.execute_reply.started":"2022-12-22T20:54:52.338675Z","shell.execute_reply":"2022-12-22T20:54:52.360916Z"},"trusted":true},"execution_count":null,"outputs":[]}]}