{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<center><img src=\"https://images.ctfassets.net/81iqaqpfd8fy/3Wp4SEgzagcICaSqcIMOQM/5721655abf93a19521dad8a35d747f2d/Erupting_Volcano.jpg?h=620&w=1024\"></center>\n<h1><center>INGV - Volcanic Eruption Prediction</center></h1>\n<h1><center>화산 분출 예측을 위한 센서 데이터 분석</center></h1>","metadata":{"_uuid":"31378f16-72cb-4d86-a21c-1259fb9574de","_cell_guid":"80c8a7b6-efa0-4d43-894e-01f642034577","jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-11-07T08:08:17.803696Z","iopub.execute_input":"2022-11-07T08:08:17.804223Z","iopub.status.idle":"2022-11-07T08:08:17.835877Z","shell.execute_reply.started":"2022-11-07T08:08:17.804119Z","shell.execute_reply":"2022-11-07T08:08:17.834399Z"}}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:05:55.665846Z","iopub.execute_input":"2022-11-08T06:05:55.666314Z","iopub.status.idle":"2022-11-08T06:05:55.694188Z","shell.execute_reply.started":"2022-11-08T06:05:55.666223Z","shell.execute_reply":"2022-11-08T06:05:55.692666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob as gb\n\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\nimport seaborn as sb\n\nimport plotly.express as px\nfrom plotly.subplots import make_subplots\nimport plotly.graph_objs as pgo","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:05:55.695926Z","iopub.execute_input":"2022-11-08T06:05:55.696422Z","iopub.status.idle":"2022-11-08T06:05:57.991899Z","shell.execute_reply.started":"2022-11-08T06:05:55.696390Z","shell.execute_reply":"2022-11-08T06:05:57.990509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. 데이터 확인","metadata":{}},{"cell_type":"markdown","source":"## 파일 수 확인","metadata":{}},{"cell_type":"code","source":"train_folder = gb.glob('../input/predict-volcanic-eruptions-ingv-oe/train/*')\nlen(train_folder)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:05:57.993821Z","iopub.execute_input":"2022-11-08T06:05:57.994291Z","iopub.status.idle":"2022-11-08T06:05:58.603383Z","shell.execute_reply.started":"2022-11-08T06:05:57.994247Z","shell.execute_reply":"2022-11-08T06:05:58.602561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_folder = gb.glob('../input/predict-volcanic-eruptions-ingv-oe/test/*')\nlen(test_folder)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:05:58.605807Z","iopub.execute_input":"2022-11-08T06:05:58.606630Z","iopub.status.idle":"2022-11-08T06:05:59.245089Z","shell.execute_reply.started":"2022-11-08T06:05:58.606577Z","shell.execute_reply":"2022-11-08T06:05:59.244274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train.csv 확인","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv(filepath_or_buffer='../input/predict-volcanic-eruptions-ingv-oe/train.csv')\ntrain_data","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:05:59.247919Z","iopub.execute_input":"2022-11-08T06:05:59.248324Z","iopub.status.idle":"2022-11-08T06:05:59.297317Z","shell.execute_reply.started":"2022-11-08T06:05:59.248288Z","shell.execute_reply":"2022-11-08T06:05:59.296427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sample Submission 확인","metadata":{}},{"cell_type":"code","source":"sample_submission = pd.read_csv('../input/predict-volcanic-eruptions-ingv-oe/sample_submission.csv')\nsample_submission","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:05:59.298579Z","iopub.execute_input":"2022-11-08T06:05:59.299314Z","iopub.status.idle":"2022-11-08T06:05:59.329124Z","shell.execute_reply.started":"2022-11-08T06:05:59.299281Z","shell.execute_reply":"2022-11-08T06:05:59.328352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train segment_id 확인","metadata":{}},{"cell_type":"code","source":"train_folder[0]","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:05:59.330623Z","iopub.execute_input":"2022-11-08T06:05:59.331133Z","iopub.status.idle":"2022-11-08T06:05:59.337107Z","shell.execute_reply.started":"2022-11-08T06:05:59.331104Z","shell.execute_reply":"2022-11-08T06:05:59.336265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sequence = pd.read_csv(train_folder[0])\nsequence","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:05:59.338774Z","iopub.execute_input":"2022-11-08T06:05:59.339294Z","iopub.status.idle":"2022-11-08T06:05:59.485788Z","shell.execute_reply.started":"2022-11-08T06:05:59.339263Z","shell.execute_reply":"2022-11-08T06:05:59.484813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sequence.describe()","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:05:59.489881Z","iopub.execute_input":"2022-11-08T06:05:59.490629Z","iopub.status.idle":"2022-11-08T06:05:59.576048Z","shell.execute_reply.started":"2022-11-08T06:05:59.490589Z","shell.execute_reply":"2022-11-08T06:05:59.574569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sensor 값 확인하기","metadata":{}},{"cell_type":"code","source":"def show_sensor(df):\n    fig, ax = plt.subplots(10, 1, constrained_layout=True)\n    fig.set_size_inches((16, 15))\n    \n    for i in range(1, 11):\n        ax[i - 1].plot(df['sensor_{id}'.format(id=i)].values)\n        ax[i - 1].set_title('Sensor_{id}'.format(id=i))\n        ax[i - 1].set_xlabel('time')","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:05:59.577924Z","iopub.execute_input":"2022-11-08T06:05:59.578812Z","iopub.status.idle":"2022-11-08T06:05:59.585635Z","shell.execute_reply.started":"2022-11-08T06:05:59.578709Z","shell.execute_reply":"2022-11-08T06:05:59.584793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_sensor(sequence)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:05:59.586966Z","iopub.execute_input":"2022-11-08T06:05:59.587444Z","iopub.status.idle":"2022-11-08T06:06:01.603185Z","shell.execute_reply.started":"2022-11-08T06:05:59.587417Z","shell.execute_reply":"2022-11-08T06:06:01.601915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_sensor(sequence.fillna(0))","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:06:01.604495Z","iopub.execute_input":"2022-11-08T06:06:01.604856Z","iopub.status.idle":"2022-11-08T06:06:03.319258Z","shell.execute_reply.started":"2022-11-08T06:06:01.604822Z","shell.execute_reply":"2022-11-08T06:06:03.317347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. 데이터 분석 - Basic EDA","metadata":{}},{"cell_type":"markdown","source":"## Train.csv\n- `time_to_eruption`을 h:m:s(시:분:초)로 변환","metadata":{}},{"cell_type":"code","source":"train_data","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:06:03.321254Z","iopub.execute_input":"2022-11-08T06:06:03.321671Z","iopub.status.idle":"2022-11-08T06:06:03.335624Z","shell.execute_reply.started":"2022-11-08T06:06:03.321636Z","shell.execute_reply":"2022-11-08T06:06:03.333986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- 0.01초 간격으로 샘플링되어 있음 (100분의 1초)","metadata":{}},{"cell_type":"code","source":"train_data['hhmmss'] = train_data['time_to_eruption'].apply(lambda x: pd.Timedelta(seconds=(x / 100)))\ntrain_data['hhmmss']","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:06:03.338169Z","iopub.execute_input":"2022-11-08T06:06:03.338683Z","iopub.status.idle":"2022-11-08T06:06:03.395846Z","shell.execute_reply.started":"2022-11-08T06:06:03.338645Z","shell.execute_reply":"2022-11-08T06:06:03.394957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16, 8))\nplt.hist(train_data['hhmmss'] / pd.Timedelta(days=1))\nplt.xlabel('Time between eruptions (days)')\nplt.ylabel('$ of eruptions')","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:06:03.397230Z","iopub.execute_input":"2022-11-08T06:06:03.397513Z","iopub.status.idle":"2022-11-08T06:06:03.590926Z","shell.execute_reply.started":"2022-11-08T06:06:03.397482Z","shell.execute_reply":"2022-11-08T06:06:03.590020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(train_data['hhmmss'] / pd.Timedelta(hours=1)).hist()","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:06:03.592302Z","iopub.execute_input":"2022-11-08T06:06:03.593103Z","iopub.status.idle":"2022-11-08T06:06:03.780333Z","shell.execute_reply.started":"2022-11-08T06:06:03.593070Z","shell.execute_reply":"2022-11-08T06:06:03.777918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.histogram(train_data, x='time_to_eruption', nbins=10, width=800, height=600, title='Time to eruption distribution')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:06:03.782591Z","iopub.execute_input":"2022-11-08T06:06:03.783184Z","iopub.status.idle":"2022-11-08T06:06:04.997324Z","shell.execute_reply.started":"2022-11-08T06:06:03.783109Z","shell.execute_reply":"2022-11-08T06:06:04.996394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.line(train_data, y='time_to_eruption', width=800, height=500, title='Time to eruption for all volcanos')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:06:04.998678Z","iopub.execute_input":"2022-11-08T06:06:04.999092Z","iopub.status.idle":"2022-11-08T06:06:05.092841Z","shell.execute_reply.started":"2022-11-08T06:06:04.999056Z","shell.execute_reply":"2022-11-08T06:06:05.091325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train 셋 분석","metadata":{}},{"cell_type":"code","source":"sensors = set()\nobservations = set()\nnan_columns = list()\nmissed_groups = list()\nfor_train_df = list()\n\nfor item in train_folder:\n    name = int(item.split('.')[-2].split('/')[-1])\n    \n    at_least_one_missed = 0\n    frag = pd.read_csv(item)\n    missed_group = list()\n    missed_percents = list()\n    \n    for col in frag.columns:\n        missed_percents.append(frag[col].isnull().sum() / len(frag))\n        if frag[col].isnull().all() == True:\n            at_least_one_missed = 1\n            nan_columns.append(col)\n            missed_group.append(col)\n            \n        if len(missed_group) > 0:\n            missed_groups.append(missed_group)\n            \n    sensors.add(len(frag.columns))\n    observations.add(len(frag))\n    for_train_df.append([name, at_least_one_missed] + missed_percents)\n\nprint('고유한(Unique) 센서 수 집합 (frag별 column 개수로 구분): {value}'.format(value=sensors))\nprint('고유한(Unique) 행(row) 수 집합 (frag별 행 수 구분): {value}'.format(value=observations))\nprint('값이 비어있는(missed) 센서(column) 수: {value}'.format(value=len(nan_columns)))","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:06:05.094675Z","iopub.execute_input":"2022-11-08T06:06:05.095157Z","iopub.status.idle":"2022-11-08T06:15:04.956449Z","shell.execute_reply.started":"2022-11-08T06:06:05.095115Z","shell.execute_reply":"2022-11-08T06:15:04.954056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for_train_df[:10]","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:15:51.603370Z","iopub.execute_input":"2022-11-08T06:15:51.603997Z","iopub.status.idle":"2022-11-08T06:15:51.614719Z","shell.execute_reply.started":"2022-11-08T06:15:51.603935Z","shell.execute_reply":"2022-11-08T06:15:51.613550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"absent_sensors = dict()\nfor item in nan_columns:\n    if item in absent_sensors:\n        absent_sensors[item] += 1\n    else:\n        absent_sensors[item] = 0","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:16:01.864710Z","iopub.execute_input":"2022-11-08T06:16:01.865106Z","iopub.status.idle":"2022-11-08T06:16:01.871248Z","shell.execute_reply.started":"2022-11-08T06:16:01.865074Z","shell.execute_reply":"2022-11-08T06:16:01.870339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"absent_df = pd.DataFrame(absent_sensors.items(), columns=['Sensor', 'Missed sensors(columns)'])\n\nfig = px.bar(absent_df, x='Sensor', y='Missed sensors(columns)', width=800, height=500, title='Number of missed sensors(columns) in training dataset')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-08T02:16:49.030788Z","iopub.execute_input":"2022-11-08T02:16:49.031891Z","iopub.status.idle":"2022-11-08T02:16:49.135463Z","shell.execute_reply.started":"2022-11-08T02:16:49.031839Z","shell.execute_reply":"2022-11-08T02:16:49.134152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test 셋 분석","metadata":{}},{"cell_type":"code","source":"sensors = set()\nobservations = set()\nnan_columns = list()\nmissed_groups = list()\nfor_test_df = list()\n\nfor item in test_folder:\n    name = int(item.split('.')[-2].split('/')[-1])\n    \n    at_least_one_missed = 0\n    frag = pd.read_csv(item)\n    missed_group = list()\n    missed_percents = list()\n    \n    for col in frag.columns:\n        missed_percents.append(frag[col].isnull().sum() / len(frag))\n        if frag[col].isnull().all() == True:\n            at_least_one_missed = 1\n            nan_columns.append(col)\n            missed_group.append(col)\n            \n        if len(missed_group) > 0:\n            missed_groups.append(missed_group)\n            \n    sensors.add(len(frag.columns))\n    observations.add(len(frag))\n    for_test_df.append([name, at_least_one_missed] + missed_percents)\n\nprint('고유한(Unique) 센서 수 집합 (frag별 column 개수로 구분): {value}'.format(value=sensors))\nprint('고유한(Unique) 행(row) 수 집합 (frag별 행 수 구분): {value}'.format(value=observations))\nprint('값이 비어있는(missed) 센서(column) 수: {value}'.format(value=len(nan_columns)))","metadata":{"execution":{"iopub.status.busy":"2022-11-08T02:16:49.139133Z","iopub.execute_input":"2022-11-08T02:16:49.139524Z","iopub.status.idle":"2022-11-08T02:24:50.902328Z","shell.execute_reply.started":"2022-11-08T02:16:49.139490Z","shell.execute_reply":"2022-11-08T02:24:50.901070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for_test_df[:10]","metadata":{"execution":{"iopub.status.busy":"2022-11-08T02:24:50.904073Z","iopub.execute_input":"2022-11-08T02:24:50.904527Z","iopub.status.idle":"2022-11-08T02:24:51.186129Z","shell.execute_reply.started":"2022-11-08T02:24:50.904484Z","shell.execute_reply":"2022-11-08T02:24:51.184961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"absent_sensors = dict()\nfor item in nan_columns:\n    if item in absent_sensors:\n        absent_sensors[item] += 1\n    else:\n        absent_sensors[item] = 0","metadata":{"execution":{"iopub.status.busy":"2022-11-08T02:24:51.187861Z","iopub.execute_input":"2022-11-08T02:24:51.188638Z","iopub.status.idle":"2022-11-08T02:24:51.196610Z","shell.execute_reply.started":"2022-11-08T02:24:51.188591Z","shell.execute_reply":"2022-11-08T02:24:51.195789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"absent_df = pd.DataFrame(absent_sensors.items(), columns=['Sensor', 'Missed sensors(columns)'])\n\nfig = px.bar(absent_df, x='Sensor', y='Missed sensors(columns)', width=800, height=500, title='Number of missed sensors(columns) in test dataset')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-08T02:24:51.197923Z","iopub.execute_input":"2022-11-08T02:24:51.198297Z","iopub.status.idle":"2022-11-08T02:24:51.268061Z","shell.execute_reply.started":"2022-11-08T02:24:51.198265Z","shell.execute_reply":"2022-11-08T02:24:51.267054Z"},"trusted":true},"execution_count":null,"outputs":[]}]}