{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport plotly.express as px\nfrom plotly.subplots import make_subplots\nimport plotly.graph_objects as go\nimport glob\nfrom tqdm import tqdm\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Investigating train.csv"},{"metadata":{},"cell_type":"markdown","source":"`train.csv` is containing target value for sensors data. for example this `[1136037770 ,12262005]` tells us given 10 mins of 10 sensors data in `1136037770.csv`, 12262005 time (in some unit) remain to eruption."},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/predict-volcanic-eruptions-ingv-oe/train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('info: \\n', train.info())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('info: \\n', train.info())\nprint('-+-'*30)\nprint('Statistics: \\n',train['time_to_eruption'].describe(),\n      '\\nskewness:', train['time_to_eruption'].skew(),\n      '\\nkurtosis: ', train['time_to_eruption'].kurtosis(),\n      '\\nIQR:', train['time_to_eruption'].quantile(0.75) - train['time_to_eruption'].quantile(.25),\n      '\\nrange: ', train['time_to_eruption'].max() - train['time_to_eruption'].min())\nprint('-+-'*30)\nprint('train.head:\\n',train.head())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"There is no missing value (of course!).\n\nIdeas to investigate:\n1. sort `time_to_eruption` values and see the relation with volano activity and remaining time.\n\n\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"#Let's look at the histogram of the target value\npx.histogram(train,\n             x='time_to_eruption',\n             nbins=200)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"It seems, `time_to_eruption` uniformly distributed (roughly)."},{"metadata":{"trusted":true},"cell_type":"code","source":"px.line(train, \n        x=train.index, \n        y='time_to_eruption',\n        log_y=True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Let's sort values by `time_to_eruption`"},{"metadata":{"trusted":true},"cell_type":"code","source":"sorted_df = train.sort_values(by='time_to_eruption', ascending=False)\nsorted_df.reset_index(inplace=True)\n# sorted_df.drop('index', axis='columns', inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"px.line(sorted_df, \n        x=sorted_df.index,\n        y=(sorted_df['time_to_eruption']))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"px.line(sorted_df, \n        x=sorted_df.index,\n        y=(sorted_df['time_to_eruption']),\n        log_y=True\n       )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sorted_df['step'] = sorted_df['time_to_eruption'].shift(-1) - sorted_df['time_to_eruption']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"px.line(sorted_df, x=sorted_df.index,\n        y=sorted_df['step'])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Investigating Sensors Data"},{"metadata":{},"cell_type":"markdown","source":"Let's See what a single file look's like."},{"metadata":{"trusted":true},"cell_type":"code","source":"sensor_path = '/kaggle/input/predict-volcanic-eruptions-ingv-oe/train'\n\ndef read_sensor_data(path=sensor_path, fname='1000015382.csv', dtype='Int16'):\n    df = pd.read_csv(path+'/'+fname, dtype='Int16')\n    return df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sensor_df = read_sensor_data()\nsensor_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sensor_df.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sensor_df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_sensor_data(df, size=(1000, 1000), fixed_range=[-5000, 5000]):\n    fig = make_subplots(rows=10, \n                        cols=1,\n                        shared_xaxes=True,\n                        vertical_spacing=0.03,\n                        subplot_titles = df.columns.to_list())\n    \n    for i in range(df.shape[1]):\n        fig.add_trace(go.Scatter(x=df.index,\n                                 y=df.iloc[:, i].fillna(0),\n                                 mode='lines',\n                                 name=df.columns[i]),\n                        row=i+1,\n                        col=1)\n        fig.update_yaxes(range=fixed_range)\n        \n    fig.layout.update(\n        {'width':size[0],\n         'height':size[1],\n         'showlegend':False})\n    return fig\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plot_sensor_data(sensor_df, fixed_range=None)\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_sensors_hist(df, shape=[5, 2]):\n\n    fig = make_subplots(rows=shape[0], \n                        cols=shape[1],\n                        vertical_spacing=0.09,\n                        specs=[[{\"secondary_y\": True}]*shape[1] for i in range(shape[0])],\n                        subplot_titles = df.columns.to_list())\n    \n    for i in range(df.shape[1]):\n        row = int(i % shape[0] + 1)\n        col = int(i // shape[0] + 1)\n\n        fig.add_trace(go.Histogram(x=df.iloc[:, i].fillna(0),\n                                   name=df.columns[i]),\n                      row=row,\n                      col=col)\n        fig.add_trace(go.Histogram(x=df.iloc[:, i].fillna(0),\n                                   name=df.columns[i],\n                                   cumulative={'enabled':True},\n                                   histnorm='probability',\n                                   opacity=0.5),\n\n                      row=row,\n                      col=col,\n                      secondary_y=True)\n\n    fig.layout.update(\n        {'width': shape[1]*280,\n         'height':shape[0]*280,\n         'showlegend':False})\n    return fig","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plot_sensors_hist(sensor_df, shape=[2,5])\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Preparing Data"},{"metadata":{},"cell_type":"markdown","source":"If an entire sensor is `Null` we will drop it."},{"metadata":{"trusted":true},"cell_type":"code","source":"# 15G training data!\n!du -h -d 1 /kaggle/input/predict-volcanic-eruptions-ingv-oe/train/","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sensors_files = glob.glob(f\"{sensor_path}/*\")\nprint(len(sensors_files))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nmeta_df = pd.DataFrame(np.zeros((10,2)), columns=['full_null', 'partial_null'], \n             index=[f'sensor_{i+1}' for i in range(10)])\nfor f in tqdm(sensors_files):\n    df = pd.read_csv(f, dtype='Int16')\n    \n    partial_null = df.columns[df.isnull().any()].to_list()\n    full_null = df.columns[df.isnull().all()].to_list()\n    \n    if partial_null:\n        meta_df.loc[partial_null,'partial_null'] +=1\n    \n    if full_null:\n        meta_df.loc[full_null,'full_null'] +=1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# let's save this data and to not go with the process again\n# meta_df.to_csv('/kaggle/working/null_data.csv')\nmeta_df = pd.read_csv('/kaggle/working/null_data.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"meta_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = go.Figure(data=[go.Bar(x=meta_df.index, \n                             y=meta_df['full_null'], \n                             name='full_null'),\n                      go.Bar(x=meta_df.index, \n                             y=meta_df['partial_null'],\n                             name='partial_null')\n])\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":" meta_df.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}