{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport glob\nimport math\n\nimport plotly.express as px\nfrom plotly.subplots import make_subplots\nimport plotly.graph_objs as go\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error as mse\nfrom sklearn.feature_selection import RFE\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.tree import DecisionTreeRegressor\nimport os\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom pylab import rcParams\nimport optuna\nfrom optuna.samplers import TPESampler\nfrom tqdm import tqdm\nimport tensorflow as tf\nimport tensorflow.keras.backend as K\nfrom tensorflow.keras.layers import Input, Dense\nfrom sklearn.model_selection import KFold\nfrom tensorflow.keras.callbacks import EarlyStopping","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train = pd.read_csv(\"../input/predict-volcanic-eruptions-ingv-oe/train.csv\")\nsample_submission = pd.read_csv(\"../input/predict-volcanic-eruptions-ingv-oe/sample_submission.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.histogram(\n    train, \n    x=\"time_to_eruption\",\n    width=800,\n    height=500,\n    nbins=100,\n    title='Time to eruption distribution'\n)\n\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.distplot(train['time_to_eruption'], bins=40)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['time_to_eruption'].describe()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"For each train[\"segment_id\"] we have a csv that contains information of the sensors logs fro 10 minutes"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_files = glob.glob(\"../input/predict-volcanic-eruptions-ingv-oe/train/*\")\nlen(train_files),train.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"Lets randomly pick a segment_id to see its sensors"},{"metadata":{"trusted":true},"cell_type":"code","source":"df_sensor = pd.read_csv('../input/predict-volcanic-eruptions-ingv-oe/train/2005284845.csv')\ndf_sensor.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_sensor.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_sensor.info()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"As we can see, the sensor_4 is the one having more Null-values,latter on we will use different startegies in order to fill those Null values"},{"metadata":{"trusted":true},"cell_type":"code","source":"rcParams['figure.figsize'] = 12, 20\ndf_sensor.plot(kind = 'line', subplots = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_sensor_2 = pd.read_csv(\"../input/predict-volcanic-eruptions-ingv-oe/test/1000213997.csv\")\ndf_sensor_2.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_sensor_2.info()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"As we can see in this test sensor, one of the columns is missing completely, this issue can be seen in many others factors, and will be a key components if we want to predict with a model correctly"},{"metadata":{"trusted":true},"cell_type":"code","source":"rcParams['figure.figsize'] = 12, 20\ndf_sensor_2.plot(kind = 'line', subplots = True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"By plotting two we can see that some peaks in the different sensors coincide"},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_stats(files_folder, segs_df):\n    files_dir = data_dir + files_folder\n    \n    df = segs_df.copy()\n    df.set_index('segment_id', inplace = True)\n\n    for file in tqdm(os.listdir(files_dir)):\n        file_df = pd.read_csv(files_dir + file, dtype=\"Int16\")\n\n        for sensor in file_df.columns:\n            df.loc[int(file[:-4]), '%s_max'%(sensor)] = file_df.loc[:, sensor].max()\n            df.loc[int(file[:-4]), '%s_min'%(sensor)] = file_df.loc[:, sensor].min()\n            df.loc[int(file[:-4]), '%s_std'%(sensor)] = file_df.loc[:, sensor].std()\n            df.loc[int(file[:-4]), '%s_skew'%(sensor)] = file_df.loc[:, sensor].skew()\n            df.loc[int(file[:-4]), '%s_kurtosis'%(sensor)] = file_df.loc[:, sensor].kurtosis()\n\n            df.loc[int(file[:-4]), '%s_nan'%(sensor)] = file_df.loc[:, sensor].isna().sum()\n\n            df.loc[int(file[:-4]), '%s_99q'%(sensor)] = file_df.loc[:, sensor].quantile(.99)\n            df.loc[int(file[:-4]), '%s_01q'%(sensor)] = file_df.loc[:, sensor].quantile(.01)\n\n            df.loc[int(file[:-4]), '%s_amplitude'%(sensor)] = df.loc[int(file[:-4]), '%s_max'%(sensor)] -\\\n                                                              df.loc[int(file[:-4]), '%s_min'%(sensor)]\n\n            df.loc[int(file[:-4]), '%s_neg_peaks'%(sensor)] = (file_df.loc[:, sensor] <=\\\n                                                               df.loc[int(file[:-4]), '%s_01q'%(sensor)]).sum()\n            df.loc[int(file[:-4]), '%s_pos_peaks'%(sensor)] = (file_df.loc[:, sensor] >=\\\n                                                               df.loc[int(file[:-4]), '%s_99q'%(sensor)]).sum()\n\n    df.reset_index(inplace = True)\n\n    return df\n\n                       ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_dir = '../input/predict-volcanic-eruptions-ingv-oe/'\nstats_train_df = get_stats('train/', train)\n\ntest_df = pd.DataFrame(columns=['segment_id'])\ntest_df['segment_id'] = [int(file_name[:-4]) for file_name in os.listdir(data_dir + 'test/')]\n\nstats_test_df = get_stats('test/', test_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"files_dir = data_dir + 'train/'\nfiles_dir","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tqdm()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}