{"cells":[{"metadata":{},"cell_type":"markdown","source":"I'm using the results obtained from a blend: \n* https://www.kaggle.com/domizianostingi/volcanic-blend"},{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install ProgressBar","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport glob \nfrom progressbar import ProgressBar\nfrom sklearn.decomposition import PCA as pca\nimport os ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"\ndef prepare(name,features):\n    train_df = pd.read_csv('../input/predict-volcanic-eruptions-ingv-oe/train.csv')\n    index=[]\n    df=[]  # 4431, 10 * number of featuers\n    frag = glob.glob(\"../input/predict-volcanic-eruptions-ingv-oe/{}/*\".format(name)) # len = 4431\n    \n    pbar = ProgressBar()\n    for i in pbar(frag[0:4]):\n        for j in features:\n            df = np.append(df,pd.read_csv(i).agg(j))\n    \n    df = pd.DataFrame(df.reshape(len(frag[0:4]),len(features)*10))  \n    \n    col_name=[]\n    num = [0,1,2,3,4,5,6,7,8,9]\n    for j in features:\n        for i in num:\n            col_name=np.append(col_name,'{}_{}'.format(j,i))\n            \n    df.columns=col_name \n    \n    for i in range(0,len(frag[0:4])):\n        index = np.append(index,os.path.splitext(frag[i].split('{}/'.format(name))[1])[0])\n        \n    df['segment_id']=index\n    df['segment_id']=df['segment_id'].astype(int)\n    \n    if name == 'train': \n        df = pd.merge(df, train_df, on =['segment_id'],how='left') \n        \n    return(df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"features = (('mean'),('std'),('min'),('max'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\ntrain = prepare('train',features)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\ntest = prepare('test',features)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def fill_na(data):\n    for i in data.columns:\n        data[i] = data[i].fillna(np.mean(train[i]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fill_na(train)\nfill_na(test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pseudo = pd.read_csv('../input/volcanic-blend/submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pseudo_test = pd.merge(test,pseudo, on =['segment_id'], how='left')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pseudo_train = pd.concat((train,pseudo_test),axis=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y = pseudo_train['time_to_eruption']\nx = pseudo_train.iloc[:,0:-2]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.metrics import mean_absolute_error as mae\nfrom  sklearn.tree import DecisionTreeClassifier\nfrom  sklearn.model_selection import train_test_split\nimport xgboost as xgb\n\nmodel = xgb.XGBRegressor(n_estimators=500000,max_depth=8,learning_rate=0.05,alpha=0.1,SUBSAMPLE=0.6)#,tree_method='gpu_hist'\nXt, Xv, Yt, Yv = train_test_split(x, y, test_size =0.2, shuffle=False)\neval_set = [(Xv,Yv)]\nmodel.fit(Xt, Yt,early_stopping_rounds=25,eval_metric='mae', eval_set=eval_set, verbose=False)\nprediction = model.predict(test.iloc[:,0:-1])\nprint('mean absolute error: ', mae(prediction,pseudo['time_to_eruption']))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_df = pd.DataFrame(test['segment_id'])\nsub_df = pd.concat([sub_df,pd.Series(prediction)],axis=1)\nsample_submission=pd.read_csv('../input/predict-volcanic-eruptions-ingv-oe/sample_submission.csv')\n        \nsample_submission = pd.merge(sample_submission,sub_df, on =['segment_id'])\nsample_submission = sample_submission.drop(columns=['time_to_eruption'])\nsample_submission.columns = ['segment_id', 'time_to_eruption']\nsample_submission.to_csv('sample_submission.csv', header=True, index=False)\nprint('saved')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}