{"cells":[{"metadata":{},"cell_type":"markdown","source":"#  REF:https://www.kaggle.com/jiweiliu/dask-with-simple-xgb"},{"metadata":{"trusted":true},"cell_type":"code","source":"from glob import glob\nfrom collections import Counter\nimport os\nimport sys\n\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nfrom sklearn.preprocessing import StandardScaler,OneHotEncoder\nfrom sklearn.model_selection import KFold\nimport xgboost as xgb","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_building_floor(fname):\n    xx = fname.split('/')\n    return xx[-3],xx[-2]\n\ndef get_test_building(name):\n    with open(name) as f:\n        for c,line in enumerate(f):\n            if c==1:\n                x = line.split()[1].split(':')[1]\n                return x  \n\ndef get_floor_target(floor):\n    floor = floor.lower()\n    if floor in ['bf','bm']:\n        return None\n    elif floor == 'b':\n        return -1\n    if floor.startswith('f'):\n        return int(floor[1])\n    elif floor.endswith('f'):\n        return int(floor[0])\n    elif floor.startswith('b'):\n        return -int(floor[1])\n    elif floor.endswith('b'):\n        return -int(floor[0])\n    else:\n        return None\n        \nACOLS = ['timestamp','x','y','z']\n        \nFIELDS = {\n    'acce': ACOLS,\n    'acce_uncali': ACOLS,\n    'gyro': ACOLS,\n    'gyro_uncali': ACOLS,\n    'magn': ACOLS,\n    'magn_uncali': ACOLS,\n    'ahrs': ACOLS,\n    'wifi': ['timestamp','ssid','bssid','rssi','last_timestamp'],\n    'ibeacon': ['timestamp','code','rssi'],\n    'waypoint': ['timestamp','x','y']\n}\n\nNFEAS = {\n    'acce': 3,\n    'acce_uncali': 3,\n    'gyro': 3,\n    'gyro_uncali': 3,\n    'magn': 3,\n    'magn_uncali': 3,\n    'ahrs': 3,\n    'wifi': 1,\n    'ibeacon': 1,\n    'waypoint': 3\n}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"PATH = '../input/indoor-location-navigation'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def mpe(yp, y):\n    e1 = (yp[:,0] - y[:,0])**2 + (yp[:,1] - y[:,1])**2\n    e2 = 15*np.abs(yp[:,2] - y[:,2])\n    return np.mean(e1**0.5 + e2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"db = pd.read_csv('../input/extracteddata/feature.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cols = ['F' + str(i) for i in range(244)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X = np.asarray(db[cols])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.read_csv('../input/extracteddata/target.csv')\ndf = df[df.columns[1:]]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_files = glob(f'{PATH}/test/*.txt')\nlen(test_files)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\ntest_b = []\nfor name in tqdm(test_files):\n    test_b.append(get_test_building(name))\ntest_b = np.array(test_b)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"db = pd.read_csv('../input/extracteddata/feature_test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\nXt = np.asarray(db[cols])\nXt.shape","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Train XGB"},{"metadata":{"trusted":true},"cell_type":"code","source":"params2 = {\n        'booster' : 'gbtree',\n        'objective': 'reg:linear',\n        'eval_metric': 'mae',\n        'eta':0.1,\n        'depth':7,\n        'nthread':2,\n        'verbosity': 0,\n    }","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"params = {\n        'booster' : 'gblinear',\n        'objective': 'reg:squarederror',\n        'eval_metric': 'rmse',\n        ''\n        'eta':0.1,\n        'depth':7,\n        'nthread':2,\n        'verbosity': 0,\n    }","execution_count":null,"outputs":[]},{"metadata":{"jupyter":{"outputs_hidden":true},"trusted":true},"cell_type":"code","source":"N = 5\ndtest = xgb.DMatrix(data=Xt)\nysub = np.zeros([Xt.shape[0],3])\n\nkf = KFold(n_splits=N,shuffle=True,random_state=42)\n\nmsgs = []\nfor i,(train_index, test_index) in enumerate(kf.split(X)):\n    X_train, X_test = X[train_index], X[test_index]\n    yps = np.zeros([X_test.shape[0],3])\n    yrs = yps.copy()\n    for c,col in enumerate(['w_x','w_y','floors']):\n        y = df[col].values\n        y_train, y_test = y[train_index], y[test_index]\n        print(y_train.shape)            \n        dtrain = xgb.DMatrix(data=X_train, label=y_train)\n        dvalid = xgb.DMatrix(data=X_test, label=y_test)\n        watchlist = [(dtrain, 'train'), (dvalid, 'eval')] \n        if c != 2:\n            clf = xgb.train(params, dtrain=dtrain,\n                        num_boost_round=70,evals=watchlist,\n                        early_stopping_rounds=10,\n                       verbose_eval=100)\n        else:\n            clf = xgb.train(params2, dtrain=dtrain,\n                        num_boost_round=70,evals=watchlist,\n                        early_stopping_rounds=10,\n                       verbose_eval=100)            \n            \n        yp = clf.predict(dvalid)\n        yps[:,c] = yp\n        yrs[:,c] = y_test\n        ysub[:,c] += clf.predict(dtest)\n    msg = f'Fold {i}: MPE {mpe(yps, yrs):.4f}'\n    print(msg)\n    msgs.append(msg)\nysub = ysub/N","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"msgs","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub = pd.read_csv(f'{PATH}/sample_submission.csv')\nsub.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub['site'] = sub['site_path_timestamp'].apply(lambda x: x.split('_')[0])\nsub.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_map = {i:j for i,j in zip(test_b, test_files)}\nsub['filename'] = sub['site'].apply(lambda x: test_map[x])\nsub.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ds = pd.DataFrame(ysub,columns=['x','y','floor'])\nds.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ds['filename'] = test_files\nds.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub = sub.drop(['x','y','floor'],axis=1).merge(ds,on='filename',how='left')\nprint(sub.shape)\nsub.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i in sub.columns:\n    print(i,sub[i].isnull().sum())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub['floor'] = sub['floor'].astype('int')\nsub.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub.drop(['site','filename'],axis=1).to_csv('submission.csv',index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}