{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport glob\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm\nimport gc\n\n!ls ../input/indoor-location-navigation","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"floor_map = {\n             \"F1\":0, \"F2\": 1, \"F3\":2, \"F4\":3, \"F5\":4, \"F6\":5, \"F7\":6,\"F8\":7, \"F9\":8, \"F10\":9,\n             \"1F\":0, \"2F\": 1, \"3F\":2, \"4F\":3, \"5F\":4, \"6F\":5, \"7F\":6,\"8F\":7, \"9F\":8, \"10F\":9,\n            \n            \"B\":0, \"B1\":-1, \"B2\":-2, \"B3\":-3, \"B4\":-4, \"B5\":-5, \"B6\":-6, \"B7\":-7, \"B8\":-8, \"B9\":-9,\n                   \"1B\":-1, \"2B\":-2, \"3B\":-3, \"4B\":-4, \"5B\":-5, \"6B\":-6, \"7B\":-7, \"8B\":-8, \"9B\":-9,\n    \n            \"G\":0, \"L2\": 1, \"L3\":2, \"L4\":3, \"L5\":4, \"L6\":5, \"L7\":6,\"L8\":7, \"L9\":8, \"L10\":9,\n            \n            \"L1\":0, \"LG1\": 1, \"LG2\":2, \"LM\":3, \"M\":4, \"P1\":0, \"P2\":1,\n            }\n\nfloor_map","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"trainfiles = glob.glob('../input/indoor-location-navigation/train/*/*/*')\ntestfiles = glob.glob('../input/indoor-location-navigation/test/*')\nlen(trainfiles), len(testfiles), trainfiles[0], testfiles[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"floor, floor_count = np.unique( [ f.split('/')[-2] for f in trainfiles ], return_counts=True )\nfloor, floor_count","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!head ../input/indoor-location-navigation/train/5cd56c0ce2acfd2d33b6ab27/B1/5d09a625bd54340008acddb9.txt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_dict( filename ):\n    with open(filename) as f:\n        lines = f.readlines()\n\n    proc = {}\n    SiteID = ''\n    count = 0\n    for l in lines:\n        l = l.replace('\\n','')\n        \n        if l[0]!='#':\n            val = l.split('\\t')\n            \n            ts = val[0]\n            if ts not in proc:\n                proc[ts] = {}\n                proc[ts]['wp0'] = np.nan\n                proc[ts]['wp1'] = np.nan\n                proc[ts]['wifi0'] = np.nan\n                proc[ts]['wifi1'] = np.nan\n                proc[ts]['ts'] = ts\n                proc[ts]['SiteID'] = SiteID\n\n            if val[1] == 'TYPE_WAYPOINT':\n                proc[ts]['wp0']= float(val[2])\n                proc[ts]['wp1']= float(val[3])\n\n            elif val[1] == 'TYPE_WIFI':\n                proc[ts]['wifi0']= val[2]\n                proc[ts]['wifi1']= val[2]\n\n            count+=1\n                \n                \n        else:\n            val = l.split('\\t')\n            #print(val)\n            if len(val)>=2:\n                if val[1][:7] == 'SiteID:':\n                    SiteID = val[1].split(':')[-1]\n\n    return proc","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!ls ../input/gibatewifidataset1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# TRAIN = []\n# for fn in tqdm(trainfiles):\n#     path = load_dict( fn )\n#     #print( len(path), fn )\n\n#     dt = pd.DataFrame( {\n#         'SiteID':[path[f]['SiteID'] for f in path],\n#         'ts':[path[f]['ts'] for f in path],\n#         'wifi0':[path[f]['wifi0'] for f in path],\n#         'wifi1':[path[f]['wifi1'] for f in path],\n#         'wp0':[path[f]['wp0'] for f in path],\n#         'wp1':[path[f]['wp1'] for f in path],\n#     } )\n\n#     dt['floor'] = fn.split('/')[-2]\n\n#     dt['wifi0'] = dt['wifi0'].fillna(method='ffill')\n#     dt['wifi1'] = dt['wifi1'].fillna(method='ffill')\n#     dt['wifi0'] = dt['wifi0'].fillna(method='bfill')\n#     dt['wifi1'] = dt['wifi1'].fillna(method='bfill')\n\n#     dt['wp0'] = dt['wp0'].interpolate()\n#     dt['wp1'] = dt['wp1'].interpolate()\n    \n#     grp = dt.groupby( ['wifi0','wifi1','floor'] )['wp0','wp1'].agg('mean').reset_index()\n#     grp['SiteID'] = dt['SiteID'].values[0]\n    \n#     TRAIN.append(grp)\n# gc.collect()\n\n# TRAIN = pd.concat( TRAIN, sort=False )\n# TRAIN.to_csv('train_files.csv', index=False)\n# TRAIN.shape\n\nTRAIN = pd.read_csv( '../input/gibatewifidataset1/train_files.csv' )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"TRAIN.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!head -n 20 ../input/indoor-location-navigation/test/52ad8c760ff9978d0949deed.txt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!ls ../input/indoor-location-navigation","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub = pd.read_csv( '../input/indoor-location-navigation/sample_submission.csv' )\nprint( sub.shape )\nsub['SiteID'] = sub['site_path_timestamp'].apply( lambda x: x.split('_')[0] )\nsub['path'] = sub['site_path_timestamp'].apply( lambda x: x.split('_')[1] )\nsub['ts'] = sub['site_path_timestamp'].apply( lambda x: x.split('_')[2] )\n#sub.iloc[1,0]\n\nprint( sub.nunique() )\n\nsub.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(testfiles), testfiles[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"TEST = []\nfor fn in tqdm(testfiles):\n    path = load_dict( fn )\n    #print( len(path), fn )\n\n    dt = pd.DataFrame( {\n        'SiteID':[path[f]['SiteID'] for f in path],\n        'ts':[path[f]['ts'] for f in path],\n        'wifi0':[path[f]['wifi0'] for f in path],\n        'wifi1':[path[f]['wifi1'] for f in path],\n        'wp0':[path[f]['wp0'] for f in path],\n        'wp1':[path[f]['wp1'] for f in path],\n    } )\n\n    dt['floor'] = fn.split('/')[-2]\n\n    dt['wifi0'] = dt['wifi0'].fillna(method='ffill')\n    dt['wifi1'] = dt['wifi1'].fillna(method='ffill')\n    dt['wifi0'] = dt['wifi0'].fillna(method='bfill')\n    dt['wifi1'] = dt['wifi1'].fillna(method='bfill')\n\n    dt['wp0'] = dt['wp0'].interpolate()\n    dt['wp1'] = dt['wp1'].interpolate()\n    \n    dt['path'] = fn.split('/')[-1].split('.')[0]\n    \n    TEST.append(dt)\ngc.collect()\n\nTEST = pd.concat( TEST, sort=False )\nTEST.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"TEST['site_path_timestamp'] = TEST['SiteID'] + '_' + TEST['path'] + '_' + TEST['ts']\nTEST['istest'] = 0\nTEST.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub['SiteID'] = sub['site_path_timestamp'].apply(lambda x: x.split('_')[0] )\nsub['path'] = sub['site_path_timestamp'].apply(lambda x: x.split('_')[1] )\nsub['ts'] = sub['site_path_timestamp'].apply(lambda x: x.split('_')[2] )\nsub['istest'] = 1\nsub.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(TEST.shape)\nTESTall = pd.concat( (TEST,sub), sort=False )\ngc.collect()\nprint(TESTall.shape)\nTESTall.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"TESTall = TESTall.sort_values( ['site_path_timestamp'] )\nTESTall['wifi0'] = TESTall['wifi0'].fillna(method='bfill')\nTESTall['wifi0'] = TESTall['wifi0'].fillna(method='ffill')\nTESTall['wifi1'] = TESTall['wifi1'].fillna(method='bfill')\nTESTall['wifi1'] = TESTall['wifi1'].fillna(method='ffill')\nTESTall.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"TRAIN.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"TRAIN.nunique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"TRAIN['fold'] = TRAIN['SiteID'].apply( lambda x: hash(x)%10 )\nTRAIN['fold'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"TRAIN['lvl'] = TRAIN['floor'].map(floor_map)\nTRAIN.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = TRAIN.loc[ TRAIN.fold!=0 ].copy()\nvalid = TRAIN.loc[ TRAIN.fold==0 ].copy()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dt0 = train.groupby(['wifi0'])['wp0'].agg('mean').reset_index()\ndt1 = train.groupby(['wifi0'])['wp1'].agg('mean').reset_index()\ndt2 = train.groupby(['wifi0'])['lvl'].agg('mean').reset_index()\ndt0, dt1, dt2","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"valid['yx'] = valid[['wifi0']].merge( dt0, on=['wifi0'], how='left' )['wp0'].values\nvalid['yy'] = valid[['wifi0']].merge( dt1, on=['wifi0'], how='left' )['wp1'].values\nvalid['ylvl'] = valid[['wifi0']].merge( dt2, on=['wifi0'], how='left' )['lvl'].values\n\nvalid['yx'].isnull().mean(), valid['yy'].isnull().mean()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# the metric used in this competition\ndef comp_metric(xhat, yhat, fhat, x, y, f):\n    intermediate = np.sqrt(np.power(xhat - x,2) + np.power(yhat-y,2)) + 15 * np.abs(fhat-f)\n    return intermediate.sum()/xhat.shape[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"valid = valid.loc[ valid.yx.notnull() ].copy()\nprint( valid.shape )\nvalid.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"comp_metric( valid.yx, valid.yy, valid.ylvl, valid.wp0, valid.wp1, valid.lvl  )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dt0 = TRAIN.groupby(['wifi0'])['wp0'].agg('mean').reset_index()\ndt1 = TRAIN.groupby(['wifi0'])['wp1'].agg('mean').reset_index()\ndt2 = TRAIN.groupby(['wifi0'])['lvl'].agg('mean').reset_index()\n\nTESTall['x'] = TESTall[['wifi0']].merge( dt0, on=['wifi0'], how='left' )['wp0'].values\nTESTall['y'] = TESTall[['wifi0']].merge( dt1, on=['wifi0'], how='left' )['wp1'].values\nTESTall['floor'] = TESTall[['wifi0']].merge( dt2, on=['wifi0'], how='left' )['lvl'].values\nTESTall.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"TESTall['x'] = TESTall['x'].interpolate()\nTESTall['x'] = TESTall['x'].fillna( method='bfill' )\nTESTall['x'] = TESTall['x'].fillna( method='ffill' )\n\nTESTall['y'] = TESTall['y'].interpolate()\nTESTall['y'] = TESTall['y'].fillna( method='bfill' )\nTESTall['y'] = TESTall['y'].fillna( method='ffill' )\n\nTESTall['floor'] = TESTall['floor'].interpolate()\nTESTall['floor'] = TESTall['floor'].fillna( method='bfill' )\nTESTall['floor'] = TESTall['floor'].fillna( method='ffill' )\nTESTall['floor'] = TESTall['floor'].round().astype(np.int32)\n\nTESTall.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub = TESTall.loc[ TESTall.istest==1, ['site_path_timestamp','x','y','floor'] ]\nsub.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub.to_csv( 'submission.csv', index=False )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}