{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<img src=\"https://i.imgur.com/oiG2jZl.png\">\n<center><h1>🧭Indoor Location and Navigation🧭</h1></center>\n\n# 1. Introduction\n> 📌 **Goal**: Predicting the indoor position of smartphones 📱 based on a *real-time* sensor 🎯.\n\n> We'll also learn how to use the **GitHub Repository** available through this competition and call the custom functions **without** copy-pasting them into the notebook.\n\n### Libraries📚","metadata":{}},{"cell_type":"code","source":"# CPU libraries\nimport os\nimport json\nfrom glob import glob\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport plotly.graph_objs as go\n\nfrom PIL import Image, ImageOps\nfrom skimage import io\nfrom skimage.color import rgba2rgb, rgb2xyz\nfrom tqdm import tqdm\nfrom dataclasses import dataclass\nfrom math import floor, ceil","metadata":{"execution":{"iopub.status.busy":"2023-05-06T13:53:45.308733Z","iopub.execute_input":"2023-05-06T13:53:45.309271Z","iopub.status.idle":"2023-05-06T13:53:46.786186Z","shell.execute_reply.started":"2023-05-06T13:53:45.309152Z","shell.execute_reply":"2023-05-06T13:53:46.785122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How 1 path looks\nbase = '../input/indoor-location-navigation'\npath = f'{base}/train/5a0546857ecc773753327266/B1/5e15730aa280850006f3d005.txt'\n\nwith open(path) as p:\n    lines = p.readlines()\n\nprint(\"No. Lines in 1 example: {:,}\". format(len(lines)), \"\\n\" +\n      \"Example (5 lines): \", lines[0:5])","metadata":{"execution":{"iopub.status.busy":"2023-05-06T13:42:35.256118Z","iopub.execute_input":"2023-05-06T13:42:35.256481Z","iopub.status.idle":"2023-05-06T13:42:35.297704Z","shell.execute_reply.started":"2023-05-06T13:42:35.256448Z","shell.execute_reply":"2023-05-06T13:42:35.296677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp -r ../input/indoor-locationnavigation-2021/indoor-location-competition-20-master/indoor-location-competition-20-master/* ./","metadata":{"execution":{"iopub.status.busy":"2023-05-06T13:42:37.727017Z","iopub.execute_input":"2023-05-06T13:42:37.727546Z","iopub.status.idle":"2023-05-06T13:43:07.146355Z","shell.execute_reply.started":"2023-05-06T13:42:37.727509Z","shell.execute_reply":"2023-05-06T13:43:07.144991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# wifi feature extract","metadata":{}},{"cell_type":"code","source":"# pull out all the buildings actually used in the test set, given current method we don't need the other ones\nssubm = pd.read_csv('../input/indoor-location-navigation/sample_submission.csv')\n\n# only 24 of the total buildings are used in the test set, \n# this allows us to greatly reduce the intial size of the dataset\n\nssubm_df = ssubm[\"site_path_timestamp\"].apply(lambda x: pd.Series(x.split(\"_\")))\nused_buildings = sorted(ssubm_df[0].value_counts().index.tolist())\n\n# dictionary used to map the floor codes to the values used in the submission file. \nfloor_map = {\"B2\":-2, \"B1\":-1, \"F1\":0, \"F2\": 1, \"F3\":2, \"F4\":3, \"F5\":4, \"F6\":5, \"F7\":6,\"F8\":7, \"F9\":8,\n             \"1F\":0, \"2F\":1, \"3F\":2, \"4F\":3, \"5F\":4, \"6F\":5, \"7F\":6, \"8F\": 7, \"9F\":8}","metadata":{"execution":{"iopub.status.busy":"2023-05-05T06:55:31.555655Z","iopub.execute_input":"2023-05-05T06:55:31.556406Z","iopub.status.idle":"2023-05-05T06:55:34.08151Z","shell.execute_reply.started":"2023-05-05T06:55:31.556364Z","shell.execute_reply":"2023-05-05T06:55:34.080327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_path = '../input/indoor-location-navigation/'\nbssid = dict()\nfrom tqdm import tqdm\nfor building in tqdm(used_buildings):\n    folders = sorted(glob(os.path.join(base_path,'train/'+building+'/*')))\n    print(building)\n    wifi = list()\n    for i, folder in enumerate(folders):\n        for file in glob(folder + '/*'):\n            with open(file) as p:\n                lines = p.readlines()\n            idx = pd.Series(lines).apply(lambda x : x.strip().split('\\t')[1] == 'TYPE_WIFI')\n            # bssid 만 뽑기\n            wifi.extend(pd.Series(lines)[idx].apply(lambda x : x.strip().split('\\t')[3]).values)\n\n    v_c = pd.Series(wifi).value_counts() \n#     top_bssid = v_c[v_c > 3000].index.tolist()\n    top_bssid = v_c.index.to_list()[:10]\n\n    print(len(top_bssid))\n    bssid[building] = top_bssid","metadata":{"execution":{"iopub.status.busy":"2023-05-05T06:55:36.38265Z","iopub.execute_input":"2023-05-05T06:55:36.383075Z","iopub.status.idle":"2023-05-05T07:07:12.513232Z","shell.execute_reply.started":"2023-05-05T06:55:36.383035Z","shell.execute_reply":"2023-05-05T07:07:12.51148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bssid = pd.Series(wifi).value_counts().index.values\ncount = pd.Series(wifi).value_counts().values\nwifi_c = pd.DataFrame({\"bssid\":bssid, \n              \"count\":count})\nwifi_c.index = wifi_c[\"bssid\"]","metadata":{"execution":{"iopub.status.busy":"2023-05-01T04:47:57.539895Z","iopub.execute_input":"2023-05-01T04:47:57.540488Z","iopub.status.idle":"2023-05-01T04:47:57.546923Z","shell.execute_reply.started":"2023-05-01T04:47:57.540453Z","shell.execute_reply":"2023-05-01T04:47:57.545787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(\"bssid_1000.json\", \"w\") as f:\n    json.dump(bssid, f)\n\nwith open(\"bssid_1000.json\") as f:\n    bssid = json.load(f)","metadata":{"execution":{"iopub.status.busy":"2023-04-30T07:44:41.050299Z","iopub.execute_input":"2023-04-30T07:44:41.05076Z","iopub.status.idle":"2023-04-30T07:44:41.061949Z","shell.execute_reply.started":"2023-04-30T07:44:41.050717Z","shell.execute_reply":"2023-04-30T07:44:41.060743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 전처리 코드 wifi\nbuilding_dfs = dict()\n\nfor building in used_buildings[:1]:\n    folders = sorted(glob(os.path.join(base_path,'train', building +'/*')))\n    dfs = list()\n    index = sorted(bssid[building])\n    \n    for folder in folders:\n        floor = floor_map[folder.split('/')[-1]]\n        files = glob(os.path.join(folder, \"*.txt\"))\n        \n        for file in tqdm(files):\n            wifi = list()\n            waypoint = list()\n            with open(file) as f:\n                txt = f.readlines()\n\n            idx = pd.Series(txt).apply(lambda x : x.strip().split()[1] == 'TYPE_WAYPOINT')\n            waypoint.extend(pd.Series(txt)[idx].apply(lambda x : np.array(x.strip().split())[[0,2,3]]).values)\n            \n            idx = pd.Series(txt).apply(lambda x : x.strip().split()[1] == 'TYPE_WIFI')\n            wifi.extend(pd.Series(txt)[idx].apply(lambda x : np.array(x.strip().split())[[0,2,3,4,5,6]]))\n\n            df = pd.DataFrame(np.array(wifi))  \n\n            for gid, g in df.groupby(0):\n                near_idx = np.argmin(abs(np.vstack(waypoint)[:, 0].astype(int) - int(gid)))\n                g = g.drop_duplicates(subset=2)\n                tmp = g.iloc[:,2:4]\n                feat = tmp.set_index(2).reindex(index).replace(np.nan, -999).T\n                feat[\"x\"] = float(waypoint[near_idx][1])\n                feat[\"y\"] = float(waypoint[near_idx][2])\n                feat[\"f\"] = floor\n                feat['time'] = waypoint[near_idx][0]\n                feat[\"path\"] = file.split('/')[-1].split('.')[0] \n                dfs.append(feat)\n\n    building_df = pd.concat(dfs)\n    building_dfs[building] = df\n    building_df.to_csv(building+\"_1000_train.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-05-05T02:39:07.629418Z","iopub.execute_input":"2023-05-05T02:39:07.629843Z","iopub.status.idle":"2023-05-05T02:41:16.779922Z","shell.execute_reply.started":"2023-05-05T02:39:07.629788Z","shell.execute_reply":"2023-05-05T02:41:16.778923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"building_df","metadata":{"execution":{"iopub.status.busy":"2023-05-05T02:49:08.902486Z","iopub.execute_input":"2023-05-05T02:49:08.902956Z","iopub.status.idle":"2023-05-05T02:49:08.939519Z","shell.execute_reply.started":"2023-05-05T02:49:08.902912Z","shell.execute_reply":"2023-05-05T02:49:08.938563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 만들어진 데이터 확인\n\nbase_dir = \"../input/indoor-navigation-and-location-wifi-features/wifi_features\"\ntrain_dir = \"/train/*_train.csv\"\ntest_dir = \"/test/*_test.csv\"\n\n# Paths for train & test files\ntrain_paths = sorted(glob(base_dir + train_dir))\ntest_paths = sorted(glob(base_dir + test_dir))\nsample_subm = pd.read_csv('../input/indoor-location-navigation/sample_submission.csv',\n                          index_col=0)\n\nprint(\"Len Train Files: {}\".format(len(train_paths)), \"\\n\" +\n      \"Len Test Files: {}\".format(len(test_paths)))","metadata":{"execution":{"iopub.status.busy":"2023-05-05T02:10:16.137367Z","iopub.execute_input":"2023-05-05T02:10:16.137935Z","iopub.status.idle":"2023-05-05T02:10:16.153666Z","shell.execute_reply.started":"2023-05-05T02:10:16.137899Z","shell.execute_reply":"2023-05-05T02:10:16.152249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(train_paths[0], index_col=0)\ntest_df = pd.read_csv(test_paths[0], index_col=0).iloc[:, :-1]","metadata":{"execution":{"iopub.status.busy":"2023-05-05T02:10:16.154717Z","iopub.status.idle":"2023-05-05T02:10:16.155251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# still run plot","metadata":{}},{"cell_type":"code","source":"# How 1 path looks\nbase = '../input/indoor-location-navigation'\npath = f'{base}/train/5cd56b5ae2acfd2d33b58549/5F/5d06134c4a19c000086c4324.txt'\nwith open(path) as p:\n    lines = p.readlines()\n\nprint(\"No. Lines in 1 example: {:,}\". format(len(lines)), \"\\n\" +\n      \"Example (5 lines): \", lines[0:5])","metadata":{"execution":{"iopub.status.busy":"2023-05-06T13:43:07.148526Z","iopub.execute_input":"2023-05-06T13:43:07.148905Z","iopub.status.idle":"2023-05-06T13:43:07.281717Z","shell.execute_reply.started":"2023-05-06T13:43:07.148868Z","shell.execute_reply":"2023-05-06T13:43:07.280857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_path = glob(f'{base}/train/5cd56b5ae2acfd2d33b58549/5F/*')","metadata":{"execution":{"iopub.status.busy":"2023-05-06T13:43:07.283263Z","iopub.execute_input":"2023-05-06T13:43:07.283888Z","iopub.status.idle":"2023-05-06T13:43:07.298152Z","shell.execute_reply.started":"2023-05-06T13:43:07.283849Z","shell.execute_reply":"2023-05-06T13:43:07.296865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"waypoint = []\nfor path in sample_path:\n    with open(path) as p:\n        lines = p.readlines()\n    idx = pd.Series(lines).apply(lambda x : x.strip().split('\\t')[1] == 'TYPE_WAYPOINT')\n    waypoint.append(pd.Series(lines)[idx].apply(lambda x : x.strip().split('\\t')).values)\n    print(len(waypoint))","metadata":{"execution":{"iopub.status.busy":"2023-05-06T10:13:25.652352Z","iopub.execute_input":"2023-05-06T10:13:25.6528Z","iopub.status.idle":"2023-05-06T10:13:26.983893Z","shell.execute_reply.started":"2023-05-06T10:13:25.652763Z","shell.execute_reply":"2023-05-06T10:13:26.982456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a = np.vstack(waypoint[0])","metadata":{"execution":{"iopub.status.busy":"2023-05-06T10:18:59.33185Z","iopub.execute_input":"2023-05-06T10:18:59.332217Z","iopub.status.idle":"2023-05-06T10:18:59.337354Z","shell.execute_reply.started":"2023-05-06T10:18:59.332187Z","shell.execute_reply":"2023-05-06T10:18:59.336019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a","metadata":{"execution":{"iopub.status.busy":"2023-05-06T10:20:11.865246Z","iopub.execute_input":"2023-05-06T10:20:11.865648Z","iopub.status.idle":"2023-05-06T10:20:11.872855Z","shell.execute_reply.started":"2023-05-06T10:20:11.865616Z","shell.execute_reply":"2023-05-06T10:20:11.871575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"previous = a[1:, -2:]\nfuture = a[:-1, -2:]","metadata":{"execution":{"iopub.status.busy":"2023-05-06T10:19:58.265419Z","iopub.execute_input":"2023-05-06T10:19:58.265846Z","iopub.status.idle":"2023-05-06T10:19:58.271808Z","shell.execute_reply.started":"2023-05-06T10:19:58.265805Z","shell.execute_reply":"2023-05-06T10:19:58.270349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"previous, future","metadata":{"execution":{"iopub.status.busy":"2023-05-06T10:20:08.390963Z","iopub.execute_input":"2023-05-06T10:20:08.39143Z","iopub.status.idle":"2023-05-06T10:20:08.399002Z","shell.execute_reply.started":"2023-05-06T10:20:08.391384Z","shell.execute_reply":"2023-05-06T10:20:08.397582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"waypoint = []\nidx = pd.Series(lines).apply(lambda x : x.strip().split('\\t')[1] == 'TYPE_WAYPOINT')\nwaypoint.extend(pd.Series(lines)[idx].apply(lambda x : x.strip().split('\\t')).values)\nlen(waypoint)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T10:02:59.416332Z","iopub.execute_input":"2023-05-06T10:02:59.41709Z","iopub.status.idle":"2023-05-06T10:02:59.540791Z","shell.execute_reply.started":"2023-05-06T10:02:59.417033Z","shell.execute_reply":"2023-05-06T10:02:59.539458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"waypoint","metadata":{"execution":{"iopub.status.busy":"2023-05-06T10:03:02.031289Z","iopub.execute_input":"2023-05-06T10:03:02.0317Z","iopub.status.idle":"2023-05-06T10:03:02.041302Z","shell.execute_reply.started":"2023-05-06T10:03:02.031651Z","shell.execute_reply":"2023-05-06T10:03:02.040142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import Libraries\n# ~~~~\n# Data\n# ~~~~\nbase_dir = \"../input/indoor-navigation-and-location-wifi-features/wifi_features\"\ntrain_dir = \"/train/*_train.csv\"\ntest_dir = \"/test/*_test.csv\"\n\n# Paths for train & test files\ntrain_paths = sorted(glob(base_dir + train_dir))\ntest_paths = sorted(glob(base_dir + test_dir))\nsample_subm = pd.read_csv('../input/indoor-location-navigation/sample_submission.csv',\n                          index_col=0)\n\nprint(\"Len Train Files: {}\".format(len(train_paths)), \"\\n\" +\n      \"Len Test Files: {}\".format(len(test_paths)))","metadata":{"execution":{"iopub.status.busy":"2023-05-05T11:04:51.513305Z","iopub.execute_input":"2023-05-05T11:04:51.513607Z","iopub.status.idle":"2023-05-05T11:04:51.600745Z","shell.execute_reply.started":"2023-05-05T11:04:51.513577Z","shell.execute_reply":"2023-05-05T11:04:51.599915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pull out all the buildings actually used in the test set, given current method we don't need the other ones\nssubm = pd.read_csv('../input/indoor-location-navigation/sample_submission.csv')\n\n# only 24 of the total buildings are used in the test set, \n# this allows us to greatly reduce the intial size of the dataset\n\nssubm_df = ssubm[\"site_path_timestamp\"].apply(lambda x: pd.Series(x.split(\"_\")))\nused_buildings = sorted(ssubm_df[0].value_counts().index.tolist())\n\n# dictionary used to map the floor codes to the values used in the submission file. \nfloor_map = {\"B2\":-2, \"B1\":-1, \"F1\":0, \"F2\": 1, \"F3\":2, \"F4\":3, \"F5\":4, \"F6\":5, \"F7\":6,\"F8\":7, \"F9\":8,\n             \"1F\":0, \"2F\":1, \"3F\":2, \"4F\":3, \"5F\":4, \"6F\":5, \"7F\":6, \"8F\": 7, \"9F\":8}","metadata":{"execution":{"iopub.status.busy":"2023-05-05T11:05:20.329705Z","iopub.execute_input":"2023-05-05T11:05:20.330383Z","iopub.status.idle":"2023-05-05T11:05:22.808676Z","shell.execute_reply.started":"2023-05-05T11:05:20.330324Z","shell.execute_reply":"2023-05-05T11:05:22.807581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(train_paths[0], index_col=0)\ntest_df = pd.read_csv(test_paths[0], index_col=0).iloc[:, :-1]","metadata":{"execution":{"iopub.status.busy":"2023-05-05T11:04:51.601813Z","iopub.execute_input":"2023-05-05T11:04:51.602269Z","iopub.status.idle":"2023-05-05T11:04:53.4336Z","shell.execute_reply.started":"2023-05-05T11:04:51.60223Z","shell.execute_reply":"2023-05-05T11:04:53.431886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# idx = pd.Series(lines).apply(lambda x : x.strip().split('\\t')[1] == 'TYPE_BEACON')\n# pd.Series(lines)[idx].apply(lambda x : x.strip().split('\\t'))","metadata":{"execution":{"iopub.status.busy":"2023-05-06T10:00:55.461439Z","iopub.execute_input":"2023-05-06T10:00:55.461815Z","iopub.status.idle":"2023-05-06T10:00:55.466092Z","shell.execute_reply.started":"2023-05-06T10:00:55.461784Z","shell.execute_reply":"2023-05-06T10:00:55.464718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_path = '../input/indoor-location-navigation/'\nbssid = dict()\nfrom tqdm import tqdm\nstill_run = []\nfor building in tqdm(used_buildings):\n    folders = sorted(glob(os.path.join(base_path,'train/'+building+'/*')))\n    print(building)\n    for i, folder in enumerate(folders):\n        waypoint = list()\n        for file in glob(folder + '/*'):\n            with open(file) as p:\n                lines = p.readlines()\n            idx = pd.Series(lines).apply(lambda x : x.strip().split('\\t')[1] == 'TYPE_WAYPOINT')\n            # bssid 만 뽑기\n            waypoint.extend(pd.Series(lines)[idx].apply(lambda x : x.strip().split('\\t')).values)\n        x_y = np.vstack(waypoint)[:, 2:4].astype(float)\n        still_run.extend((x_y[:-1] == x_y[1:]).all(axis=1))","metadata":{"execution":{"iopub.status.busy":"2023-05-05T11:05:28.640349Z","iopub.execute_input":"2023-05-05T11:05:28.640864Z","iopub.status.idle":"2023-05-05T11:06:14.119761Z","shell.execute_reply.started":"2023-05-05T11:05:28.640816Z","shell.execute_reply":"2023-05-05T11:06:14.118072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.Series(still_run).value_counts()\n\nf, axs = plt.subplots(1,figsize=(15,10))\nplt.rcParams.update({\"font.size\":25})\nobject_cnt = s_r.value_counts().sort_values(ascending=False)\n# name_list = [crop_aware_name_decoder[i] for i in object_cnt.index.tolist()]\nname_list = ['moving', 'still']\naxs.bar(name_list, object_cnt.values, color=['#d4dddd' if i%2==0 else '#F5DEB3' for i in range(len(name_list))])\nfor x,y,z in zip(range(len(name_list)), object_cnt.values, object_cnt.values/object_cnt.sum()*100):        \n    axs.annotate('%d\\n(%.2f%%)' %(int(y),z), xy=(x,y+100), textcoords='data', ha = 'center') \naxs.axis(ymin=0,ymax=int(max(object_cnt)*1.1))\naxs.set_xticklabels(name_list, rotation = 0,fontsize = 30)\n# axs.set_title(\"Class\")\nf.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T10:32:19.449976Z","iopub.execute_input":"2023-05-01T10:32:19.45049Z","iopub.status.idle":"2023-05-01T10:32:19.75242Z","shell.execute_reply.started":"2023-05-01T10:32:19.450439Z","shell.execute_reply":"2023-05-01T10:32:19.75117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"###  지하 지상 분석","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\npd.set_option(\"display.max_colwidth\", 50)\n    \nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport glob\nimport os\nimport sys\nimport math\nimport gc\nimport json \nimport re\nfrom tqdm import tqdm\n\nimport warnings # Supress warnings \nwarnings.filterwarnings('ignore')\n\nPATH = '../input/indoor-location-navigation/'\n\nsubmission = pd.read_csv(f'{PATH}sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-05-04T05:20:31.364684Z","iopub.execute_input":"2023-05-04T05:20:31.365063Z","iopub.status.idle":"2023-05-04T05:20:31.393818Z","shell.execute_reply.started":"2023-05-04T05:20:31.365025Z","shell.execute_reply":"2023-05-04T05:20:31.392752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Edited from https://www.kaggle.com/devinanzelmo/wifi-features\nsubmission['building'] = submission[\"site_path_timestamp\"].apply(lambda x: x.split(\"_\")[0])\nprint(f\"The test set convers {submission['building'].nunique()} buildings with {len(submission)} pathes to predict.\")\nused_buildings = list(submission['building'].unique())","metadata":{"execution":{"iopub.status.busy":"2023-05-04T05:20:31.395876Z","iopub.execute_input":"2023-05-04T05:20:31.396294Z","iopub.status.idle":"2023-05-04T05:20:31.416247Z","shell.execute_reply.started":"2023-05-04T05:20:31.396247Z","shell.execute_reply":"2023-05-04T05:20:31.415209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"b_f = {\n    \"B\":{\n        \"wifi\":[],\n        \"ibeacon\":[],\n        \"acc\":[],\n        \"acc_u\":[],\n        \"gro\":[],\n        \"gro_u\":[],\n        \"mc\":[],\n        \"mc_u\":[],\n        \"ro\":[],\n    },\n    \"F\":{\n        \"wifi\":[],\n        \"ibeacon\":[],\n        \"acc\":[],\n        \"acc_u\":[],\n        \"gro\":[],\n        \"gro_u\":[],\n        \"mc\":[],\n        \"mc_u\":[],\n        \"ro\":[],\n    }\n}","metadata":{"execution":{"iopub.status.busy":"2023-05-04T05:20:31.41782Z","iopub.execute_input":"2023-05-04T05:20:31.418474Z","iopub.status.idle":"2023-05-04T05:20:31.425651Z","shell.execute_reply.started":"2023-05-04T05:20:31.41843Z","shell.execute_reply":"2023-05-04T05:20:31.424555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.choice(used_buildings, 5, replace=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-04T05:21:41.619306Z","iopub.execute_input":"2023-05-04T05:21:41.619901Z","iopub.status.idle":"2023-05-04T05:21:41.626061Z","shell.execute_reply.started":"2023-05-04T05:21:41.619863Z","shell.execute_reply":"2023-05-04T05:21:41.624972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.DataFrame()\nbuilding_dict = {}\nfor building in np.random.choice(used_buildings, 5, replace=False):\n    for floor in os.listdir(f\"{PATH}/train/{building}\"):\n        for path in os.listdir(f\"{PATH}/train/{building}/{floor}\"):\n            file = f\"{PATH}/train/{building}/{floor}/{path}\"\n                                                                                                                                                                                        \n            with open(file) as p:\n                lines = p.readlines()\n            \n            if floor in [\"B1\", \"B2\"]:\n                # wifi\n                idx = pd.Series(lines).apply(lambda x : x.strip().split('\\t')[1] == 'TYPE_WIFI')\n                if idx.sum(): \n                    b_f['B']['wifi'].extend(pd.Series(lines)[idx].apply(lambda x : np.array(x.strip().split('\\t'))[4]).values)\n                # ibeacon\n\n                idx = pd.Series(lines).apply(lambda x : x.strip().split('\\t')[1] == 'TYPE_BEACON')\n                if idx.sum():    \n                    b_f['B']['ibeacon'].extend(np.vstack(pd.Series(lines)[idx].apply(lambda x : np.array(x.strip().split('\\t'))[[6, 7]]).values))\n\n                # acc\n                idx = pd.Series(lines).apply(lambda x : x.strip().split('\\t')[1] == 'TYPE_ACCELEROMETER')\n                if idx.sum():\n                    b_f['B']['acc'].extend(np.vstack(pd.Series(lines)[idx].apply(lambda x : np.array(x.strip().split('\\t'))[[2, 3, 4]]).values))\n                idx = pd.Series(lines).apply(lambda x : x.strip().split('\\t')[1] == 'TYPE_ACCELEROMETER_UNCALIBRATED')\n                if idx.sum():\n                    b_f['B']['acc_u'].extend(np.vstack(pd.Series(lines)[idx].apply(lambda x : np.array(x.strip().split('\\t'))[[2, 3, 4]]).values))\n\n                # gro\n                idx = pd.Series(lines).apply(lambda x : x.strip().split('\\t')[1] == 'TYPE_GYROSCOPE')\n                if idx.sum():\n                    b_f['B']['gro'].extend(np.vstack(pd.Series(lines)[idx].apply(lambda x : np.array(x.strip().split('\\t'))[[2, 3, 4]]).values))\n                idx = pd.Series(lines).apply(lambda x : x.strip().split('\\t')[1] == 'TYPE_GYROSCOPE_UNCALIBRATED')\n                if idx.sum():\n                    b_f['B']['gro_u'].extend(np.vstack(pd.Series(lines)[idx].apply(lambda x : np.array(x.strip().split('\\t'))[[2, 3, 4]]).values))\n\n                # mc\n                idx = pd.Series(lines).apply(lambda x : x.strip().split('\\t')[1] == 'TYPE_MAGNETIC_FIELD')\n                if idx.sum():\n                    b_f['B']['mc'].extend(np.vstack(pd.Series(lines)[idx].apply(lambda x : np.array(x.strip().split('\\t'))[[2, 3, 4]]).values))\n                idx = pd.Series(lines).apply(lambda x : x.strip().split('\\t')[1] == 'TTYPE_MAGNETIC_FIELD_UNCALIBRATED')\n                if idx.sum(): \n                    b_f['B']['mc_u'].extend(np.vstack(pd.Series(lines)[idx].apply(lambda x : np.array(x.strip().split('\\t'))[[2, 3, 4]]).values))\n\n                # ro\n                idx = pd.Series(lines).apply(lambda x : x.strip().split('\\t')[1] == 'TYPE_ROTATION_VECTOR')\n                if idx.sum():\n                    b_f['B']['ro'].extend(np.vstack(pd.Series(lines)[idx].apply(lambda x : np.array(x.strip().split('\\t'))[[2, 3, 4]]).values))\n            else:\n                # wifi\n                idx = pd.Series(lines).apply(lambda x : x.strip().split('\\t')[1] == 'TYPE_WIFI')\n                if idx.sum(): \n                    b_f['F']['wifi'].extend(pd.Series(lines)[idx].apply(lambda x : np.array(x.strip().split('\\t'))[4]).values)\n                \n                # ibeacon\n                idx = pd.Series(lines).apply(lambda x : x.strip().split('\\t')[1] == 'TYPE_BEACON')\n                if idx.sum():    \n                    b_f['F']['ibeacon'].extend(np.vstack(pd.Series(lines)[idx].apply(lambda x : np.array(x.strip().split('\\t'))[[6, 7]]).values))\n                \n                # acc\n                idx = pd.Series(lines).apply(lambda x : x.strip().split('\\t')[1] == 'TYPE_ACCELEROMETER')\n                if idx.sum():\n                    b_f['F']['acc'].extend(np.vstack(pd.Series(lines)[idx].apply(lambda x : np.array(x.strip().split('\\t'))[[2, 3, 4]]).values))\n                idx = pd.Series(lines).apply(lambda x : x.strip().split('\\t')[1] == 'TYPE_ACCELEROMETER_UNCALIBRATED')\n                if idx.sum():\n                    b_f['F']['acc_u'].extend(np.vstack(pd.Series(lines)[idx].apply(lambda x : np.array(x.strip().split('\\t'))[[2, 3, 4]]).values))\n\n                # gro\n                idx = pd.Series(lines).apply(lambda x : x.strip().split('\\t')[1] == 'TYPE_GYROSCOPE')\n                if idx.sum():\n                    b_f['F']['gro'].extend(np.vstack(pd.Series(lines)[idx].apply(lambda x : np.array(x.strip().split('\\t'))[[2, 3, 4]]).values))\n                idx = pd.Series(lines).apply(lambda x : x.strip().split('\\t')[1] == 'TYPE_GYROSCOPE_UNCALIBRATED')\n                if idx.sum():\n                    b_f['F']['gro_u'].extend(np.vstack(pd.Series(lines)[idx].apply(lambda x : np.array(x.strip().split('\\t'))[[2, 3, 4]]).values))\n\n                # mc\n                idx = pd.Series(lines).apply(lambda x : x.strip().split('\\t')[1] == 'TYPE_MAGNETIC_FIELD')\n                if idx.sum():\n                    b_f['F']['mc'].extend(np.vstack(pd.Series(lines)[idx].apply(lambda x : np.array(x.strip().split('\\t'))[[2, 3, 4]]).values))\n                idx = pd.Series(lines).apply(lambda x : x.strip().split('\\t')[1] == 'TTYPE_MAGNETIC_FIELD_UNCALIBRATED')\n                if idx.sum(): \n                    b_f['F']['mc_u'].extend(np.vstack(pd.Series(lines)[idx].apply(lambda x : np.array(x.strip().split('\\t'))[[2, 3, 4]]).values))\n\n                # ro\n                idx = pd.Series(lines).apply(lambda x : x.strip().split('\\t')[1] == 'TYPE_ROTATION_VECTOR')\n                if idx.sum():\n                    b_f['F']['ro'].extend(np.vstack(pd.Series(lines)[idx].apply(lambda x : np.array(x.strip().split('\\t'))[[2, 3, 4]]).values))\n","metadata":{"execution":{"iopub.status.busy":"2023-05-04T05:21:45.42323Z","iopub.execute_input":"2023-05-04T05:21:45.4236Z","iopub.status.idle":"2023-05-04T05:34:17.948082Z","shell.execute_reply.started":"2023-05-04T05:21:45.423565Z","shell.execute_reply":"2023-05-04T05:34:17.946102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# .style.background_gradient(cmap='Blues', vmin=0, vmax=0.5).format(\"{:.0f}\").set_caption('')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# B/F 층별 패턴 분석\n# wifi, ibeacon, acc, gro, mc, ro","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"B_data = b_f['B']\nF_data = b_f['F']","metadata":{"execution":{"iopub.status.busy":"2023-05-04T05:34:54.999056Z","iopub.execute_input":"2023-05-04T05:34:54.999733Z","iopub.status.idle":"2023-05-04T05:34:55.003964Z","shell.execute_reply.started":"2023-05-04T05:34:54.999695Z","shell.execute_reply":"2023-05-04T05:34:55.002982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wifi_BF = pd.DataFrame({\"B\":pd.Series(B_data['wifi'], dtype=np.int),\n             \"F\":pd.Series(F_data['wifi'], dtype=np.int)})\nwifi_BF.to_csv(\"wifi.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-05-04T05:38:59.954656Z","iopub.execute_input":"2023-05-04T05:38:59.955161Z","iopub.status.idle":"2023-05-04T05:39:10.879773Z","shell.execute_reply.started":"2023-05-04T05:38:59.955118Z","shell.execute_reply":"2023-05-04T05:39:10.878665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BF_floor = [\"B\"] * len(B_data['wifi']) + [\"F\"] * len(F_data['wifi'])\nsignal = pd.concat((pd.Series(B_data['wifi'], dtype=np.int),\n                    pd.Series(F_data['wifi'], dtype=np.int)), axis=0)\nwifi_BF = pd.DataFrame({\"Floor\":BF_floor,\n             \"Signal\":signal})","metadata":{"execution":{"iopub.status.busy":"2023-05-04T05:42:51.478114Z","iopub.execute_input":"2023-05-04T05:42:51.478491Z","iopub.status.idle":"2023-05-04T05:42:51.726193Z","shell.execute_reply.started":"2023-05-04T05:42:51.478456Z","shell.execute_reply":"2023-05-04T05:42:51.724964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wifi_BF.groupby(\"Floor\").mean().style.background_gradient(cmap='Blues', vmin=-50, vmax=0).format(\"{:.0f}\").set_caption('')","metadata":{"execution":{"iopub.status.busy":"2023-05-04T05:49:52.279244Z","iopub.execute_input":"2023-05-04T05:49:52.279645Z","iopub.status.idle":"2023-05-04T05:49:53.372792Z","shell.execute_reply.started":"2023-05-04T05:49:52.279608Z","shell.execute_reply":"2023-05-04T05:49:53.372053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# wifi signal\nsns.boxplot(x='Floor', y='Signal', data=wifi_BF)","metadata":{"execution":{"iopub.status.busy":"2023-05-04T05:45:21.794354Z","iopub.execute_input":"2023-05-04T05:45:21.794728Z","iopub.status.idle":"2023-05-04T05:45:27.82652Z","shell.execute_reply.started":"2023-05-04T05:45:21.794698Z","shell.execute_reply":"2023-05-04T05:45:27.825365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ibeacon","metadata":{"execution":{"iopub.status.busy":"2023-05-04T05:57:21.050002Z","iopub.execute_input":"2023-05-04T05:57:21.050406Z","iopub.status.idle":"2023-05-04T05:57:21.054801Z","shell.execute_reply.started":"2023-05-04T05:57:21.050373Z","shell.execute_reply":"2023-05-04T05:57:21.053631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BF_floor = [\"B\"] * len(B_data['ibeacon']) + [\"F\"] * len(F_data['ibeacon'])\nsignal = np.concatenate((np.vstack(B_data['ibeacon'])[:, 0].astype(np.float),\n               np.vstack(F_data['ibeacon'])[:, 0].astype(np.float)))\ndistance = np.concatenate((np.vstack(B_data['ibeacon'])[:, 1].astype(np.float),\n               np.vstack(F_data['ibeacon'])[:, 1].astype(np.float)))\nibeacon_BF = pd.DataFrame({\"Floor\":BF_floor,\n             \"Signal\":signal,\n                          \"Distance\":distance})\nibeacon_BF.to_csv(\"ibeacon.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-05-04T05:55:14.554067Z","iopub.execute_input":"2023-05-04T05:55:14.554417Z","iopub.status.idle":"2023-05-04T05:55:16.390688Z","shell.execute_reply.started":"2023-05-04T05:55:14.554387Z","shell.execute_reply":"2023-05-04T05:55:16.389866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### site별 층 존재","metadata":{}},{"cell_type":"code","source":"from glob import glob\nall_floors = glob(\"../input/indoor-location-navigation/metadata/*/*\")","metadata":{"execution":{"iopub.status.busy":"2023-05-02T09:08:49.707563Z","iopub.execute_input":"2023-05-02T09:08:49.707972Z","iopub.status.idle":"2023-05-02T09:08:51.36983Z","shell.execute_reply.started":"2023-05-02T09:08:49.707937Z","shell.execute_reply":"2023-05-02T09:08:51.368888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.DataFrame()\nbuilding_dict = {}\n\nall_floors = glob(\"../input/indoor-location-navigation/metadata/*/*\")\nfloor_no = []\n\n# Extract only the floor number\nfor floor in all_floors:\n    no = floor.split(\"/\")[5]\n    sity = floor.split(\"/\")[4]\n    building_dict['building'] = sity\n    building_dict['floor'] = no\n    train_df = train_df.append(building_dict, ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-02T09:08:51.372051Z","iopub.execute_input":"2023-05-02T09:08:51.372379Z","iopub.status.idle":"2023-05-02T09:08:53.870632Z","shell.execute_reply.started":"2023-05-02T09:08:51.372347Z","shell.execute_reply":"2023-05-02T09:08:53.869739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s_f = []\nfor sity in train_df['building'].unique():\n    st = []\n    for f in train_df.floor.unique():\n        if f in train_df[train_df['building'] == sity].floor.values:\n            st.append(1)\n        else:\n            st.append(0)\n    s_f.append(st)","metadata":{"execution":{"iopub.status.busy":"2023-04-30T15:50:22.165753Z","iopub.execute_input":"2023-04-30T15:50:22.166236Z","iopub.status.idle":"2023-04-30T15:50:23.76725Z","shell.execute_reply.started":"2023-04-30T15:50:22.166205Z","shell.execute_reply":"2023-04-30T15:50:23.766325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.floor = train_df.floor.apply(lambda x : convert[x] if x in list(convert.keys()) else x)","metadata":{"execution":{"iopub.status.busy":"2023-04-30T15:50:09.490127Z","iopub.execute_input":"2023-04-30T15:50:09.490597Z","iopub.status.idle":"2023-04-30T15:50:09.498268Z","shell.execute_reply.started":"2023-04-30T15:50:09.490558Z","shell.execute_reply":"2023-04-30T15:50:09.497103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"convert = {'1F' : 'F1', '2F' : 'F2', '3F' : 'F3', \n '4F' : 'F4', '5F' : 'F5', \n '6F' : 'F6', '7F' : 'F7',\n '8F' : 'F8', '9F' : 'F9',\n 'B'  : 'B1', 'B1' : 'B1',\n 'B2' : 'B2', 'B3' : 'B3', \n 'BF' : 'B1', 'BM' : 'B1', \n 'L1' : 'F1', 'L2' : 'F2', \n 'L3' : 'F3', 'L4' : 'F4',\n 'L5' : 'F5', 'L6' : 'F6',\n 'L7' : 'F7', 'L8' : 'F8',\n 'L9' : 'F9', 'L10': 'F10',\n 'L11': 'F11','G' : 'F1',\n 'LG1': 'F1', 'LG2': 'F2', \n 'LM' : 'F1', 'M'  : 'F1',\n 'P1' : 'F1', 'P2' : 'F2'}","metadata":{"execution":{"iopub.status.busy":"2023-04-30T15:47:43.984899Z","iopub.execute_input":"2023-04-30T15:47:43.985288Z","iopub.status.idle":"2023-04-30T15:47:43.991907Z","shell.execute_reply.started":"2023-04-30T15:47:43.985256Z","shell.execute_reply":"2023-04-30T15:47:43.99091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s_f_df = pd.DataFrame(s_f, columns=train_df.floor.unique(), index=[f\"bulid {i+1}\" for i in range(204)])","metadata":{"execution":{"iopub.status.busy":"2023-04-30T15:52:13.802401Z","iopub.execute_input":"2023-04-30T15:52:13.802778Z","iopub.status.idle":"2023-04-30T15:52:13.811549Z","shell.execute_reply.started":"2023-04-30T15:52:13.802743Z","shell.execute_reply":"2023-04-30T15:52:13.810225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s_c = [\"B3\", \"B2\", \"B1\", \"F1\", \"F2\", \"F3\", \"F4\", \"F5\", \"F6\", \"F7\", \"F8\", \"F9\", \"F10\", \"F11\"]","metadata":{"execution":{"iopub.status.busy":"2023-04-30T15:54:53.100994Z","iopub.execute_input":"2023-04-30T15:54:53.101391Z","iopub.status.idle":"2023-04-30T15:54:53.106216Z","shell.execute_reply.started":"2023-04-30T15:54:53.101351Z","shell.execute_reply":"2023-04-30T15:54:53.105007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s_f_df.loc[:, s_c].T[::-1].style.background_gradient(cmap='Blues', vmin=0, vmax=0.5).format(\"{:.0f}\").set_caption('Number Training Pathes per Floor for Buildings in Test Set')","metadata":{"execution":{"iopub.status.busy":"2023-04-30T15:55:29.060745Z","iopub.execute_input":"2023-04-30T15:55:29.061099Z","iopub.status.idle":"2023-04-30T15:55:29.648427Z","shell.execute_reply.started":"2023-04-30T15:55:29.06107Z","shell.execute_reply":"2023-04-30T15:55:29.647283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 빌딩-층별 path count\ntrain_df = pd.DataFrame()\nbuilding_dict = {}\nfor building in build_name:\n    building_dict['building'] = building\n    for floor in os.listdir(f\"{PATH}/train/{building}\"):\n        building_dict['floor'] = floor\n#         building_dict['floor_enc'] = encode_floor(floor)\n        for path in os.listdir(f\"{PATH}/train/{building}/{floor}\"):\n            building_dict['path'] = path\n            train_df = train_df.append(building_dict, ignore_index=True)\n\n# display(train_df.head().style.set_caption('Training Data'))\n# train_df.to_csv(\"train_df.csv\", index=False)\n\n# # Visualize\n# building_overview_df = train_df.groupby(['building', 'floor_enc']).path.nunique().to_frame().reset_index(drop=False)\n# building_overview_df = building_overview_df.pivot(index='building', columns='floor_enc')['path'].fillna(0)\n# building_overview_df.style.background_gradient(cmap='Blues', vmin=0, vmax=150).format(\"{:.0f}\").set_caption('Number Training Pathes per Floor for Buildings in Test Set')","metadata":{"execution":{"iopub.status.busy":"2023-04-30T15:11:24.781343Z","iopub.execute_input":"2023-04-30T15:11:24.781706Z","iopub.status.idle":"2023-04-30T15:12:55.494707Z","shell.execute_reply.started":"2023-04-30T15:11:24.781675Z","shell.execute_reply":"2023-04-30T15:12:55.493769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir(f\"{PATH}/train/{building}\")","metadata":{"execution":{"iopub.status.busy":"2023-04-30T15:09:36.18376Z","iopub.execute_input":"2023-04-30T15:09:36.184139Z","iopub.status.idle":"2023-04-30T15:09:36.192203Z","shell.execute_reply.started":"2023-04-30T15:09:36.184107Z","shell.execute_reply":"2023-04-30T15:09:36.191241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"site2f = (building_overview_df > 0).astype(int)\nsite2f.columns = [\"B2\", \"B1\", \"F1\", \"F2\", \"F3\", \"F4\", \"F5\", \"F6\", \"F7\", \"F8\", \"F9\"]\nsite2f.index = [f\"building{i+1}\"for i in range(len(site2f.index))]\nsite2f.T[::-1].style.background_gradient(cmap='Blues', vmin=0, vmax=0.5).format(\"{:.0f}\").set_caption('Number Training Pathes per Floor for Buildings in Test Set')","metadata":{"execution":{"iopub.status.busy":"2023-04-30T11:56:52.571348Z","iopub.execute_input":"2023-04-30T11:56:52.571793Z","iopub.status.idle":"2023-04-30T11:56:52.661341Z","shell.execute_reply.started":"2023-04-30T11:56:52.571756Z","shell.execute_reply":"2023-04-30T11:56:52.660165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### ibeacon","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\npd.set_option(\"display.max_colwidth\", 50)\n    \nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport glob\nimport os\nimport sys\nimport math\nimport gc\nimport json \nimport re\nfrom tqdm import tqdm\n\nimport warnings # Supress warnings \nwarnings.filterwarnings('ignore')\n\nPATH = '../input/indoor-location-navigation/'\n\nsubmission = pd.read_csv(f'{PATH}sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-05-05T12:18:18.863381Z","iopub.execute_input":"2023-05-05T12:18:18.863823Z","iopub.status.idle":"2023-05-05T12:18:18.912982Z","shell.execute_reply.started":"2023-05-05T12:18:18.863776Z","shell.execute_reply":"2023-05-05T12:18:18.912145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assign each iBeacon distance to a waypoint position in building.\nfrom dataclasses import dataclass\n\n@dataclass\nclass ReadData:\n    # Removed other signal info acquisition for simplicity reasons ...\n    ibeacon: np.ndarray\n    waypoint: np.ndarray\n\n\ndef read_data_file(data_filename):\n    # Removed other signal info acquisition for simplicity reasons ...\n    ibeacon = []\n    waypoint = []\n\n    with open(data_filename, 'r', encoding='utf-8') as file:\n        lines = file.readlines()\n\n    for line_data in lines:\n        line_data = line_data.strip()\n        if not line_data or line_data[0] == '#':\n            continue\n\n        line_data = line_data.split('\\t')\n\n        # Removed other signal info acquisition for simplicity reasons ...\n\n        if line_data[1] == 'TYPE_BEACON':\n            ts = line_data[0]\n            uuid = line_data[2]\n            major = line_data[3]\n            minor = line_data[4]\n            rssi = line_data[6]\n            distance = line_data[7] # Newly added\n            \n            ibeacon_data = [ts, '_'.join([uuid, major, minor]), rssi, distance]\n            \n            ibeacon.append(ibeacon_data)\n            continue\n\n        if line_data[1] == 'TYPE_WAYPOINT':\n            waypoint.append([int(line_data[0]), float(line_data[2]), float(line_data[3])])\n\n    # Removed other signal info acquisition for simplicity reasons ...\n\n    ibeacon = np.array(ibeacon)\n    waypoint = np.array(waypoint)\n\n    return ReadData(ibeacon, waypoint)","metadata":{"execution":{"iopub.status.busy":"2023-05-05T12:18:18.914517Z","iopub.execute_input":"2023-05-05T12:18:18.914791Z","iopub.status.idle":"2023-05-05T12:18:19.241255Z","shell.execute_reply.started":"2023-05-05T12:18:18.914763Z","shell.execute_reply":"2023-05-05T12:18:19.240161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def encode_floor(floor):\n    \"\"\"\n    Encodes floor string to integer\n    \"\"\"\n    # Get floor number\n    encoded_floor = int(re.findall('[0-9]+', floor)[0])\n    \n    if floor.find('B') == 0:\n        # Multiply with -1 incase of B floor\n        encoded_floor = (-1)*encoded_floor\n    else: \n        # Subtract one floor to start at 0 for F1\n        encoded_floor = encoded_floor -1\n        \n    return encoded_floor\n\n# The following code is copied and edited from https://www.kaggle.com/jiweiliu/wifi-label-encode\nACOLS = ['timestamp','x','y','z']\n        \nFIELDS = {\n    'acce': ACOLS,\n    'acce_uncali': ACOLS,\n    'gyro': ACOLS,\n    'gyro_uncali': ACOLS,\n    'magn': ACOLS,\n    'magn_uncali': ACOLS,\n    'ahrs': ACOLS,\n    'wifi': ['timestamp','ssid','bssid','rssi','last_timestamp'],\n    'ibeacon': ['timestamp','id','rssi', 'distance'],\n    'waypoint': ['timestamp','x','y']\n}\n\ndef create_dummy_df(cols):\n    \"\"\"\n    Edited from https://www.kaggle.com/jiweiliu/wifi-label-encode\n    \"\"\"\n    df = pd.DataFrame()\n    for col in cols:\n        df[col] = [0]\n        if col in ['ssid','bssid']:\n            df[col] = df[col].map(str)\n    return df\n\ndef to_frame(data, col):    \n    \"\"\"\n    Edited from https://www.kaggle.com/jiweiliu/wifi-label-encode\n    \"\"\"\n    cols = FIELDS[col]\n    if data.shape[0]>0:\n        df = pd.DataFrame(data, columns=cols)\n    else:\n        df = create_dummy_df(cols)\n    for col in df.columns:\n        if 'timestamp' in col:\n            df[col] = df[col].astype('int64')\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-05T07:16:24.729947Z","iopub.execute_input":"2023-05-05T07:16:24.7306Z","iopub.status.idle":"2023-05-05T07:16:24.820775Z","shell.execute_reply.started":"2023-05-05T07:16:24.730561Z","shell.execute_reply":"2023-05-05T07:16:24.819982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['building'] = submission[\"site_path_timestamp\"].apply(lambda x: x.split(\"_\")[0])\nprint(f\"The test set convers {submission['building'].nunique()} buildings with {len(submission)} pathes to predict.\")\nused_buildings = list(submission['building'].unique())","metadata":{"execution":{"iopub.status.busy":"2023-05-05T07:16:25.272271Z","iopub.execute_input":"2023-05-05T07:16:25.272931Z","iopub.status.idle":"2023-05-05T07:16:25.297135Z","shell.execute_reply.started":"2023-05-05T07:16:25.272866Z","shell.execute_reply":"2023-05-05T07:16:25.29605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.DataFrame()\nbuilding_dict = {}\nfor building in used_buildings:\n    building_dict['building'] = building\n    for floor in os.listdir(f\"{PATH}/train/{building}\"):\n        building_dict['floor'] = floor\n        building_dict['floor_enc'] = encode_floor(floor)\n        for path in os.listdir(f\"{PATH}/train/{building}/{floor}\"):\n            building_dict['path'] = path\n            train_df = train_df.append(building_dict, ignore_index=True)\n\ndisplay(train_df.head().style.set_caption('Training Data'))\ntrain_df.to_csv(\"train_df.csv\", index=False)\n\n# Visualize\nbuilding_overview_df = train_df.groupby(['building', 'floor_enc']).path.nunique().to_frame().reset_index(drop=False)\nbuilding_overview_df = building_overview_df.pivot(index='building', columns='floor_enc')['path'].fillna(0)\nbuilding_overview_df.style.background_gradient(cmap='Blues', vmin=0, vmax=150).format(\"{:.0f}\").set_caption('Number Training Pathes per Floor for Buildings in Test Set')","metadata":{"execution":{"iopub.status.busy":"2023-05-05T07:16:27.309546Z","iopub.execute_input":"2023-05-05T07:16:27.309956Z","iopub.status.idle":"2023-05-05T07:17:14.939575Z","shell.execute_reply.started":"2023-05-05T07:16:27.309913Z","shell.execute_reply":"2023-05-05T07:17:14.938221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_building = '5a0546857ecc773753327266'\nsample_floor = 'B1'\nsample_path = f'{PATH}train/{sample_building}/{sample_floor}/5e15731fa280850006f3d013.txt'\nprint(f\"Sample path: {sample_path}\")\n\npath = read_data_file(sample_path)\n\nibeacon_df = to_frame(path.ibeacon,'ibeacon')\ndisplay(ibeacon_df.head().style.set_caption('DataFrame Head of iBecon Data'))\ndisplay(ibeacon_df.id.value_counts())\n\nwaypoint_df = to_frame(path.waypoint,'waypoint')\ndisplay(waypoint_df.head().style.set_caption('DataFrame Head of Waypoint Data'))","metadata":{"execution":{"iopub.status.busy":"2023-05-04T10:14:32.29279Z","iopub.execute_input":"2023-05-04T10:14:32.293137Z","iopub.status.idle":"2023-05-04T10:14:32.364473Z","shell.execute_reply.started":"2023-05-04T10:14:32.293103Z","shell.execute_reply":"2023-05-04T10:14:32.363434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get iBeacon timestamps\nibeacon_df_grouped = ibeacon_df.groupby('timestamp')['id'].unique().to_frame().reset_index(drop=False)\ndisplay(ibeacon_df_grouped.head().style.set_caption(\"iBeacon Timestamps\"))\n\n# Combine with waypoint data\nwaypoint_ble_combined = pd.concat([ibeacon_df_grouped, waypoint_df])\nwaypoint_ble_combined = waypoint_ble_combined.sort_values(by='timestamp').reset_index(drop=True)\n\ndisplay(waypoint_ble_combined.head().style.set_caption('Combined Dataframe of Waypoint and iBeacon Timestamps Sorted by Time'))\n\nfig, ax = plt.subplots(nrows=2, ncols=1, figsize=(14, 6))\nsns.lineplot(x=waypoint_ble_combined.timestamp, y=waypoint_ble_combined.x, marker='o', ax=ax[0], label = 'before interpolation', color='orange')\nsns.lineplot(x=waypoint_ble_combined.timestamp, y=waypoint_ble_combined.y, marker='o', ax=ax[1], label = 'before interpolation', color='orange')\n\n# Interpolate waypoint\nwaypoint_ble_combined.x = waypoint_ble_combined.x.interpolate()\nwaypoint_ble_combined.y = waypoint_ble_combined.y.interpolate()\n\nsns.lineplot(x=waypoint_ble_combined.timestamp, y=waypoint_ble_combined.x, marker='o', ax=ax[0], label = 'after interpolation', color='blue')\nsns.lineplot(x=waypoint_ble_combined.timestamp, y=waypoint_ble_combined.y, marker='o', ax=ax[1], label = 'afte interpolationr', color='blue')\n\nplt.show()\n\ndisplay(waypoint_ble_combined.head().style.set_caption(\"Lookup Table with Interpolated Waypoint Data for iBeacon Timestamps\"))","metadata":{"execution":{"iopub.status.busy":"2023-05-04T10:14:32.365977Z","iopub.execute_input":"2023-05-04T10:14:32.366317Z","iopub.status.idle":"2023-05-04T10:14:33.217812Z","shell.execute_reply.started":"2023-05-04T10:14:32.366281Z","shell.execute_reply":"2023-05-04T10:14:33.216752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ble_df_with_waypoint = ibeacon_df.merge(waypoint_ble_combined[['timestamp', 'x', 'y']], on='timestamp', how='left')\ndisplay(ble_df_with_waypoint.head().style.set_caption(\"iBeacon Dataframe with added waypoint data\"))","metadata":{"execution":{"iopub.status.busy":"2023-05-04T10:14:33.219659Z","iopub.execute_input":"2023-05-04T10:14:33.220076Z","iopub.status.idle":"2023-05-04T10:14:33.244431Z","shell.execute_reply.started":"2023-05-04T10:14:33.220029Z","shell.execute_reply":"2023-05-04T10:14:33.243348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ble_list = pd.DataFrame()\ndf = train_df[train_df.building == sample_building]\n\nfiles = df[df.floor == sample_floor].path.unique()\n\nfor file in (files):\n\n    path = read_data_file(f\"{PATH}train/{sample_building}/{sample_floor}/{file}\")\n    \n    # Get path's wifi and waypoint data\n    ibeacon_df = to_frame(path.ibeacon,'ibeacon')\n    waypoint_df = to_frame(path.waypoint,'waypoint')\n\n    # Combine wifi and waypoint dataframe to one dataframe\n    ibeacon_df_grouped = ibeacon_df.groupby('timestamp')['id'].nunique().to_frame().reset_index(drop=False)\n    waypoint_ble_combined = pd.concat([ibeacon_df_grouped, waypoint_df])\n    waypoint_ble_combined = waypoint_ble_combined.sort_values(by='timestamp').reset_index(drop=True)\n\n    # Interpolate waypoint data for wifi timestamps\n    waypoint_ble_combined.x = waypoint_ble_combined.x.interpolate()\n    waypoint_ble_combined.y = waypoint_ble_combined.y.interpolate()\n\n    # Map wifi timestamps to wifi dataframe\n    ble_df_with_waypoint = ibeacon_df.merge(waypoint_ble_combined[['timestamp', 'x', 'y']], on='timestamp', how='left')\n\n\n    # Append to dataframe holding all wifi data for this floor\n    ble_list = ble_list.append(ble_df_with_waypoint)\n\ndisplay(ble_list.head().style.set_caption(f\"All iBeacon data for floor {sample_floor} in building {sample_building}\"))","metadata":{"execution":{"iopub.status.busy":"2023-05-04T10:14:33.247382Z","iopub.execute_input":"2023-05-04T10:14:33.247761Z","iopub.status.idle":"2023-05-04T10:14:36.762775Z","shell.execute_reply.started":"2023-05-04T10:14:33.247724Z","shell.execute_reply":"2023-05-04T10:14:36.761776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_ibeacon_signal(path, building, floor, identifier):\n    # Prepare width_meter & height_meter (taken from the .json file)\n    floor_plan_filename = f\"{path}/metadata/{building}/{floor}/floor_image.png\"\n    json_plan_filename = f\"{path}/metadata/{building}/{floor}/floor_info.json\"\n\n    # Load and extract floor dimensions in meter\n    with open(json_plan_filename) as json_file:\n        json_data = json.load(json_file)\n\n    width_meter = json_data[\"map_info\"][\"width\"]\n    height_meter = json_data[\"map_info\"][\"height\"]\n\n    # Load floor map as image\n    floor_img = plt.imread(f\"{path}/metadata/{sample_building}/{sample_floor}/floor_image.png\")\n\n    # Get datapoints for single beacon\n    ble_df = ble_list[ble_list.id ==identifier]\n\n    # Scale x, y, and distance to match floor with sclaer = floor_img.shape[0] / height_meter = floor_img.shape[1] / width_meter)\n    ble_df[\"x_scaled\"] = ble_df[\"x\"] * floor_img.shape[0] / height_meter\n    ble_df[\"y_scaled\"] = (ble_df[\"y\"] * -1 * floor_img.shape[1] / width_meter) + floor_img.shape[0]\n    ble_df[\"distance_scaled\"] = ble_df[\"distance\"].astype(float) * floor_img.shape[0] / height_meter\n\n    # Plot distance to beacon from x,y position as circle \n    fig, ax = plt.subplots(figsize=(12, 12))\n    plt.imshow(floor_img)\n    for i in range(len(ble_df)):\n        circle=plt.Circle((ble_df[\"x_scaled\"].values[i], ble_df[\"y_scaled\"].values[i]), ble_df[\"distance_scaled\"].values[i], color='dodgerblue', fill=False)\n        ax.add_patch(circle)\n    ax.set_title(f\"Building {building}, Floor {floor} \\n Beacon ID {identifier}\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-02T02:05:49.369505Z","iopub.execute_input":"2023-05-02T02:05:49.369882Z","iopub.status.idle":"2023-05-02T02:05:49.379512Z","shell.execute_reply.started":"2023-05-02T02:05:49.369848Z","shell.execute_reply":"2023-05-02T02:05:49.378494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_ibeacon_signal(PATH, sample_building, sample_floor, \"89cb11b04122cef23388b0da06bd426c1f48a9b5_cfc84f0752adc96b489f71195d91a946c5f6d3e8_8159618423dfa22f1ca0b62543e2f18eef630ce8\")","metadata":{"execution":{"iopub.status.busy":"2023-05-02T02:05:55.147446Z","iopub.execute_input":"2023-05-02T02:05:55.147839Z","iopub.status.idle":"2023-05-02T02:06:06.563855Z","shell.execute_reply.started":"2023-05-02T02:05:55.147802Z","shell.execute_reply":"2023-05-02T02:06:06.562719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ble_list = pd.DataFrame()\ndf = train_df[train_df.building == sample_building]\n\nfor floor in tqdm(df.floor.unique()):\n    files = df[df.floor == floor].path.unique()\n\n    for file in (files):\n\n        path = read_data_file(f\"{PATH}train/{sample_building}/{floor}/{file}\")\n\n        # Get path's wifi and waypoint data\n        ibeacon_df = to_frame(path.ibeacon,'ibeacon')\n        waypoint_df = to_frame(path.waypoint,'waypoint')\n\n        # Combine wifi and waypoint dataframe to one dataframe\n        ibeacon_df_grouped = ibeacon_df.groupby('timestamp')['id'].nunique().to_frame().reset_index(drop=False)\n        waypoint_ble_combined = pd.concat([ibeacon_df_grouped, waypoint_df])\n        waypoint_ble_combined = waypoint_ble_combined.sort_values(by='timestamp').reset_index(drop=True)\n\n        # Interpolate waypoint data for wifi timestamps\n        waypoint_ble_combined.x = waypoint_ble_combined.x.interpolate()\n        waypoint_ble_combined.y = waypoint_ble_combined.y.interpolate()\n\n        # Map wifi timestamps to wifi dataframe\n        ble_df_with_waypoint = ibeacon_df.merge(waypoint_ble_combined[['timestamp', 'x', 'y']], on='timestamp', how='left')\n\n        ble_df_with_waypoint['floor'] = floor\n        # Append to dataframe holding all wifi data for this floor\n        ble_list = ble_list.append(ble_df_with_waypoint)\n\nble_list.distance = ble_list.distance.astype(float)","metadata":{"execution":{"iopub.status.busy":"2023-05-04T10:14:36.764722Z","iopub.execute_input":"2023-05-04T10:14:36.76505Z","iopub.status.idle":"2023-05-04T10:15:06.441909Z","shell.execute_reply.started":"2023-05-04T10:14:36.765018Z","shell.execute_reply":"2023-05-04T10:15:06.440877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# full \nfull_df = pd.DataFrame()\nfor sample_building in used_buildings:\n    ble_list = pd.DataFrame()\n    df = train_df[train_df.building == sample_building]\n\n    for floor in tqdm(df.floor.unique()):\n        files = df[df.floor == floor].path.unique()\n\n        for file in (files):\n\n            path = read_data_file(f\"{PATH}train/{sample_building}/{floor}/{file}\")\n\n            # Get path's wifi and waypoint data\n            ibeacon_df = to_frame(path.ibeacon,'ibeacon')\n            waypoint_df = to_frame(path.waypoint,'waypoint')\n\n            # Combine ibeacon and waypoint dataframe to one dataframe\n            ibeacon_df_grouped = ibeacon_df.groupby('timestamp')['id'].nunique().to_frame().reset_index(drop=False)\n            waypoint_ble_combined = pd.concat([ibeacon_df_grouped, waypoint_df])\n            waypoint_ble_combined = waypoint_ble_combined.sort_values(by='timestamp').reset_index(drop=True)\n\n            # Interpolate waypoint data for ibeacon timestamps\n            waypoint_ble_combined.x = waypoint_ble_combined.x.interpolate()\n            waypoint_ble_combined.y = waypoint_ble_combined.y.interpolate()\n\n            # Map wifi timestamps to ibeacon dataframe\n            ble_df_with_waypoint = ibeacon_df.merge(waypoint_ble_combined[['timestamp', 'x', 'y']], on='timestamp', how='left')\n\n            ble_df_with_waypoint['floor'] = floor\n            # Append to dataframe holding all ibeacon data for this floor\n            ble_list = ble_list.append(ble_df_with_waypoint)\n\n    ble_list.distance = ble_list.distance.astype(float)\n    ble_list['site'] = [sample_building] * ble_list.shape[0]\n    full_df = pd.concat((full_df, ble_list), axis=0)","metadata":{"execution":{"iopub.status.busy":"2023-05-05T07:17:21.978563Z","iopub.execute_input":"2023-05-05T07:17:21.978995Z","iopub.status.idle":"2023-05-05T07:30:59.211834Z","shell.execute_reply.started":"2023-05-05T07:17:21.978959Z","shell.execute_reply":"2023-05-05T07:30:59.210781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"full_df.to_csv(\"ibeacon.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-05-04T11:07:14.954906Z","iopub.execute_input":"2023-05-04T11:07:14.955348Z","iopub.status.idle":"2023-05-04T11:07:32.211368Z","shell.execute_reply.started":"2023-05-04T11:07:14.955304Z","shell.execute_reply":"2023-05-04T11:07:32.21028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ble_list.to_csv(f\"{sample_building}_ibeacon.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-05-04T10:17:06.111771Z","iopub.execute_input":"2023-05-04T10:17:06.11221Z","iopub.status.idle":"2023-05-04T10:17:07.132378Z","shell.execute_reply.started":"2023-05-04T10:17:06.112158Z","shell.execute_reply":"2023-05-04T10:17:07.131223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"beacon_floor_mapping = ble_list[ble_list.distance < 3].groupby(['id', 'floor']).floor.count().to_frame()\nbeacon_floor_mapping.columns = ['num_signals']\nbeacon_floor_mapping = beacon_floor_mapping.reset_index(drop=False)\nbeacon_floor_mapping = beacon_floor_mapping.pivot(index = 'id', columns='floor').num_signals.fillna(0)\nbeacon_floor_mapping['located_on'] = beacon_floor_mapping.idxmax(axis=1)\nbeacon_floor_mapping.reset_index(drop=False, inplace=True)\nbeacon_floor_mapping.style.set_properties(subset=['id'], **{'width': '300px'}).background_gradient(subset=[\"B1\", \"F1\", \"F2\", \"F3\", \"F4\"], cmap='Blues', vmin=0, vmax=1500).format({\"B1\": \"{:.0f}\", \"F1\": \"{:.0f}\", \"F2\": \"{:.0f}\", \"F3\": \"{:.0f}\", \"F4\": \"{:.0f}\", \"F5\": \"{:.0f}\", }).set_caption('Number Strong iBeacon Signals in Sample Building')","metadata":{"execution":{"iopub.status.busy":"2023-05-02T02:09:56.091151Z","iopub.execute_input":"2023-05-02T02:09:56.092081Z","iopub.status.idle":"2023-05-02T02:09:56.188492Z","shell.execute_reply.started":"2023-05-02T02:09:56.092018Z","shell.execute_reply":"2023-05-02T02:09:56.187637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = read_data_file(sample_path)\n\n# Get path's wifi and waypoint data\nibeacon_df = to_frame(path.ibeacon,'ibeacon')\nwaypoint_df = to_frame(path.waypoint,'waypoint')\n\n# Combine wifi and waypoint dataframe to one dataframe\nibeacon_df_grouped = ibeacon_df.groupby('timestamp')['id'].nunique().to_frame().reset_index(drop=False)\nwaypoint_ble_combined = pd.concat([ibeacon_df_grouped, waypoint_df])\nwaypoint_ble_combined = waypoint_ble_combined.sort_values(by='timestamp').reset_index(drop=True)\n\n# Interpolate waypoint data for wifi timestamps\nwaypoint_ble_combined.x = waypoint_ble_combined.x.interpolate()\nwaypoint_ble_combined.y = waypoint_ble_combined.y.interpolate()\n\n# Map wifi timestamps to wifi dataframe\nble_df_with_waypoint = ibeacon_df.merge(waypoint_ble_combined[['timestamp', 'x', 'y']], on='timestamp', how='left')\n\n# Merge with lookup table\nX = ble_df_with_waypoint.merge(beacon_floor_mapping[['id', 'located_on']], on='id', how='left')\nX.distance = X.distance.astype(float)\ndisplay(X.head())","metadata":{"execution":{"iopub.status.busy":"2023-05-02T02:09:58.649284Z","iopub.execute_input":"2023-05-02T02:09:58.649887Z","iopub.status.idle":"2023-05-02T02:09:58.748074Z","shell.execute_reply.started":"2023-05-02T02:09:58.649825Z","shell.execute_reply":"2023-05-02T02:09:58.747111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assign each iBeacon distance to a waypoint position in building.\nfrom dataclasses import dataclass\n\n@dataclass\nclass ReadData:\n    # Removed other signal info acquisition for simplicity reasons ...\n    ibeacon: np.ndarray\n    waypoint: np.ndarray\n\n\ndef read_data_file(data_filename):\n    # Removed other signal info acquisition for simplicity reasons ...\n    ibeacon = []\n    waypoint = []\n\n    with open(data_filename, 'r', encoding='utf-8') as file:\n        lines = file.readlines()\n\n    for line_data in lines:\n        line_data = line_data.strip()\n        if not line_data or line_data[0] == '#':\n            continue\n\n        line_data = line_data.split('\\t')\n\n        # Removed other signal info acquisition for simplicity reasons ...\n\n        if line_data[1] == 'TYPE_BEACON':\n            ts = line_data[0]\n            uuid = line_data[2]\n            major = line_data[3]\n            minor = line_data[4]\n            rssi = line_data[6]\n            distance = line_data[7] # Newly added\n            \n            ibeacon_data = [ts, '_'.join([uuid, major, minor]), rssi, distance]\n            \n            ibeacon.append(ibeacon_data)\n            continue\n\n        if line_data[1] == 'TYPE_WAYPOINT':\n            waypoint.append([int(line_data[0]), float(line_data[2]), float(line_data[3])])\n\n    # Removed other signal info acquisition for simplicity reasons ...\n\n    ibeacon = np.array(ibeacon)\n    waypoint = np.array(waypoint)\n\n    return ReadData(ibeacon, waypoint)\n\ndef encode_floor(floor):\n    \"\"\"\n    Encodes floor string to integer\n    \"\"\"\n    # Get floor number\n    encoded_floor = int(re.findall('[0-9]+', floor)[0])\n    \n    if floor.find('B') == 0:\n        # Multiply with -1 incase of B floor\n        encoded_floor = (-1)*encoded_floor\n    else: \n        # Subtract one floor to start at 0 for F1\n        encoded_floor = encoded_floor -1\n        \n    return encoded_floor\n\n# The following code is copied and edited from https://www.kaggle.com/jiweiliu/wifi-label-encode\nACOLS = ['timestamp','x','y','z']\n        \nFIELDS = {\n    'acce': ACOLS,\n    'acce_uncali': ACOLS,\n    'gyro': ACOLS,\n    'gyro_uncali': ACOLS,\n    'magn': ACOLS,\n    'magn_uncali': ACOLS,\n    'ahrs': ACOLS,\n    'wifi': ['timestamp','ssid','bssid','rssi','last_timestamp'],\n    'ibeacon': ['timestamp','id','rssi', 'distance'],\n    'waypoint': ['timestamp','x','y']\n}\n\ndef create_dummy_df(cols):\n    \"\"\"\n    Edited from https://www.kaggle.com/jiweiliu/wifi-label-encode\n    \"\"\"\n    df = pd.DataFrame()\n    for col in cols:\n        df[col] = [0]\n        if col in ['ssid','bssid']:\n            df[col] = df[col].map(str)\n    return df\n\ndef to_frame(data, col):    \n    \"\"\"\n    Edited from https://www.kaggle.com/jiweiliu/wifi-label-encode\n    \"\"\"\n    cols = FIELDS[col]\n    if data.shape[0]>0:\n        df = pd.DataFrame(data, columns=cols)\n    else:\n        df = create_dummy_df(cols)\n    for col in df.columns:\n        if 'timestamp' in col:\n            df[col] = df[col].astype('int64')\n    return df\n\ndef building2ibeacon_feature_extract(building_name:str):\n    ble_list = pd.DataFrame()\n    df = train_df[train_df.building == sample_building]\n\n    for floor in tqdm(df.floor.unique()):\n        files = df[df.floor == floor].path.unique()\n\n        for file in (files):\n            path = read_data_file(f\"{PATH}train/{sample_building}/{floor}/{file}\")\n\n            ibeacon_df = to_frame(path.ibeacon,'ibeacon')\n            waypoint_df = to_frame(path.waypoint,'waypoint')\n\n            ibeacon_df_grouped = ibeacon_df.groupby('timestamp')['id'].nunique().to_frame().reset_index(drop=False)\n            waypoint_ble_combined = pd.concat([ibeacon_df_grouped, waypoint_df])\n            waypoint_ble_combined = waypoint_ble_combined.sort_values(by='timestamp').reset_index(drop=True)\n\n            # Interpolate waypoint data for wifi timestamps\n            waypoint_ble_combined.x = waypoint_ble_combined.x.interpolate()\n            waypoint_ble_combined.y = waypoint_ble_combined.y.interpolate()\n\n            ble_df_with_waypoint = ibeacon_df.merge(waypoint_ble_combined[['timestamp', 'x', 'y']], on='timestamp', how='left')\n\n            ble_df_with_waypoint['floor'] = floor\n            ble_list = ble_list.append(ble_df_with_waypoint)\n\n    ble_list.distance = ble_list.distance.astype(float)\n    return ble_list","metadata":{"execution":{"iopub.status.busy":"2023-05-02T02:20:48.040409Z","iopub.execute_input":"2023-05-02T02:20:48.040847Z","iopub.status.idle":"2023-05-02T02:20:48.064197Z","shell.execute_reply.started":"2023-05-02T02:20:48.040806Z","shell.execute_reply":"2023-05-02T02:20:48.063078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\npd.set_option(\"display.max_colwidth\", 50)\n    \nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport glob\nimport os\nimport sys\nimport math\nimport gc\nimport json \nimport re\nfrom tqdm import tqdm\nimport warnings \nwarnings.filterwarnings('ignore')\n\nPATH = '../input/indoor-location-navigation/'\n\nsubmission = pd.read_csv(f'{PATH}sample_submission.csv')\n\ntrain_df = pd.DataFrame()\nbuilding_dict = {}\nfor building in used_buildings:\n    building_dict['building'] = building\n    for floor in os.listdir(f\"{PATH}/train/{building}\"):\n        building_dict['floor'] = floor\n        building_dict['floor_enc'] = encode_floor(floor)\n        for path in os.listdir(f\"{PATH}/train/{building}/{floor}\"):\n            building_dict['path'] = path\n            train_df = train_df.append(building_dict, ignore_index=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2023-05-02T02:25:37.945086Z","iopub.execute_input":"2023-05-02T02:25:37.945768Z","iopub.status.idle":"2023-05-02T02:25:37.963421Z","shell.execute_reply.started":"2023-05-02T02:25:37.945712Z","shell.execute_reply":"2023-05-02T02:25:37.962321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ibeacon_feature = building2ibeacon_feature_extract(train_df.building[0])","metadata":{"execution":{"iopub.status.busy":"2023-05-02T02:26:33.479319Z","iopub.execute_input":"2023-05-02T02:26:33.479738Z","iopub.status.idle":"2023-05-02T02:27:01.571681Z","shell.execute_reply.started":"2023-05-02T02:26:33.479694Z","shell.execute_reply":"2023-05-02T02:27:01.570705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test ibeacon feature","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### sample 확인","metadata":{}},{"cell_type":"code","source":"# from glob import glob\n\n# floor_encoder = {\n#     \"B1\":-1,\n#     \"F1\":0,\n#     \"F2\":1,\n#     \"F3\":2,\n#     \"F4\":3,\n# }\n\n# floor_decoder = {v:k for k,v in floor_encoder.items()}\n\n# train_dir = \"/kaggle/input/indoor-location-navigation/train\"\n# city = train_paths[0].split(\"/\")[-1][:-15]\n# floor = floor_decoder[train_df.f.values[0]]\n# p_t = train_df.path.values[0]\n# glob(os.path.join(train_dir, city) + \"/*\")","metadata":{"execution":{"iopub.status.busy":"2023-05-06T09:06:59.294086Z","iopub.execute_input":"2023-05-06T09:06:59.294545Z","iopub.status.idle":"2023-05-06T09:06:59.29999Z","shell.execute_reply.started":"2023-05-06T09:06:59.294508Z","shell.execute_reply":"2023-05-06T09:06:59.298332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base = '../input/indoor-location-navigation'\npath = f'{base}/train/5cd56b5ae2acfd2d33b58549/5F/5d06134c4a19c000086c4324.txt'","metadata":{"execution":{"iopub.status.busy":"2023-05-06T09:07:22.73152Z","iopub.execute_input":"2023-05-06T09:07:22.731935Z","iopub.status.idle":"2023-05-06T09:07:22.736836Z","shell.execute_reply.started":"2023-05-06T09:07:22.7319Z","shell.execute_reply":"2023-05-06T09:07:22.735312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from main import calibrate_magnetic_wifi_ibeacon_to_position\n\nmwi_datas = calibrate_magnetic_wifi_ibeacon_to_position([path])","metadata":{"execution":{"iopub.status.busy":"2023-05-06T09:07:41.362639Z","iopub.execute_input":"2023-05-06T09:07:41.363516Z","iopub.status.idle":"2023-05-06T09:07:45.921802Z","shell.execute_reply.started":"2023-05-06T09:07:41.363465Z","shell.execute_reply":"2023-05-06T09:07:45.920932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-05-01T04:10:09.423031Z","iopub.execute_input":"2023-05-01T04:10:09.423463Z","iopub.status.idle":"2023-05-01T04:10:42.71525Z","shell.execute_reply.started":"2023-05-01T04:10:09.423428Z","shell.execute_reply":"2023-05-01T04:10:42.71437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### model","metadata":{}},{"cell_type":"code","source":"from numpy import sqrt, power\nfrom numpy import abs as absolute","metadata":{"execution":{"iopub.status.busy":"2023-05-06T09:10:37.407713Z","iopub.execute_input":"2023-05-06T09:10:37.408129Z","iopub.status.idle":"2023-05-06T09:10:37.413113Z","shell.execute_reply.started":"2023-05-06T09:10:37.408095Z","shell.execute_reply":"2023-05-06T09:10:37.411822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mean_position_error(x_pred, y_pred, f_pred, x_true, y_true, f_true, p=15):\n    '''Custom function to evaluate Mean Position Error.\n    x: x coordinate of the waypoint position; dtype list()\n    y: y coordinate of the waypoint position; dtype list()\n    f: exact floor or the building; dtype list()\n    p: floor penalty, set to 15 (always)'''\n    \n    N = len(x_true)\n    #1\n    formula = sqrt( power(x_pred - x_true, 2) + power(y_pred - y_true, 2) )\n    #2\n    formula = formula + p * absolute(f_pred - f_true)\n    #3\n    formula = formula.sum() / N\n    \n    return formula","metadata":{"execution":{"iopub.status.busy":"2023-05-06T09:17:20.850306Z","iopub.execute_input":"2023-05-06T09:17:20.850691Z","iopub.status.idle":"2023-05-06T09:17:20.857717Z","shell.execute_reply.started":"2023-05-06T09:17:20.850659Z","shell.execute_reply":"2023-05-06T09:17:20.856211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import Libraries\nimport lightgbm as lgb\n\n# ~~~~\n# Data\n# ~~~~\nbase_dir = \"../input/indoor-navigation-and-location-wifi-features/wifi_features\"\ntrain_dir = \"/train/*_train.csv\"\ntest_dir = \"/test/*_test.csv\"\n\n\n# Paths for train & test files\ntrain_paths = sorted(glob(base_dir + train_dir))\ntest_paths = sorted(glob(base_dir + test_dir))\nsample_subm = pd.read_csv('../input/indoor-location-navigation/sample_submission.csv',\n                          index_col=0)\n\nprint(\"Len Train Files: {}\".format(len(train_paths)), \"\\n\" +\n      \"Len Test Files: {}\".format(len(test_paths)))","metadata":{"execution":{"iopub.status.busy":"2023-05-06T09:15:24.563256Z","iopub.execute_input":"2023-05-06T09:15:24.563647Z","iopub.status.idle":"2023-05-06T09:15:24.597792Z","shell.execute_reply.started":"2023-05-06T09:15:24.563617Z","shell.execute_reply":"2023-05-06T09:15:24.596433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_lgbm(train_perc=0.70, version=1, n_estimators=150, num_leaves=127):\n    f = open(f\"lgbm_logs_{version}.txt\", \"w+\")\n    lgbm_predictions = []\n    reals = []\n    preds = []\n    \n    k = 1\n    for train_path, test_path in zip(train_paths, test_paths):\n        # --- Read in data ---\n        train_df = pd.read_csv(train_path, index_col=0)\n        train_df = train_df.sample(frac=1, random_state=10)\n\n        test_df = pd.read_csv(test_path, index_col=0).iloc[:, :-1]\n        train_size = int(len(train_df) * train_perc)\n\n        # --- Data Validation ---\n        # Train features + targets\n        X_train = train_df.iloc[:train_size, :-4]\n        y_train_x = train_df.iloc[:train_size, -4]\n        y_train_y = train_df.iloc[:train_size, -3]\n        y_train_f = train_df.iloc[:train_size, -2]\n\n        # Valid features + targets\n        X_valid = train_df.iloc[train_size:, :-4]\n        y_valid_x = train_df.iloc[train_size:, -4]\n        y_valid_y = train_df.iloc[train_size:, -3]\n        y_valid_f = train_df.iloc[train_size:, -2]\n\n        # --- Model Training ---\n        lgbm_x = lgb.LGBMRegressor(n_estimators=n_estimators, num_leaves=num_leaves)\n        lgbm_x.fit(X_train, y_train_x)\n\n        lgbm_y = lgb.LGBMRegressor(n_estimators=n_estimators, num_leaves=num_leaves)\n        lgbm_y.fit(X_train, y_train_y)\n\n        lgbm_f = lgb.LGBMClassifier(n_estimators=n_estimators, num_leaves=num_leaves)\n        lgbm_f.fit(X_train, y_train_f)\n\n\n        # --- Model Validation Predictions ---\n        preds_x = lgbm_x.predict(X_valid)\n        preds_y = lgbm_y.predict(X_valid)\n        preds_f = lgbm_f.predict(X_valid).astype(int)\n        \n        mpe = mean_position_error(preds_x, preds_y, preds_f,\n                                  y_valid_x, y_valid_y, y_valid_f)\n        reals.extend(y_valid_f)\n        preds.extend(preds_f)\n        print(\"{} | MPE: {}\".format(k, mpe))\n        # Save logs\n        with open(f\"lgbm_logs_{version}.txt\", 'a+') as f:\n            print(\"{} | MPE: {}\".format(k, mpe), file=f)\n        \n        k+=1\n        # --- Model Test Predictions ---\n        test_preds_x = lgbm_x.predict(test_df)\n        test_preds_y = lgbm_y.predict(test_df)\n        test_preds_f = lgbm_f.predict(test_df).astype(int)\n\n        all_test_preds = pd.DataFrame({'floor' : test_preds_f,\n                                       'x' : test_preds_x, \n                                       'y' : test_preds_y})\n        lgbm_predictions.append(all_test_preds)\n    \n    return lgbm_predictions, reals, preds","metadata":{"execution":{"iopub.status.busy":"2023-05-06T09:29:46.607882Z","iopub.execute_input":"2023-05-06T09:29:46.608302Z","iopub.status.idle":"2023-05-06T09:29:46.622783Z","shell.execute_reply.started":"2023-05-06T09:29:46.608249Z","shell.execute_reply":"2023-05-06T09:29:46.621429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(train_paths[0], index_col=0)\ntrain_df = train_df.sample(frac=1, random_state=10)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T10:23:17.261289Z","iopub.execute_input":"2023-05-06T10:23:17.261728Z","iopub.status.idle":"2023-05-06T10:23:18.144425Z","shell.execute_reply.started":"2023-05-06T10:23:17.261688Z","shell.execute_reply":"2023-05-06T10:23:18.143209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2023-05-06T10:25:28.595558Z","iopub.execute_input":"2023-05-06T10:25:28.596192Z","iopub.status.idle":"2023-05-06T10:25:28.632163Z","shell.execute_reply.started":"2023-05-06T10:25:28.596143Z","shell.execute_reply":"2023-05-06T10:25:28.631099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.drop(\"path\", axis=1).drop_duplicates(subset=None, keep='first', inplace=False, ignore_index=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T10:25:18.972195Z","iopub.execute_input":"2023-05-06T10:25:18.972616Z","iopub.status.idle":"2023-05-06T10:25:19.582007Z","shell.execute_reply.started":"2023-05-06T10:25:18.972584Z","shell.execute_reply":"2023-05-06T10:25:19.580837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t = train_df.drop(\"path\", axis=1).drop_duplicates()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T10:27:23.493331Z","iopub.execute_input":"2023-05-06T10:27:23.493803Z","iopub.status.idle":"2023-05-06T10:27:23.849612Z","shell.execute_reply.started":"2023-05-06T10:27:23.493761Z","shell.execute_reply":"2023-05-06T10:27:23.848671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_size = int(len(t) * 0.7)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T10:27:31.926628Z","iopub.execute_input":"2023-05-06T10:27:31.926959Z","iopub.status.idle":"2023-05-06T10:27:31.932374Z","shell.execute_reply.started":"2023-05-06T10:27:31.926931Z","shell.execute_reply":"2023-05-06T10:27:31.930846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = t.iloc[:train_size, :]\nvalid = t.iloc[train_size:, :]","metadata":{"execution":{"iopub.status.busy":"2023-05-06T10:28:08.252057Z","iopub.execute_input":"2023-05-06T10:28:08.252476Z","iopub.status.idle":"2023-05-06T10:28:08.258766Z","shell.execute_reply.started":"2023-05-06T10:28:08.252437Z","shell.execute_reply":"2023-05-06T10:28:08.25725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.to_csv(\"train_site1.csv\", index=False)\nvalid.to_csv(\"valid_site1.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T10:29:41.729197Z","iopub.execute_input":"2023-05-06T10:29:41.729655Z","iopub.status.idle":"2023-05-06T10:29:44.61971Z","shell.execute_reply.started":"2023-05-06T10:29:41.72962Z","shell.execute_reply":"2023-05-06T10:29:44.618616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgbm_predictions, reals, preds = train_lgbm()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T09:29:49.335177Z","iopub.execute_input":"2023-05-06T09:29:49.335669Z","iopub.status.idle":"2023-05-06T09:42:16.800964Z","shell.execute_reply.started":"2023-05-06T09:29:49.335628Z","shell.execute_reply":"2023-05-06T09:42:16.799926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\nprint(classification_report(reals, preds))","metadata":{"execution":{"iopub.status.busy":"2023-05-06T09:43:42.340616Z","iopub.execute_input":"2023-05-06T09:43:42.341061Z","iopub.status.idle":"2023-05-06T09:43:42.564233Z","shell.execute_reply.started":"2023-05-06T09:43:42.341024Z","shell.execute_reply":"2023-05-06T09:43:42.562967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from dataclasses import dataclass\n\nimport numpy as np\n\n\n@dataclass\nclass ReadData:\n    acce: np.ndarray\n    acce_uncali: np.ndarray\n    gyro: np.ndarray\n    gyro_uncali: np.ndarray\n    magn: np.ndarray\n    magn_uncali: np.ndarray\n    ahrs: np.ndarray\n    wifi: np.ndarray\n    ibeacon: np.ndarray\n    waypoint: np.ndarray\n\n\ndef read_data_file(data_filename):\n    acce = []\n    acce_uncali = []\n    gyro = []\n    gyro_uncali = []\n    magn = []\n    magn_uncali = []\n    ahrs = []\n    wifi = []\n    ibeacon = []\n    waypoint = []\n\n    with open(data_filename, 'r', encoding='utf-8') as file:\n        lines = file.readlines()\n\n    for line_data in lines:\n        line_data = line_data.strip()\n        if not line_data or line_data[0] == '#':\n            continue\n\n        line_data = line_data.split('\\t')\n\n        if line_data[1] == 'TYPE_ACCELEROMETER':\n            acce.append([int(line_data[0]), float(line_data[2]), float(line_data[3]), float(line_data[4])])\n            continue\n\n        if line_data[1] == 'TYPE_ACCELEROMETER_UNCALIBRATED':\n            acce_uncali.append([int(line_data[0]), float(line_data[2]), float(line_data[3]), float(line_data[4])])\n            continue\n\n        if line_data[1] == 'TYPE_GYROSCOPE':\n            gyro.append([int(line_data[0]), float(line_data[2]), float(line_data[3]), float(line_data[4])])\n            continue\n\n        if line_data[1] == 'TYPE_GYROSCOPE_UNCALIBRATED':\n            gyro_uncali.append([int(line_data[0]), float(line_data[2]), float(line_data[3]), float(line_data[4])])\n            continue\n\n        if line_data[1] == 'TYPE_MAGNETIC_FIELD':\n            magn.append([int(line_data[0]), float(line_data[2]), float(line_data[3]), float(line_data[4])])\n            continue\n\n        if line_data[1] == 'TYPE_MAGNETIC_FIELD_UNCALIBRATED':\n            magn_uncali.append([int(line_data[0]), float(line_data[2]), float(line_data[3]), float(line_data[4])])\n            continue\n\n        if line_data[1] == 'TYPE_ROTATION_VECTOR':\n            ahrs.append([int(line_data[0]), float(line_data[2]), float(line_data[3]), float(line_data[4])])\n            continue\n\n        if line_data[1] == 'TYPE_WIFI':\n            \n            sys_ts = line_data[0]\n            ssid = line_data[2]\n            bssid = line_data[3]\n            rssi = line_data[4]\n            freq = line_data[5]\n            lastseen_ts = line_data[6]\n            \n            wifi_data = [sys_ts, ssid, bssid, rssi, lastseen_ts]\n            wifi.append(wifi_data)\n            continue\n\n        if line_data[1] == 'TYPE_BEACON':\n            ts = line_data[0]\n            uuid = line_data[2]\n            major = line_data[3]\n            minor = line_data[4]\n            tx_power = line_data[5]\n            rssi = line_data[6]\n            distance = line_data[7]\n            mac_add = line_data[8]\n            sts = line_data[9]\n            \n            # ibeacon_data = [ts, '_'.join([uuid, major, minor]),tx_power, rssi, distance,mac_add,sts]\n            ibeacon_data = [ts, uuid, major, minor, tx_power, rssi, distance,mac_add,sts]\n            ibeacon.append(ibeacon_data)\n            continue\n\n        if line_data[1] == 'TYPE_WAYPOINT':\n            waypoint.append([int(line_data[0]), float(line_data[2]), float(line_data[3])])\n\n    acce = np.array(acce)\n    acce_uncali = np.array(acce_uncali)\n    gyro = np.array(gyro)\n    gyro_uncali = np.array(gyro_uncali)\n    magn = np.array(magn)\n    magn_uncali = np.array(magn_uncali)\n    ahrs = np.array(ahrs)\n    wifi = np.array(wifi)\n    ibeacon = np.array(ibeacon)\n    waypoint = np.array(waypoint)\n\n    return ReadData(acce, acce_uncali, gyro, gyro_uncali, magn, magn_uncali, ahrs, wifi, ibeacon, waypoint)","metadata":{"execution":{"iopub.status.busy":"2023-05-02T08:45:33.216003Z","iopub.execute_input":"2023-05-02T08:45:33.216641Z","iopub.status.idle":"2023-05-02T08:45:33.247506Z","shell.execute_reply.started":"2023-05-02T08:45:33.216596Z","shell.execute_reply":"2023-05-02T08:45:33.246067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sensor_type = [\n\"TYPE_ACCELEROMETER\",\n\"TYPE_ACCELEROMETER_UNCALIBRATED\",\n\"TYPE_GYROSCOPE\",\n\"TYPE_GYROSCOPE_UNCALIBRATED\",\n\"TYPE_MAGNETIC_FIELD\",\n\"TYPE_MAGNETIC_FIELD_UNCALIBRATED\",\n\"TYPE_ROTATION_VECTOR\",\n\"TYPE_WIFI\",\n\"TYPE_BEACON\",\n\"TYPE_WAYPOINT\",]","metadata":{"execution":{"iopub.status.busy":"2023-04-30T14:46:59.85691Z","iopub.execute_input":"2023-04-30T14:46:59.857271Z","iopub.status.idle":"2023-04-30T14:46:59.86204Z","shell.execute_reply.started":"2023-04-30T14:46:59.85724Z","shell.execute_reply":"2023-04-30T14:46:59.861044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from io_f import read_data_file\npath = train_dir + \"/\" + city + \"/\"+ floor + \"/\"+ p_t + \".txt\"\n# Read in 1 random example\nsample_file = read_data_file(path)","metadata":{"execution":{"iopub.status.busy":"2023-04-30T07:13:41.118505Z","iopub.execute_input":"2023-04-30T07:13:41.119231Z","iopub.status.idle":"2023-04-30T07:13:41.151305Z","shell.execute_reply.started":"2023-04-30T07:13:41.11919Z","shell.execute_reply":"2023-04-30T07:13:41.150051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_file.waypoint.shape, sample_file.ibeacon.shape, sample_file.wifi.shape, sample_file.acce.shape, sample_file.ahrs.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-30T05:09:29.79177Z","iopub.execute_input":"2023-04-30T05:09:29.792122Z","iopub.status.idle":"2023-04-30T05:09:29.799038Z","shell.execute_reply.started":"2023-04-30T05:09:29.792093Z","shell.execute_reply":"2023-04-30T05:09:29.798076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id : 사람 -> 층 -> data","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# generate all the training data \nbuilding_dfs = dict()\n\nfor building in used_buildings:\n    \n    ### 빌딩 나누기\n    folders = sorted(glob.glob(os.path.join(base_path,'train', building +'/*')))\n    dfs = list()\n    index = sorted(bssid[building])\n    print(building)\n    \n    \n    # 층 나누기\n    for folder in folders:\n        floor = floor_map[folder.split('/')[-1]]\n        files = glob.glob(os.path.join(folder, \"*.txt\"))\n        print(floor)\n        \n        # path\n        for file in files:\n            wifi = list()\n            waypoint = list()\n            with open(file) as f:\n                txt = f.readlines()\n            \n            \n            for line in txt:\n                line = line.strip().split()\n                if line[1] == \"TYPE_WAYPOINT\":\n                    waypoint.append(line)\n                if line[1] == \"TYPE_WIFI\":\n                    wifi.append(line)\n\n            df = pd.DataFrame(np.array(wifi))    \n\n            # generate a feature, and label for each wifi block\n            for gid, g in df.groupby(0):\n                dists = list()\n                for e, k in enumerate(waypoint):\n                    dist = abs(int(gid) - int(k[0]))\n                    dists.append(dist)\n                nearest_wp_index = np.argmin(dists)\n                \n                g = g.drop_duplicates(subset=3)\n                tmp = g.iloc[:,3:5]\n                feat = tmp.set_index(3).reindex(index).replace(np.nan, -999).T\n                feat[\"x\"] = float(waypoint[nearest_wp_index][2])\n                feat[\"y\"] = float(waypoint[nearest_wp_index][3])\n                feat[\"f\"] = floor\n                feat[\"path\"] = file.split('/')[-1].split('.')[0] # useful for crossvalidation\n                dfs.append(feat)\n                \n    building_df = pd.concat(dfs)\n    building_dfs[building] = df\n    building_df.to_csv(building+\"_1000_train.csv\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acce: 속도 +\nacce_uncali: 속도 -\ngyro: 회전 +\ngyro_uncali: 회전 -\nmagn: 마그넷 + \nmagn_uncali: 마그넷 -\nahrs: np.ndarray\n------------------------------\n    \nwifi: np.ndarray\nibeacon: np.ndarray\nwaypoint: np.ndarray","metadata":{"execution":{"iopub.status.busy":"2023-04-29T11:06:36.16422Z","iopub.execute_input":"2023-04-29T11:06:36.164667Z","iopub.status.idle":"2023-04-29T11:06:36.171555Z","shell.execute_reply.started":"2023-04-29T11:06:36.164629Z","shell.execute_reply":"2023-04-29T11:06:36.170478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wifi_df = pd.DataFrame(sample_file.wifi)\nwifi_df.columns = ['timestamp', 'ssid', 'bssid', 'rssi', 'last_seen_timestamp']\nwifi_df = wifi_df.pivot(index='timestamp', columns=['ssid', 'bssid'])['rssi']\nwifi_df.reset_index(drop=False, inplace=True)\nwifi_df.style.set_caption('WiFi')","metadata":{"execution":{"iopub.status.busy":"2023-04-29T10:57:03.311242Z","iopub.execute_input":"2023-04-29T10:57:03.31172Z","iopub.status.idle":"2023-04-29T10:57:03.516784Z","shell.execute_reply.started":"2023-04-29T10:57:03.311683Z","shell.execute_reply":"2023-04-29T10:57:03.515926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from io_f import read_data_file\n\n# Read in 1 random example\nsample_file = read_data_file(path)\nsample_file.waypoint.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-30T14:47:55.681552Z","iopub.execute_input":"2023-04-30T14:47:55.681965Z","iopub.status.idle":"2023-04-30T14:47:55.75293Z","shell.execute_reply.started":"2023-04-30T14:47:55.681932Z","shell.execute_reply":"2023-04-30T14:47:55.751308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ssubm = pd.read_csv('../input/indoor-location-navigation/sample_submission.csv')\n\nssubm_df = ssubm[\"site_path_timestamp\"].apply(lambda x: pd.Series(x.split(\"_\")))\nused_buildings = sorted(ssubm_df[0].value_counts().index.tolist())\n\n# dictionary used to map the floor codes to the values used in the submission file. \nfloor_map = {\"B2\":-2, \"B1\":-1, \"F1\":0, \"F2\": 1, \"F3\":2, \"F4\":3, \"F5\":4, \"F6\":5, \"F7\":6,\"F8\":7, \"F9\":8,\n             \"1F\":0, \"2F\":1, \"3F\":2, \"4F\":3, \"5F\":4, \"6F\":5, \"7F\":6, \"8F\": 7, \"9F\":8}","metadata":{"execution":{"iopub.status.busy":"2023-04-23T08:49:04.336998Z","iopub.execute_input":"2023-04-23T08:49:04.337459Z","iopub.status.idle":"2023-04-23T08:49:06.824022Z","shell.execute_reply.started":"2023-04-23T08:49:04.337421Z","shell.execute_reply":"2023-04-23T08:49:06.822616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"/kaggle/input/indoor-location-navigation/sample_submission.csv","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base = \"/kaggle/input\"\n\nfeature_dir = f\"{base}/indoor-navigation-and-location-wifi-features/wifi_features\"\ntrain_files = sorted(glob.glob(os.path.join(feature_dir, 'train/*.csv')))\ntest_files = sorted(glob.glob(os.path.join(feature_dir, 'test/*.csv')))\nsubm = pd.read_csv(f'{base}/indoor-location-navigation/sample_submission.csv', index_col=0)","metadata":{"execution":{"iopub.status.busy":"2023-04-24T04:37:24.778759Z","iopub.execute_input":"2023-04-24T04:37:24.779127Z","iopub.status.idle":"2023-04-24T04:37:24.814448Z","shell.execute_reply.started":"2023-04-24T04:37:24.779091Z","shell.execute_reply":"2023-04-24T04:37:24.813678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv(train_files[0])","metadata":{"execution":{"iopub.status.busy":"2023-04-24T04:38:50.575137Z","iopub.execute_input":"2023-04-24T04:38:50.575526Z","iopub.status.idle":"2023-04-24T04:38:52.397914Z","shell.execute_reply.started":"2023-04-24T04:38:50.575488Z","shell.execute_reply":"2023-04-24T04:38:52.397178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import Libraries\n# ~~~~\n# Data\n# ~~~~\nbase_dir = \"../input/indoor-navigation-and-location-wifi-features/wifi_features\"\ntrain_dir = \"/train/*_train.csv\"\ntest_dir = \"/test/*_test.csv\"\n\n\n# Paths for train & test files\ntrain_paths = sorted(glob(base_dir + train_dir))\ntest_paths = sorted(glob(base_dir + test_dir))\nsample_subm = pd.read_csv('../input/indoor-location-navigation/sample_submission.csv',\n                          index_col=0)\n\nprint(\"Len Train Files: {}\".format(len(train_paths)), \"\\n\" +\n      \"Len Test Files: {}\".format(len(test_paths)))","metadata":{"execution":{"iopub.status.busy":"2023-04-30T14:48:27.628632Z","iopub.execute_input":"2023-04-30T14:48:27.629002Z","iopub.status.idle":"2023-04-30T14:48:27.68654Z","shell.execute_reply.started":"2023-04-30T14:48:27.628969Z","shell.execute_reply":"2023-04-30T14:48:27.685488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_paths[0]","metadata":{"execution":{"iopub.status.busy":"2023-04-24T04:45:16.315027Z","iopub.execute_input":"2023-04-24T04:45:16.315392Z","iopub.status.idle":"2023-04-24T04:45:16.321555Z","shell.execute_reply.started":"2023-04-24T04:45:16.315361Z","shell.execute_reply":"2023-04-24T04:45:16.320556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_paths[0]","metadata":{"execution":{"iopub.status.busy":"2023-04-24T04:45:21.302891Z","iopub.execute_input":"2023-04-24T04:45:21.303605Z","iopub.status.idle":"2023-04-24T04:45:21.311083Z","shell.execute_reply.started":"2023-04-24T04:45:21.303553Z","shell.execute_reply":"2023-04-24T04:45:21.30969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = pd.read_csv(train_paths[0])","metadata":{"execution":{"iopub.status.busy":"2023-04-24T04:42:41.356963Z","iopub.execute_input":"2023-04-24T04:42:41.357317Z","iopub.status.idle":"2023-04-24T04:42:42.460531Z","shell.execute_reply.started":"2023-04-24T04:42:41.357287Z","shell.execute_reply":"2023-04-24T04:42:42.459515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = sample.sample(frac=1, random_state=10)","metadata":{"execution":{"iopub.status.busy":"2023-04-24T04:42:53.936301Z","iopub.execute_input":"2023-04-24T04:42:53.936942Z","iopub.status.idle":"2023-04-24T04:42:54.003898Z","shell.execute_reply.started":"2023-04-24T04:42:53.936878Z","shell.execute_reply":"2023-04-24T04:42:54.0029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### waypath","metadata":{}},{"cell_type":"code","source":"def visualize_trajectory(trajectory, floor_plan_filename, width_meter, height_meter, title=None, mode='lines + markers + text', show=False):\n    fig = go.Figure()\n\n    # add trajectory\n    size_list = [6] * trajectory.shape[0]\n    size_list[0] = 10\n    size_list[-1] = 10\n\n    color_list = ['rgba(4, 174, 4, 0.5)'] * trajectory.shape[0]\n    color_list[0] = 'rgba(12, 5, 235, 1)'\n    color_list[-1] = 'rgba(235, 5, 5, 1)'\n\n    position_count = {}\n    text_list = []\n    for i in range(trajectory.shape[0]):\n        if str(trajectory[i]) in position_count:\n            position_count[str(trajectory[i])] += 1\n        else:\n            position_count[str(trajectory[i])] = 0\n        text_list.append('        ' * position_count[str(trajectory[i])] + f'{i}')\n    text_list[0] = 'Start Point: 0'\n    text_list[-1] = f'End Point: {trajectory.shape[0] - 1}'\n\n    fig.add_trace(\n        go.Scattergl(\n            x=trajectory[:, 0],\n            y=trajectory[:, 1],\n            mode=mode,\n            marker=dict(size=size_list, color=color_list),\n            line=dict(shape='linear', color='rgb(100, 10, 100)', width=2, dash='dot'),\n            text=text_list,\n            textposition=\"top center\",\n            name='trajectory',\n        ))\n\n    # add floor plan\n    floor_plan = Image.open(floor_plan_filename)\n    fig.update_layout(images=[\n        go.layout.Image(\n            source=floor_plan,\n            xref=\"x\",\n            yref=\"y\",\n            x=0,\n            y=height_meter,\n            sizex=width_meter,\n            sizey=height_meter,\n            sizing=\"contain\",\n            opacity=1,\n            layer=\"below\",\n        )\n    ])\n\n    # configure\n    fig.update_xaxes(autorange=False, range=[0, width_meter])\n    fig.update_yaxes(autorange=False, range=[0, height_meter], scaleanchor=\"x\", scaleratio=1)\n    fig.update_layout(\n        title=go.layout.Title(\n            text=title or \"No title.\",\n            xref=\"paper\",\n            x=0,\n        ),\n        autosize=True,\n        width=900,\n        height=200 + 900 * height_meter / width_meter,\n        template=\"plotly_white\",\n    )\n\n#     if show:\n#         fig.show()\n\n    return fig","metadata":{"execution":{"iopub.status.busy":"2023-05-01T08:04:05.031161Z","iopub.execute_input":"2023-05-01T08:04:05.031755Z","iopub.status.idle":"2023-05-01T08:04:05.052771Z","shell.execute_reply.started":"2023-05-01T08:04:05.03171Z","shell.execute_reply":"2023-05-01T08:04:05.051403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from visualize_f import visualize_trajectory, visualize_heatmap\nbase = '../input/indoor-location-navigation'\npath = f'{base}/train/5cd56b5ae2acfd2d33b58548/1F/5cf20b20718b08000848aa00.txt'\n\n#/kaggle/input/indoor-location-navigation/train/5cd56b5ae2acfd2d33b58548/2F/5cf214b8c852a70008c01605.txt\n# Read in a sample\nexample = read_data_file(path)\n\n# ~~~~~~~~~\n\n# Returns timestamp, x, y values\ntrajectory = example.waypoint\n# Removes timestamp (we only need the coordinates)\ntrajectory = trajectory[:, 1:3]\n\n# Prepare floor_plan coresponding with our example\nsite = path.split(\"/\")[4]\nfloorNo = path.split(\"/\")[5]\nfloor_plan_filename = f'{base}/metadata/{site}/{floorNo}/floor_image.png'\n\n# Prepare width_meter & height_meter\n### (taken from the .json file)\njson_plan_filename = f'{base}/metadata/{site}/{floorNo}/floor_info.json'\nwith open(json_plan_filename) as json_file:\n    json_data = json.load(json_file)\n    \nwidth_meter = json_data[\"map_info\"][\"width\"]\nheight_meter = json_data[\"map_info\"][\"height\"]\n\n# Title\ntitle = \"Example of Waypoint\"\n\n# ~~~~~~~~~\n\n# Finally, let's plot\nvisualize_trajectory(trajectory = trajectory,\n                     floor_plan_filename = floor_plan_filename,\n                     width_meter = width_meter,\n                     height_meter = height_meter,\n                     title = title,\n)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T08:04:05.16834Z","iopub.execute_input":"2023-05-01T08:04:05.168846Z","iopub.status.idle":"2023-05-01T08:04:06.576899Z","shell.execute_reply.started":"2023-05-01T08:04:05.168799Z","shell.execute_reply":"2023-05-01T08:04:06.575957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from main import calibrate_magnetic_wifi_ibeacon_to_position\nfrom main import extract_magnetic_strength\nfrom visualize_f import visualize_heatmap\n\n# Extracting the magnetic strength\nmwi_datas = calibrate_magnetic_wifi_ibeacon_to_position([path])\nmagnetic_strength = extract_magnetic_strength(mwi_datas)\n\nheat_positions = np.array(list(magnetic_strength.keys()))\nheat_values = np.array(list(magnetic_strength.values()))","metadata":{"execution":{"iopub.status.busy":"2023-04-30T17:10:09.325745Z","iopub.execute_input":"2023-04-30T17:10:09.32626Z","iopub.status.idle":"2023-04-30T17:10:10.126984Z","shell.execute_reply.started":"2023-04-30T17:10:09.326217Z","shell.execute_reply":"2023-04-30T17:10:10.125682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# GitHub Libraries\nfrom main import extract_wifi_rssi, extract_wifi_count\n\n# Get WiFi data\nwifi_rssi = extract_wifi_rssi(mwi_datas)\nprint(f'This floor has {len(wifi_rssi.keys())} wifi aps (access points).')\n\nwifi_counts = extract_wifi_count(mwi_datas)\nheat_positions = np.array(list(wifi_counts.keys()))\nheat_values = np.array(list(wifi_counts.values()))\n# filter out positions that no wifi detected\nmask = heat_values != 0\nheat_positions = heat_positions[mask]\nheat_values = heat_values[mask]\n\n# The heatmap\nvisualize_heatmap(heat_positions, \n                  heat_values, \n                  floor_plan_filename, \n                  width_meter, \n                  height_meter, \n                  colorbar_title='count', \n                  title=f'WiFi Count',\n                  g_size=755,\n                  colorscale='temps')","metadata":{"execution":{"iopub.status.busy":"2023-04-30T17:10:10.129534Z","iopub.execute_input":"2023-04-30T17:10:10.130035Z","iopub.status.idle":"2023-04-30T17:10:10.267498Z","shell.execute_reply.started":"2023-04-30T17:10:10.129982Z","shell.execute_reply":"2023-04-30T17:10:10.266308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from main import extract_ibeacon_rssi\n\n# Getting the iBeacon data\nibeacon_rssi = extract_ibeacon_rssi(mwi_datas)\nprint(f'This floor has {len(ibeacon_rssi.keys())} ibeacons.')\nibeacon_ummids = list(ibeacon_rssi.keys())\ntarget_ibeacon = ibeacon_ummids[0]\nheat_positions = np.array(list(ibeacon_rssi[target_ibeacon].keys()))\nheat_values = np.array(list(ibeacon_rssi[target_ibeacon].values()))[:, 0]\n\n# The heatmap\nvisualize_heatmap(heat_positions, \n                  heat_values, \n                  floor_plan_filename, \n                  width_meter, \n                  height_meter, \n                  colorbar_title='dBm', \n                  title='iBeacon RSSE',\n                  colorscale='temps')","metadata":{"execution":{"iopub.status.busy":"2023-04-30T17:10:10.269105Z","iopub.execute_input":"2023-04-30T17:10:10.269724Z","iopub.status.idle":"2023-04-30T17:10:10.394732Z","shell.execute_reply.started":"2023-04-30T17:10:10.269671Z","shell.execute_reply":"2023-04-30T17:10:10.393616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_lgbm(train_perc=0.75, version=1, n_estimators=150, num_leaves=127):\n    '''\n    Training loop\n    '''\n\n    f = open(f\"lgbm_logs_{version}.txt\", \"w+\")\n    lgbm_predictions = []\n    \n    # Log in W&B\n    wandb.log({'n_estimators': n_estimators, 'num_leaves': num_leaves})\n    \n    k = 1\n    for train_path, test_path in zip(train_paths, test_paths):\n\n\n        # --- Read in data ---\n        train_df = pd.read_csv(train_path, index_col=0)\n        train_df = train_df.sample(frac=1, random_state=10)\n\n        # Erase last column (which is \"site_path_timestamp\")\n        test_df = pd.read_csv(test_path, index_col=0).iloc[:, :-1]\n\n        # Sample out training and validation data\n        ### we need to be careful to choose same information for ALL 3 models\n        ### 1 for x, 1 for y and 1 for floor\n\n        train_size = int(len(train_df) * train_perc)\n\n\n        # --- Data Validation ---\n        # Train features + targets\n        X_train = train_df.iloc[:train_size, :-4]\n        y_train_x = train_df.iloc[:train_size, -4]\n        y_train_y = train_df.iloc[:train_size, -3]\n        y_train_f = train_df.iloc[:train_size, -2]\n\n        # Valid features + targets\n        X_valid = train_df.iloc[train_size:, :-4]\n        y_valid_x = train_df.iloc[train_size:, -4]\n        y_valid_y = train_df.iloc[train_size:, -3]\n        y_valid_f = train_df.iloc[train_size:, -2]\n\n\n        # --- Model Training ---\n        lgbm_x = lgb.LGBMRegressor(n_estimators=n_estimators, num_leaves=num_leaves)\n        lgbm_x.fit(X_train, y_train_x)\n\n        lgbm_y = lgb.LGBMRegressor(n_estimators=n_estimators, num_leaves=num_leaves)\n        lgbm_y.fit(X_train, y_train_y)\n\n        lgbm_f = lgb.LGBMClassifier(n_estimators=n_estimators, num_leaves=num_leaves)\n        lgbm_f.fit(X_train, y_train_f)\n\n\n        # --- Model Validation Predictions ---\n        preds_x = lgbm_x.predict(X_valid)\n        preds_y = lgbm_y.predict(X_valid)\n        preds_f = lgbm_f.predict(X_valid).astype(int)\n        \n        mpe = mean_position_error(preds_x, preds_y, preds_f,\n                                  y_valid_x, y_valid_y, y_valid_f)\n        print(\"{} | MPE: {}\".format(k, mpe))\n        # Save logs\n        with open(f\"lgbm_logs_{version}.txt\", 'a+') as f:\n            print(\"{} | MPE: {}\".format(k, mpe), file=f)\n        \n        # Log MPE of this experiment\n        wandb.log({'MPE' : mpe, 'step' : k})\n        \n        k+=1\n\n\n        # --- Model Test Predictions ---\n        test_preds_x = lgbm_x.predict(test_df)\n        test_preds_y = lgbm_y.predict(test_df)\n        test_preds_f = lgbm_f.predict(test_df).astype(int)\n\n        all_test_preds = pd.DataFrame({'floor' : test_preds_f,\n                                       'x' : test_preds_x, \n                                       'y' : test_preds_y})\n        lgbm_predictions.append(all_test_preds)\n    \n    \n    return lgbm_predictions","metadata":{"execution":{"iopub.status.busy":"2023-04-30T11:30:02.449893Z","iopub.execute_input":"2023-04-30T11:30:02.450525Z","iopub.status.idle":"2023-04-30T11:30:02.467072Z","shell.execute_reply.started":"2023-04-30T11:30:02.450486Z","shell.execute_reply":"2023-04-30T11:30:02.466026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(train_path[0], index_col=0)\ntrain_df = train_df.sample(frac=1, random_state=10)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T10:22:25.612515Z","iopub.execute_input":"2023-05-06T10:22:25.612932Z","iopub.status.idle":"2023-05-06T10:22:25.634208Z","shell.execute_reply.started":"2023-05-06T10:22:25.612899Z","shell.execute_reply":"2023-05-06T10:22:25.632437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wifi_sample_path = '/kaggle/input/indoor-navigation-and-location-wifi-features/wifi_features/train/5d27099f03f801723c32511d_1000_train.csv'","metadata":{"execution":{"iopub.status.busy":"2023-04-23T09:27:07.018921Z","iopub.execute_input":"2023-04-23T09:27:07.01932Z","iopub.status.idle":"2023-04-23T09:27:07.025743Z","shell.execute_reply.started":"2023-04-23T09:27:07.019287Z","shell.execute_reply":"2023-04-23T09:27:07.024298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(wifi_sample_path)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T09:27:07.437447Z","iopub.execute_input":"2023-04-23T09:27:07.438156Z","iopub.status.idle":"2023-04-23T09:27:07.590521Z","shell.execute_reply.started":"2023-04-23T09:27:07.438103Z","shell.execute_reply":"2023-04-23T09:27:07.58969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-04-23T09:27:08.071643Z","iopub.execute_input":"2023-04-23T09:27:08.072319Z","iopub.status.idle":"2023-04-23T09:27:08.104273Z","shell.execute_reply.started":"2023-04-23T09:27:08.072269Z","shell.execute_reply":"2023-04-23T09:27:08.103405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_file.wifi","metadata":{"execution":{"iopub.status.busy":"2023-04-19T04:14:39.938207Z","iopub.execute_input":"2023-04-19T04:14:39.938998Z","iopub.status.idle":"2023-04-19T04:14:39.946826Z","shell.execute_reply.started":"2023-04-19T04:14:39.938955Z","shell.execute_reply":"2023-04-19T04:14:39.945427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = glob.glob('../input/indoor-location-navigation/train/*/*/*')[0]","metadata":{"execution":{"iopub.status.busy":"2023-04-18T08:12:44.455997Z","iopub.execute_input":"2023-04-18T08:12:44.456429Z","iopub.status.idle":"2023-04-18T08:12:45.154638Z","shell.execute_reply.started":"2023-04-18T08:12:44.45639Z","shell.execute_reply":"2023-04-18T08:12:45.153339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from io_f import read_data_file\n\nfor path in glob.glob('../input/indoor-location-navigation/train/*/*/*')[:1000]:\n    _, _, _, _, building, floor, f_name = path.split(\"/\")\n    data = read_data_file(path)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_, _, _, _, building, floor, f_name = path.split(\"/\")","metadata":{"execution":{"iopub.status.busy":"2023-04-18T08:13:53.840022Z","iopub.execute_input":"2023-04-18T08:13:53.84045Z","iopub.status.idle":"2023-04-18T08:13:53.845915Z","shell.execute_reply.started":"2023-04-18T08:13:53.840414Z","shell.execute_reply":"2023-04-18T08:13:53.84505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datetime.fromtimestamp(float(sample_file.wifi[:, 0][0])/1000)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:38:14.188151Z","iopub.execute_input":"2023-04-18T07:38:14.188564Z","iopub.status.idle":"2023-04-18T07:38:14.195927Z","shell.execute_reply.started":"2023-04-18T07:38:14.188529Z","shell.execute_reply":"2023-04-18T07:38:14.194978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"float(sample_file.waypoint[:, 0][0]) in sample_file.wifi[:, 0].astype(float)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T04:21:34.95484Z","iopub.execute_input":"2023-04-19T04:21:34.955226Z","iopub.status.idle":"2023-04-19T04:21:34.96398Z","shell.execute_reply.started":"2023-04-19T04:21:34.955195Z","shell.execute_reply":"2023-04-19T04:21:34.962914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path.split(\"/\")[-1][:-4]","metadata":{"execution":{"iopub.status.busy":"2023-04-19T04:30:11.955817Z","iopub.execute_input":"2023-04-19T04:30:11.956391Z","iopub.status.idle":"2023-04-19T04:30:11.963858Z","shell.execute_reply.started":"2023-04-19T04:30:11.956352Z","shell.execute_reply":"2023-04-19T04:30:11.962488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(sample_file.wifi[:, 1] == path.split(\"/\")[-1][:-4]).sum()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T04:31:02.686299Z","iopub.execute_input":"2023-04-19T04:31:02.686721Z","iopub.status.idle":"2023-04-19T04:31:02.695251Z","shell.execute_reply.started":"2023-04-19T04:31:02.686679Z","shell.execute_reply":"2023-04-19T04:31:02.693856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_file.wifi.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-19T04:31:43.212875Z","iopub.execute_input":"2023-04-19T04:31:43.213272Z","iopub.status.idle":"2023-04-19T04:31:43.220075Z","shell.execute_reply.started":"2023-04-19T04:31:43.213239Z","shell.execute_reply":"2023-04-19T04:31:43.219195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(sample_file.wifi[:, 0].astype(float) == float(sample_file.waypoint[:, 0][0])).sum()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T04:31:08.671516Z","iopub.execute_input":"2023-04-19T04:31:08.671926Z","iopub.status.idle":"2023-04-19T04:31:08.679602Z","shell.execute_reply.started":"2023-04-19T04:31:08.671895Z","shell.execute_reply":"2023-04-19T04:31:08.678599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from main import calibrate_magnetic_wifi_ibeacon_to_position\nfrom main import extract_magnetic_strength\n\nmwi_datas = calibrate_magnetic_wifi_ibeacon_to_position([path])","metadata":{"execution":{"iopub.status.busy":"2023-04-19T04:13:06.467424Z","iopub.execute_input":"2023-04-19T04:13:06.468072Z","iopub.status.idle":"2023-04-19T04:13:10.787406Z","shell.execute_reply.started":"2023-04-19T04:13:06.468032Z","shell.execute_reply":"2023-04-19T04:13:10.786234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from main import extract_wifi_rssi, extract_wifi_count\nwifi_rssi = extract_wifi_rssi(mwi_datas)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:58:22.993923Z","iopub.execute_input":"2023-04-18T07:58:22.994585Z","iopub.status.idle":"2023-04-18T07:58:23.008367Z","shell.execute_reply.started":"2023-04-18T07:58:22.994521Z","shell.execute_reply":"2023-04-18T07:58:23.006529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datetime.fromtimestamp(float(sample_file.wifi[0][0])/1000)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:41:58.054405Z","iopub.execute_input":"2023-04-18T07:41:58.054774Z","iopub.status.idle":"2023-04-18T07:41:58.062702Z","shell.execute_reply.started":"2023-04-18T07:41:58.054744Z","shell.execute_reply":"2023-04-18T07:41:58.061595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from datetime import datetime\n\ndatetime.fromtimestamp(sample_file.waypoint[0][0]/1000.0)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T07:22:57.902431Z","iopub.execute_input":"2023-04-18T07:22:57.903015Z","iopub.status.idle":"2023-04-18T07:22:57.909957Z","shell.execute_reply.started":"2023-04-18T07:22:57.902959Z","shell.execute_reply":"2023-04-18T07:22:57.908977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How 1 path looks\nbase = '../input/indoor-location-navigation'\npath = f'{base}/train/5dd1116b94e4900006126296.txt'\n\nwith open(path) as p:\n    lines = p.readlines()\n\nprint(\"No. Lines in 1 example: {:,}\". format(len(lines)), \"\\n\" +\n      \"Example (5 lines): \", lines[0:5])","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:09:54.488283Z","iopub.execute_input":"2023-05-06T15:09:54.488697Z","iopub.status.idle":"2023-05-06T15:09:54.509001Z","shell.execute_reply.started":"2023-05-06T15:09:54.488662Z","shell.execute_reply":"2023-05-06T15:09:54.507428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 여기부터","metadata":{}},{"cell_type":"code","source":"from dataclasses import dataclass\n\nimport numpy as np\n\n\n@dataclass\nclass ReadData:\n    acce: np.ndarray\n    acce_uncali: np.ndarray\n    gyro: np.ndarray\n    gyro_uncali: np.ndarray\n    magn: np.ndarray\n    magn_uncali: np.ndarray\n    ahrs: np.ndarray\n    wifi: np.ndarray\n    ibeacon: np.ndarray\n    waypoint: np.ndarray\n\n\ndef read_data_file(data_filename):\n    acce = []\n    acce_uncali = []\n    gyro = []\n    gyro_uncali = []\n    magn = []\n    magn_uncali = []\n    ahrs = []\n    wifi = []\n    ibeacon = []\n    waypoint = []\n\n    with open(data_filename, 'r', encoding='utf-8') as file:\n        lines = file.readlines()\n\n    for line_data in lines:\n        line_data = line_data.strip()\n        if not line_data or line_data[0] == '#':\n            continue\n\n        line_data = line_data.split('\\t')\n\n        if line_data[1] == 'TYPE_ACCELEROMETER':\n            acce.append([int(line_data[0]), float(line_data[2]), float(line_data[3]), float(line_data[4])])\n            continue\n\n        if line_data[1] == 'TYPE_ACCELEROMETER_UNCALIBRATED':\n            acce_uncali.append([int(line_data[0]), float(line_data[2]), float(line_data[3]), float(line_data[4])])\n            continue\n\n        if line_data[1] == 'TYPE_GYROSCOPE':\n            gyro.append([int(line_data[0]), float(line_data[2]), float(line_data[3]), float(line_data[4])])\n            continue\n\n        if line_data[1] == 'TYPE_GYROSCOPE_UNCALIBRATED':\n            gyro_uncali.append([int(line_data[0]), float(line_data[2]), float(line_data[3]), float(line_data[4])])\n            continue\n\n        if line_data[1] == 'TYPE_MAGNETIC_FIELD':\n            magn.append([int(line_data[0]), float(line_data[2]), float(line_data[3]), float(line_data[4])])\n            continue\n\n        if line_data[1] == 'TYPE_MAGNETIC_FIELD_UNCALIBRATED':\n            magn_uncali.append([int(line_data[0]), float(line_data[2]), float(line_data[3]), float(line_data[4])])\n            continue\n\n        if line_data[1] == 'TYPE_ROTATION_VECTOR':\n            ahrs.append([int(line_data[0]), float(line_data[2]), float(line_data[3]), float(line_data[4])])\n            continue\n\n        if line_data[1] == 'TYPE_WIFI':\n            \n            sys_ts = line_data[0]\n            ssid = line_data[2]\n            bssid = line_data[3]\n            rssi = line_data[4]\n            freq = line_data[5]\n            lastseen_ts = line_data[6]\n            \n            wifi_data = [sys_ts, ssid, bssid, rssi, lastseen_ts]\n            wifi.append(wifi_data)\n            continue\n\n        if line_data[1] == 'TYPE_BEACON':\n            ts = line_data[0]\n            uuid = line_data[2]\n            major = line_data[3]\n            minor = line_data[4]\n            tx_power = line_data[5]\n            rssi = line_data[6]\n            distance = line_data[7]\n            mac_add = line_data[8]\n            sts = line_data[9]\n            \n            # ibeacon_data = [ts, '_'.join([uuid, major, minor]),tx_power, rssi, distance,mac_add,sts]\n            ibeacon_data = [ts, uuid, major, minor, tx_power, rssi, distance,mac_add,sts]\n            ibeacon.append(ibeacon_data)\n            continue\n\n        if line_data[1] == 'TYPE_WAYPOINT':\n            waypoint.append([int(line_data[0]), float(line_data[2]), float(line_data[3])])\n\n    acce = np.array(acce)\n    acce_uncali = np.array(acce_uncali)\n    gyro = np.array(gyro)\n    gyro_uncali = np.array(gyro_uncali)\n    magn = np.array(magn)\n    magn_uncali = np.array(magn_uncali)\n    ahrs = np.array(ahrs)\n    wifi = np.array(wifi)\n    ibeacon = np.array(ibeacon)\n    waypoint = np.array(waypoint)\n\n    return ReadData(acce, acce_uncali, gyro, gyro_uncali, magn, magn_uncali, ahrs, wifi, ibeacon, waypoint)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:40:19.949606Z","iopub.execute_input":"2023-05-06T15:40:19.950054Z","iopub.status.idle":"2023-05-06T15:40:19.975424Z","shell.execute_reply.started":"2023-05-06T15:40:19.950013Z","shell.execute_reply":"2023-05-06T15:40:19.974157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base = '../input/indoor-location-navigation'\npath = f'{base}/test/5dd1116b94e4900006126296.txt'","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:40:20.343501Z","iopub.execute_input":"2023-05-06T15:40:20.343875Z","iopub.status.idle":"2023-05-06T15:40:20.349168Z","shell.execute_reply.started":"2023-05-06T15:40:20.343842Z","shell.execute_reply":"2023-05-06T15:40:20.348035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = read_data_file('/kaggle/input/indoor-location-navigation/train/5a0546857ecc773753327266/F2/5d134069ffe23f000860512b.txt')","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:40:20.725507Z","iopub.execute_input":"2023-05-06T15:40:20.725908Z","iopub.status.idle":"2023-05-06T15:40:22.030746Z","shell.execute_reply.started":"2023-05-06T15:40:20.725854Z","shell.execute_reply":"2023-05-06T15:40:22.029054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(sample.wifi.shape) # ssid\tbssid\tRSSI\tfrequency\tlast seen timestamp\t\t\t\n\n'''\nsys_ts = line_data[0]\nssid = line_data[2]\nbssid = line_data[3]\nrssi = line_data[4]\nfreq = line_data[5]\nlastseen_ts = line_data[6]\n'''","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:40:22.033700Z","iopub.execute_input":"2023-05-06T15:40:22.034245Z","iopub.status.idle":"2023-05-06T15:40:22.042259Z","shell.execute_reply.started":"2023-05-06T15:40:22.034192Z","shell.execute_reply":"2023-05-06T15:40:22.041058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample.wifi[0]","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:40:22.044113Z","iopub.execute_input":"2023-05-06T15:40:22.044464Z","iopub.status.idle":"2023-05-06T15:40:22.063718Z","shell.execute_reply.started":"2023-05-06T15:40:22.044431Z","shell.execute_reply":"2023-05-06T15:40:22.062229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sdf = pd.DataFrame(sample.wifi)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:40:22.065225Z","iopub.execute_input":"2023-05-06T15:40:22.065685Z","iopub.status.idle":"2023-05-06T15:40:22.113586Z","shell.execute_reply.started":"2023-05-06T15:40:22.065645Z","shell.execute_reply":"2023-05-06T15:40:22.112263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sdf","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:40:22.115820Z","iopub.execute_input":"2023-05-06T15:40:22.116242Z","iopub.status.idle":"2023-05-06T15:40:22.132622Z","shell.execute_reply.started":"2023-05-06T15:40:22.116204Z","shell.execute_reply":"2023-05-06T15:40:22.131648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wap = sdf[sdf[2]=='849955db4cb21ba4003b71e5ce616a4fbf3f68e9']","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:40:22.402052Z","iopub.execute_input":"2023-05-06T15:40:22.402611Z","iopub.status.idle":"2023-05-06T15:40:22.414197Z","shell.execute_reply.started":"2023-05-06T15:40:22.402573Z","shell.execute_reply":"2023-05-06T15:40:22.413230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wap[0].nunique()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:40:22.816351Z","iopub.execute_input":"2023-05-06T15:40:22.816765Z","iopub.status.idle":"2023-05-06T15:40:22.825385Z","shell.execute_reply.started":"2023-05-06T15:40:22.816727Z","shell.execute_reply":"2023-05-06T15:40:22.824032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wap.sort_values([0]).reset_index(drop=True,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:40:23.153516Z","iopub.execute_input":"2023-05-06T15:40:23.153882Z","iopub.status.idle":"2023-05-06T15:40:23.158958Z","shell.execute_reply.started":"2023-05-06T15:40:23.153841Z","shell.execute_reply":"2023-05-06T15:40:23.158113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.options.display.float_format = '{:.5f}'.format","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:40:23.705194Z","iopub.execute_input":"2023-05-06T15:40:23.706129Z","iopub.status.idle":"2023-05-06T15:40:23.712396Z","shell.execute_reply.started":"2023-05-06T15:40:23.706082Z","shell.execute_reply":"2023-05-06T15:40:23.710915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wap","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:40:24.226282Z","iopub.execute_input":"2023-05-06T15:40:24.226838Z","iopub.status.idle":"2023-05-06T15:40:24.249666Z","shell.execute_reply.started":"2023-05-06T15:40:24.226801Z","shell.execute_reply":"2023-05-06T15:40:24.248660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sway = pd.DataFrame(sample.waypoint)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:40:24.736619Z","iopub.execute_input":"2023-05-06T15:40:24.737041Z","iopub.status.idle":"2023-05-06T15:40:24.742152Z","shell.execute_reply.started":"2023-05-06T15:40:24.737005Z","shell.execute_reply":"2023-05-06T15:40:24.741059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.options.display.float_format = '{:.5f}'.format\nsway","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:40:25.424793Z","iopub.execute_input":"2023-05-06T15:40:25.425613Z","iopub.status.idle":"2023-05-06T15:40:25.440431Z","shell.execute_reply.started":"2023-05-06T15:40:25.425565Z","shell.execute_reply":"2023-05-06T15:40:25.438880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sway = sway.sort_values([0])\nwap = wap.sort_values([0])","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:40:25.993704Z","iopub.execute_input":"2023-05-06T15:40:25.994345Z","iopub.status.idle":"2023-05-06T15:40:26.001103Z","shell.execute_reply.started":"2023-05-06T15:40:25.994301Z","shell.execute_reply":"2023-05-06T15:40:25.999947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wap.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:40:26.535064Z","iopub.execute_input":"2023-05-06T15:40:26.535420Z","iopub.status.idle":"2023-05-06T15:40:26.551388Z","shell.execute_reply.started":"2023-05-06T15:40:26.535388Z","shell.execute_reply":"2023-05-06T15:40:26.549965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sway.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:40:26.982969Z","iopub.execute_input":"2023-05-06T15:40:26.983327Z","iopub.status.idle":"2023-05-06T15:40:26.997722Z","shell.execute_reply.started":"2023-05-06T15:40:26.983297Z","shell.execute_reply":"2023-05-06T15:40:26.996410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# wap[0] = wap[0].apply(lambda x: datetime.fromtimestamp(int(x)/1000))\n# sway[0] = sway[0].apply(lambda x: datetime.fromtimestamp(int(x)/1000))","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:40:27.696252Z","iopub.execute_input":"2023-05-06T15:40:27.696654Z","iopub.status.idle":"2023-05-06T15:40:27.701469Z","shell.execute_reply.started":"2023-05-06T15:40:27.696621Z","shell.execute_reply":"2023-05-06T15:40:27.700270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wap[0] = wap[0].apply(lambda x: int(x))\nsway[0] = sway[0].apply(lambda x: int(x))","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:40:28.640168Z","iopub.execute_input":"2023-05-06T15:40:28.640544Z","iopub.status.idle":"2023-05-06T15:40:28.648577Z","shell.execute_reply.started":"2023-05-06T15:40:28.640506Z","shell.execute_reply":"2023-05-06T15:40:28.647056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sway.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:40:29.608047Z","iopub.execute_input":"2023-05-06T15:40:29.608398Z","iopub.status.idle":"2023-05-06T15:40:29.619364Z","shell.execute_reply.started":"2023-05-06T15:40:29.608368Z","shell.execute_reply":"2023-05-06T15:40:29.617962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wap.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:40:30.108462Z","iopub.execute_input":"2023-05-06T15:40:30.108843Z","iopub.status.idle":"2023-05-06T15:40:30.122401Z","shell.execute_reply.started":"2023-05-06T15:40:30.108812Z","shell.execute_reply":"2023-05-06T15:40:30.121270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wap['x'] = ''\nwap['y'] = ''","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:40:30.908845Z","iopub.execute_input":"2023-05-06T15:40:30.909240Z","iopub.status.idle":"2023-05-06T15:40:30.917481Z","shell.execute_reply.started":"2023-05-06T15:40:30.909206Z","shell.execute_reply":"2023-05-06T15:40:30.916054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wap.reset_index(drop=True,inplace=True)\nsway.reset_index(drop=True,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:40:31.692179Z","iopub.execute_input":"2023-05-06T15:40:31.692600Z","iopub.status.idle":"2023-05-06T15:40:31.698553Z","shell.execute_reply.started":"2023-05-06T15:40:31.692566Z","shell.execute_reply":"2023-05-06T15:40:31.697183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def nearest(items,pivot):\n    dif = (items).apply(lambda x: abs(x-pivot))\n    near = np.argmin(dif)\n    return near","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:45:13.814524Z","iopub.execute_input":"2023-05-06T15:45:13.815222Z","iopub.status.idle":"2023-05-06T15:45:13.820123Z","shell.execute_reply.started":"2023-05-06T15:45:13.815180Z","shell.execute_reply":"2023-05-06T15:45:13.819185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.argmin(items.apply(lambda x: abs(x-100)))","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:45:59.250480Z","iopub.execute_input":"2023-05-06T15:45:59.250858Z","iopub.status.idle":"2023-05-06T15:45:59.260703Z","shell.execute_reply.started":"2023-05-06T15:45:59.250813Z","shell.execute_reply":"2023-05-06T15:45:59.259468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"items = sway[[0]]\n\nfor i in range(len(wap)):\n    row_time = int(wap.iloc[i,0])\n    # print(items,row_time)\n    nearest = nearest(items,row_time)\n    # print(nearest)\n    wap['x'][i] = sway[1][nearest]    \n    wap['y'][i] = sway[2][nearest]\n    print('done')","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:44:54.569648Z","iopub.execute_input":"2023-05-06T15:44:54.570125Z","iopub.status.idle":"2023-05-06T15:44:54.617640Z","shell.execute_reply.started":"2023-05-06T15:44:54.570050Z","shell.execute_reply":"2023-05-06T15:44:54.616053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wap","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:53.539141Z","iopub.execute_input":"2023-05-06T16:06:53.539559Z","iopub.status.idle":"2023-05-06T16:06:53.563405Z","shell.execute_reply.started":"2023-05-06T16:06:53.539526Z","shell.execute_reply":"2023-05-06T16:06:53.562171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datetime.fromtimestamp(int(1561454341262)/1000)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:20:23.089564Z","iopub.execute_input":"2023-05-06T15:20:23.090024Z","iopub.status.idle":"2023-05-06T15:20:23.096513Z","shell.execute_reply.started":"2023-05-06T15:20:23.089984Z","shell.execute_reply":"2023-05-06T15:20:23.095556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datetime.fromtimestamp(int(1561454339274)/1000)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T15:20:29.658123Z","iopub.execute_input":"2023-05-06T15:20:29.658715Z","iopub.status.idle":"2023-05-06T15:20:29.666495Z","shell.execute_reply.started":"2023-05-06T15:20:29.658677Z","shell.execute_reply":"2023-05-06T15:20:29.664979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## I. Light GBM","metadata":{}},{"cell_type":"code","source":"# Import Libraries\nimport lightgbm as lgb\n\n# ~~~~\n# Data\n# ~~~~\nbase_dir = \"../input/indoor-navigation-and-location-wifi-features/wifi_features\"\ntrain_dir = \"/train/*_train.csv\"\ntest_dir = \"/test/*_test.csv\"\n\n\n# Paths for train & test files\ntrain_paths = sorted(glob.glob(base_dir + train_dir))\ntest_paths = sorted(glob.glob(base_dir + test_dir))\nsample_subm = pd.read_csv('../input/indoor-location-navigation/sample_submission.csv',\n                          index_col=0)\n\nprint(\"Len Train Files: {}\".format(len(train_paths)), \"\\n\" +\n      \"Len Test Files: {}\".format(len(test_paths)))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize new experiment (LGBM)\nrun = wandb.init(project=\"indoor-location-kaggle\", name=\"lgbm_train\")\n\nwandb.log({'Len Train Files' : len(train_paths),\n           'Len Test Files' : len(test_paths)})","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Schema of the \"Training loop\" (inspired by [Jiwei Liu's work](https://www.kaggle.com/jiweiliu)):\n<img src=\"https://i.imgur.com/UQmdRcz.png\" width=750>\n\n#### Code Below","metadata":{}},{"cell_type":"code","source":"def train_lgbm(train_perc=0.75, version=1, n_estimators=150, num_leaves=127):\n    '''\n    Training loop\n    '''\n\n    f = open(f\"lgbm_logs_{version}.txt\", \"w+\")\n    lgbm_predictions = []\n    \n    # Log in W&B\n    wandb.log({'n_estimators': n_estimators, 'num_leaves': num_leaves})\n    \n    k = 1\n    for train_path, test_path in zip(train_paths, test_paths):\n\n\n        # --- Read in data ---\n        train_df = pd.read_csv(train_path, index_col=0)\n        train_df = train_df.sample(frac=1, random_state=10)\n\n        # Erase last column (which is \"site_path_timestamp\")\n        test_df = pd.read_csv(test_path, index_col=0).iloc[:, :-1]\n\n        # Sample out training and validation data\n        ### we need to be careful to choose same information for ALL 3 models\n        ### 1 for x, 1 for y and 1 for floor\n\n        train_size = int(len(train_df) * train_perc)\n\n\n        # --- Data Validation ---\n        # Train features + targets\n        X_train = train_df.iloc[:train_size, :-4]\n        y_train_x = train_df.iloc[:train_size, -4]\n        y_train_y = train_df.iloc[:train_size, -3]\n        y_train_f = train_df.iloc[:train_size, -2]\n\n        # Valid features + targets\n        X_valid = train_df.iloc[train_size:, :-4]\n        y_valid_x = train_df.iloc[train_size:, -4]\n        y_valid_y = train_df.iloc[train_size:, -3]\n        y_valid_f = train_df.iloc[train_size:, -2]\n\n\n        # --- Model Training ---\n        lgbm_x = lgb.LGBMRegressor(n_estimators=n_estimators, num_leaves=num_leaves)\n        lgbm_x.fit(X_train, y_train_x)\n\n        lgbm_y = lgb.LGBMRegressor(n_estimators=n_estimators, num_leaves=num_leaves)\n        lgbm_y.fit(X_train, y_train_y)\n\n        lgbm_f = lgb.LGBMClassifier(n_estimators=n_estimators, num_leaves=num_leaves)\n        lgbm_f.fit(X_train, y_train_f)\n\n\n        # --- Model Validation Predictions ---\n        preds_x = lgbm_x.predict(X_valid)\n        preds_y = lgbm_y.predict(X_valid)\n        preds_f = lgbm_f.predict(X_valid).astype(int)\n        \n        mpe = mean_position_error(preds_x, preds_y, preds_f,\n                                  y_valid_x, y_valid_y, y_valid_f)\n        print(\"{} | MPE: {}\".format(k, mpe))\n        # Save logs\n        with open(f\"lgbm_logs_{version}.txt\", 'a+') as f:\n            print(\"{} | MPE: {}\".format(k, mpe), file=f)\n        \n        # Log MPE of this experiment\n        wandb.log({'MPE' : mpe, 'step' : k})\n        \n        k+=1\n\n\n        # --- Model Test Predictions ---\n        test_preds_x = lgbm_x.predict(test_df)\n        test_preds_y = lgbm_y.predict(test_df)\n        test_preds_f = lgbm_f.predict(test_df).astype(int)\n\n        all_test_preds = pd.DataFrame({'floor' : test_preds_f,\n                                       'x' : test_preds_x, \n                                       'y' : test_preds_y})\n        lgbm_predictions.append(all_test_preds)\n    \n    \n    return lgbm_predictions","metadata":{"_kg_hide-input":true,"_kg_hide-output":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training","metadata":{}},{"cell_type":"code","source":"# Uncomment line below to train your own model\n# lgbm_predictions = train_lgbm(train_perc = 0.75, version=1)\n\n# # Logs from my training:\nprint(open('../input/indoor-locationnavigation-2021/lgbm_logs_1.txt', \"r\").read())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> ❗**Attention**: *5th* dataframe had a BIG error (jumped from ~4 on average to 18). This case HAS to be taken into consideration, as the models seems to be underfitting. The *10th, 13th, 21st* have big MPE as well.\n\n### Submission LGBM","metadata":{}},{"cell_type":"code","source":"# Uncomment line below to make your own submission\n# make_submission(lgbm_predictions, sample_subm, name=\"lgbm_base.csv\")\n\n# My submission:\nlgbm_predictions = pd.read_csv(\"../input/indoor-locationnavigation-2021/lgbm_base.csv\")\nlgbm_predictions.to_csv(\"lgbm_base.csv\", index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ~ END of EXPERIMENT ~\nwandb.finish()\n# ~~~~~~~~~~~~~~~~~~~~~","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## II. XGBoost - Faster with RAPIDS\n\n> 📌 **Note**: I will use a combination of **RAPIDS** libraries on GPU and XGBoost as one of my base models. More information on this open source suite of libraries [here](https://rapids.ai/).","metadata":{}},{"cell_type":"code","source":"# Libraries\nimport cudf\nimport cupy\nimport cuml\nimport xgboost\n\n# Adjust floor function\n### As the Multiclass XGBoost takes only labels between [0, n)\n### But we have negative floor values\ndef adjust_floor(df, col_name):\n    '''Adjusts the floor to be >= 0.\n    Also returns the number fo classes (also used to complete classification).'''\n    num_classes = df[col_name].nunique()\n    smallest = df[col_name].unique().min()\n    df[col_name] = df[col_name] - smallest\n    \n    return df[col_name], num_classes, smallest","metadata":{"_kg_hide-input":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize new experiment (XGB)\nrun = wandb.init(project=\"indoor-location-kaggle\", name=\"xgb_train\")\n\nwandb.log({'Len Train Files' : len(train_paths),\n           'Len Test Files' : len(test_paths)})","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-05-01T07:43:56.893149Z","iopub.execute_input":"2023-05-01T07:43:56.893539Z","iopub.status.idle":"2023-05-01T07:43:57.002639Z","shell.execute_reply.started":"2023-05-01T07:43:56.893507Z","shell.execute_reply":"2023-05-01T07:43:57.001387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_xgb(train_perc=0.75, version=1):\n    '''\n    Training loop\n    '''\n\n    f = open(f\"xgb_logs_{version}.txt\", \"w+\")\n    xgb_predictions = []\n    \n    \n    k = 1\n    for train_path, test_path in zip(train_paths, test_paths):\n\n\n        # --- Read in data ---\n        train_df = cudf.read_csv(train_path, index_col=0)\n        train_df = train_df.sample(frac=1, random_state=10)\n        train_df[\"f\"], num_classes, smallest = adjust_floor(train_df, 'f')\n\n        # Erase last column (which is \"site_path_timestamp\")\n        test_df = cudf.read_csv(test_path, index_col=0).iloc[:, :-1]\n\n        # Sample out training and validation data\n        ### we need to be careful to choose same information for ALL 3 models\n        ### 1 for x, 1 for y and 1 for floor\n\n        train_size = int(len(train_df) * train_perc)\n\n\n        # --- Data Validation ---\n        # Train features + targets\n        X_train = train_df.iloc[:train_size, :-4]\n        y_train_x = train_df.iloc[:train_size, -4]\n        y_train_y = train_df.iloc[:train_size, -3]\n        y_train_f = train_df.iloc[:train_size, -2]\n\n        # Valid features + targets\n        X_valid = train_df.iloc[train_size:, :-4]\n        y_valid_x = cupy.asanyarray(train_df.iloc[train_size:, -4])\n        y_valid_y = cupy.asanyarray(train_df.iloc[train_size:, -3])\n        y_valid_f = cupy.asanyarray(train_df.iloc[train_size:, -2])\n        \n        \n        # --- Parameters ---\n        regr_params = {'max_depth' : 4, 'max_leaves' : 2**4, \n                       'tree_method' : 'gpu_hist', 'objective' : 'reg:squarederror',\n                       'grow_policy' : 'lossguide', 'colsample_bynode': 0.8}\n        classif_params = {'max_depth' : 4, 'max_leaves' : 2**4,\n                          'tree_method' : 'gpu_hist', 'objective' : 'multi:softmax',\n                          'num_class' : num_classes, 'grow_policy' : 'lossguide',\n                          'colsample_bynode': 0.8, 'verbosity' : 0}\n        \n        # Log once to W&B\n        if k == 1:\n            wandb.log(regr_params)\n            wandb.log(classif_params)\n\n\n        # --- Model Training ---\n        trainMatrix_x = xgboost.DMatrix(data=X_train, label=y_train_x)\n        xgboost_x = xgboost.train(params=regr_params, dtrain=trainMatrix_x)\n\n        trainMatrix_y = xgboost.DMatrix(data=X_train, label=y_train_y)\n        xgboost_y = xgboost.train(params=regr_params, dtrain=trainMatrix_y)\n\n        trainMatrix_f = xgboost.DMatrix(data=X_train, label=y_train_f)\n        xgboost_f = xgboost.train(params=classif_params, dtrain=trainMatrix_f)\n\n\n        # --- Model Validation Predictions ---\n        preds_x = cupy.asanyarray(xgboost_x.predict(xgboost.DMatrix(X_valid)))\n        preds_y = cupy.asanyarray(xgboost_y.predict(xgboost.DMatrix(X_valid)))\n        preds_f = cupy.asanyarray(xgboost_f.predict(xgboost.DMatrix(X_valid)).astype(int))\n\n        mpe = mean_position_error_gpu(preds_x, preds_y, preds_f,\n                                      y_valid_x, y_valid_y, y_valid_f)\n        print(\"{} | MPE: {}\".format(k, mpe))\n        # Save logs\n        with open(f\"xgb_logs_{version}.txt\", 'a+') as f:\n            print(\"{} | MPE: {}\".format(k, mpe), file=f)\n        \n        # Log MPE of this experiment\n        wandb.log({'MPE' : mpe, 'step' : k})\n        \n        k+=1\n\n\n        # --- Model Test Predictions ---\n        test_preds_x = xgboost_x.predict(xgboost.DMatrix(test_df))\n        test_preds_y = xgboost_y.predict(xgboost.DMatrix(test_df))\n        test_preds_f = xgboost_f.predict(xgboost.DMatrix(test_df)).astype(int) + smallest\n\n        all_test_preds = pd.DataFrame({'floor' : test_preds_f,\n                                       'x' : test_preds_x, \n                                       'y' : test_preds_y})\n        xgb_predictions.append(all_test_preds)\n    \n    \n    return xgb_predictions","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training","metadata":{}},{"cell_type":"code","source":"# Uncomment line below to train your own model\n# xgb_predictions = train_xgb(train_perc=0.75, version=1)\n\n# Logs from my training:\nprint(open('../input/indoor-locationnavigation-2021/xgb_logs_1.txt', \"r\").read())","metadata":{"_kg_hide-input":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Submission","metadata":{}},{"cell_type":"code","source":"# Uncomment line below to make your own submission\n# make_submission(xgb_predictions, sample_subm, name=\"xgb_base.csv\")\n\n# My submission:\nxgb_predictions = pd.read_csv(\"../input/indoor-locationnavigation-2021/xgb_base.csv\")\nxgb_predictions.to_csv(\"xgb_base.csv\", index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ~ END of EXPERIMENT ~\nwandb.finish()\n# ~~~~~~~~~~~~~~~~~~~~~","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. Save W&B Submissions and Logs\n\n> We can save the predictions and logs to W&B.","metadata":{}},{"cell_type":"code","source":"# Submissions\n### we can save the predictions in W&B\nrun = wandb.init(project='indoor-location-kaggle', name='submissions')\nartifact = wandb.Artifact(name='submissions', \n                          type='dataset')\n\nartifact.add_file(\"../input/indoor-locationnavigation-2021/lgbm_base.csv\")\nartifact.add_file(\"../input/indoor-locationnavigation-2021/xgb_base.csv\")\n\nwandb.log_artifact(artifact)\nwandb.finish()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Logs\n### we can save the predictions in W&B\nrun = wandb.init(project='indoor-location-kaggle', name='training_logs')\nartifact = wandb.Artifact(name='training_logs', \n                          type='dataset')\n\nartifact.add_file(\"../input/indoor-locationnavigation-2021/lgbm_logs_1.txt\")\nartifact.add_file(\"../input/indoor-locationnavigation-2021/xgb_logs_1.txt\")\n\nwandb.log_artifact(artifact)\nwandb.finish()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> The Artifact section of the project:\n<img src=\"https://i.imgur.com/KqzJuFL.png\">\n\n# 🥙 Blending Stirring Cooking \n\n> First let's compare the 2 model's predictions. (needs more work)","metadata":{}},{"cell_type":"code","source":"# # Read in data\n# lgb_preds = pd.read_csv(\"../input/indoor-locationnavigation-2021/lgbm_base.csv\")\n# xgb_preds = pd.read_csv(\"../input/indoor-locationnavigation-2021/xgb_base.csv\")\n\n# sample_submission = pd.read_csv(\"../input/indoor-location-navigation/sample_submission.csv\")\n\n# # Sample Submission\n# sample_submission[\"x\"] = lgb_preds[\"x\"] * 0.9 + xgb_preds[\"x\"] * 0.1\n# sample_submission[\"y\"] = lgb_preds[\"y\"] * 0.9 + xgb_preds[\"y\"] * 0.1\n\n# sample_submission.to_csv(\"blend1.csv\", index=False)","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<img src=\"https://i.imgur.com/cUQXtS7.png\">\n\n# ⌨️🎨 Specs on how I prepped & trained \n### (on my local machine)\n* Z8 G4 Workstation 🖥\n* 2 CPUs & 96GB Memory 💾\n* NVIDIA Quadro RTX 8000 🎮\n* RAPIDS version 0.17 🏃🏾‍♀️","metadata":{}}]}