{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Floor estimation by WIFI timestamp","metadata":{}},{"cell_type":"markdown","source":"The floor of test data can be estimated from the timestamp of TYPE_WIFI or TYPE_BEACON. This is because the timestamp of the test data is a relative value, but timestamp of them areabsolute values and can be compared with the train data.  \nImagine an investigator walking around the floor of a building. Do one floor and go to the next floor ... It is considered that the timestamps of them are close to each other for the data on the same floor.   \nIf this inference is correct, the test data floor is likely to be the same as train data with a value close to its TYPE_WIFI timestamp.","metadata":{}},{"cell_type":"markdown","source":"<h1 style='background:navy; border:0; color:white'><center>1. Preparation</center></h1>","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom glob import glob\nimport re\nimport os\nimport sys\nimport shutil\nimport warnings; warnings.simplefilter('ignore')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_DIR = '/kaggle/input/indoor-location-navigation/train'  # directory where train data is stored\nORG_TEST_DIR = '/kaggle/input/indoor-location-navigation/test'  # directory where test data is stored\nTEST_DIR = '/kaggle/working/test/'  # directory for classifying and storing test data by siteID","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### In order to predict the floor for each site, divide the test into folders by siteID.","metadata":{}},{"cell_type":"code","source":"test_files = [file for file in glob(ORG_TEST_DIR + '/*.txt')]\n\n# Create parent directory\nos.mkdir(TEST_DIR)\n\n# Create a directory for each siteID and copy the test files\nfor file in test_files:\n    with open(file, encoding='utf-8') as f:\n        for line in f:\n            if 'SiteID:' in line:\n                siteid = line.split('\\t')[1].replace('SiteID:', '')\n                site_dir = os.path.join(TEST_DIR, siteid)\n                if not os.path.isdir(site_dir):\n                    os.mkdir(site_dir)\n                break\n    shutil.copy(file, site_dir)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Define functions","metadata":{}},{"cell_type":"code","source":"# function to convert floor to number\ndef floor2int(floor):\n    if re.fullmatch('F\\d', floor) or re.fullmatch('\\dF', floor):\n        return int(re.sub('F', '', floor)) - 1\n    elif re.fullmatch('B\\d', floor) or re.fullmatch('\\dB', floor):\n        return -int(re.sub('B', '', floor))\n    elif re.fullmatch('L\\d', floor) or re.fullmatch('\\dL', floor):\n        return int(re.sub('L', '', floor)) - 1\n    \n# function to extract timestamp of TYPE_WIFI\ndef extract_wifi_time(file):\n    wifi_times = []\n    with open(file, encoding='utf-8') as f:\n        for line in f:\n            if 'TYPE_WIFI' in line:\n                wifi_times.append(line.split('\\t')[6].strip())\n    return wifi_times\n\n# function to extract timestamp of TYPE_WIFI (version 2)\ndef extract_wifi_time2(file):\n    wifi_times = []\n    with open(file, encoding='utf-8') as f:\n        for line in f:\n            if line.split('\\t')[1] == 'TYPE_WIFI':\n                wifi_times.append(line.split('\\t')[6].strip())\n                break\n    return wifi_times\n\n# function to plot\ndef stripplot(data, ylim=(0,1)):\n    fig, ax = plt.subplots(1, 1, figsize=(25, 6))\n    ax.ticklabel_format(useOffset=False, style='plain')\n    _ = sns.stripplot(data=data, x='floor_or_path', y='time', ax=ax)\n    # set axis\n    plt.xticks(rotation=45)\n    if ylim != (0,1):\n        ax.set_ylim(ylim)\n    # set yticks\n    start, end = ax.get_ylim()\n    stepsize = int((end - start) / 25)\n    ax.yaxis.set_ticks(pd.np.arange(start, end, stepsize))\n    ax.grid()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style='background:navy; border:0; color:white'><center>2. Example</center></h1>","metadata":{}},{"cell_type":"markdown","source":"Let's check how the timestamp of TYPE_WIFI is distributed in the data of one site. Here, as an example, the data of siteid: 5d27097f03f801723c320d97 is used.","metadata":{}},{"cell_type":"code","source":"%%time\n### create a DataFrame for each floor\nsiteid = '5d27097f03f801723c320d97'\nfloors = os.listdir(os.path.join(TRAIN_DIR, siteid))\ntest_files = [file for file in glob(os.path.join(TEST_DIR, siteid, '*.txt'))]\n\n# convert train data to DataFrame\ntrain_df = pd.DataFrame(columns=['time', 'floor_or_path'])\nfor floor in floors:\n    train_files = [file for file in glob(os.path.join(TRAIN_DIR, siteid, floor, '*.txt'))]\n    for train_file in train_files:\n        train_wifi_time = extract_wifi_time(train_file)\n        train_df = pd.concat([train_df, pd.DataFrame(data={'time': train_wifi_time, 'floor_or_path': [floor]*len(train_wifi_time)})])\n    \n# convert test data to DataFrame\ntest_df = pd.DataFrame(columns=['time', 'floor_or_path'])\nfor test_file in test_files:\n    test_wifi_time = extract_wifi_time(test_file)\n    file_name = test_file.split('/')[5].replace('.txt', '')\n    test_df = pd.concat([test_df, pd.DataFrame(data={'time': test_wifi_time, 'floor_or_path': [file_name] * len(test_wifi_time)})])\n\n# combine train_df and test_df\ndf = pd.concat([train_df, test_df], ignore_index=True)\ndf['time'] = df['time'].astype('int64')\ndf.drop_duplicates(inplace=True, ignore_index=True)\ndf = df.sort_values('time').reset_index(drop=True)\ndf","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's plot it. The floor name is train data and the path name is test data.\nstripplot(data=df)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It seems that one floor may be surveyed multiple times.","metadata":{}},{"cell_type":"code","source":"# Timestamp zooms in on small elements for clarity (excluding the last four)\ndf_tmp = df.query(\"floor_or_path in ['F1', 'B1', 'B2', '7727672abec7d70216173223', '698dc1d1a1885908e8fbfe4c', 'F2', '4fbd93217986b45372ebedd4']\")\nstripplot(data=df_tmp, ylim=(1573970000000, 1573990000000))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The time zones do not overlap between different floors. Also, the test data appears to be included in the timestamp of either train data. The floor of 7727 ~ and 698d ~ seems to be B2, and 4fbd seems to be F2.","metadata":{}},{"cell_type":"code","source":"# Try expanding the elements with large timestamps\ndf_tmp = df.query(\"floor_or_path not in ['F1', 'B1', 'B2', '7727672abec7d70216173223', '698dc1d1a1885908e8fbfe4c', 'F2', '4fbd93217986b45372ebedd4']\")\nstripplot(data=df_tmp, ylim=(1574042000000, 1574060000000))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"After all, the time zones do not overlap between different floors. Also, any test data appears to be included in the timestamp of any train data.  \nea6f ~ to b760 ~ seems to be F3, and d505 ~ to bf55 ~ floor seems to be F4.","metadata":{}},{"cell_type":"markdown","source":"<h1 style='background:navy; border:0; color:white'><center>3. Accuracy</center></h1>","metadata":{}},{"cell_type":"code","source":"# create DataFrame of train and test data\nsiteids = os.listdir(TEST_DIR)\ndf_train_files = pd.DataFrame(columns=['siteid', 'floor', 'path'])\ndf_test_files = pd.DataFrame(columns=['siteid', 'path'])\n\nfor siteid in siteids:\n    # train\n    train_files = glob(os.path.join(TRAIN_DIR, siteid, '*/*.txt'))\n    df_tmp = pd.DataFrame([file.split('/') for file in train_files])\n    df_tmp.drop([0,1,2,3,4], axis=1, inplace=True)\n    df_tmp.columns = ['siteid', 'floor', 'path']\n    df_tmp['path'] = df_tmp['path'].map(lambda x: x.replace('.txt', ''))\n    df_tmp['file_path'] = train_files\n    df_train_files = pd.concat([df_train_files, df_tmp])\n    \n    # test\n    test_files = glob(os.path.join(TEST_DIR, siteid, '*'))\n    df_tmp = pd.DataFrame([file.split('/') for file in test_files])\n    df_tmp.drop([0,1,2,3], axis=1, inplace=True)\n    df_tmp.columns = ['siteid', 'path']\n    df_tmp['path'] = df_tmp['path'].map(lambda x: x.replace('.txt', ''))\n    df_tmp['file_path'] = test_files\n    df_test_files = pd.concat([df_test_files, df_tmp])\n    \ndf_train_files = df_train_files.reset_index(drop=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len_floor_paths = 0\naccs = 0\n\nfor siteid in siteids:\n    df_train_files_onesite = df_train_files[df_train_files['siteid'] == siteid].reset_index(drop=True)\n    # Create a path and wifi_time pair\n    paths = df_train_files_onesite['path']\n    tmp_df = pd.DataFrame(columns=['path', 'wifi_time'])\n    for i, file in enumerate(df_train_files_onesite['file_path']):\n        wifi_time = extract_wifi_time2(file)\n        path = [paths[i]] * len(wifi_time)\n        tmp_df = pd.concat([tmp_df, pd.DataFrame(data={'path': path, 'wifi_time': wifi_time})])\n    tmp_df['wifi_time'] = tmp_df['wifi_time'].astype('float')\n    tmp_df = tmp_df.sort_values(by='wifi_time')\n    val_df = pd.merge(df_train_files_onesite, tmp_df, on='path')\n    floor_path_df = val_df.sort_values('wifi_time')[['floor', 'path', 'wifi_time']].set_index('path')\n    floor_path_df['floor_shift'] = floor_path_df.shift(1)['floor']\n    len_floor_paths += len(floor_path_df)\n    accs += len(floor_path_df[floor_path_df['floor'] == floor_path_df['floor_shift']])\n    \nprint('Accuracy:', accs/len_floor_paths)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style='background:navy; border:0; color:white'><center>4. Floor estimation</center></h1>","metadata":{}},{"cell_type":"markdown","source":"Estimate the floor by arranging each site in the order of timestamp and filling the floor of test data with the floor of the previous train data.","metadata":{}},{"cell_type":"code","source":"%%time\n# add dataset column\ndf_train_files['dataset'] = 'train'\ndf_test_files['dataset'] = 'test'\n\n# combine df_train_files and df_test_files\ndf_all = pd.concat([df_train_files, df_test_files]).reset_index(drop=True)\n\n# add wifi_time\npaths = df_all['path']\n# create a path and wifi_time pair\ntmp_df = pd.DataFrame(columns=['path', 'wifi_time'])\nfor i, file in enumerate(df_all['file_path']):\n    wifi_time = extract_wifi_time2(file)\n    path = [paths[i]] * len(wifi_time)\n    tmp_df = pd.concat([tmp_df, pd.DataFrame(data={'path': path, 'wifi_time': wifi_time})])\ntmp_df['wifi_time'] = tmp_df['wifi_time'].astype('float')\n\nwifi_time_df = pd.merge(df_all, tmp_df, on='path')\nwifi_time_df = wifi_time_df.sort_values('wifi_time')\nwifi_time_df = wifi_time_df.reset_index(drop=True)\n\nfloor_path_df = wifi_time_df[['floor', 'path', 'dataset']].fillna(method='ffill')\nfloor_path_df = floor_path_df[floor_path_df['dataset'] == 'test']\nfloor_path_df = floor_path_df.drop('dataset', axis=1).reset_index(drop=True)\nfloor_path_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# convert floor names to numbers\nfloor_path_df['floor'] = floor_path_df['floor'].map(floor2int)\nfloor_path_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Create submission file","metadata":{}},{"cell_type":"code","source":"submission_org = pd.read_csv('/kaggle/input/indoor-location-navigation/sample_submission.csv')\n# separate site_path and timestamp\ntmp_df = submission_org['site_path_timestamp'].str.split('_', expand=True)\ntmp_df.columns = ['site', 'path', 'timestamp']\nsubmission = pd.concat([tmp_df, submission_org[['site_path_timestamp', 'floor', 'x', 'y']]], axis=1)\n# merge with floor_path_df\nsubmission = submission.merge(floor_path_df, on='path')\nsubmission.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.drop(['site', 'path', 'timestamp', 'floor_x'], axis=1, inplace=True)\nsubmission.columns = ['site_path_timestamp', 'x', 'y', 'floor']\nsubmission = submission[['site_path_timestamp', 'floor', 'x', 'y']]\nsubmission","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The submission file has been created. Of course, you need to estimate x and y separately.","metadata":{}}]}