{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Two-step snapping. \n1. Snap to grid like https://www.kaggle.com/robikscube/indoor-navigation-snap-to-grid-post-processing with some conservative threshold\n\n2. Snap remaining to white space like https://www.kaggle.com/rafaelcartenet/how-to-use/data ","metadata":{}},{"cell_type":"markdown","source":"Paul's version. Forked from here: https://www.kaggle.com/robikscube/indoor-navigation-snap-to-grid-post-processing \n\nPrimary modification is to get \"fair\" CV set. The set of waypoints needs to be calculated from out-of-fold data only.","metadata":{}},{"cell_type":"code","source":"# Helper Functions\nimport pandas as pd\nimport numpy as np\nimport pickle\nimport os\n\nfrom tqdm import tqdm\n\nimport json\nimport matplotlib.pylab as plt\n\nfrom scipy.spatial.distance import cdist\n\nfrom shapely.ops import nearest_points\nfrom shapely.geometry import Point","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"THRESH = 6\n# input_path = \"../input/05-interpolate-output\"\n# input_path = \"../input/m1018-05-v4-output\"\n\n# cv_input_path = \"../input/m1018-05-v4-output/val_w_preds.csv\"\n# sub_input_path = \"../input/m1018-05-v4-output/submission.csv\"\n\n# cv_input_path =  \"../input/03-10-leaky-anchor-output/output_cv.csv\"\n# sub_input_path = \"../input/03-10-leaky-anchor-output/output_sub.csv\"\n\ncv_input_path =  \"../input/04-cost-min-output/submission_cv.csv\"\nsub_input_path = \"../input/04-cost-min-output/submission.csv\"\n\nshape_dir = \"/kaggle/input/indoor-location-navigation-scaled-geojson/scaled_geojson\"\n\nsmooth_snap_changes = False\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def split_col(df):\n    df = pd.concat([\n        df['site_path_timestamp'].str.split('_', expand=True) \\\n        .rename(columns={0:'site',\n                         1:'path',\n                         2:'timestamp'}),\n        df\n    ], axis=1).copy()\n    return df\n\nfloor_map = {\"B2\":-2, \"B1\":-1, \"F1\":0, \"F2\": 1, \"F3\":2,\n             \"F4\":3, \"F5\":4, \"F6\":5, \"F7\":6,\"F8\":7,\"F9\":8,\n             \"1F\":0, \"2F\":1, \"3F\":2, \"4F\":3, \"5F\":4, \"6F\":5,\n             \"7F\":6, \"8F\": 7, \"9F\":8}\n\ndef plot_preds(\n    site,\n    floorNo,\n    sub=None,\n    true_locs=None,\n    base=\"../input/indoor-location-navigation\",\n    show_train=True,\n    show_preds=True,\n    fix_labels=True,\n    map_floor=None\n):\n    \"\"\"\n    Plots predictions on floorplan map.\n    \n    map_floor : use a different floor's map\n    \"\"\"\n    floor_int = floor_map[floorNo]\n    \n    if map_floor is None:\n        map_floor = floorNo\n    # Prepare width_meter & height_meter (taken from the .json file)\n    floor_plan_filename = f\"{base}/metadata/{site}/{map_floor}/floor_image.png\"\n    json_plan_filename = f\"{base}/metadata/{site}/{map_floor}/floor_info.json\"\n    with open(json_plan_filename) as json_file:\n        json_data = json.load(json_file)\n\n    width_meter = json_data[\"map_info\"][\"width\"]\n    height_meter = json_data[\"map_info\"][\"height\"]\n\n    floor_img = plt.imread(f\"{base}/metadata/{site}/{map_floor}/floor_image.png\")\n\n    fig, ax = plt.subplots(figsize=(12, 12))\n    plt.imshow(floor_img)\n\n    if show_train:\n#         true_locs = true_locs.query('site == @site and floor == @floor_int').copy()\n        true_locs = true_locs.query('site == @site').copy()\n        true_locs = true_locs[true_locs['floor'] == floor_int]\n        true_locs[\"x_\"] = true_locs[\"x\"] * floor_img.shape[0] / height_meter\n        true_locs[\"y_\"] = (\n            true_locs[\"y\"] * -1 * floor_img.shape[1] / width_meter\n        ) + floor_img.shape[0]\n#         true_locs.query(\"site == @site and floor == @floor_int\").groupby(\"path\").plot(\n        true_locs.plot(\n            x=\"x_\",\n            y=\"y_\",\n            style=\"+\",\n            ax=ax,\n            label=\"train waypoint location\",\n            color=\"grey\",\n            alpha=0.5,\n        )\n\n    if show_preds:\n        sub = sub.query('site == @site and floorNo == @floorNo').copy()\n        sub[\"x_\"] = sub[\"x\"] * floor_img.shape[0] / height_meter\n        sub[\"y_\"] = (\n            sub[\"y\"] * -1 * floor_img.shape[1] / width_meter\n        ) + floor_img.shape[0]\n        for path, path_data in sub.query(\n            \"site == @site and floorNo == @floorNo\"\n        ).groupby(\"path\"):\n            path_data.plot(\n                x=\"x_\",\n                y=\"y_\",\n                style=\".-\",\n                ax=ax,\n                title=f\"{site} - floor - {floorNo}\",\n                alpha=1,\n                label=path,\n            )\n    if fix_labels:\n        handles, labels = ax.get_legend_handles_labels()\n        by_label = dict(zip(labels, handles))\n        plt.legend(\n            by_label.values(), by_label.keys(), loc=\"center left\", bbox_to_anchor=(1, 0.5)\n        )\n    return fig, ax\n\ndef sub_process(sub, train_waypoints):\n    train_waypoints['isTrainWaypoint'] = True\n    sub = split_col(sub[['site_path_timestamp','floor','x','y']]).copy()\n    sub = sub.merge(train_waypoints[['site','floorNo','floor']].drop_duplicates(), how='left')\n    sub = sub.merge(\n        train_waypoints[['x','y','site','floor','isTrainWaypoint']].drop_duplicates(),\n        how='left',\n        on=['site','x','y','floor']\n    )\n    sub['isTrainWaypoint'] = sub['isTrainWaypoint'].fillna(False)\n    return sub.copy()\n\ndef add_xy(df):\n    df['xy'] = [(x, y) for x,y in zip(df['x'], df['y'])]\n    return df\n\ndef closest_point(point, points):\n    \"\"\" Find closest point from a list of points. \"\"\"\n    return points[cdist([point], points).argmin()]\n\ndef closest_point_shape(point, floor_shape):\n    p = Point(point)\n    nearest_p, _ = nearest_points(floor_shape, p)\n\n    return (nearest_p.xy[0][0], nearest_p.xy[1][0])\n\ndef snap_to_grid(sub, threshold):\n    \"\"\"\n    Snap to grid if within a threshold.\n    \n    x, y are the predicted points.\n    x_close, y_close are the closest grid points.\n    x_new, y_new are the new predictions after post processing.\n    \"\"\"\n    sub['x_new'] = sub['x']\n    sub['y_new'] = sub['y']\n    sub.loc[sub['dist'] < threshold, 'x_new'] = sub.loc[sub['dist'] < threshold]['x_close']\n    sub.loc[sub['dist'] < threshold, 'y_new'] = sub.loc[sub['dist'] < threshold]['y_close']\n    return sub.copy()\n\n\ndef double_snap_to_grid(sub_in, threshold, snap_to_grid=True, snap_to_hallway=True):\n    \"\"\"\n    Snap to grid if within a threshold.\n    \n    x, y are the predicted points.\n    x_close, y_close are the closest grid points.\n    x_new, y_new are the new predictions after post processing.\n    \"\"\"\n    \n    sub = sub_in.copy()\n    \n    sub['x_new'] = sub['x']\n    sub['y_new'] = sub['y']\n    \n    if snap_to_hallway:\n        sub['x_new'] = sub['x_shape']\n        sub['y_new'] = sub['y_shape']\n        \n    if snap_to_grid:\n        sub.loc[sub['dist'] < threshold, 'x_new'] = sub.loc[sub['dist'] < threshold]['x_close']\n        sub.loc[sub['dist'] < threshold, 'y_new'] = sub.loc[sub['dist'] < threshold]['y_close']\n        \n    return sub","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 1: Identify training waypoints\nAs an example I'll plot the training waypoints on the map for a given floor.","metadata":{}},{"cell_type":"code","source":"sub = pd.read_csv(sub_input_path)\nif \"site\" not in sub.columns:\n    sub = split_col(pd.read_csv(sub_input_path))","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if \"site_path_timestamp\" not in sub.columns:\n    \n    site_path_timestamp_list = pd.read_csv(\"../input/indoor-location-navigation/sample_submission.csv\")[['site_path_timestamp']]\n    site_path_timestamp_list = split_col(site_path_timestamp_list)\n    site_path_timestamp_list['time'] = site_path_timestamp_list['timestamp'].astype(int)\n    site_path_timestamp_list = site_path_timestamp_list.drop('timestamp', axis=1)\n    \n    sub = sub.merge(site_path_timestamp_list, on=['site', 'path', 'time'], how=\"inner\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sites = list(sub['site'].drop_duplicates().values)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# pre-read all shape files\nall_sites_shapes = {}\nfor site in sites:\n    site_shapes = {}\n    floors = os.listdir(f\"{shape_dir}/{site}/\")\n    for f_str in floors:\n        f_int = floor_map[f_str]\n        with open(f\"{shape_dir}/{site}/{f_str}/shapely_geometry.pkl\", 'rb') as f:\n            site_shapes[f_int] = pickle.load(f)\n            \n    all_sites_shapes[site] = site_shapes\n        ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_waypoints_raw = pd.read_csv('../input/indoor-location-train-waypoints/train_waypoints.csv')\n# train_waypoints_handlabeled = pd.read_csv('../input/indoor-navigation-hand-labeled-waypoints/waypoint_by_hand.csv')\n\n# train_waypoints = (\n#     train_waypoints_raw[['site', 'floor', 'x', 'y']]\n#     .append(train_waypoints_handlabeled)\n#     .reset_index(drop=True)\n# )\n\ntrain_waypoints = train_waypoints_raw\n\nsub = sub_process(sub, train_waypoints_raw)\n# Plot the training Data For an example Floor\n# example_site = '5dbc1d84c1eb61796cf7c010'\n# example_floorNo = 'F3'\n\n# example_site = '5a0546857ecc773753327266'\n# example_floorNo = 'F1'\n\nexample_site = \"5da138764db8ce0c98bcaa46\"\nexample_floorNo = 'F1'\n\n# plot_preds(example_site, example_floorNo, sub,\n#            train_waypoints, show_preds=True)\n# plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 2: Find the closest \"grid\" point for each prediction.\n\nWe can find the closest \"grid\" point to our predictions using the `cdist` function in scipy.","metadata":{}},{"cell_type":"code","source":"sub = add_xy(sub)\ntrain_waypoints = add_xy(train_waypoints)\n\nds = []\nfor (site, myfloor), d in tqdm(sub.groupby(['site','floor'])):\n    true_floor_locs = train_waypoints.loc[(train_waypoints['floor'] == myfloor) &\n                                          (train_waypoints['site'] == site)] \\\n        .reset_index(drop=True)\n    \n    file = f\"{shape_dir}/{site}/{myfloor}/shapely_geometry.pkl\"\n\n    floor_shape = all_sites_shapes[site][myfloor]\n    \n    if len(true_floor_locs) == 0:\n        print(f'Skipping {site} {myfloor}')\n        continue\n    d['matched_point'] = [closest_point(x, list(true_floor_locs['xy'])) for x in d['xy']]\n    d['matched_point_shape'] = [closest_point_shape(x, floor_shape) for x in d['xy']]\n    d['x_close'] = d['matched_point'].apply(lambda x: x[0])\n    d['y_close'] = d['matched_point'].apply(lambda x: x[1])\n    d['x_shape'] = d['matched_point_shape'].apply(lambda x: x[0])\n    d['y_shape'] = d['matched_point_shape'].apply(lambda x: x[1])\n    ds.append(d)\n\nsub = pd.concat(ds)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sub","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 3: Apply a Threshold and \"Snap to Grid\"\n\nI've found a threshold of 3-8 works well on the LB. But this most likely will be a function of how good your predictions are to start with.","metadata":{}},{"cell_type":"code","source":"# After a snap, look at the x and y movements, and smooth them out.\n# The idea is that snapping can undo some benefits of the cost min procedure.\n# E.g., don't want to shift one step far west, and the next step in the path far east.\ndef smooth_the_diffs(df_snaps_in, ma_win=7):\n    # df_snaps is pd.df with columns x, y, x_new, and y_new.\n    # This function returns same df, but with x_new and y_new modified.\n    # Recommended to snap before AND AFTER running smooth_the_diff.\n\n    df_snaps = df_snaps_in.copy()#.sort_values(['site','path','timestamp'])\n    \n    df_snaps['x_diff'] = df_snaps['x_new'] - df_snaps['x']\n    df_snaps['y_diff'] = df_snaps['y_new'] - df_snaps['y']\n\n    # https://stackoverflow.com/questions/53339021/python-pandas-calculate-moving-average-within-group\n    df_snaps['x_diff_ma'] = df_snaps.groupby(['site', 'path'])['x_diff'].transform(lambda x: x.rolling(ma_win, min_periods=1, center=True).mean())\n    df_snaps['y_diff_ma'] = df_snaps.groupby(['site', 'path'])['y_diff'].transform(lambda x: x.rolling(ma_win, min_periods=1, center=True).mean())\n\n    df_snaps['x_new'] = df_snaps['x'] + df_snaps['x_diff_ma']\n    df_snaps['y_new'] = df_snaps['y'] + df_snaps['y_diff_ma']\n    \n    return df_snaps","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the distances\nsub['dist'] = np.sqrt( (sub.x-sub.x_close)**2 + (sub.y-sub.y_close)**2 )\n\nsub_snaps = double_snap_to_grid(sub, threshold=THRESH)\n\nif smooth_snap_changes:\n    sub_snaps = smooth_the_diffs(sub_snaps, ma_win=3)\n\nsub_pp = sub_snaps[['site_path_timestamp','floor','x_new','y_new','site','path','floorNo']] \\\n    .rename(columns={'x_new':'x', 'y_new':'y'})","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# a = sub_snaps.query(\"site == '5da138764db8ce0c98bcaa46' and path == 'c7ace6aca412926c8c25155e'\")\n# # a['tmp'] = a.groupby(['site', 'path'])['x_diff'].transform(lambda x: x.rolling(3, min_periods=1, center=True).mean())\n# # # df['moving'] = df.groupby('object')['value'].transform(lambda x: x.rolling(10, 1).mean())\n# # aa = a.query(\"path == 'c7ace6aca412926c8c25155e'\")\n# plt.plot(a['timestamp'], a['x_diff'])\n# plt.plot(a['timestamp'], a['x_diff_ma'])\n# plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# help(x_diffs.rolling)\n\n# sub_pp.groupby(['site', 'floorNo'])[['x']].count().sort_values('x', ascending=False).head(10)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets take a look at the predictions before and after post processing.","metadata":{}},{"cell_type":"code","source":"# Example of raw predictions\nplot_preds(example_site, example_floorNo, sub,\n           train_waypoints, show_preds=True)\nplt.show()\n\n# Plot example after post processing\nplot_preds(example_site, example_floorNo, sub_pp,\n           train_waypoints, show_preds=True)\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluate The Change in Predictions","metadata":{}},{"cell_type":"code","source":"sub_snaps['dist_pp_change'] = np.sqrt(((sub_snaps['x'] - sub_snaps['x_new']) ** 2) + ((sub_snaps['y'] - sub_snaps['y_new']) ** 2))\nfig, axs = plt.subplots(1, 2, figsize=(15, 5))\nsub_snaps['dist_pp_change'].plot(kind='hist', bins=30,\n                           ax=axs[0],\n                           title='Distance Changed by Post Processing')\nsub_snaps.query('dist_pp_change > 0.1')['dist_pp_change'] \\\n    .plot(kind='hist', bins=30, ax=axs[1],\n          title='Distance Changed (Excluding <0.1 Change)')\n\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = sub_snaps.groupby(['site','floorNo'])['dist_pp_change'].mean() \\\n    .reset_index() \\\n    .sort_values('dist_pp_change', ascending=False) \\\n    .set_index(['site','floorNo']).head(10)\n\nx","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Final Step: Save Post Processed Submission.","metadata":{}},{"cell_type":"code","source":"sub_pp[['site_path_timestamp','floor','x','y']] \\\n    .to_csv('submission_snap_to_grid.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Repeat everything for CV","metadata":{}},{"cell_type":"code","source":"cv = pd.read_csv(cv_input_path)\nif 'floor' in cv.columns:\n    cv = cv.drop('floor', axis=1)\n\nif 'f' in cv.columns:\n    cv = cv.rename(columns={'f': 'floor_int'})\n    \nfold_labels = pickle.load(open(\"../input/iln-fold-labels/fold_labels.pkl\", 'rb'))\nfold_labels_waypoints = pickle.load(open(\"../input/iln-fold-labels/fold_labels_waypoints.pkl\", 'rb'))\n\n# FOLDS = [0, 1, 2, 3, 4]\nFOLDS = [0]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cv_results_list = []\nfor fold in FOLDS:\n    in_fold_paths = fold_labels[fold_labels['fold'] == fold]\n    out_fold_waypoints = fold_labels_waypoints[fold_labels_waypoints['fold'] != fold]\n\n    train_waypoints = out_fold_waypoints.drop('floor', axis=1).rename(columns = {\"floor_int\": \"floor\"})[['site','path','x','y','floor']]\n    \n    in_sample = (\n        in_fold_paths[['site','path']]\n        .merge(\n            (\n                cv\n                .rename(columns = {\n                    \"x\": \"x_act\",\n                    \"y\": \"y_act\",\n                    \"floor_int\": \"f_act\"\n                })\n                .rename(columns = {\n                    \"x_pred\": \"x\",\n                    \"y_pred\": \"y\",\n                    \"f_pred\": \"floor\"\n                })\n            ), \n            on = ['site', 'path'], \n            how = 'inner'\n        )[['site','path','x','y','floor','time','x_act','y_act','f_act']]\n    )\n    \n    in_sample = add_xy(in_sample)\n    train_waypoints = add_xy(train_waypoints)\n    \n    in_sample_list = []\n    for (site, myfloor), d in tqdm(in_sample.groupby(['site','floor'])):\n        true_floor_locs = train_waypoints.loc[(train_waypoints['floor'] == myfloor) &\n                                              (train_waypoints['site'] == site)] \\\n            .reset_index(drop=True)\n        \n        floor_shape = all_sites_shapes[site][myfloor]\n        \n        d['matched_point_shape'] = [closest_point_shape(x, floor_shape) for x in d['xy']]\n        \n        if len(true_floor_locs) == 0:\n            print(f'Skipping {site} {myfloor}')\n            d['matched_point'] = d['matched_point_shape']\n\n        else:\n            d['matched_point'] = [closest_point(x, list(true_floor_locs['xy'])) for x in d['xy']]\n            \n        d['x_close'] = d['matched_point'].apply(lambda x: x[0])\n        d['y_close'] = d['matched_point'].apply(lambda x: x[1])\n        d['x_shape'] = d['matched_point_shape'].apply(lambda x: x[0])\n        d['y_shape'] = d['matched_point_shape'].apply(lambda x: x[1])\n        \n    \n        in_sample_list.append(d)\n\n    in_sample = pd.concat(in_sample_list)\n    \n    # Calculate the distances\n    in_sample['dist'] = np.sqrt( (in_sample.x-in_sample.x_close)**2 + (in_sample.y-in_sample.y_close)**2 )\n\n    in_sample_pp = double_snap_to_grid(in_sample, threshold=THRESH)\n\n    #rename back to convention I've been using. \n    in_sample_pp = (\n        in_sample_pp[['time','floor','x_new','y_new','site','path','x_act','y_act','f_act']]\n        .rename(columns={\n            'x_new':'x_pred', \n            'y_new':'y_pred',\n            'floor': 'f_pred',\n        })\n        .rename(columns={\n            'x_act':'x', \n            'y_act':'y',\n            'f_act': 'floor_int',\n        })\n    )\n    \n    cv_results_list.append(in_sample_pp)\n    \ncv_results = pd.concat(cv_results_list)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df = cv_results #laziness, code already written w/ sub_df\n#Performance Metrics\nsub_df['score_xy'] = ((sub_df['x_pred'] - sub_df['x'])**2 + (sub_df['y_pred'] - sub_df['y'])**2)**0.5\nsub_df['score_f'] = np.abs(sub_df['f_pred'] - sub_df['floor_int'])\nprint(f\"score_xy: {sub_df['score_xy'].mean()}\")\nprint(f\"score_f: {sub_df['score_f'].mean()}\")\n\nprint(f\"floor accuracy: {np.mean(sub_df['f_pred'] == sub_df['floor_int'])}\")\n\nsub_df.to_csv(\"val_w_preds.csv\", index=False)\n\nplt.scatter(sub_df['x'], sub_df['x_pred'])\nplt.show()\n\nplt.scatter(sub_df['y'], sub_df['y_pred'])\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}