{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-08-02T17:16:50.816839Z","iopub.execute_input":"2021-08-02T17:16:50.817434Z","iopub.status.idle":"2021-08-02T17:16:52.184224Z","shell.execute_reply.started":"2021-08-02T17:16:50.817334Z","shell.execute_reply":"2021-08-02T17:16:52.183302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport glob\nimport os\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm\nfrom pathlib import Path\nimport plotly.express as px\npd.options.mode.chained_assignment = None","metadata":{"execution":{"iopub.status.busy":"2021-08-02T17:16:52.185909Z","iopub.execute_input":"2021-08-02T17:16:52.186303Z","iopub.status.idle":"2021-08-02T17:16:54.08399Z","shell.execute_reply.started":"2021-08-02T17:16:52.186264Z","shell.execute_reply":"2021-08-02T17:16:54.083087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = Path(\"../input/google-smartphone-decimeter-challenge\")\ntrain_df = pd.read_csv(data_dir / \"baseline_locations_train.csv\")\ndf_sub = pd.read_csv(data_dir / 'baseline_locations_test.csv')\ndf_sub.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-02T17:16:54.085444Z","iopub.execute_input":"2021-08-02T17:16:54.085888Z","iopub.status.idle":"2021-08-02T17:16:54.680316Z","shell.execute_reply.started":"2021-08-02T17:16:54.085856Z","shell.execute_reply":"2021-08-02T17:16:54.679506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get all ground truth dataframe\ngt_df = pd.DataFrame()\nfor (collection_name, phone_name), df in tqdm(train_df.groupby([\"collectionName\", \"phoneName\"])):\n    path = data_dir / f\"train/{collection_name}/{phone_name}/ground_truth.csv\"\n    df = pd.read_csv(path)  \n    gt_df = pd.concat([gt_df, df]).reset_index(drop=True)   \ngt_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-02T17:16:54.681588Z","iopub.execute_input":"2021-08-02T17:16:54.68198Z","iopub.status.idle":"2021-08-02T17:16:56.693213Z","shell.execute_reply.started":"2021-08-02T17:16:54.68195Z","shell.execute_reply":"2021-08-02T17:16:56.69185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install osmnx momepy geopandas\nfrom shapely.geometry import Point\nimport osmnx as ox\nimport momepy\nimport geopandas as gpd\nfrom geopy import distance","metadata":{"execution":{"iopub.status.busy":"2021-08-02T17:16:56.694855Z","iopub.execute_input":"2021-08-02T17:16:56.695216Z","iopub.status.idle":"2021-08-02T17:17:12.318551Z","shell.execute_reply.started":"2021-08-02T17:16:56.695184Z","shell.execute_reply":"2021-08-02T17:17:12.317264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def iterate(target_gt_df, phone):\n    \n    target_gt_df[\"geometry\"] = [Point(p) for p in target_gt_df[[\"lngDeg\", \"latDeg\"]].to_numpy()]\n    target_gt_gdf = gpd.GeoDataFrame(target_gt_df, geometry=target_gt_df[\"geometry\"])\n    \n    offset = 0.1**5\n    bbox = target_gt_gdf.bounds + [-offset, -offset, offset, offset]\n    east = bbox[\"minx\"].min()\n    west = bbox[\"maxx\"].max()\n    south = bbox[\"miny\"].min()\n    north = bbox[\"maxy\"].max()\n    G = ox.graph.graph_from_bbox(north, south, east, west, network_type='drive')\n    \n    ox.folium.plot_graph_folium(G)\n    nodes, edges = momepy.nx_to_gdf(G)\n    edges = edges.dropna(subset=[\"geometry\"]).reset_index(drop=True)\n    hits = bbox.apply(lambda row: list(edges.sindex.intersection(row)), axis=1)\n    tmp = pd.DataFrame({\n    \"pt_idx\": np.repeat(hits.index, hits.apply(len)),\n    \"line_i\": np.concatenate(hits.values)})\n    tmp = tmp.join(edges.reset_index(drop=True), on=\"line_i\")\n    tmp = tmp.join(target_gt_gdf.geometry.rename(\"point\"), on=\"pt_idx\")\n    tmp = gpd.GeoDataFrame(tmp, geometry=\"geometry\", crs=target_gt_gdf.crs)\n    \n    tmp[\"snap_dist\"] = tmp.geometry.distance(gpd.GeoSeries(tmp.point))\n    tolerance = 0.0005  \n    tmp = tmp.loc[tmp.snap_dist <= tolerance]\n    tmp = tmp.sort_values(by=[\"snap_dist\"])\n    closest = tmp.groupby(\"pt_idx\").first()\n    closest = gpd.GeoDataFrame(closest, geometry=\"geometry\")\n    closest = closest.drop_duplicates(\"line_i\").reset_index(drop=True)\n    \n    line_points_list = []\n    split = 50  \n    for dist in range(0, split, 1):\n        dist = dist/split\n        line_points = closest[\"geometry\"].interpolate(dist, normalized=True)\n        line_points_list.append(line_points)\n    line_points = pd.concat(line_points_list).reset_index(drop=True)\n    line_points = line_points.reset_index().rename(columns={0:\"geometry\"})\n    line_points[\"lngDeg\"] = line_points[\"geometry\"].x\n    line_points[\"latDeg\"] = line_points[\"geometry\"].y\n\n    for row in df_sub.index:\n        if df_sub[\"collectionName\"].loc[row] != phone:\n            continue\n        row = 1000\n        currLat = df_sub['latDeg'].loc[row]\n        currLng = df_sub['lngDeg'].loc[row]\n        testPoint = (currLat, currLng)\n        latTemp = 0\n        lngTemp = 0\n        lowestDist = 100000000000\n        subset = line_points[(line_points['latDeg'] > currLat - 0.0005) & (line_points['latDeg'] < currLat + 0.0005) & (line_points['lngDeg'] > currLng - 0.0005) & (line_points['lngDeg'] < currLng + 0.0005)]\n        for row1 in subset.index:\n            roadPoint = (subset['latDeg'].loc[row1], subset['lngDeg'].loc[row1])\n            dist = distance.distance(testPoint, roadPoint).miles * 16093.4\n            if dist < lowestDist:\n                lowestDist = dist\n                latTemp = subset['latDeg'].loc[row1]\n                lngTemp = subset['lngDeg'].loc[row1]\n        df_sub['latDeg'].loc[row] = latTemp\n        df_sub['lngDeg'].loc[row] = lngTemp","metadata":{"execution":{"iopub.status.busy":"2021-08-02T17:17:12.322554Z","iopub.execute_input":"2021-08-02T17:17:12.322919Z","iopub.status.idle":"2021-08-02T17:17:12.345661Z","shell.execute_reply.started":"2021-08-02T17:17:12.322885Z","shell.execute_reply":"2021-08-02T17:17:12.344265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"phones = df_sub['collectionName'].unique()\nfor phone in phones:\n    subset = df_sub[df_sub[\"collectionName\"]==phone].reset_index(drop=True)\n    iterate(subset, phone)","metadata":{"execution":{"iopub.status.busy":"2021-08-02T17:17:12.348569Z","iopub.execute_input":"2021-08-02T17:17:12.348886Z","iopub.status.idle":"2021-08-02T17:39:49.808677Z","shell.execute_reply.started":"2021-08-02T17:17:12.348855Z","shell.execute_reply":"2021-08-02T17:39:49.807442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub = df_sub.drop(['heightAboveWgs84EllipsoidM', 'collectionName', 'phoneName'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-08-02T17:39:49.81104Z","iopub.execute_input":"2021-08-02T17:39:49.811397Z","iopub.status.idle":"2021-08-02T17:39:49.819539Z","shell.execute_reply.started":"2021-08-02T17:39:49.81135Z","shell.execute_reply":"2021-08-02T17:39:49.81841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-08-02T17:39:49.824898Z","iopub.execute_input":"2021-08-02T17:39:49.825335Z","iopub.status.idle":"2021-08-02T17:39:50.399444Z","shell.execute_reply.started":"2021-08-02T17:39:49.825292Z","shell.execute_reply":"2021-08-02T17:39:50.398269Z"},"trusted":true},"execution_count":null,"outputs":[]}]}