{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"I shamelessly took almost everything and adapted from last years wonderful notebook: https://www.kaggle.com/code/nayuts/let-s-visualize-dataset-to-understand still a work in progress","metadata":{}},{"cell_type":"code","source":"\n!pip install pynmea2\n\nimport glob\nimport itertools\nimport json\nimport os\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport geopandas as gpd\nfrom geopandas import GeoDataFrame\nimport geoplot as gplt\nfrom IPython.display import Video\nfrom matplotlib import animation\nimport matplotlib.pyplot as plt\nimport numpy as np \nimport pandas as pd\nimport plotly.express as px\nimport pynmea2\nimport requests\nimport seaborn\nfrom shapely.geometry import Point, shape\nimport shapely.wkt\n\n%matplotlib inline\n\nDATA_PATH = \"../input/smartphone-decimeter-2022/\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-05-06T10:02:34.270397Z","iopub.execute_input":"2022-05-06T10:02:34.27091Z","iopub.status.idle":"2022-05-06T10:02:44.496491Z","shell.execute_reply.started":"2022-05-06T10:02:34.270855Z","shell.execute_reply":"2022-05-06T10:02:44.495631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(DATA_PATH + \"sample_submission.csv\")\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-06T10:02:44.498445Z","iopub.execute_input":"2022-05-06T10:02:44.498702Z","iopub.status.idle":"2022-05-06T10:02:44.578464Z","shell.execute_reply.started":"2022-05-06T10:02:44.498659Z","shell.execute_reply":"2022-05-06T10:02:44.57766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Devices can be one thing or multiple things. Data collection trials are separated as collectionName like this","metadata":{}},{"cell_type":"code","source":"!ls ../input/smartphone-decimeter-2022/train","metadata":{"execution":{"iopub.status.busy":"2022-05-06T10:02:44.579546Z","iopub.execute_input":"2022-05-06T10:02:44.579798Z","iopub.status.idle":"2022-05-06T10:02:45.344122Z","shell.execute_reply.started":"2022-05-06T10:02:44.579771Z","shell.execute_reply":"2022-05-06T10:02:45.343363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Under each collection Name, the data of the device is stored.","metadata":{}},{"cell_type":"code","source":"!ls ../input/smartphone-decimeter-2022/train/2020-05-15-US-MTV-1","metadata":{"execution":{"iopub.status.busy":"2022-05-06T10:02:45.346665Z","iopub.execute_input":"2022-05-06T10:02:45.346949Z","iopub.status.idle":"2022-05-06T10:02:46.112508Z","shell.execute_reply.started":"2022-05-06T10:02:45.346915Z","shell.execute_reply":"2022-05-06T10:02:46.111515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In addition, the data collected from each device, groundtruth, and supplemental data are stored under it.","metadata":{}},{"cell_type":"code","source":"!ls ../input/smartphone-decimeter-2022/train/2020-05-15-US-MTV-1/GooglePixel4XL","metadata":{"execution":{"iopub.status.busy":"2022-05-06T10:02:46.115225Z","iopub.execute_input":"2022-05-06T10:02:46.115486Z","iopub.status.idle":"2022-05-06T10:02:46.8741Z","shell.execute_reply.started":"2022-05-06T10:02:46.115454Z","shell.execute_reply":"2022-05-06T10:02:46.872995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The supplemental data contains the raw data that was measured, and I'll show way to read nmea.","metadata":{}},{"cell_type":"code","source":"!ls ../input/smartphone-decimeter-2022/train/2020-05-15-US-MTV-1/GooglePixel4XL/supplemental","metadata":{"execution":{"iopub.status.busy":"2022-05-06T10:02:46.875965Z","iopub.execute_input":"2022-05-06T10:02:46.876246Z","iopub.status.idle":"2022-05-06T10:02:47.642921Z","shell.execute_reply.started":"2022-05-06T10:02:46.876209Z","shell.execute_reply":"2022-05-06T10:02:47.642123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ../input/smartphone-decimeter-2022/train/2020-05-15-US-MTV-1/GooglePixel4XL/device_gnss.csv","metadata":{"execution":{"iopub.status.busy":"2022-05-06T10:02:47.6442Z","iopub.execute_input":"2022-05-06T10:02:47.644452Z","iopub.status.idle":"2022-05-06T10:02:47.649795Z","shell.execute_reply.started":"2022-05-06T10:02:47.64442Z","shell.execute_reply":"2022-05-06T10:02:47.648876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sample_trail_gnss = pd.read_csv(DATA_PATH + \"train/2020-05-15-US-MTV-1/GooglePixel4XL/device_gnss.csv\")\ndf_sample_trail_gt = pd.read_csv(DATA_PATH + \"train/2020-05-15-US-MTV-1/GooglePixel4XL//ground_truth.csv\")\ndf_sample_trail_imu = pd.read_csv(DATA_PATH + \"train/2020-05-15-US-MTV-1/GooglePixel4XL//device_imu.csv\")\n\ndf_sample_trail_gnss.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-06T10:02:47.651177Z","iopub.execute_input":"2022-05-06T10:02:47.651403Z","iopub.status.idle":"2022-05-06T10:02:49.361648Z","shell.execute_reply.started":"2022-05-06T10:02:47.651375Z","shell.execute_reply":"2022-05-06T10:02:49.360799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sample_trail_gnss.columns","metadata":{"execution":{"iopub.status.busy":"2022-05-06T10:02:49.362699Z","iopub.execute_input":"2022-05-06T10:02:49.362917Z","iopub.status.idle":"2022-05-06T10:02:49.370272Z","shell.execute_reply.started":"2022-05-06T10:02:49.362892Z","shell.execute_reply":"2022-05-06T10:02:49.369362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sample_trail_gt.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-06T10:02:49.372976Z","iopub.execute_input":"2022-05-06T10:02:49.373249Z","iopub.status.idle":"2022-05-06T10:02:49.392084Z","shell.execute_reply.started":"2022-05-06T10:02:49.37321Z","shell.execute_reply":"2022-05-06T10:02:49.390677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sample_trail_imu.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-06T10:02:49.393443Z","iopub.execute_input":"2022-05-06T10:02:49.393708Z","iopub.status.idle":"2022-05-06T10:02:49.411457Z","shell.execute_reply.started":"2022-05-06T10:02:49.393678Z","shell.execute_reply":"2022-05-06T10:02:49.410537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# How to check track in detail?\n\nWe can use plotly to see our model or ground truth like this. To see trafic, you should adjust map centor and scale.","metadata":{}},{"cell_type":"code","source":"def visualize_trafic(df, center, zoom=9):\n    fig = px.scatter_mapbox(df,\n                            \n                            # Here, plotly gets, (x,y) coordinates\n                            lat=\"LatitudeDegrees\",\n                            lon=\"LongitudeDegrees\",\n                            \n                            #Here, plotly detects color of series\n                            color=\"phone\",\n                            labels=\"phone\",\n                            \n                            zoom=zoom,\n                            center=center,\n                            height=600,\n                            width=800)\n    fig.update_layout(mapbox_style='stamen-terrain')\n    fig.update_layout(margin={\"r\": 0, \"t\": 0, \"l\": 0, \"b\": 0})\n    fig.update_layout(title_text=\"GPS trafic\")\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-06T10:02:49.413341Z","iopub.execute_input":"2022-05-06T10:02:49.413871Z","iopub.status.idle":"2022-05-06T10:02:49.425111Z","shell.execute_reply.started":"2022-05-06T10:02:49.413828Z","shell.execute_reply":"2022-05-06T10:02:49.424511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sample_trail_gt['phone'] = 'GooglePixel4XL'\ncenter = {\"lat\":37.423576, \"lon\":-122.094132}\nvisualize_trafic(df_sample_trail_gt, center)","metadata":{"execution":{"iopub.status.busy":"2022-05-06T10:02:49.426037Z","iopub.execute_input":"2022-05-06T10:02:49.42661Z","iopub.status.idle":"2022-05-06T10:02:49.514051Z","shell.execute_reply.started":"2022-05-06T10:02:49.426575Z","shell.execute_reply":"2022-05-06T10:02:49.513075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sample_trail_gt2 = pd.read_csv(DATA_PATH + \"train/2020-05-21-US-MTV-1/GooglePixel4/ground_truth.csv\")\ndf_sample_trail_gt2['phone'] = 'GooglePixel4'\n\n\n# Since plotly looks at the phoneName of the dataframe,\n# you can visualize multiple series of data by simply concatting dataframes.\ndf_sample_trail_gt3 = pd.concat([df_sample_trail_gt, df_sample_trail_gt2])\n\ncenter = {\"lat\":37.423576, \"lon\":-122.094132}\nvisualize_trafic(df_sample_trail_gt3, center)","metadata":{"execution":{"iopub.status.busy":"2022-05-06T10:02:49.515301Z","iopub.execute_input":"2022-05-06T10:02:49.515536Z","iopub.status.idle":"2022-05-06T10:02:49.606444Z","shell.execute_reply.started":"2022-05-06T10:02:49.515509Z","shell.execute_reply":"2022-05-06T10:02:49.605613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# How to check large amounts of tracks?\n\nEarlier we saw how to use plotly to map data on OpenStreetMap. This time, since there is a certain amount of tracking data in the train data alone, I will also show you how to use geopandas to get a quick overview as a regular diagram.\n\nFrom here on, the cells will be hidden for a while, because the procedure is necessary for visualization and we will get tired of following everything. If you have interest, please open and check them accordingly.\n\nFirst, I'll download shape file lof bayarea.\n","metadata":{}},{"cell_type":"code","source":"#Download geojson file of US San Francisco Bay Area.\nr = requests.get(\"https://data.sfgov.org/api/views/wamw-vt4s/rows.json?accessType=DOWNLOAD\")\nr.raise_for_status()\n\n#get geojson from response\ndata = r.json()\n\n#get polygons that represents San Francisco Bay Area.\nshapes = []\nfor d in data[\"data\"]:\n    shapes.append(shapely.wkt.loads(d[8]))\n    \n#Convert list of porygons to geopandas dataframe.\ngdf_bayarea = pd.DataFrame()\n\n#I'll use only 6 and 7th object.\nfor shp in shapes[5:7]:\n    tmp = pd.DataFrame(shp, columns=[\"geometry\"])\n    gdf_bayarea = pd.concat([gdf_bayarea, tmp])\ngdf_bayarea = GeoDataFrame(gdf_bayarea)","metadata":{"execution":{"iopub.status.busy":"2022-05-06T10:02:49.60796Z","iopub.execute_input":"2022-05-06T10:02:49.608432Z","iopub.status.idle":"2022-05-06T10:02:51.085964Z","shell.execute_reply.started":"2022-05-06T10:02:49.608387Z","shell.execute_reply":"2022-05-06T10:02:51.085025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"collection_names = [item.split(\"/\")[-1] for item in glob.glob(\"../input/smartphone-decimeter-2022/train/*\")]\n\ngdfs = []\nfor collection_name in collection_names:\n    gdfs_each_collectionName = []\n    csv_paths = glob.glob(f\"../input/smartphone-decimeter-2022/train/{collection_name}/*/ground_truth.csv\")\n    for csv_path in csv_paths:\n        df_gt = pd.read_csv(csv_path)\n        df_gt['collectionName'] = collection_name\n        df_gt[\"geometry\"] = [Point(lngDeg, latDeg) for lngDeg, latDeg in zip(df_gt[\"LongitudeDegrees\"], df_gt[\"LatitudeDegrees\"])]\n        gdfs_each_collectionName.append(GeoDataFrame(df_gt))\n    gdfs.append(gdfs_each_collectionName)","metadata":{"execution":{"iopub.status.busy":"2022-05-06T10:02:51.087179Z","iopub.execute_input":"2022-05-06T10:02:51.087403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"colors = ['blue', 'green', 'purple', 'orange']\nfor collectionName, gdfs_each_collectionName in zip(collection_names, gdfs):\n    fig, axs = plt.subplots(1, 2, figsize=(15, 5))\n    gdf_bayarea.plot(figsize=(10,10), color='none', edgecolor='gray', zorder=3, ax=axs[0])\n    for i, gdf in enumerate(gdfs_each_collectionName):\n        g1 = gdf.plot(color=colors[i], ax=axs[0])\n        g1.set_title(f\"Phone track of {collectionName} with map\")\n        g2 = gdf.plot(color=colors[i], ax=axs[1])\n        g2.set_title(f\"Phone track of {collectionName}\")\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are several tracks that have the same form of data with different collectionName. It is easy to understand the positional relationship by overlapping them. There are two roads extending from the northwest to the southeast, and they seem to run along those roads all the time, or occasionally go off those roads. The tracks wandering around the grid-like paths seem to be collected farther southeast than those paths.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(15, 5))\n\nfor collectionName, gdfs_each_collectionName in zip(collection_names, gdfs):   \n    for i, gdf in enumerate(gdfs_each_collectionName):\n        gdf.plot(color=colors[i], ax=ax, markersize=5, alpha=0.5)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_tracks = pd.DataFrame()\n\nfor collectionName, gdfs_each_collectionName in zip(collection_names, gdfs):   \n    for i, gdf in enumerate(gdfs_each_collectionName):\n        all_tracks = pd.concat([all_tracks, gdf])\n        # Tracks they have same collectionName is also same\n        break\n        \nfig = px.scatter_mapbox(all_tracks,\n                            \n                        # Here, plotly gets, (x,y) coordinates\n                        lat=\"LongitudeDegrees\",\n                        lon=\"LatitudeDegrees\",\n                            \n                        #Here, plotly detects color of series\n                        color=\"collectionName\",\n                        labels=\"collectionName\",\n                            \n                        zoom=9,\n                        center={\"lat\":37.423576, \"lon\":-122.094132},\n                        height=600,\n                        width=800)\nfig.update_layout(mapbox_style='stamen-terrain')\nfig.update_layout(margin={\"r\": 0, \"t\": 0, \"l\": 0, \"b\": 0})\nfig.update_layout(title_text=\"GPS trafic\")\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# How to check tracks in animation?¶\n\nI am sure that you all would like to see animations of what the data was driving at. I have also prepared code to check the x, y coordinate movement in gif, so try it out.\n\nThe process takes some minutes (about 2 or 3 minutes for my implement).\n\nNote: The x, y coordinates have been thinned out (to 1/10) because the processing uses much memory. You can adjust them if necessary.","metadata":{}},{"cell_type":"code","source":"def create_gif_track(df, git_path):\n    \"\"\" Create git animation of phone track.\n    \"\"\"\n\n    fig, ax = plt.subplots()\n\n    imgs = []\n    df[\"geometry\"] = [Point(lngDeg, latDeg) for lngDeg, latDeg in zip(df[\"LongitudeDegrees\"], df[\"LatitudeDegrees\"])]\n    gdf = GeoDataFrame(df)\n    gdf.plot(color=\"lightskyblue\", ax=ax)\n\n    # Here, (x,y) coordinates are thinned out!!!\n    for i in range(0, len(gdf), 10):\n        # plot data\n        p = ax.plot(gdf.iloc[i][\"LongitudeDegrees\"], gdf.iloc[i][\"LatitudeDegrees\"], \n                    color = 'dodgerblue', marker = 'o', markersize = 8)\n        imgs.append(p)\n\n    # Create animation & save it\n    ani = animation.ArtistAnimation(fig, imgs, interval=200)\n    ani.save(git_path, writer='imagemagick', dpi = 300)\n    \n\ndef create_gif_track_on_map(df, gdf_map, git_path):\n    \"\"\" Create git animation of phone track on bayarea map.\n    \"\"\"\n\n    fig, ax = plt.subplots()\n    df[\"geometry\"] = [Point(lngDeg, latDeg) for lngDeg, latDeg in zip(df[\"LongitudeDegrees\"], df[\"LatitudeDegrees\"])]\n    gdf = GeoDataFrame(df)\n    gdf.plot(color=\"lightskyblue\", ax=ax)\n    imgs = []  \n    gdf_map.plot(color='none', edgecolor='gray', zorder=3, ax=ax)\n    \n    # Here, (x,y) coordinates are thinned out!!!\n    for i in range(0, len(gdf), 10):\n        # plot data on map\n        p = ax.plot(gdf.iloc[i][\"LongitudeDegrees\"], gdf.iloc[i][\"LatitudeDegrees\"], \n                    color = 'dodgerblue', marker = 'o', markersize = 8)\n        imgs.append(p)\n\n    # Create animation & save it\n    ani = animation.ArtistAnimation(fig, imgs, interval=200)\n    ani.save(git_path, writer='imagemagick', dpi = 300)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nGooglePixel4XLgt = pd.read_csv(DATA_PATH + \"train/2020-05-15-US-MTV-1/GooglePixel4XL/ground_truth.csv\")\n\ncreate_gif_track(GooglePixel4XLgt, \"./GooglePixel4XLgt.gif\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}