{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"I noticed that some of the ground truths are wrong. Disscusion is [here](https://www.kaggle.com/competitions/smartphone-decimeter-2022/discussion/334654).\n\n**2022-07-04 Update:** Baseline errors were visualized for all train data.","metadata":{}},{"cell_type":"code","source":"!pip install pymap3d\nimport numpy as np\nimport pandas as pd\nimport pymap3d as pm\nimport glob as gl\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:38:56.467514Z","iopub.execute_input":"2022-07-04T14:38:56.467923Z","iopub.status.idle":"2022-07-04T14:39:07.233161Z","shell.execute_reply.started":"2022-07-04T14:38:56.467891Z","shell.execute_reply":"2022-07-04T14:39:07.231694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot baseline error (Baseline-GroundTruth)\ndef plot_baseline_error(path):\n    drive, phone = path.split('/')[-3:-1]\n\n    # Read data\n    gnss_df = pd.read_csv(f'{path}/device_gnss.csv')  # GNSS data\n    gt_df = pd.read_csv(f'{path}/ground_truth.csv')  # ground truth\n\n    # Ground truth\n    llh_gt = gt_df[['LatitudeDegrees', 'LongitudeDegrees', 'AltitudeMeters']].to_numpy()\n    \n    # Baseline\n    xyz_bl = gnss_df.groupby('utcTimeMillis')[\n        ['WlsPositionXEcefMeters', 'WlsPositionYEcefMeters', 'WlsPositionZEcefMeters']].mean().to_numpy()\n    llh_bl = np.array(pm.ecef2geodetic(xyz_bl[:, 0], xyz_bl[:, 1], xyz_bl[:, 2])).T\n    \n    # Baseline-GroundTruth\n    err_llh = llh_bl - llh_gt\n\n    # Plot distance error\n    fig = plt.figure(figsize=(8, 6))\n    ax = fig.add_subplot(1, 1, 1, aspect='equal', title=f'{drive}/{phone}')\n\n    plt.plot(err_llh[:,1],err_llh[:,0],marker='.',linewidth = 0)\n    plt.grid()\n    plt.xlabel('Longitude error [deg]')\n    plt.ylabel('Latitude error [deg]')\n    ax.set_aspect('equal', adjustable='box')\n    plt.xlim([-1e-4, 1e-4])\n    plt.ylim([-1e-4, 1e-4])","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:39:07.236184Z","iopub.execute_input":"2022-07-04T14:39:07.236693Z","iopub.status.idle":"2022-07-04T14:39:07.250084Z","shell.execute_reply.started":"2022-07-04T14:39:07.236643Z","shell.execute_reply":"2022-07-04T14:39:07.248656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"GNSS pseudorange-based positioning error distributions are usually **Gaussian** if there is no multipath.\nHowever, for some train data, the distribution of baseline errors has **multiple peaks**.\nThis is thought to be due to the inclusion of errors in ground truth.\nIt is fatal to the competition that ground truth is unreliable. \nFor a more detailed discussion, see the discussion [here](https://www.kaggle.com/competitions/smartphone-decimeter-2022/discussion/334654).","metadata":{}},{"cell_type":"code","source":"# Check ground truth\npath = '/kaggle/input/smartphone-decimeter-2022/train'\nfor drive in sorted(gl.glob(f'{path}/*/')):\n    phone = gl.glob(f'{drive}/*/')\n    \n    # Plot baseline error\n    plot_baseline_error(phone[0])","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:39:07.251733Z","iopub.execute_input":"2022-07-04T14:39:07.252276Z","iopub.status.idle":"2022-07-04T14:39:55.626431Z","shell.execute_reply.started":"2022-07-04T14:39:07.252232Z","shell.execute_reply":"2022-07-04T14:39:55.625293Z"},"trusted":true},"execution_count":null,"outputs":[]}]}