{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Import the Library**","metadata":{}},{"cell_type":"code","source":"%config Completer.use_jedi = False","metadata":{"execution":{"iopub.status.busy":"2022-05-10T07:10:17.438569Z","iopub.execute_input":"2022-05-10T07:10:17.438893Z","iopub.status.idle":"2022-05-10T07:10:17.454688Z","shell.execute_reply.started":"2022-05-10T07:10:17.438861Z","shell.execute_reply":"2022-05-10T07:10:17.453946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nfrom dataclasses import dataclass\nimport numpy as np\nimport pandas as pd\nfrom tqdm.notebook import tqdm\nfrom scipy.interpolate import InterpolatedUnivariateSpline\n","metadata":{"execution":{"iopub.status.busy":"2022-05-10T07:39:34.500917Z","iopub.execute_input":"2022-05-10T07:39:34.501980Z","iopub.status.idle":"2022-05-10T07:39:35.115967Z","shell.execute_reply.started":"2022-05-10T07:39:34.501928Z","shell.execute_reply":"2022-05-10T07:39:35.114983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Baseline**\n## **Data Preparation**\n1 地心坐标系（ECEF）和WGS-84坐标系（WGS84）互转\n\n    GeographicCoordinate System: GCS_WGS_1984\n    AngularUnit: Degree (0.0174532925199433)\n    PrimeMeridian: Greenwich (0.0)\n    Datum: D_WGS_1984\n    Spheroid:WGS_1984\n    Semimajor Axis: 6378137.0\n    Semiminor Axis: 6356752.314245179\n    Inverse Flattening:298.257223563\n2 ECEF 转换经纬度\n\n3 预测经纬度与真实经纬度对比","metadata":{}},{"cell_type":"code","source":"WGS84_SEMI_MAJOR_AXIS = 6378137.0\nWGS84_SEMI_MINOR_AXIS = 6356752.314245\n\nWGS84_SQUARED_FIRST_ECCENTRICITY  = 6.69437999013e-3\nWGS84_SQUARED_SECOND_ECCENTRICITY = 6.73949674226e-3\n\nHAVERSINE_RADIUS = 6_371_000\n\n@dataclass\nclass ECEF:\n    x: np.array\n    y: np.array\n    z: np.array\n\n    def to_numpy(self):\n        return np.stack([self.x, self.y, self.z], axis=0)\n\n    @staticmethod\n    def from_numpy(pos):\n        x, y, z = [np.squeeze(w) for w in np.split(pos, 3, axis=-1)]\n        return ECEF(x=x, y=y, z=z)\n\n@dataclass\nclass BLH:\n    lat : np.array\n    lng : np.array\n    hgt : np.array\n\ndef ECEF_to_BLH(ecef):\n    a = WGS84_SEMI_MAJOR_AXIS\n    b = WGS84_SEMI_MINOR_AXIS\n    e2  = WGS84_SQUARED_FIRST_ECCENTRICITY\n    e2_ = WGS84_SQUARED_SECOND_ECCENTRICITY\n    x = ecef.x\n    y = ecef.y\n    z = ecef.z\n    r = np.sqrt(x**2 + y**2)\n    t = np.arctan2(z * (a/b), r)\n    B = np.arctan2(z + (e2_*b)*np.sin(t)**3, r - (e2*a)*np.cos(t)**3)\n    L = np.arctan2(y, x)\n    n = a / np.sqrt(1 - e2*np.sin(B)**2)\n    H = (r / np.cos(B)) - n\n    return BLH(lat=B, lng=L, hgt=H)\n\ndef haversine_distance(blh_1, blh_2):\n    dlat = blh_2.lat - blh_1.lat\n    dlng = blh_2.lng - blh_1.lng\n    a = np.sin(dlat/2)**2 + np.cos(blh_1.lat) * np.cos(blh_2.lat) * np.sin(dlng/2)**2\n    dist = 2 * HAVERSINE_RADIUS * np.arcsin(np.sqrt(a))\n    return dist\n\ndef pandas_haversine_distance(df1, df2):\n    blh1 = BLH(\n        lat=np.deg2rad(df1['LatitudeDegrees'].to_numpy()),\n        lng=np.deg2rad(df1['LongitudeDegrees'].to_numpy()),\n        hgt=0,\n    )\n    blh2 = BLH(\n        lat=np.deg2rad(df2['LatitudeDegrees'].to_numpy()),\n        lng=np.deg2rad(df2['LongitudeDegrees'].to_numpy()),\n        hgt=0,\n    )\n    return haversine_distance(blh1, blh2)","metadata":{"execution":{"iopub.status.busy":"2022-05-10T07:39:46.528555Z","iopub.execute_input":"2022-05-10T07:39:46.529256Z","iopub.status.idle":"2022-05-10T07:39:46.549057Z","shell.execute_reply.started":"2022-05-10T07:39:46.529213Z","shell.execute_reply":"2022-05-10T07:39:46.548016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ecef_to_lat_lng(tripID, gnss_df, UnixTimeMillis):\n    ecef_columns = ['WlsPositionXEcefMeters', 'WlsPositionYEcefMeters', 'WlsPositionZEcefMeters']\n    columns = ['utcTimeMillis'] + ecef_columns\n    ecef_df = (gnss_df.drop_duplicates(subset='utcTimeMillis')[columns]\n               .dropna().reset_index(drop=True))\n    ecef = ECEF.from_numpy(ecef_df[ecef_columns].to_numpy())\n    blh  = ECEF_to_BLH(ecef)\n\n    TIME = ecef_df['utcTimeMillis'].to_numpy()\n    lat = InterpolatedUnivariateSpline(TIME, blh.lat, ext=3)(UnixTimeMillis)\n    lng = InterpolatedUnivariateSpline(TIME, blh.lng, ext=3)(UnixTimeMillis)\n    return pd.DataFrame({\n        'tripId' : tripID,\n        'UnixTimeMillis'   : UnixTimeMillis,\n        'LatitudeDegrees'  : np.degrees(lat),\n        'LongitudeDegrees' : np.degrees(lng),\n    })","metadata":{"execution":{"iopub.status.busy":"2022-05-10T07:42:09.769823Z","iopub.execute_input":"2022-05-10T07:42:09.770164Z","iopub.status.idle":"2022-05-10T07:42:09.779099Z","shell.execute_reply.started":"2022-05-10T07:42:09.770130Z","shell.execute_reply":"2022-05-10T07:42:09.778113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calc_score(tripID, pred_df, gt_df):\n    d = pandas_haversine_distance(pred_df, gt_df)\n    score = np.mean([np.quantile(d, 0.50), np.quantile(d, 0.95)])    \n    return score","metadata":{"execution":{"iopub.status.busy":"2022-05-10T07:46:01.760927Z","iopub.execute_input":"2022-05-10T07:46:01.761305Z","iopub.status.idle":"2022-05-10T07:46:01.768554Z","shell.execute_reply.started":"2022-05-10T07:46:01.761270Z","shell.execute_reply":"2022-05-10T07:46:01.767229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Get Baseline**","metadata":{}},{"cell_type":"code","source":"INPUT_PATH = '../input/smartphone-decimeter-2022'","metadata":{"execution":{"iopub.status.busy":"2022-05-10T07:40:01.118629Z","iopub.execute_input":"2022-05-10T07:40:01.118941Z","iopub.status.idle":"2022-05-10T07:40:01.123917Z","shell.execute_reply.started":"2022-05-10T07:40:01.118911Z","shell.execute_reply":"2022-05-10T07:40:01.122801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%capture --no-stdout\n\npred_dfs  = []\nscore_list = []\n\nfor dirname in tqdm(sorted(glob.glob(f'{INPUT_PATH}/train/*/*'))):\n    drive, phone = dirname.split('/')[-2:]\n    tripID  = f'{drive}/{phone}'\n    gnss_df = pd.read_csv(f'{dirname}/device_gnss.csv')\n    gt_df   = pd.read_csv(f'{dirname}/ground_truth.csv')\n    pred_df = ecef_to_lat_lng(tripID, gnss_df, gt_df['UnixTimeMillis'].to_numpy())\n    pred_dfs.append(pred_df)\n    score = calc_score(tripID, pred_df, gt_df)\n#     print(f'{tripID:<45}: score = {score:.3f}')\n    score_list.append(score)","metadata":{"execution":{"iopub.status.busy":"2022-05-10T07:59:47.715141Z","iopub.execute_input":"2022-05-10T07:59:47.715470Z","iopub.status.idle":"2022-05-10T08:02:24.128354Z","shell.execute_reply.started":"2022-05-10T07:59:47.715434Z","shell.execute_reply":"2022-05-10T08:02:24.127489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df = pd.read_csv(f'{INPUT_PATH}/sample_submission.csv')\npred_dfs  = []\nfor dirname in tqdm(sorted(glob.glob(f'{INPUT_PATH}/test/*/*'))):\n    drive, phone = dirname.split('/')[-2:]\n    tripID  = f'{drive}/{phone}'\n    gnss_df = pd.read_csv(f'{dirname}/device_gnss.csv')\n    UnixTimeMillis = sample_df[sample_df['tripId'] == tripID]['UnixTimeMillis'].to_numpy()\n    pred_dfs.append(ecef_to_lat_lng(tripID, gnss_df, UnixTimeMillis))\nbaseline_test_df = pd.concat(pred_dfs)\nbaseline_test_df.to_csv('baseline_test.csv', index=False)\nbaseline_test_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-05-10T08:20:11.211840Z","iopub.execute_input":"2022-05-10T08:20:11.212286Z","iopub.status.idle":"2022-05-10T08:21:04.674480Z","shell.execute_reply.started":"2022-05-10T08:20:11.212247Z","shell.execute_reply":"2022-05-10T08:21:04.673333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Exploratory Data Analysis**","metadata":{}}]}