{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"In this competition, baseline locations are provided in the ECEF(Earth-Centered Earth-Fixed) coordinate system, so the coordinate system must be converted for submission.","metadata":{}},{"cell_type":"code","source":"import glob\nfrom dataclasses import dataclass\nimport numpy as np\nimport pandas as pd\nfrom tqdm.notebook import tqdm\nfrom scipy.interpolate import InterpolatedUnivariateSpline#\n\nINPUT_PATH = '../input/smartphone-decimeter-2022'\n\nWGS84_SEMI_MAJOR_AXIS = 6378137.0\nWGS84_SEMI_MINOR_AXIS = 6356752.314245\nWGS84_SQUARED_FIRST_ECCENTRICITY  = 6.69437999013e-3\nWGS84_SQUARED_SECOND_ECCENTRICITY = 6.73949674226e-3\n\nHAVERSINE_RADIUS = 6_371_000","metadata":{"execution":{"iopub.status.busy":"2022-06-07T07:39:47.477628Z","iopub.execute_input":"2022-06-07T07:39:47.477853Z","iopub.status.idle":"2022-06-07T07:39:47.807102Z","shell.execute_reply.started":"2022-06-07T07:39:47.477827Z","shell.execute_reply":"2022-06-07T07:39:47.806409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@dataclass\nclass ECEF:\n    x: np.array\n    y: np.array\n    z: np.array\n\n    def to_numpy(self):\n        return np.stack([self.x, self.y, self.z], axis=0)\n\n    @staticmethod\n    def from_numpy(pos):\n        x, y, z = [np.squeeze(w) for w in np.split(pos, 3, axis=-1)]\n        return ECEF(x=x, y=y, z=z)\n\n@dataclass\nclass BLH:\n    lat : np.array\n    lng : np.array\n    hgt : np.array\n\ndef ECEF_to_BLH(ecef):\n    #ここが計算の肝。ただ変更することは特にない？ない？\n    a = WGS84_SEMI_MAJOR_AXIS\n    b = WGS84_SEMI_MINOR_AXIS\n    e2  = WGS84_SQUARED_FIRST_ECCENTRICITY\n    e2_ = WGS84_SQUARED_SECOND_ECCENTRICITY\n    x = ecef.x\n    y = ecef.y\n    z = ecef.z\n    r = np.sqrt(x**2 + y**2)\n    t = np.arctan2(z * (a/b), r)\n    B = np.arctan2(z + (e2_*b)*np.sin(t)**3, r - (e2*a)*np.cos(t)**3)\n    L = np.arctan2(y, x)\n    n = a / np.sqrt(1 - e2*np.sin(B)**2)\n    H = (r / np.cos(B)) - n\n    return BLH(lat=B, lng=L, hgt=H)\n\ndef haversine_distance(blh_1, blh_2):\n    dlat = blh_2.lat - blh_1.lat\n    dlng = blh_2.lng - blh_1.lng\n    a = np.sin(dlat/2)**2 + np.cos(blh_1.lat) * np.cos(blh_2.lat) * np.sin(dlng/2)**2\n    dist = 2 * HAVERSINE_RADIUS * np.arcsin(np.sqrt(a))\n    return dist\n\ndef pandas_haversine_distance(df1, df2):\n    blh1 = BLH(\n        lat=np.deg2rad(df1['LatitudeDegrees'].to_numpy()),\n        lng=np.deg2rad(df1['LongitudeDegrees'].to_numpy()),\n        hgt=0,\n    )\n    blh2 = BLH(\n        lat=np.deg2rad(df2['LatitudeDegrees'].to_numpy()),\n        lng=np.deg2rad(df2['LongitudeDegrees'].to_numpy()),\n        hgt=0,\n    )\n    return haversine_distance(blh1, blh2)","metadata":{"execution":{"iopub.status.busy":"2022-06-07T07:39:47.8085Z","iopub.execute_input":"2022-06-07T07:39:47.808727Z","iopub.status.idle":"2022-06-07T07:39:47.825597Z","shell.execute_reply.started":"2022-06-07T07:39:47.8087Z","shell.execute_reply":"2022-06-07T07:39:47.824737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ecef_to_lat_lng(tripID, gnss_df, UnixTimeMillis):\n    ecef_columns = ['WlsPositionXEcefMeters', 'WlsPositionYEcefMeters', 'WlsPositionZEcefMeters']\n    columns = ['utcTimeMillis'] + ecef_columns\n    ecef_df = (gnss_df.drop_duplicates(subset='utcTimeMillis')[columns]\n               .dropna().reset_index(drop=True))\n    \n    ecef = ECEF.from_numpy(ecef_df[ecef_columns].to_numpy())\n    blh  = ECEF_to_BLH(ecef)\n\n    TIME = ecef_df['utcTimeMillis'].to_numpy()\n    #受信時刻と求めたい時間の線形補間\n    lat = InterpolatedUnivariateSpline(TIME, blh.lat, ext=3)(UnixTimeMillis)\n    lng = InterpolatedUnivariateSpline(TIME, blh.lng, ext=3)(UnixTimeMillis)\n    hgt = InterpolatedUnivariateSpline(TIME, blh.hgt, ext=3)(UnixTimeMillis)\n    ecefx = InterpolatedUnivariateSpline(TIME, ecef_df.WlsPositionXEcefMeters, ext=3)(UnixTimeMillis)\n    ecefy = InterpolatedUnivariateSpline(TIME, ecef_df.WlsPositionYEcefMeters, ext=3)(UnixTimeMillis)\n    ecefz = InterpolatedUnivariateSpline(TIME, ecef_df.WlsPositionZEcefMeters, ext=3)(UnixTimeMillis)\n\n\n    return pd.DataFrame({\n        'tripId' : tripID,\n        'UnixTimeMillis'   : UnixTimeMillis,\n        'LatitudeDegrees'  : np.degrees(lat),\n        'LongitudeDegrees' : np.degrees(lng),\n        'higth' : hgt,\n        'WlsPositionXEcefMeters' : ecefx,\n        'WlsPositionYEcefMeters' : ecefy,\n        'WlsPositionZEcefMeters' : ecefz,\n    })\n\ndef calc_score(tripID, pred_df, gt_df):\n    d = pandas_haversine_distance(pred_df, gt_df)\n    score = np.mean([np.quantile(d, 0.50), np.quantile(d, 0.95)])    \n    return score","metadata":{"execution":{"iopub.status.busy":"2022-06-07T07:41:42.572212Z","iopub.execute_input":"2022-06-07T07:41:42.572515Z","iopub.status.idle":"2022-06-07T07:41:42.585824Z","shell.execute_reply.started":"2022-06-07T07:41:42.572473Z","shell.execute_reply":"2022-06-07T07:41:42.584599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture --no-stdout\n\npred_dfs  = []\nscore_list = []\nfor dirname in sorted(glob.glob(f'{INPUT_PATH}/train/*/*')):\n    drive, phone = dirname.split('/')[-2:]\n    tripID  = f'{drive}/{phone}'\n    gnss_df = pd.read_csv(f'{dirname}/device_gnss.csv')\n    gt_df   = pd.read_csv(f'{dirname}/ground_truth.csv')\n    pred_df = ecef_to_lat_lng(tripID, gnss_df, gt_df['UnixTimeMillis'].to_numpy())\n    pred_dfs.append(pred_df)\n    score = calc_score(tripID, pred_df, gt_df)\n    print(f'{tripID:<45}: score = {score:.3f}')\n    score_list.append(score)","metadata":{"execution":{"iopub.status.busy":"2022-06-07T07:41:43.469262Z","iopub.execute_input":"2022-06-07T07:41:43.469571Z","iopub.status.idle":"2022-06-07T07:44:17.528365Z","shell.execute_reply.started":"2022-06-07T07:41:43.469524Z","shell.execute_reply":"2022-06-07T07:44:17.526774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gnss_df.columns","metadata":{"execution":{"iopub.status.busy":"2022-06-07T07:44:17.53014Z","iopub.execute_input":"2022-06-07T07:44:17.530378Z","iopub.status.idle":"2022-06-07T07:44:17.539233Z","shell.execute_reply.started":"2022-06-07T07:44:17.530349Z","shell.execute_reply":"2022-06-07T07:44:17.538382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"baseline_train_df = pd.concat(pred_dfs)\nbaseline_train_df.to_csv('baseline_train.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-06-07T07:44:17.540398Z","iopub.execute_input":"2022-06-07T07:44:17.540728Z","iopub.status.idle":"2022-06-07T07:44:22.062903Z","shell.execute_reply.started":"2022-06-07T07:44:17.540693Z","shell.execute_reply":"2022-06-07T07:44:22.062016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_score = np.mean(score_list)\nprint(f'mean_score = {mean_score:.3f}')","metadata":{"execution":{"iopub.status.busy":"2022-06-07T07:44:22.065166Z","iopub.execute_input":"2022-06-07T07:44:22.065469Z","iopub.status.idle":"2022-06-07T07:44:22.071341Z","shell.execute_reply.started":"2022-06-07T07:44:22.065429Z","shell.execute_reply":"2022-06-07T07:44:22.070363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df = pd.read_csv(f'{INPUT_PATH}/sample_submission.csv')\npred_dfs  = []\nfor dirname in tqdm(sorted(glob.glob(f'{INPUT_PATH}/test/*/*'))):\n    drive, phone = dirname.split('/')[-2:]\n    tripID  = f'{drive}/{phone}'\n    gnss_df = pd.read_csv(f'{dirname}/device_gnss.csv')\n    UnixTimeMillis = sample_df[sample_df['tripId'] == tripID]['UnixTimeMillis'].to_numpy()\n    pred_dfs.append(ecef_to_lat_lng(tripID, gnss_df, UnixTimeMillis))\nbaseline_test_df = pd.concat(pred_dfs)\nbaseline_test_df.to_csv('baseline_test.csv', index=False)\nbaseline_test_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-06-07T07:44:22.072765Z","iopub.execute_input":"2022-06-07T07:44:22.073111Z","iopub.status.idle":"2022-06-07T07:45:04.679224Z","shell.execute_reply.started":"2022-06-07T07:44:22.073068Z","shell.execute_reply":"2022-06-07T07:45:04.678604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}