{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction\nThis notebook ...","metadata":{}},{"cell_type":"markdown","source":"# Baseline submission\nBaseline location convertion code is based on @saitodevel01 [code](https://www.kaggle.com/code/saitodevel01/gsdc2-baseline-submission), thanks for a good start. ","metadata":{}},{"cell_type":"markdown","source":"In this competition, baseline locations are provided in the ECEF(Earth-Centered Earth-Fixed) coordinate system, so the coordinate system must be converted for submission.","metadata":{}},{"cell_type":"code","source":"import glob\nfrom dataclasses import dataclass\nimport numpy as np\nimport pandas as pd\nfrom tqdm.notebook import tqdm\nfrom scipy.interpolate import InterpolatedUnivariateSpline\n\nINPUT_PATH = '../input/smartphone-decimeter-2022'\n\nWGS84_SEMI_MAJOR_AXIS = 6378137.0\nWGS84_SEMI_MINOR_AXIS = 6356752.314245\nWGS84_SQUARED_FIRST_ECCENTRICITY  = 6.69437999013e-3\nWGS84_SQUARED_SECOND_ECCENTRICITY = 6.73949674226e-3\n\nHAVERSINE_RADIUS = 6_371_000","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:17:49.656851Z","iopub.execute_input":"2022-06-17T13:17:49.657728Z","iopub.status.idle":"2022-06-17T13:17:50.006196Z","shell.execute_reply.started":"2022-06-17T13:17:49.657691Z","shell.execute_reply":"2022-06-17T13:17:50.005269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@dataclass\nclass ECEF:\n    x: np.array\n    y: np.array\n    z: np.array\n\n    def to_numpy(self):\n        return np.stack([self.x, self.y, self.z], axis=0)\n\n    @staticmethod\n    def from_numpy(pos):\n        x, y, z = [np.squeeze(w) for w in np.split(pos, 3, axis=-1)]\n        return ECEF(x=x, y=y, z=z)\n\n@dataclass\nclass BLH:\n    lat : np.array\n    lng : np.array\n    hgt : np.array\n\ndef ECEF_to_BLH(ecef):\n    a = WGS84_SEMI_MAJOR_AXIS\n    b = WGS84_SEMI_MINOR_AXIS\n    e2  = WGS84_SQUARED_FIRST_ECCENTRICITY\n    e2_ = WGS84_SQUARED_SECOND_ECCENTRICITY\n    x = ecef.x\n    y = ecef.y\n    z = ecef.z\n    r = np.sqrt(x**2 + y**2)\n    t = np.arctan2(z * (a/b), r)\n    B = np.arctan2(z + (e2_*b)*np.sin(t)**3, r - (e2*a)*np.cos(t)**3)\n    L = np.arctan2(y, x)\n    n = a / np.sqrt(1 - e2*np.sin(B)**2)\n    H = (r / np.cos(B)) - n\n    return BLH(lat=B, lng=L, hgt=H)\n\ndef haversine_distance(blh_1, blh_2):\n    dlat = blh_2.lat - blh_1.lat\n    dlng = blh_2.lng - blh_1.lng\n    a = np.sin(dlat/2)**2 + np.cos(blh_1.lat) * np.cos(blh_2.lat) * np.sin(dlng/2)**2\n    dist = 2 * HAVERSINE_RADIUS * np.arcsin(np.sqrt(a))\n    return dist\n\ndef pandas_haversine_distance(df1, df2):\n    blh1 = BLH(\n        lat=np.deg2rad(df1['LatitudeDegrees'].to_numpy()),\n        lng=np.deg2rad(df1['LongitudeDegrees'].to_numpy()),\n        hgt=0,\n    )\n    blh2 = BLH(\n        lat=np.deg2rad(df2['LatitudeDegrees'].to_numpy()),\n        lng=np.deg2rad(df2['LongitudeDegrees'].to_numpy()),\n        hgt=0,\n    )\n    return haversine_distance(blh1, blh2)","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:17:56.563533Z","iopub.execute_input":"2022-06-17T13:17:56.563854Z","iopub.status.idle":"2022-06-17T13:17:56.578715Z","shell.execute_reply.started":"2022-06-17T13:17:56.563821Z","shell.execute_reply":"2022-06-17T13:17:56.577583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ecef_to_lat_lng(tripID, gnss_df, UnixTimeMillis):\n    \"\"\"\n    Convert Earth-Centered Earth-Fixed to Latitude and Longitude.\n    \n    Returns dataframe with tripID, position, and time. \n    \"\"\"\n    ecef_columns = ['WlsPositionXEcefMeters', 'WlsPositionYEcefMeters', 'WlsPositionZEcefMeters']\n    columns = ['utcTimeMillis'] + ecef_columns\n    ecef_df = (gnss_df.drop_duplicates(subset='utcTimeMillis')[columns]\n               .dropna().reset_index(drop=True))\n    ecef = ECEF.from_numpy(ecef_df[ecef_columns].to_numpy())\n    blh  = ECEF_to_BLH(ecef)\n\n    TIME = ecef_df['utcTimeMillis'].to_numpy()\n    lat = InterpolatedUnivariateSpline(TIME, blh.lat, ext=3)(UnixTimeMillis)\n    lng = InterpolatedUnivariateSpline(TIME, blh.lng, ext=3)(UnixTimeMillis)\n    return pd.DataFrame({\n        'tripId' : tripID,\n        'UnixTimeMillis'   : UnixTimeMillis,\n        'LatitudeDegrees'  : np.degrees(lat),\n        'LongitudeDegrees' : np.degrees(lng),\n    })\n\ndef calc_score(tripID, pred_df, gt_df):\n    d = pandas_haversine_distance(pred_df, gt_df)\n    score = np.mean([np.quantile(d, 0.50), np.quantile(d, 0.95)])    \n    return score","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:17:58.971412Z","iopub.execute_input":"2022-06-17T13:17:58.971694Z","iopub.status.idle":"2022-06-17T13:17:58.980057Z","shell.execute_reply.started":"2022-06-17T13:17:58.971659Z","shell.execute_reply":"2022-06-17T13:17:58.979152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture --no-stdout\n\npred_dfs  = []\nimu_dfs = []\ngnss_dfs = []\nscore_list = []\nfor dirname in sorted(glob.glob(f'{INPUT_PATH}/train/*/*')):\n    drive, phone = dirname.split('/')[-2:]\n    tripID  = f'{drive}/{phone}'\n    \n    # Load .csv files to dataframes\n    gnss_df = pd.read_csv(f'{dirname}/device_gnss.csv')\n    imu_df = pd.read_csv(f'{dirname}/device_imu.csv')\n    gt_df   = pd.read_csv(f'{dirname}/ground_truth.csv')\n    \n    # Append dfs in list\n    gnss_dfs.append(gnss_df)\n    imu_dfs.append(imu_df)\n    \n    # Convert ECEF to lat/long data\n    pred_df = ecef_to_lat_lng(tripID, gnss_df, gt_df['UnixTimeMillis'].to_numpy())\n    pred_dfs.append(pred_df)\n    \n    # Calc score\n    score = calc_score(tripID, pred_df, gt_df)\n    print(f'{tripID:<45}: score = {score:.3f}')\n    score_list.append(score)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T17:33:39.102319Z","iopub.execute_input":"2022-06-16T17:33:39.103135Z","iopub.status.idle":"2022-06-16T17:36:43.337689Z","shell.execute_reply.started":"2022-06-16T17:33:39.103083Z","shell.execute_reply":"2022-06-16T17:36:43.336753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check head of raw gnss data\ngnss_train_df = pd.concat(gnss_dfs)\ngnss_train_df.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check head of raw imu data\nimu_train_df = pd.concat(imu_dfs)\nimu_train_df.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"baseline_train_df = pd.concat(pred_dfs)\nbaseline_train_df.to_csv('baseline_train.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T17:36:43.338932Z","iopub.execute_input":"2022-06-16T17:36:43.339171Z","iopub.status.idle":"2022-06-16T17:36:45.489861Z","shell.execute_reply.started":"2022-06-16T17:36:43.339141Z","shell.execute_reply":"2022-06-16T17:36:45.488788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_score = np.mean(score_list)\nprint(f'mean_score = {mean_score:.3f}')","metadata":{"execution":{"iopub.status.busy":"2022-06-16T17:36:45.491202Z","iopub.execute_input":"2022-06-16T17:36:45.491492Z","iopub.status.idle":"2022-06-16T17:36:45.496257Z","shell.execute_reply.started":"2022-06-16T17:36:45.491459Z","shell.execute_reply":"2022-06-16T17:36:45.495481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df = pd.read_csv(f'{INPUT_PATH}/sample_submission.csv')\npred_dfs  = []\nfor dirname in tqdm(sorted(glob.glob(f'{INPUT_PATH}/test/*/*'))):\n    drive, phone = dirname.split('/')[-2:]\n    tripID  = f'{drive}/{phone}'\n    gnss_df = pd.read_csv(f'{dirname}/device_gnss.csv')\n    UnixTimeMillis = sample_df[sample_df['tripId'] == tripID]['UnixTimeMillis'].to_numpy()\n    pred_dfs.append(ecef_to_lat_lng(tripID, gnss_df, UnixTimeMillis))\nbaseline_test_df = pd.concat(pred_dfs)\nbaseline_test_df.to_csv('baseline_test.csv', index=False)\nbaseline_test_df.to_csv('submission.csv', index=False) # Commented out for further post-processing than only baseline","metadata":{"execution":{"iopub.status.busy":"2022-06-16T17:36:45.497487Z","iopub.execute_input":"2022-06-16T17:36:45.497753Z","iopub.status.idle":"2022-06-16T17:37:25.915172Z","shell.execute_reply.started":"2022-06-16T17:36:45.497725Z","shell.execute_reply":"2022-06-16T17:37:25.91397Z"},"trusted":true},"execution_count":null,"outputs":[]}]}