{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport glob\nimport os\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm\nfrom pathlib import Path\nimport plotly.express as px\nimport math","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-17T23:36:46.39804Z","iopub.execute_input":"2021-07-17T23:36:46.398411Z","iopub.status.idle":"2021-07-17T23:36:46.403736Z","shell.execute_reply.started":"2021-07-17T23:36:46.398379Z","shell.execute_reply":"2021-07-17T23:36:46.402846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = Path(\"../input/google-smartphone-decimeter-challenge\")\ntrain_df = pd.read_csv(data_dir / \"baseline_locations_train.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-07-17T23:36:46.405321Z","iopub.execute_input":"2021-07-17T23:36:46.405693Z","iopub.status.idle":"2021-07-17T23:36:46.689909Z","shell.execute_reply.started":"2021-07-17T23:36:46.405664Z","shell.execute_reply":"2021-07-17T23:36:46.688831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get all ground truth dataframe\ngt_df = pd.DataFrame()\nfor (collection_name, phone_name), df in tqdm(train_df.groupby([\"collectionName\", \"phoneName\"])):\n    path = data_dir / f\"train/{collection_name}/{phone_name}/ground_truth.csv\"\n    df = pd.read_csv(path)  \n    gt_df = pd.concat([gt_df, df]).reset_index(drop=True)   \ngt_df.head()\n\ngt_df['phone'] = gt_df['collectionName'] + '_' + gt_df['phoneName']\n\ntrain_data = pd.read_csv('../input/google-smartphone-decimeter-challenge/baseline_locations_train.csv')","metadata":{"execution":{"iopub.status.busy":"2021-07-17T23:36:46.691466Z","iopub.execute_input":"2021-07-17T23:36:46.691786Z","iopub.status.idle":"2021-07-17T23:36:50.075444Z","shell.execute_reply.started":"2021-07-17T23:36:46.691749Z","shell.execute_reply":"2021-07-17T23:36:50.074736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calc_haversine(lat1, lon1, lat2, lon2):\n    \"\"\"Calculates the great circle distance between two points\n    on the earth. Inputs are array-like and specified in decimal degrees.\n    \"\"\"\n    RADIUS = 6_367_000\n    lat1, lon1, lat2, lon2 = map(np.radians, [lat1, lon1, lat2, lon2])\n    dlat = lat2 - lat1\n    dlon = lon2 - lon1\n    a = np.sin(dlat/2)**2 + \\\n        np.cos(lat1) * np.cos(lat2) * np.sin(dlon/2)**2\n    dist = 2 * RADIUS * np.arcsin(a**0.5)\n    return dist\n\ndef percentile50(x):\n    return np.percentile(x, 50)\ndef percentile95(x):\n    return np.percentile(x, 95)\n\ndef get_train_score(df, gt):\n    gt = gt.rename(columns={'latDeg':'latDeg_gt', 'lngDeg':'lngDeg_gt'})\n    df = df.merge(gt, on=['collectionName', 'phoneName', 'millisSinceGpsEpoch'], how='inner')\n    # calc_distance_error\n    df['err'] = calc_haversine(df['latDeg_gt'], df['lngDeg_gt'], df['latDeg'], df['lngDeg'])\n    # calc_evaluate_score\n    df['phone'] = df['collectionName'] + '_' + df['phoneName']\n    res = df.groupby('phone')['err'].agg([percentile50, percentile95])\n    res['p50_p90_mean'] = (res['percentile50'] + res['percentile95']) / 2 \n    score = res['p50_p90_mean'].mean()\n    return score","metadata":{"execution":{"iopub.status.busy":"2021-07-17T23:36:50.076785Z","iopub.execute_input":"2021-07-17T23:36:50.07717Z","iopub.status.idle":"2021-07-17T23:36:50.087884Z","shell.execute_reply.started":"2021-07-17T23:36:50.077141Z","shell.execute_reply":"2021-07-17T23:36:50.086856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gt_df['gt'] = 1\ntrain_data['gt'] = 0","metadata":{"execution":{"iopub.status.busy":"2021-07-17T23:36:52.540422Z","iopub.execute_input":"2021-07-17T23:36:52.540683Z","iopub.status.idle":"2021-07-17T23:36:52.557851Z","shell.execute_reply.started":"2021-07-17T23:36:52.540657Z","shell.execute_reply":"2021-07-17T23:36:52.557024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now let us introduce some helper functions which would help us to get lat/lng. For more info on that you may want to look for converting a degrees-minutes-seconds coordinate pairs to decimal degrees.","metadata":{}},{"cell_type":"code","source":"def dm(x):\n    degrees = x // 100\n    minutes = x - 100*degrees\n\n    return degrees, minutes\n\ndef decimal_degrees(degrees, minutes):\n    return degrees + minutes/60 ","metadata":{"execution":{"iopub.status.busy":"2021-07-17T23:36:57.012266Z","iopub.execute_input":"2021-07-17T23:36:57.012747Z","iopub.status.idle":"2021-07-17T23:36:57.01905Z","shell.execute_reply.started":"2021-07-17T23:36:57.01268Z","shell.execute_reply":"2021-07-17T23:36:57.017851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_path = Path(\"../input/android-smartphones-high-accuracy-datasets/training\")\n\ntruths = (data_path).rglob('*chipset.nmea')\n    # returns a generator\n    \ndf_list = []\n\nfor t in tqdm(truths, total=34):\n    \n    df_phone = pd.read_table(t, sep=',', header=None, index_col=1, parse_dates=True, comment='*')\n    \n    df_phone['phone'] = t\n    print(t)\n    \n    df_list.append(df_phone)\n    \ndf_truth = pd.concat(df_list, ignore_index=True)\n\n#The next steps are just input data processing. \n#To understand it better, you may want to run it line by line and compare with the previous line.\n\ndf_truth = df_truth[[2,4, 'phone']]\ndf_truth.rename(columns={2: 'latDeg', 4: 'lngDeg'}, inplace=True)\n\ndf_truth = df_truth[df_truth.latDeg != 'A']\ndf_truth[\"latDeg\"] = pd.to_numeric(df_truth[\"latDeg\"], downcast=\"float\")\ndf_truth[\"lngDeg\"] = pd.to_numeric(df_truth[\"lngDeg\"], downcast=\"float\")\ndf_truth['latDeg'] = decimal_degrees(*dm(df_truth['latDeg']))\ndf_truth['lngDeg'] = 0 - decimal_degrees(*dm(df_truth['lngDeg']))\n\ndf_truth['gt'] = 2\n\ndf_truth","metadata":{"execution":{"iopub.status.busy":"2021-07-17T23:36:57.020395Z","iopub.execute_input":"2021-07-17T23:36:57.020883Z","iopub.status.idle":"2021-07-17T23:36:58.198742Z","shell.execute_reply.started":"2021-07-17T23:36:57.020849Z","shell.execute_reply":"2021-07-17T23:36:58.197729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now let us see that newly added data is indeed new and wasnt included as the basic training datasets.","metadata":{}},{"cell_type":"code","source":"ccc = pd.concat([df_truth, gt_df],ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2021-07-17T23:36:58.227644Z","iopub.execute_input":"2021-07-17T23:36:58.228261Z","iopub.status.idle":"2021-07-17T23:36:58.259936Z","shell.execute_reply.started":"2021-07-17T23:36:58.228212Z","shell.execute_reply":"2021-07-17T23:36:58.257794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.scatter_mapbox(ccc,\n\n                    # Here, plotly gets, (x,y) coordinates\n                    lat=\"latDeg\",\n                    lon=\"lngDeg\",\n                    #text='phoneName',\n\n                    #Here, plotly detects color of series\n                    color=\"gt\",\n                    #labels=\"collectionName\",\n\n                    zoom=9,\n                    center={\"lat\":37.423576, \"lon\":-122.094132},\n                    height=600,\n                    width=800)\nfig.update_layout(mapbox_style='open-street-map')\nfig.update_layout(margin={\"r\": 0, \"t\": 0, \"l\": 0, \"b\": 0})\nfig.update_layout(title_text=\"GPS trafic\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T23:36:58.261421Z","iopub.execute_input":"2021-07-17T23:36:58.261881Z","iopub.status.idle":"2021-07-17T23:36:58.269303Z","shell.execute_reply.started":"2021-07-17T23:36:58.261837Z","shell.execute_reply":"2021-07-17T23:36:58.268347Z"},"trusted":true},"execution_count":null,"outputs":[]}]}