{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Visualization of final submissions\n\nIn this notebook I visualize our final submissions. ( Private:2.872 , Public:3.487 )\n\n","metadata":{}},{"cell_type":"code","source":"from pathlib import Path\nimport numpy as np\nimport pandas as pd\nimport copy\nimport plotly.express as px\nimport plotly.graph_objects as go\nimport pyproj\nimport json\nimport bisect\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom matplotlib.collections import LineCollection\nfrom matplotlib.colors import ListedColormap, BoundaryNorm\nimport pickle\nimport random\nfrom tqdm.notebook import tqdm\nimport lightgbm as lgb\nimport xgboost as xgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestRegressor\nfrom catboost import CatBoostRegressor\nimport warnings\nwarnings.simplefilter('ignore')\npd.set_option('display.max_rows',30)\npd.set_option('display.max_columns',None)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-08-07T22:53:39.742998Z","iopub.execute_input":"2021-08-07T22:53:39.743382Z","iopub.status.idle":"2021-08-07T22:53:42.948398Z","shell.execute_reply.started":"2021-08-07T22:53:39.743344Z","shell.execute_reply":"2021-08-07T22:53:42.947657Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def visualize_trafic(df, center={\"lat\":37.423576, \"lon\":-122.094132}, zoom=9):\n    fig = px.scatter_mapbox(df,\n                            \n                            # Here, plotly gets, (x,y) coordinates\n                            lat=\"latDeg\",\n                            lon=\"lngDeg\",\n                            \n                            #Here, plotly detects color of series\n                            color=\"phoneName\",\n                            labels=\"phoneName\",\n                            \n                            zoom=zoom,\n                            center=center,\n                            height=500,\n                            width=750)\n    fig.update_layout(mapbox_style='stamen-terrain')\n    fig.update_layout(margin={\"r\": 0, \"t\": 0, \"l\": 0, \"b\": 0})\n    fig.update_layout(title_text=\"GPS trafic\")\n    fig.show()\n    \ndef visualize_collection(df, collection, center={\"lat\":37.423576, \"lon\":-122.094132}):\n    df_traj = df[df['collectionName'] == collection]\n    center = {\"lat\":37.423576, \"lon\":-122.094132}\n    visualize_trafic(df_traj, center)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-07T23:09:35.110873Z","iopub.execute_input":"2021-08-07T23:09:35.111229Z","iopub.status.idle":"2021-08-07T23:09:35.119539Z","shell.execute_reply.started":"2021-08-07T23:09:35.111199Z","shell.execute_reply":"2021-08-07T23:09:35.118557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calc_haversine(lat1, lon1, lat2, lon2):\n    \"\"\"Calculates the great circle distance between two points\n    on the earth. Inputs are array-like and specified in decimal degrees.\n    \"\"\"\n    RADIUS = 6_367_000\n    lat1, lon1, lat2, lon2 = map(np.radians, [lat1, lon1, lat2, lon2])\n    dlat = lat2 - lat1\n    dlon = lon2 - lon1\n    a = np.sin(dlat/2)**2 + \\\n        np.cos(lat1) * np.cos(lat2) * np.sin(dlon/2)**2\n    dist = 2 * RADIUS * np.arcsin(a**0.5)\n    return dist","metadata":{"execution":{"iopub.status.busy":"2021-08-07T22:53:42.959907Z","iopub.execute_input":"2021-08-07T22:53:42.960316Z","iopub.status.idle":"2021-08-07T22:53:42.973291Z","shell.execute_reply.started":"2021-08-07T22:53:42.960277Z","shell.execute_reply":"2021-08-07T22:53:42.97262Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def percentile50(x):\n    return np.percentile(x, 50)\ndef percentile95(x):\n    return np.percentile(x, 95)\n\ndef get_train_score(df, gt):\n    gt = gt.rename(columns={'latDeg':'latDeg_gt', 'lngDeg':'lngDeg_gt'})\n    df = df.merge(gt, on=['collectionName', 'phoneName', 'millisSinceGpsEpoch'], how='inner')\n    # calc_distance_error\n    df['err'] = calc_haversine(df['latDeg_gt'], df['lngDeg_gt'], df['latDeg'], df['lngDeg'])\n    # calc_evaluate_score\n    df['phone'] = df['collectionName'] + '_' + df['phoneName']\n    res = df.groupby('phone')['err'].agg([percentile50, percentile95])\n    res['p50_p90_mean'] = (res['percentile50'] + res['percentile95']) / 2 \n    score = res['p50_p90_mean'].mean()\n    return score","metadata":{"execution":{"iopub.status.busy":"2021-08-07T22:53:47.880247Z","iopub.execute_input":"2021-08-07T22:53:47.88075Z","iopub.status.idle":"2021-08-07T22:53:47.887505Z","shell.execute_reply.started":"2021-08-07T22:53:47.880716Z","shell.execute_reply":"2021-08-07T22:53:47.886857Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def eval_all(df_pred, df_gt):\n    scores = []\n    compared_cols = [\"latDeg_truth\",\"lngDeg_truth\",\"latDeg_pred\",\"lngDeg_pred\"]\n    collections = sorted(df_gt['collectionName'].unique())\n    for collection in collections:\n        df_pred_col = df_pred[df_pred['collectionName'] == collection]\n        df_gt_col = df_gt[df_gt['collectionName'] == collection]\n        \n        score = get_train_score(df_pred_col, df_gt_col)\n        \n        df_merged = pd.merge_asof(df_gt_col.sort_values('millisSinceGpsEpoch'), df_pred_col.sort_values('millisSinceGpsEpoch'), \n                                  on=\"millisSinceGpsEpoch\", by=[\"collectionName\", \"phoneName\"], \n                                  direction='nearest',tolerance=100000, suffixes=('_truth', '_pred'))\n        df_merged = df_merged.sort_values(by=[\"collectionName\", \"phoneName\", \"millisSinceGpsEpoch\"], ignore_index=True)\n\n        haversine = calc_haversine(*df_merged[compared_cols].to_numpy().transpose()).mean()\n        scores.append([collection, haversine, score])\n    \n    score = get_train_score(df_pred, df_gt)\n    df_merged = pd.merge_asof(df_gt.sort_values('millisSinceGpsEpoch'), df_pred.sort_values('millisSinceGpsEpoch'), \n                              on=\"millisSinceGpsEpoch\", by=[\"collectionName\", \"phoneName\"], \n                              direction='nearest',tolerance=100000, suffixes=('_truth', '_pred'))\n    haversine = calc_haversine(*df_merged[compared_cols].to_numpy().transpose()).mean()\n    scores.append(['all', haversine, score])\n    \n    df_scores = pd.DataFrame(scores, columns=['collection', 'haversine', 'score'])\n    return df_scores","metadata":{"execution":{"iopub.status.busy":"2021-08-07T22:53:53.653516Z","iopub.execute_input":"2021-08-07T22:53:53.654033Z","iopub.status.idle":"2021-08-07T22:53:53.66366Z","shell.execute_reply.started":"2021-08-07T22:53:53.654003Z","shell.execute_reply":"2021-08-07T22:53:53.662952Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datapath = Path(\"../input/google-smartphone-decimeter-challenge/\")\nground_truths = (datapath / \"train\").rglob(\"ground_truth.csv\")\ndf_gt = pd.concat([pd.read_csv(filepath) for filepath in ground_truths], ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2021-08-07T22:53:59.53443Z","iopub.execute_input":"2021-08-07T22:53:59.534936Z","iopub.status.idle":"2021-08-07T22:54:01.861259Z","shell.execute_reply.started":"2021-08-07T22:53:59.534894Z","shell.execute_reply":"2021-08-07T22:54:01.860301Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('../input/210804-sub-c/train_submission.csv')\n\ndf_sub = pd.read_csv('../input/210804-sub-c/submission.csv')\ntmp = df_sub['phone'].apply(lambda s : pd.Series(s.split('_')))\ndf_sub['collectionName'] = tmp[0]\ndf_sub['phoneName'] = tmp[1]","metadata":{"execution":{"iopub.status.busy":"2021-08-07T23:08:48.672487Z","iopub.execute_input":"2021-08-07T23:08:48.673217Z","iopub.status.idle":"2021-08-07T23:09:11.158953Z","shell.execute_reply.started":"2021-08-07T23:08:48.673171Z","shell.execute_reply":"2021-08-07T23:09:11.158215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train","metadata":{}},{"cell_type":"code","source":"eval_all(df_train, df_gt)","metadata":{"execution":{"iopub.status.busy":"2021-08-07T23:05:57.919129Z","iopub.execute_input":"2021-08-07T23:05:57.919682Z","iopub.status.idle":"2021-08-07T23:06:01.366259Z","shell.execute_reply.started":"2021-08-07T23:05:57.919647Z","shell.execute_reply":"2021-08-07T23:06:01.365316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_trafic(df_train)","metadata":{"execution":{"iopub.status.busy":"2021-08-07T23:09:38.844069Z","iopub.execute_input":"2021-08-07T23:09:38.844428Z","iopub.status.idle":"2021-08-07T23:09:39.778616Z","shell.execute_reply.started":"2021-08-07T23:09:38.844389Z","shell.execute_reply":"2021-08-07T23:09:39.777693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"visualize_trafic(df_sub)","metadata":{"execution":{"iopub.status.busy":"2021-08-07T23:09:46.328002Z","iopub.execute_input":"2021-08-07T23:09:46.328343Z","iopub.status.idle":"2021-08-07T23:09:47.007793Z","shell.execute_reply.started":"2021-08-07T23:09:46.328314Z","shell.execute_reply":"2021-08-07T23:09:47.006732Z"},"trusted":true},"execution_count":null,"outputs":[]}]}