{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### We divided the data into three regions and evaluated them.\n\nUsing [this discussion](https://www.kaggle.com/c/google-smartphone-decimeter-challenge/discussion/245160) as a guide, we divided the data into three areas: highways, streets with trees, and city streets, and evaluated the training data in each area.\n\nThe results are as follows.\n\nhighway : 3.4528073895024414\n\ntree : 6.173261717576203\n\ndowntown : 19.432900281799608\n\nThank you for sharing this discusiion with us.\n\nhttps://www.kaggle.com/c/google-smartphone-decimeter-challenge/discussion/245160\n\nI used the following code as a reference for the evaluation script.\n\nThank you for sharing code with us.\n\nhttps://www.kaggle.com/t88take/gsdc-phones-mean-prediction#evaluate-train-score","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport pathlib\nfrom tqdm.notebook import tqdm\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-19T05:53:36.557522Z","iopub.execute_input":"2021-06-19T05:53:36.558201Z","iopub.status.idle":"2021-06-19T05:53:36.572905Z","shell.execute_reply.started":"2021-06-19T05:53:36.558088Z","shell.execute_reply":"2021-06-19T05:53:36.571731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(\"../input/google-smartphone-decimeter-challenge/baseline_locations_train.csv\")\ndf_test = pd.read_csv(\"../input/google-smartphone-decimeter-challenge/baseline_locations_test.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-06-19T05:53:36.577674Z","iopub.execute_input":"2021-06-19T05:53:36.577983Z","iopub.status.idle":"2021-06-19T05:53:36.919333Z","shell.execute_reply.started":"2021-06-19T05:53:36.577953Z","shell.execute_reply":"2021-06-19T05:53:36.918245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_collectionName = df_train[\"collectionName\"].unique()","metadata":{"execution":{"iopub.status.busy":"2021-06-19T05:53:36.920651Z","iopub.execute_input":"2021-06-19T05:53:36.920935Z","iopub.status.idle":"2021-06-19T05:53:36.937518Z","shell.execute_reply.started":"2021-06-19T05:53:36.920903Z","shell.execute_reply":"2021-06-19T05:53:36.936534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_highway = df_train[df_train['collectionName'].isin([train_collectionName[0],\n                                                           train_collectionName[1],\n                                                           train_collectionName[2],\n                                                           train_collectionName[3],\n                                                           train_collectionName[4],\n                                                           train_collectionName[5],\n                                                           train_collectionName[6],\n                                                           train_collectionName[7],\n                                                           train_collectionName[8],\n                                                           train_collectionName[9],\n                                                           train_collectionName[10],\n                                                           train_collectionName[11],\n                                                           train_collectionName[12],\n                                                           train_collectionName[13],\n                                                           train_collectionName[14],\n                                                           train_collectionName[15],\n                                                           train_collectionName[16],\n                                                           train_collectionName[17],\n                                                           train_collectionName[18],\n                                                           train_collectionName[19],\n                                                           train_collectionName[20]])]","metadata":{"execution":{"iopub.status.busy":"2021-06-19T05:53:36.938701Z","iopub.execute_input":"2021-06-19T05:53:36.938979Z","iopub.status.idle":"2021-06-19T05:53:36.958951Z","shell.execute_reply.started":"2021-06-19T05:53:36.938952Z","shell.execute_reply":"2021-06-19T05:53:36.957897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_tree = df_train[df_train['collectionName'].isin([train_collectionName[21],\n                                                          train_collectionName[22],\n                                                          train_collectionName[24],\n                                                          train_collectionName[25],\n                                                          train_collectionName[27]])]","metadata":{"execution":{"iopub.status.busy":"2021-06-19T05:53:36.961648Z","iopub.execute_input":"2021-06-19T05:53:36.961941Z","iopub.status.idle":"2021-06-19T05:53:36.975126Z","shell.execute_reply.started":"2021-06-19T05:53:36.961912Z","shell.execute_reply":"2021-06-19T05:53:36.974147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_downtown = df_train[df_train['collectionName'].isin([train_collectionName[23],\n                                                              train_collectionName[26],\n                                                              train_collectionName[28]])]","metadata":{"execution":{"iopub.status.busy":"2021-06-19T05:53:36.976753Z","iopub.execute_input":"2021-06-19T05:53:36.977110Z","iopub.status.idle":"2021-06-19T05:53:37.000100Z","shell.execute_reply.started":"2021-06-19T05:53:36.977079Z","shell.execute_reply":"2021-06-19T05:53:36.998970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ground_truth\np = pathlib.Path(\"../input/google-smartphone-decimeter-challenge\")\ngt_files = list(p.glob('train/*/*/ground_truth.csv'))\nprint('ground_truth.csv count : ', len(gt_files))\n\ngts = []\nfor gt_file in tqdm(gt_files):\n    gts.append(pd.read_csv(gt_file))\nground_truth = pd.concat(gts)","metadata":{"execution":{"iopub.status.busy":"2021-06-19T05:53:37.001772Z","iopub.execute_input":"2021-06-19T05:53:37.002325Z","iopub.status.idle":"2021-06-19T05:53:37.716411Z","shell.execute_reply.started":"2021-06-19T05:53:37.002261Z","shell.execute_reply":"2021-06-19T05:53:37.715400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calc_haversine(lat1, lon1, lat2, lon2):\n    \"\"\"Calculates the great circle distance between two points\n    on the earth. Inputs are array-like and specified in decimal degrees.\n    \"\"\"\n    RADIUS = 6_367_000\n    lat1, lon1, lat2, lon2 = map(np.radians, [lat1, lon1, lat2, lon2])\n    dlat = lat2 - lat1\n    dlon = lon2 - lon1\n    a = np.sin(dlat/2)**2 + \\\n        np.cos(lat1) * np.cos(lat2) * np.sin(dlon/2)**2\n    dist = 2 * RADIUS * np.arcsin(a**0.5)\n    return dist","metadata":{"execution":{"iopub.status.busy":"2021-06-19T05:53:37.717980Z","iopub.execute_input":"2021-06-19T05:53:37.718402Z","iopub.status.idle":"2021-06-19T05:53:37.726094Z","shell.execute_reply.started":"2021-06-19T05:53:37.718354Z","shell.execute_reply":"2021-06-19T05:53:37.725317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def percentile50(x):\n    return np.percentile(x, 50)\ndef percentile95(x):\n    return np.percentile(x, 95)","metadata":{"execution":{"iopub.status.busy":"2021-06-19T05:53:37.727344Z","iopub.execute_input":"2021-06-19T05:53:37.727665Z","iopub.status.idle":"2021-06-19T05:53:37.738113Z","shell.execute_reply.started":"2021-06-19T05:53:37.727636Z","shell.execute_reply":"2021-06-19T05:53:37.736919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_train_score(df, gt):\n    gt = gt.rename(columns={'latDeg':'latDeg_gt', 'lngDeg':'lngDeg_gt'})\n    df = df.merge(gt, on=['collectionName', 'phoneName', 'millisSinceGpsEpoch'], how='inner')\n    # calc_distance_error\n    df['err'] = calc_haversine(df['latDeg_gt'], df['lngDeg_gt'], df['latDeg'], df['lngDeg'])\n    # calc_evaluate_score\n    df['phone'] = df['collectionName'] + '_' + df['phoneName']\n    res = df.groupby('phone')['err'].agg([percentile50, percentile95])\n    res['p50_p90_mean'] = (res['percentile50'] + res['percentile95']) / 2 \n    score = res['p50_p90_mean'].mean()\n    return score,df","metadata":{"execution":{"iopub.status.busy":"2021-06-19T05:53:37.739487Z","iopub.execute_input":"2021-06-19T05:53:37.739862Z","iopub.status.idle":"2021-06-19T05:53:37.751016Z","shell.execute_reply.started":"2021-06-19T05:53:37.739828Z","shell.execute_reply":"2021-06-19T05:53:37.749989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score_highway,df_highway = get_train_score(df_train_highway, ground_truth)\nscore_tree,df_tree = get_train_score(df_train_tree, ground_truth)\nscore_downtown,df_downtown = get_train_score(df_train_downtown, ground_truth)","metadata":{"execution":{"iopub.status.busy":"2021-06-19T05:53:37.752305Z","iopub.execute_input":"2021-06-19T05:53:37.752802Z","iopub.status.idle":"2021-06-19T05:53:38.224463Z","shell.execute_reply.started":"2021-06-19T05:53:37.752770Z","shell.execute_reply":"2021-06-19T05:53:38.223438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"highway :\" , score_highway )\nprint('tree : ' ,score_tree)\nprint('downtown : ' , score_downtown)","metadata":{"execution":{"iopub.status.busy":"2021-06-19T05:53:38.225734Z","iopub.execute_input":"2021-06-19T05:53:38.226011Z","iopub.status.idle":"2021-06-19T05:53:38.233245Z","shell.execute_reply.started":"2021-06-19T05:53:38.225985Z","shell.execute_reply":"2021-06-19T05:53:38.232238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"c1,c2,c3 = \"blue\",\"green\",\"red\"\nfig, ax = plt.subplots(nrows=1, ncols=3,figsize = (25,5))\nax[0].hist(df_highway.err, bins=range(100), color=c1)\nax[0].set_title('highway')\nax[0].set_xlabel('error',fontsize = 20)\nax[0].set_ylabel('freq')\nax[0].set_ylim(0,30000)\nax[1].hist(df_tree.err, bins=range(100), color=c2)\nax[1].set_title('tree')\nax[1].set_xlabel('error',fontsize = 20)\nax[1].set_ylabel('freq')\nax[1].set_ylim(0,30000)\nax[2].hist(df_downtown.err, bins=range(100), color=c3)\nax[2].set_title('downtown')\nax[2].set_xlabel('error',fontsize = 20)\nax[2].set_ylabel('freq')\nax[2].set_ylim(0,30000)","metadata":{"execution":{"iopub.status.busy":"2021-06-19T05:53:38.234584Z","iopub.execute_input":"2021-06-19T05:53:38.234884Z","iopub.status.idle":"2021-06-19T05:53:39.177395Z","shell.execute_reply.started":"2021-06-19T05:53:38.234854Z","shell.execute_reply":"2021-06-19T05:53:39.176366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize = (10,5))\nplt.hist([df_highway.err,df_tree.err,df_downtown.err], stacked=True, bins=range(100),color=[c1,c2,c3], label=[\"highway\",\"tree\",\"downtown\"])\nplt.xlabel('error',fontsize = 15)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-19T05:54:17.372529Z","iopub.execute_input":"2021-06-19T05:54:17.372910Z","iopub.status.idle":"2021-06-19T05:54:18.338205Z","shell.execute_reply.started":"2021-06-19T05:54:17.372880Z","shell.execute_reply":"2021-06-19T05:54:18.337330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**We can see that the data in the city is small in number but has a large error. In the evaluation of this competition, it may be effective to approach the data where the error is large even if the number is small.**","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}