{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Reference\n1. https://www.kaggle.com/rohanrao/tutorial-on-reading-large-datasets\n1. https://www.kaggle.com/asobod11138/gsdc-neuralnet-keras (multi-threading)","metadata":{}},{"cell_type":"markdown","source":"# Import Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom glob import glob\nimport os\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm\nfrom pathlib import Path\nimport plotly.express as px\nfrom multiprocessing import Pool\nimport multiprocessing as multi","metadata":{"ExecuteTime":{"end_time":"2021-06-20T04:06:35.637584Z","start_time":"2021-06-20T04:06:35.622102Z"},"execution":{"iopub.status.busy":"2021-06-20T20:38:41.256754Z","iopub.execute_input":"2021-06-20T20:38:41.257381Z","iopub.status.idle":"2021-06-20T20:38:42.759915Z","shell.execute_reply.started":"2021-06-20T20:38:41.257279Z","shell.execute_reply":"2021-06-20T20:38:42.758759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Set Path and Load Dataset","metadata":{}},{"cell_type":"code","source":"PATH = Path(\"../input/google-smartphone-decimeter-challenge\")\ntrain_df = pd.read_csv(PATH / \"baseline_locations_train.csv\")\ntest_df = pd.read_csv(PATH / \"baseline_locations_test.csv\")","metadata":{"ExecuteTime":{"end_time":"2021-06-20T04:06:36.598863Z","start_time":"2021-06-20T04:06:36.350864Z"},"execution":{"iopub.status.busy":"2021-06-20T20:38:42.761549Z","iopub.execute_input":"2021-06-20T20:38:42.761861Z","iopub.status.idle":"2021-06-20T20:38:43.309087Z","shell.execute_reply.started":"2021-06-20T20:38:42.761830Z","shell.execute_reply":"2021-06-20T20:38:43.306386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df.shape)\ntrain_df.head()","metadata":{"ExecuteTime":{"end_time":"2021-06-20T04:06:37.063853Z","start_time":"2021-06-20T04:06:37.049853Z"},"execution":{"iopub.status.busy":"2021-06-20T20:38:43.312932Z","iopub.execute_input":"2021-06-20T20:38:43.313372Z","iopub.status.idle":"2021-06-20T20:38:43.348195Z","shell.execute_reply.started":"2021-06-20T20:38:43.313326Z","shell.execute_reply":"2021-06-20T20:38:43.346974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test_df.shape)\ntest_df.head()","metadata":{"ExecuteTime":{"end_time":"2021-06-20T04:06:37.528852Z","start_time":"2021-06-20T04:06:37.514853Z"},"execution":{"iopub.status.busy":"2021-06-20T20:38:43.350512Z","iopub.execute_input":"2021-06-20T20:38:43.350813Z","iopub.status.idle":"2021-06-20T20:38:43.369090Z","shell.execute_reply.started":"2021-06-20T20:38:43.350785Z","shell.execute_reply":"2021-06-20T20:38:43.368249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define Loading GnssLog.txt file Function","metadata":{}},{"cell_type":"code","source":"def gnss_log_to_dataframes(path):\n    gnss_section_names = {'Raw','UncalAccel', 'UncalGyro', 'UncalMag', 'Fix', 'Status', 'OrientationDeg'}\n    with open(path) as f_open:\n        datalines = f_open.readlines()\n\n    datas = {k: [] for k in gnss_section_names}\n    gnss_map = {k: [] for k in gnss_section_names}\n    for dataline in datalines:\n        is_header = dataline.startswith('#')\n        dataline = dataline.strip('#').strip().split(',')\n        # skip over notes, version numbers, etc\n        if is_header and dataline[0] in gnss_section_names:\n            try:\n                gnss_map[dataline[0]] = dataline[1:]\n            except:\n                pass\n        elif not is_header:\n            try:\n                datas[dataline[0]].append(dataline[1:])\n            except:\n                pass\n    results = dict()\n    for k, v in datas.items():\n        results[k] = pd.DataFrame(v, columns=gnss_map[k])\n    # pandas doesn't properly infer types from these lists by default\n    for k, df in results.items():\n        for col in df.columns:\n            if col == 'CodeType':\n                continue\n            try:\n                results[k][col] = pd.to_numeric(results[k][col])\n            except:\n                pass\n    return results","metadata":{"ExecuteTime":{"end_time":"2021-06-20T03:08:37.774226Z","start_time":"2021-06-20T03:08:37.760226Z"},"execution":{"iopub.status.busy":"2021-06-20T20:38:43.370647Z","iopub.execute_input":"2021-06-20T20:38:43.371188Z","iopub.status.idle":"2021-06-20T20:38:43.384169Z","shell.execute_reply.started":"2021-06-20T20:38:43.371156Z","shell.execute_reply":"2021-06-20T20:38:43.383275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load All Data Function","metadata":{}},{"cell_type":"code","source":"def get_addtional_data(df : pd.DataFrame, path: Path, train = True):\n    gnss_section_names = {'Raw','UncalAccel', 'UncalGyro', 'UncalMag', 'Fix', 'Status', 'OrientationDeg'}\n    section_names = {'GroundTruth', 'Derived', 'Raw','UncalAccel', 'UncalGyro', 'UncalMag', 'Fix', 'Status', 'OrientationDeg'}\n    _columns = ['latDeg', 'lngDeg', 'heightAboveWgs84EllipsoidM']\n\n    output = dict()\n    for section in section_names:\n        output[section] = pd.DataFrame()\n\n    if train:\n        start_path = \"train\"\n    else:\n        start_path = \"test\"\n        \n    for path in tqdm(glob(str(PATH / start_path / \"*/*/*\"))):\n        print(path)\n        (collectionName, phoneName) = path.split(\"/\")[-3:-1]\n        \n        file_name = path.split(\"/\")[-1]\n        \n        if(file_name.find('ground_truth') >= 0): # get ground truth data\n            _df = pd.read_csv(path)    \n            _df[['t_'+col for col in _columns]] = _df[_columns]\n            _df = _df.drop(columns=_columns)\n            output['GroundTruth'] = pd.concat([output['GroundTruth'], _df])\n            \n        elif(file_name.find('derived.csv') >= 0): # get derived data\n            _df = pd.read_csv(path)\n            output['Derived'] = pd.concat([output['Derived'], _df])\n            \n        elif(file_name.find('GnssLog.txt') >= 0): # get gnss log data (it is dict)\n            _dict = gnss_log_to_dataframes(path)\n            for key, value in _dict.items():\n                if value.shape[0] == 0: # empty log bypass\n                    continue\n                    \n                # Addtional meta data for merging original data frame\n                value['collectionName'] = collectionName \n                value['phoneName'] = phoneName\n                if (key == \"Status\") or (key == \"Fix\"):  \n                    value.rename(columns = {'UnixTimeMillis':'utcTimeMillis'}, inplace = True)\n                value[\"millisSinceGpsEpoch\"] = value[\"utcTimeMillis\"] - 315964800000\n                \n                output[key] = pd.concat([output[key], value])\n\n    for key, value in output.items():\n        if value.shape[0] == 0:\n            continue\n        df = pd.merge_asof(df.sort_values('millisSinceGpsEpoch'), \n              value.sort_values('millisSinceGpsEpoch'), \n              on=\"millisSinceGpsEpoch\", by=[\"collectionName\", \"phoneName\"], \n              direction='nearest',tolerance=100000)\n        \n    return df\n    \n                \n    \n    ","metadata":{"ExecuteTime":{"end_time":"2021-06-20T03:09:32.591208Z","start_time":"2021-06-20T03:09:32.568218Z"},"execution":{"iopub.status.busy":"2021-06-20T20:38:43.385917Z","iopub.execute_input":"2021-06-20T20:38:43.386259Z","iopub.status.idle":"2021-06-20T20:38:43.407237Z","shell.execute_reply.started":"2021-06-20T20:38:43.386223Z","shell.execute_reply":"2021-06-20T20:38:43.405622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save To Pickle File","metadata":{}},{"cell_type":"code","source":"output = get_addtional_data(train_df, PATH, train = True)\n\noutput.to_pickle(\"gsdc_train.pkl.gzip\")","metadata":{"ExecuteTime":{"end_time":"2021-06-20T03:24:23.380264Z","start_time":"2021-06-20T03:09:32.86221Z"},"scrolled":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-06-20T20:38:43.409629Z","iopub.execute_input":"2021-06-20T20:38:43.410643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = get_addtional_data(test_df, PATH, train = False)\n\noutput.to_pickle(\"gsdc_test.pkl.gzip\")","metadata":{"ExecuteTime":{"end_time":"2021-06-20T03:34:10.000606Z","start_time":"2021-06-20T03:24:23.861202Z"},"scrolled":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%clear","metadata":{"ExecuteTime":{"end_time":"2021-06-20T04:04:33.924349Z","start_time":"2021-06-20T04:04:33.619159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Pickle File","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom glob import glob\nimport os\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm\nfrom pathlib import Path\nimport plotly.express as px","metadata":{"ExecuteTime":{"end_time":"2021-06-20T04:05:47.669772Z","start_time":"2021-06-20T04:05:47.655773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH = Path(\"../input/google-smartphone-decimeter-challenge\")","metadata":{"ExecuteTime":{"end_time":"2021-06-20T04:05:47.954765Z","start_time":"2021-06-20T04:05:47.940762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_pickle(\"gsdc_train.pkl.gzip\")","metadata":{"ExecuteTime":{"end_time":"2021-06-20T04:05:48.871715Z","start_time":"2021-06-20T04:05:48.226762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_train.shape)\ndf_train.head()","metadata":{"ExecuteTime":{"end_time":"2021-06-20T04:06:00.966203Z","start_time":"2021-06-20T04:06:00.942202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_pickle(\"gsdc_test.pkl.gzip\")","metadata":{"ExecuteTime":{"end_time":"2021-06-20T04:06:06.621771Z","start_time":"2021-06-20T04:06:06.534773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_test.shape)\ndf_test.head()","metadata":{"ExecuteTime":{"end_time":"2021-06-20T04:06:07.101773Z","start_time":"2021-06-20T04:06:07.072772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}