{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd \nimport numpy as np \n","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train data \ntrn = pd.read_csv('../input/google-smartphone-decimeter-challenge/baseline_locations_train.csv')\ntrn.head()","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test data\ntst = pd.read_csv('../input/google-smartphone-decimeter-challenge/baseline_locations_test.csv')\ntst.head()","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trn.shape , tst.shape","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# find distinct values in test and train data set \nfor col in trn.columns:\n    print('{0} : (trn , tst) : {1}, {2} '.format(col, trn[col].nunique(), trn[col].nunique()))","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Check min , max and percentile of the dataset\ntrn.describe()","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tst.describe()","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# change data type of millisSinceGpsEpoch\nEpoch_offset = pd.to_datetime('1980-01-06 00:00:00')\n#print(Epoch_offset.value)\nEpoch_offset_in_ms = int(Epoch_offset.value / 1e6)\nprint(Epoch_offset_in_ms)\n","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trn['millisSinceGpsEpoch'] = pd.to_datetime(trn['millisSinceGpsEpoch'] + Epoch_offset_in_ms , unit='ms' )\ntst['millisSinceGpsEpoch'] = pd.to_datetime(tst['millisSinceGpsEpoch'] + Epoch_offset_in_ms , unit='ms' )","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trn.head()","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# latdeg and lngdeg\ntrn_lat_lng = trn[['latDeg','lngDeg']].round(3)\ntrn_lat_lng['count'] = 1\ntrn_lat_lng_f  = trn_lat_lng.groupby(['latDeg','lngDeg']).sum().reset_index()\ntrn_lat_lng_f[trn_lat_lng_f['count'] != 1]","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# latdeg and lngdeg\ntst_lat_lng = tst[['latDeg','lngDeg']].round(3)\ntst_lat_lng['count'] = 1\ntst_lat_lng_f  = tst_lat_lng.groupby(['latDeg','lngDeg']).sum().reset_index()\ntst_lat_lng_f[tst_lat_lng_f['count'] != 1]","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import folium \nlat1 = trn.latDeg[0]\nlng1 = trn.lngDeg[0]\nm = folium.Map(location=[lat1, lng1])\nfor i in range(1,len(tst_lat_lng_f)):\n    folium.Marker(location=[trn.latDeg[i], trn.lngDeg[i]],popup='Default popup Marker1').add_to(m)\nm","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Region / Model Phone / Log Analysis","metadata":{"editable":false}},{"cell_type":"code","source":"# Check first log to find data \n#('../input/google-smartphone-decimeter-challenge/train/2020-05-14-US-MTV-1/Pixel4/Pixel4_GnssLog.txt');\n\ndef ReadGNSSlog(train_test, drive_id, phone_name, log_type):\n    '''Read log file . Extract headers and then create data frame under each header'''\n    path= '../input/google-smartphone-decimeter-challenge/'+train_test+'/'+drive_id+'/'+phone_name+'/'+phone_name+'_GnssLog.txt'\n    #print('Path is {0}'.format(path))\n    header = [log_type]\n    header_dict = {i: [] for i in header}    \n    datas = {i: [] for i in header}\n    \n    with open(path) as f_open:\n        datalines = f_open.read().splitlines()\n        \n        for line in datalines:\n            #print(line)\n            is_header = line.startswith('#')\n            line = line.strip('#').strip(' ').split(',')\n            #print('after change {0}'.format(line))\n            if (is_header and line[0] in header):\n                header_dict[line[0]] = ['collectionName', 'phoneName']+line[1:]\n            elif not is_header and line[0] in header:\n               # datas[line[0]].append([drive_id,phone_name ])\n                datas[line[0]].append([drive_id,phone_name ]+ line[1:])\n        \n        results= dict()\n        for key, val in datas.items():\n            #df_name = print('df_{0}'.format(key))\n            #print(df_name)\n            #exec(f\"df_{key} = pd.DataFrame(datas[key] , columns= header_dict[key])\")\n            results[key] = pd.DataFrame(val, columns=header_dict[key])\n            print(key)\n            \n    \n        f_open.close()\n        return results","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trn_GNSS_data = trn[['collectionName', 'phoneName']].groupby(['collectionName', 'phoneName']).sum().reset_index()\ntrn_GNSS_data.head()","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results_f={}\nfor row in range(len(trn_GNSS_data)):\n    print(trn_GNSS_data.loc[row,'collectionName'] , trn_GNSS_data.loc[row,'phoneName'])\n    key = 'results_train_'+trn_GNSS_data.loc[row,'collectionName']+'_'+trn_GNSS_data.loc[row,'phoneName']\n    print('.....'+key+'....')\n    results_f[key]= ReadGNSSlog('train', trn_GNSS_data.loc[row,'collectionName'], trn_GNSS_data.loc[row,'phoneName'],'Raw')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"first = True\nfor key in results_f.keys():\n    if (first):\n        df_train_Raw = results_f[key][\"Raw\"]\n        first = False\n    else:\n        #print( df_train_Raw.head(5))\n        #print(results_f[key][\"Raw\"].head(5))\n        df_train_Raw = pd.concat([df_train_Raw, results_f[key][\"Raw\"]], ignore_index=True )\n\ndf_train_Raw.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shape of df_train_Raw\ndf_train_Raw.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_Raw.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"EDA for GNSS Logs : Raw Data","metadata":{}},{"cell_type":"code","source":"\"I am error \"\ncols = list(df_train_Raw.columns.difference(['collectionName','phoneName']))\ndf_train_Raw[cols]= df_train_Raw[cols].apply(pd.to_numeric, errors='coerce')\ndf_train_Raw = df_train_Raw.replace(np.nan, 0)\ndf_train_Raw.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Clear intermediate dictionary\nresults_f.clear()\ndf_train_Raw = df_train_Raw.replace(np.nan, 0)\ndf_train_Raw.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Info of new dataframe\ndf_train_Raw.info(verbose=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n# Calculate Arrival Time \ndf_Raw['ArrivalTime'] = df_Raw['TimeNanos'] - df_Raw['FullBiasNanos'] - df_Raw['BiasNanos']\nprint(df_Raw.ArrivalTime.describe())\ndf_Raw.ArrivalTime.hist(bins=20)\nplt.show()","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_Raw.info()","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_Raw['TransmitTime'] = df_Raw.ReceivedSvTimeNanos  - df_Raw.ReceivedSvTimeUncertaintyNanos\nprint(df_Raw['TransmitTime'].describe())\ndf_Raw['TransmitTime'].hist(bins=2)\nplt.show()","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Full Bias Nano secons is zero \ndf_Raw.info()","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Bias Uncertainity NOnno is too large\nprint(df_Raw.BiasUncertaintyNanos.describe([0.25,0.50,0.6,0.7,0.8,0.9]))\ndf_Raw.boxplot(column='BiasUncertaintyNanos')\nplt.show()\n","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_Raw['ArrivalTime'].describe([0.25,0.5,0.6,0.7,0.8,0.9]))\ndf_Raw.boxplot(column = ['ArrivalTime'])\nplt.show()","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ConstellationType                          \ndf_Raw.ConstellationType.value_counts(normalize=True)","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# TimeNanos is empty \ndf_Raw[df_Raw.TimeNanos.isnull()]","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Max value should be 65535 \ndf_Raw.State.value_counts()","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#11  ReceivedSvTimeNanos                        57497 non-null  int64  \n #12  ReceivedSvTimeUncertaintyNanos             57497 non-null  int64  \n\ndf_Raw['TransmitTime'].describe()","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#AccumulatedDateRange - Max value allocated should be 31 as range of bits is only till 5 \ndf_Raw.AccumulatedDeltaRangeState.value_counts()","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# AccumulatedDeltaRangeUncertaintyMeters     is too high should range between 0 and 1 \ndf_Raw['AccumulatedDeltaRangeUncertaintyMeters'].describe()","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_Raw.head()","metadata":{"editable":false,"trusted":true},"execution_count":null,"outputs":[]}]}