{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"conda install -c conda-forge lightgbm","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:06:22.178307Z","iopub.execute_input":"2022-11-02T12:06:22.179204Z","iopub.status.idle":"2022-11-02T12:06:54.829462Z","shell.execute_reply.started":"2022-11-02T12:06:22.179145Z","shell.execute_reply":"2022-11-02T12:06:54.827944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Import**","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport seaborn as sns\nimport os\n!pip install folium\n!pip install simdkalman\nimport pickle\nimport sys\nimport warnings\nfrom glob import glob\nimport requests\nimport folium\nfrom shapely.geometry import Point, shape\nimport shapely.wkt\nfrom geopandas import GeoDataFrame\nimport simdkalman\nimport shap\nimport xgboost\nfrom scipy.stats import spearmanr\nfrom sklearn.ensemble import (\n    ExtraTreesRegressor,\n    GradientBoostingRegressor,\n    RandomForestRegressor,\n)\nfrom sklearn.metrics import accuracy_score, mean_squared_error\nfrom tqdm.notebook import tqdm\npd.options.mode.use_inf_as_na = True","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:06:54.832269Z","iopub.execute_input":"2022-11-02T12:06:54.832716Z","iopub.status.idle":"2022-11-02T12:07:17.727339Z","shell.execute_reply.started":"2022-11-02T12:06:54.832675Z","shell.execute_reply":"2022-11-02T12:07:17.725948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dirname, _,filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname,filename))","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:17.729490Z","iopub.execute_input":"2022-11-02T12:07:17.729987Z","iopub.status.idle":"2022-11-02T12:07:18.422111Z","shell.execute_reply.started":"2022-11-02T12:07:17.729935Z","shell.execute_reply":"2022-11-02T12:07:18.421080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nos.listdir('/kaggle/input/')","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:18.425425Z","iopub.execute_input":"2022-11-02T12:07:18.426235Z","iopub.status.idle":"2022-11-02T12:07:18.434339Z","shell.execute_reply.started":"2022-11-02T12:07:18.426187Z","shell.execute_reply":"2022-11-02T12:07:18.433271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data=pd.read_csv('../input/smartphone-decimeter-2022/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:18.435802Z","iopub.execute_input":"2022-11-02T12:07:18.436593Z","iopub.status.idle":"2022-11-02T12:07:18.514073Z","shell.execute_reply.started":"2022-11-02T12:07:18.436551Z","shell.execute_reply":"2022-11-02T12:07:18.512854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:18.515603Z","iopub.execute_input":"2022-11-02T12:07:18.515971Z","iopub.status.idle":"2022-11-02T12:07:18.521454Z","shell.execute_reply.started":"2022-11-02T12:07:18.515938Z","shell.execute_reply":"2022-11-02T12:07:18.520000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:18.523232Z","iopub.execute_input":"2022-11-02T12:07:18.523606Z","iopub.status.idle":"2022-11-02T12:07:18.543644Z","shell.execute_reply.started":"2022-11-02T12:07:18.523574Z","shell.execute_reply":"2022-11-02T12:07:18.542551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.info()","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:18.545153Z","iopub.execute_input":"2022-11-02T12:07:18.545558Z","iopub.status.idle":"2022-11-02T12:07:18.569186Z","shell.execute_reply.started":"2022-11-02T12:07:18.545525Z","shell.execute_reply":"2022-11-02T12:07:18.568025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['tripId'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:18.570636Z","iopub.execute_input":"2022-11-02T12:07:18.571112Z","iopub.status.idle":"2022-11-02T12:07:18.584780Z","shell.execute_reply.started":"2022-11-02T12:07:18.571073Z","shell.execute_reply":"2022-11-02T12:07:18.583451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X=data[['UnixTimeMillis','LatitudeDegrees','LongitudeDegrees']]\ny=data['tripId']","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:18.590790Z","iopub.execute_input":"2022-11-02T12:07:18.591324Z","iopub.status.idle":"2022-11-02T12:07:18.604402Z","shell.execute_reply.started":"2022-11-02T12:07:18.591273Z","shell.execute_reply":"2022-11-02T12:07:18.602976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data=pd.DataFrame(data)\nprint(data)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:18.606032Z","iopub.execute_input":"2022-11-02T12:07:18.606551Z","iopub.status.idle":"2022-11-02T12:07:18.622586Z","shell.execute_reply.started":"2022-11-02T12:07:18.606502Z","shell.execute_reply":"2022-11-02T12:07:18.621178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" X=data.iloc[:,[2,3]].values\nprint(X)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:18.623933Z","iopub.execute_input":"2022-11-02T12:07:18.624589Z","iopub.status.idle":"2022-11-02T12:07:18.632864Z","shell.execute_reply.started":"2022-11-02T12:07:18.624554Z","shell.execute_reply":"2022-11-02T12:07:18.631655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y=data.iloc[:,0:4].values\nprint(y)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:18.634237Z","iopub.execute_input":"2022-11-02T12:07:18.634637Z","iopub.status.idle":"2022-11-02T12:07:18.661131Z","shell.execute_reply.started":"2022-11-02T12:07:18.634603Z","shell.execute_reply":"2022-11-02T12:07:18.659835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cname_ = glob('../input/smartphone-decimeter-2022/train/*')\ntmp = []\nfor i in cname_:\n    tmp.extend(glob(f'{i}/*'))\n\ncname=[]\n\nfor r in tmp:\n    cname.append([r.split('/')[4],r.split('/')[5]])\n    \ncname = pd.DataFrame(sorted(cname))\ncname","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:18.663168Z","iopub.execute_input":"2022-11-02T12:07:18.663685Z","iopub.status.idle":"2022-11-02T12:07:18.716919Z","shell.execute_reply.started":"2022-11-02T12:07:18.663638Z","shell.execute_reply":"2022-11-02T12:07:18.715673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"List of mobile phones used in train","metadata":{}},{"cell_type":"code","source":"cname[1].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:18.718802Z","iopub.execute_input":"2022-11-02T12:07:18.719958Z","iopub.status.idle":"2022-11-02T12:07:18.729688Z","shell.execute_reply.started":"2022-11-02T12:07:18.719911Z","shell.execute_reply":"2022-11-02T12:07:18.728348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"List of mobile phones used in test","metadata":{}},{"cell_type":"code","source":"cname_ = glob('../input/smartphone-decimeter-2022/test/*')\ntmp = []\nfor i in cname_:\n    tmp.extend(glob(f'{i}/*'))\n\ncname=[]\n\nfor r in tmp:\n    cname.append([r.split('/')[4],r.split('/')[5]])\n    \ncname = pd.DataFrame(sorted(cname))\ncname[1].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:18.731594Z","iopub.execute_input":"2022-11-02T12:07:18.732041Z","iopub.status.idle":"2022-11-02T12:07:18.769269Z","shell.execute_reply.started":"2022-11-02T12:07:18.731998Z","shell.execute_reply":"2022-11-02T12:07:18.767957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since the train data is from 2020 and the test is from 2021-2022, it seems that the mobile phone used is a little different.","metadata":{}},{"cell_type":"markdown","source":"Read the metadata file","metadata":{}},{"cell_type":"code","source":"import json\nraw = open('../input/smartphone-decimeter-2022/metadata/raw_state_bit_map.json', 'r')\njson.load(raw)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:18.771072Z","iopub.execute_input":"2022-11-02T12:07:18.771575Z","iopub.status.idle":"2022-11-02T12:07:18.780444Z","shell.execute_reply.started":"2022-11-02T12:07:18.771530Z","shell.execute_reply":"2022-11-02T12:07:18.779449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\nbit = open('../input/smartphone-decimeter-2022/metadata/accumulated_delta_range_state_bit_map.json', 'r')\njson.load(bit)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:18.781924Z","iopub.execute_input":"2022-11-02T12:07:18.782304Z","iopub.status.idle":"2022-11-02T12:07:18.792253Z","shell.execute_reply.started":"2022-11-02T12:07:18.782270Z","shell.execute_reply":"2022-11-02T12:07:18.791105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mapping = pd.read_csv('../input/smartphone-decimeter-2022/metadata/constellation_type_mapping.csv')\nmapping","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:18.793899Z","iopub.execute_input":"2022-11-02T12:07:18.794307Z","iopub.status.idle":"2022-11-02T12:07:18.809194Z","shell.execute_reply.started":"2022-11-02T12:07:18.794271Z","shell.execute_reply":"2022-11-02T12:07:18.807767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Read the sample file(2020-05-15-US-MTV-1)","metadata":{}},{"cell_type":"code","source":"ground = pd.read_csv('../input/smartphone-decimeter-2022/train/2020-05-15-US-MTV-1/GooglePixel4XL/ground_truth.csv')\nground","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:18.811261Z","iopub.execute_input":"2022-11-02T12:07:18.812572Z","iopub.status.idle":"2022-11-02T12:07:18.843590Z","shell.execute_reply.started":"2022-11-02T12:07:18.812520Z","shell.execute_reply":"2022-11-02T12:07:18.842436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"MessageType - \"Fix\", the prefix of sentence.\n\nProvider - \"GT\", short for ground truth.\n\n\n[Latitude/Longitude]Degrees - The WGS84 latitude, longitude (in decimal degrees) estimated by the reference GNSS receiver (NovAtel SPAN). When extracting from the NMEA file, linear interpolation has been applied to align the location to the expected non-integer timestamps.\n\n\nAltitudeMeters - The height above the WGS84 ellipsoid (in meters) estimated by the reference GNSS receiver.\n\nSpeedMps* - The speed over ground in meters per second.\n\n\nAccuracyMeters - The estimated horizontal accuracy radius in meters of this location at the 68th percentile confidence level. This means that there is a 68% chance that the true location of the device is within a distance of this uncertainty of the reported location.\n\n\nBearingDegrees - Bearing is measured in degrees clockwise from north. It ranges from 0 to 359.999 degrees.\n\n\nUnixTimeMillis - An integer number of milliseconds since the GPS epoch (1970/1/1 midnight UTC). Converted from GnssClock.","metadata":{}},{"cell_type":"code","source":"imu= pd.read_csv('../input/smartphone-decimeter-2022/train/2020-05-15-US-MTV-1/GooglePixel4XL/device_imu.csv')\nimu","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:18.845130Z","iopub.execute_input":"2022-11-02T12:07:18.845514Z","iopub.status.idle":"2022-11-02T12:07:19.525137Z","shell.execute_reply.started":"2022-11-02T12:07:18.845479Z","shell.execute_reply":"2022-11-02T12:07:19.523440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gnss = pd.read_csv('../input/smartphone-decimeter-2022/train/2020-05-15-US-MTV-1/GooglePixel4XL/device_gnss.csv')\ngnss","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:19.527028Z","iopub.execute_input":"2022-11-02T12:07:19.527605Z","iopub.status.idle":"2022-11-02T12:07:20.555866Z","shell.execute_reply.started":"2022-11-02T12:07:19.527557Z","shell.execute_reply":"2022-11-02T12:07:20.554661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"mapping the Ground True data on the map.","metadata":{}},{"cell_type":"code","source":"from folium import plugins\ndf_locs = list(ground[['LatitudeDegrees','LongitudeDegrees']].values)\nfol_map = folium.Map([ground['LatitudeDegrees'].median(), ground['LongitudeDegrees'].median()],zoom_start=11)\nheat_map = plugins.HeatMap(df_locs)\nfol_map.add_child(heat_map)\nmarkers = plugins.MarkerCluster(locations = df_locs)\nfol_map.add_child(markers)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:20.557273Z","iopub.execute_input":"2022-11-02T12:07:20.557640Z","iopub.status.idle":"2022-11-02T12:07:23.530734Z","shell.execute_reply.started":"2022-11-02T12:07:20.557608Z","shell.execute_reply":"2022-11-02T12:07:23.529860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Read the supplemental","metadata":{}},{"cell_type":"markdown","source":"gnss_log.txt","metadata":{}},{"cell_type":"code","source":"f = open('../input/smartphone-decimeter-2022/train/2020-05-15-US-MTV-1/GooglePixel4XL/supplemental/gnss_log.txt', 'r')\nlog = f.read()\nf.close()\nlog[:500]","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:23.531879Z","iopub.execute_input":"2022-11-02T12:07:23.532688Z","iopub.status.idle":"2022-11-02T12:07:23.675841Z","shell.execute_reply.started":"2022-11-02T12:07:23.532638Z","shell.execute_reply":"2022-11-02T12:07:23.674750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Txt to Pandas","metadata":{}},{"cell_type":"code","source":"    path ='../input/smartphone-decimeter-2022/train/2020-05-15-US-MTV-1/GooglePixel4XL/supplemental/gnss_log.txt'\n    gnss_section_names = {'Raw','UncalAccel', 'UncalGyro', 'UncalMag', 'Fix', 'Status', 'OrientationDeg'}\n    with open(path) as f_open:\n        datalines = f_open.readlines()\n\n    datas = {k: [] for k in gnss_section_names}\n    gnss_map = {k: [] for k in gnss_section_names}\n    for dataline in datalines:\n      if dataline !='' and dataline[0] !='':\n        is_header = dataline.startswith('#')\n        dataline = dataline.strip('#').strip().split(',')\n        # skip over notes, version numbers, etc\n        if is_header and dataline[0] in gnss_section_names:\n            gnss_map[dataline[0]] = dataline[1:]\n        elif not is_header:\n            if dataline !='' and dataline[0] !='':\n                datas[dataline[0]].append(dataline[1:])\n\n    results = dict()\n    for k, v in datas.items():\n        results[k] = pd.DataFrame(v, columns=gnss_map[k])\n    for k, df in results.items():\n        for col in df.columns:\n            if col == 'CodeType':\n                continue\n            results[k][col] = pd.to_numeric(results[k][col])","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:23.677503Z","iopub.execute_input":"2022-11-02T12:07:23.678377Z","iopub.status.idle":"2022-11-02T12:07:34.112730Z","shell.execute_reply.started":"2022-11-02T12:07:23.678325Z","shell.execute_reply":"2022-11-02T12:07:34.111563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results['Raw']","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:34.114196Z","iopub.execute_input":"2022-11-02T12:07:34.114685Z","iopub.status.idle":"2022-11-02T12:07:34.184937Z","shell.execute_reply.started":"2022-11-02T12:07:34.114650Z","shell.execute_reply":"2022-11-02T12:07:34.183735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results['UncalAccel']","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:34.187133Z","iopub.execute_input":"2022-11-02T12:07:34.188210Z","iopub.status.idle":"2022-11-02T12:07:34.210561Z","shell.execute_reply.started":"2022-11-02T12:07:34.188173Z","shell.execute_reply":"2022-11-02T12:07:34.209618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results['UncalGyro']","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:34.216999Z","iopub.execute_input":"2022-11-02T12:07:34.217719Z","iopub.status.idle":"2022-11-02T12:07:34.242068Z","shell.execute_reply.started":"2022-11-02T12:07:34.217682Z","shell.execute_reply":"2022-11-02T12:07:34.240585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results[ 'UncalMag']","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:34.243896Z","iopub.execute_input":"2022-11-02T12:07:34.244319Z","iopub.status.idle":"2022-11-02T12:07:34.269087Z","shell.execute_reply.started":"2022-11-02T12:07:34.244282Z","shell.execute_reply":"2022-11-02T12:07:34.267751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results['Fix']","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:34.270442Z","iopub.execute_input":"2022-11-02T12:07:34.270938Z","iopub.status.idle":"2022-11-02T12:07:34.283334Z","shell.execute_reply.started":"2022-11-02T12:07:34.270880Z","shell.execute_reply":"2022-11-02T12:07:34.282444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results['Status']","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:34.284489Z","iopub.execute_input":"2022-11-02T12:07:34.285679Z","iopub.status.idle":"2022-11-02T12:07:34.302283Z","shell.execute_reply.started":"2022-11-02T12:07:34.285638Z","shell.execute_reply":"2022-11-02T12:07:34.300897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results['OrientationDeg']","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:34.303832Z","iopub.execute_input":"2022-11-02T12:07:34.304298Z","iopub.status.idle":"2022-11-02T12:07:34.318619Z","shell.execute_reply.started":"2022-11-02T12:07:34.304261Z","shell.execute_reply":"2022-11-02T12:07:34.316596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"gnss_rinex.20o","metadata":{}},{"cell_type":"code","source":"f = open('../input/smartphone-decimeter-2022/train/2020-05-15-US-MTV-1/GooglePixel4XL/supplemental/gnss_rinex.20o', 'r')\nrinex = f.read()\nf.close()\nrinex[:500]","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:34.320844Z","iopub.execute_input":"2022-11-02T12:07:34.321401Z","iopub.status.idle":"2022-11-02T12:07:34.339540Z","shell.execute_reply.started":"2022-11-02T12:07:34.321330Z","shell.execute_reply":"2022-11-02T12:07:34.337980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rinex =pd.read_csv('../input/smartphone-decimeter-2022/train/2020-05-15-US-MTV-1/GooglePixel4XL/supplemental/gnss_rinex.20o')\nrinex","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:34.341150Z","iopub.execute_input":"2022-11-02T12:07:34.341954Z","iopub.status.idle":"2022-11-02T12:07:34.483480Z","shell.execute_reply.started":"2022-11-02T12:07:34.341899Z","shell.execute_reply":"2022-11-02T12:07:34.482130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"span_log.nmea","metadata":{}},{"cell_type":"code","source":"f = open('../input/smartphone-decimeter-2022/train/2020-05-15-US-MTV-1/GooglePixel4XL/supplemental/span_log.nmea', 'r')\nspan = f.read()\nf.close()\nspan[:500]","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:34.485263Z","iopub.execute_input":"2022-11-02T12:07:34.485624Z","iopub.status.idle":"2022-11-02T12:07:34.495174Z","shell.execute_reply.started":"2022-11-02T12:07:34.485593Z","shell.execute_reply":"2022-11-02T12:07:34.493883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"span = pd.read_csv('../input/smartphone-decimeter-2022/train/2020-05-15-US-MTV-1/GooglePixel4XL/supplemental/span_log.nmea')\nspan","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:34.497619Z","iopub.execute_input":"2022-11-02T12:07:34.498059Z","iopub.status.idle":"2022-11-02T12:07:34.551042Z","shell.execute_reply.started":"2022-11-02T12:07:34.497983Z","shell.execute_reply":"2022-11-02T12:07:34.549865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"presentation","metadata":{}},{"cell_type":"code","source":"sub = pd.read_csv('../input/smartphone-decimeter-2022/sample_submission.csv')\nsub","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:34.552387Z","iopub.execute_input":"2022-11-02T12:07:34.553186Z","iopub.status.idle":"2022-11-02T12:07:34.625151Z","shell.execute_reply.started":"2022-11-02T12:07:34.553152Z","shell.execute_reply":"2022-11-02T12:07:34.623681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv('../input/smartphone-decimeter-2022/train/2020-08-06-US-MTV-2/GooglePixel4/ground_truth.csv')","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:34.626739Z","iopub.execute_input":"2022-11-02T12:07:34.627196Z","iopub.status.idle":"2022-11-02T12:07:34.654623Z","shell.execute_reply.started":"2022-11-02T12:07:34.627161Z","shell.execute_reply":"2022-11-02T12:07:34.653290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Download geojson file of US San Francisco Bay Area.\nr = requests.get(\"https://data.sfgov.org/api/views/wamw-vt4s/rows.json?accessType=DOWNLOAD\")\nr.raise_for_status()\n\n#get geojson from response\ndata = r.json()\n\n#get polygons that represents San Francisco Bay Area.\nshapes = []\nfor d in data[\"data\"]:\n    shapes.append(shapely.wkt.loads(d[8]))\n    \n#Convert list of porygons to geopandas dataframe.\ngdf_bayarea = pd.DataFrame()\n\n#I'll use only 6 and 7th object.\nfor shp in shapes[5:7]:\n    tmp = pd.DataFrame(shp, columns=[\"geometry\"])\n    gdf_bayarea = pd.concat([gdf_bayarea, tmp])\ngdf_bayarea = GeoDataFrame(gdf_bayarea)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:34.656045Z","iopub.execute_input":"2022-11-02T12:07:34.656457Z","iopub.status.idle":"2022-11-02T12:07:36.170689Z","shell.execute_reply.started":"2022-11-02T12:07:34.656420Z","shell.execute_reply":"2022-11-02T12:07:36.169491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gdf_bayarea","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:36.172402Z","iopub.execute_input":"2022-11-02T12:07:36.172857Z","iopub.status.idle":"2022-11-02T12:07:36.225737Z","shell.execute_reply.started":"2022-11-02T12:07:36.172819Z","shell.execute_reply":"2022-11-02T12:07:36.224488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\ncollectionNames = [item.split(\"/\")[-1] for item in glob(\"../input/smartphone-decimeter-2022/train/*\")]\n\ngdfs = []\nfor collectionName in collectionNames:\n    gdfs_each_collectionName = []\n    csv_paths = glob(f\"../input/smartphone-decimeter-2022/train/{collectionName}/*/ground_truth.csv\")\n    for csv_path in csv_paths:\n        df_gt = pd.read_csv(csv_path)\n        df_gt[\"geometry\"] = [Point(lngDeg, latDeg) for lngDeg, latDeg in zip(df_gt[\"LatitudeDegrees\"], df_gt[\"LongitudeDegrees\"])]\n        gdfs_each_collectionName.append(GeoDataFrame(df_gt))\n    gdfs.append(gdfs_each_collectionName)\n    \ncolors = ['blue', 'green', 'purple', 'orange']","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:36.227140Z","iopub.execute_input":"2022-11-02T12:07:36.227618Z","iopub.status.idle":"2022-11-02T12:07:58.156140Z","shell.execute_reply.started":"2022-11-02T12:07:36.227584Z","shell.execute_reply":"2022-11-02T12:07:58.155157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gdfs_each_collectionName","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:58.158337Z","iopub.execute_input":"2022-11-02T12:07:58.158840Z","iopub.status.idle":"2022-11-02T12:07:58.210709Z","shell.execute_reply.started":"2022-11-02T12:07:58.158793Z","shell.execute_reply":"2022-11-02T12:07:58.209544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for collectionName, gdfs_each_collectionName in zip(collectionNames, gdfs):\n    fig, axs = plt.subplots(1, 2, figsize=(15, 5))\n    gdf_bayarea.plot(figsize=(10,10), color='none', edgecolor='gray', zorder=5, ax=axs[0])\n    for i, gdf in enumerate(gdfs_each_collectionName):\n        g2 = gdf.plot(color=colors[i], ax=axs[1])\n        g2.set_title(f\"Phone track of {collectionName}\")","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:07:58.212078Z","iopub.execute_input":"2022-11-02T12:07:58.212412Z","iopub.status.idle":"2022-11-02T12:09:26.198440Z","shell.execute_reply.started":"2022-11-02T12:07:58.212382Z","shell.execute_reply":"2022-11-02T12:09:26.197420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train,X_test,y_train,y_test=train_test_split(X,y,test_size=0.3,random_state=0)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:09:26.199729Z","iopub.execute_input":"2022-11-02T12:09:26.200864Z","iopub.status.idle":"2022-11-02T12:09:26.225276Z","shell.execute_reply.started":"2022-11-02T12:09:26.200814Z","shell.execute_reply":"2022-11-02T12:09:26.223939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\nclf=lgb.LGBMClassifier()\nclf.fit(X_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:09:26.226793Z","iopub.execute_input":"2022-11-02T12:09:26.227156Z","iopub.status.idle":"2022-11-02T12:09:27.136275Z","shell.execute_reply.started":"2022-11-02T12:09:26.227124Z","shell.execute_reply":"2022-11-02T12:09:27.134664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LGBMClassifier()","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:09:27.137574Z","iopub.status.idle":"2022-11-02T12:09:27.137990Z","shell.execute_reply.started":"2022-11-02T12:09:27.137788Z","shell.execute_reply":"2022-11-02T12:09:27.137807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred=clf.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:09:27.139092Z","iopub.status.idle":"2022-11-02T12:09:27.139489Z","shell.execute_reply.started":"2022-11-02T12:09:27.139288Z","shell.execute_reply":"2022-11-02T12:09:27.139305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\naccuracy=accuracy_score(y_pred,y_test)\nprint('LightGBM Model Accuracy score:{0:0.4f}'.format(accuracy_score(y_test,y_pred)))","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:09:27.140730Z","iopub.status.idle":"2022-11-02T12:09:27.141111Z","shell.execute_reply.started":"2022-11-02T12:09:27.140930Z","shell.execute_reply":"2022-11-02T12:09:27.140947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_train=clf.predict(X_train)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:09:27.142294Z","iopub.status.idle":"2022-11-02T12:09:27.142918Z","shell.execute_reply.started":"2022-11-02T12:09:27.142710Z","shell.execute_reply":"2022-11-02T12:09:27.142729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Training-set accuracy score:{0:0.4f}'.format(accuracy_score(y_train,y_pred_train)))","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:09:27.143844Z","iopub.status.idle":"2022-11-02T12:09:27.144212Z","shell.execute_reply.started":"2022-11-02T12:09:27.144034Z","shell.execute_reply":"2022-11-02T12:09:27.144050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Training set score:{:.4f}'.format(clf.score(X_train,y_train)))\nprint('Test set score:{:.4f}'.format(clf.score(X_test,y_test)))","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:09:27.146204Z","iopub.status.idle":"2022-11-02T12:09:27.146834Z","shell.execute_reply.started":"2022-11-02T12:09:27.146543Z","shell.execute_reply":"2022-11-02T12:09:27.146571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\ncm=confusion_matrix(y_test,y_pred)\nprint('Confusion matrix\\n\\n',cm)\nprint('\\n True Positives(TP)= ',cm[0,0])\nprint('\\n True Negatives(TN)= ',cm[1,1])\nprint('\\n False Positives(FP)= ',cm[0,1])\nprint('\\n False Negatives(FN)= ',cm[1,0])","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:09:27.148654Z","iopub.status.idle":"2022-11-02T12:09:27.149191Z","shell.execute_reply.started":"2022-11-02T12:09:27.148911Z","shell.execute_reply":"2022-11-02T12:09:27.148937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.utils.multiclass import unique_labels\nunique_labels(y_test)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:09:27.150993Z","iopub.status.idle":"2022-11-02T12:09:27.151586Z","shell.execute_reply.started":"2022-11-02T12:09:27.151253Z","shell.execute_reply":"2022-11-02T12:09:27.151285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot(y_true,y_pred):\n    labels=unique_labels(y_test)\n    columns=[f'Predicted{label}' for label in labels]\n    index=[f'Actual{label}' for label in labels]\n    table=pd.DataFrame(confusion_matrix(y_true,y_pred),columns=columns,index=index)\n    return table","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:09:27.153247Z","iopub.status.idle":"2022-11-02T12:09:27.153826Z","shell.execute_reply.started":"2022-11-02T12:09:27.153542Z","shell.execute_reply":"2022-11-02T12:09:27.153569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot(y_test,y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:09:27.155156Z","iopub.status.idle":"2022-11-02T12:09:27.155732Z","shell.execute_reply.started":"2022-11-02T12:09:27.155441Z","shell.execute_reply":"2022-11-02T12:09:27.155468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot2(y_true,y_pred):\n    labels=unique_labels(y_test)\n    column=[f'Predicted{label}' for label in labels]\n    indices=[f'Actual{label}' for label in labels]\n    table=pd.DataFrame(confusion_matrix(y_true,y_pred),columns=column,index=indices)\n    return sns.heatmap(table,annot=True,fmt='d',cmap='viridis')","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:09:27.157910Z","iopub.status.idle":"2022-11-02T12:09:27.158493Z","shell.execute_reply.started":"2022-11-02T12:09:27.158173Z","shell.execute_reply":"2022-11-02T12:09:27.158201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot2(y_test,y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:09:27.159856Z","iopub.status.idle":"2022-11-02T12:09:27.160423Z","shell.execute_reply.started":"2022-11-02T12:09:27.160114Z","shell.execute_reply":"2022-11-02T12:09:27.160142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\nprint(classification_report(y_test,y_pred))","metadata":{"execution":{"iopub.status.busy":"2022-11-02T12:09:27.162485Z","iopub.status.idle":"2022-11-02T12:09:27.163025Z","shell.execute_reply.started":"2022-11-02T12:09:27.162744Z","shell.execute_reply":"2022-11-02T12:09:27.162770Z"},"trusted":true},"execution_count":null,"outputs":[]}]}