{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport matplotlib\nimport matplotlib.pyplot as plt\n\nimport seaborn as sns\nimport os\n!pip install folium\n!pip install simdkalman\nimport pickle\nimport sys\nimport warnings\nfrom glob import glob\nimport requests\nimport folium\nfrom shapely.geometry import Point, shape\nimport shapely.wkt\nfrom geopandas import GeoDataFrame\nimport simdkalman\nimport shap\nimport xgboost\nfrom scipy.stats import spearmanr\nfrom sklearn.ensemble import (\n    ExtraTreesRegressor,\n    GradientBoostingRegressor,\n    RandomForestRegressor,\n)\nfrom sklearn.metrics import accuracy_score, mean_squared_error\nfrom tqdm.notebook import tqdm\npd.options.mode.use_inf_as_na = True\n","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:41:41.184006Z","iopub.execute_input":"2022-12-28T13:41:41.184902Z","iopub.status.idle":"2022-12-28T13:42:03.774278Z","shell.execute_reply.started":"2022-12-28T13:41:41.184849Z","shell.execute_reply":"2022-12-28T13:42:03.772944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:03.777600Z","iopub.execute_input":"2022-12-28T13:42:03.778134Z","iopub.status.idle":"2022-12-28T13:42:04.226975Z","shell.execute_reply.started":"2022-12-28T13:42:03.778083Z","shell.execute_reply":"2022-12-28T13:42:04.225539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:04.228464Z","iopub.execute_input":"2022-12-28T13:42:04.228843Z","iopub.status.idle":"2022-12-28T13:42:04.233808Z","shell.execute_reply.started":"2022-12-28T13:42:04.228813Z","shell.execute_reply":"2022-12-28T13:42:04.232533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=pd.read_csv('../input/smartphone-decimeter-2022/sample_submission.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:04.236471Z","iopub.execute_input":"2022-12-28T13:42:04.236825Z","iopub.status.idle":"2022-12-28T13:42:04.322263Z","shell.execute_reply.started":"2022-12-28T13:42:04.236795Z","shell.execute_reply":"2022-12-28T13:42:04.321058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:04.324059Z","iopub.execute_input":"2022-12-28T13:42:04.325329Z","iopub.status.idle":"2022-12-28T13:42:04.347225Z","shell.execute_reply.started":"2022-12-28T13:42:04.325285Z","shell.execute_reply":"2022-12-28T13:42:04.345697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['tripId'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:04.348804Z","iopub.execute_input":"2022-12-28T13:42:04.349203Z","iopub.status.idle":"2022-12-28T13:42:04.362787Z","shell.execute_reply.started":"2022-12-28T13:42:04.349168Z","shell.execute_reply":"2022-12-28T13:42:04.361265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X=df[['UnixTimeMillis','LatitudeDegrees','LongitudeDegrees']]\ny=df['tripId']","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:04.364771Z","iopub.execute_input":"2022-12-28T13:42:04.365165Z","iopub.status.idle":"2022-12-28T13:42:04.374308Z","shell.execute_reply.started":"2022-12-28T13:42:04.365130Z","shell.execute_reply":"2022-12-28T13:42:04.373117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data=pd.DataFrame(df)\nprint(data)","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:04.375881Z","iopub.execute_input":"2022-12-28T13:42:04.376284Z","iopub.status.idle":"2022-12-28T13:42:04.392106Z","shell.execute_reply.started":"2022-12-28T13:42:04.376223Z","shell.execute_reply":"2022-12-28T13:42:04.391064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" X=data.iloc[:,[2,3]].values\nprint(X)","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:04.393478Z","iopub.execute_input":"2022-12-28T13:42:04.394076Z","iopub.status.idle":"2022-12-28T13:42:04.405392Z","shell.execute_reply.started":"2022-12-28T13:42:04.394042Z","shell.execute_reply":"2022-12-28T13:42:04.404250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y=data.iloc[:,0:4].values\nprint(y)","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:04.409407Z","iopub.execute_input":"2022-12-28T13:42:04.410090Z","iopub.status.idle":"2022-12-28T13:42:04.437601Z","shell.execute_reply.started":"2022-12-28T13:42:04.410052Z","shell.execute_reply":"2022-12-28T13:42:04.436295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cname_ = glob('../input/smartphone-decimeter-2022/train/*')\ntmp = []\nfor i in cname_:\n    tmp.extend(glob(f'{i}/*'))\n\ncname=[]\n\nfor r in tmp:\n    cname.append([r.split('/')[4],r.split('/')[5]])\n    \ncname = pd.DataFrame(sorted(cname))\ncname","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:04.439988Z","iopub.execute_input":"2022-12-28T13:42:04.440698Z","iopub.status.idle":"2022-12-28T13:42:04.500323Z","shell.execute_reply.started":"2022-12-28T13:42:04.440659Z","shell.execute_reply":"2022-12-28T13:42:04.499470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"list of mobile phones used in training","metadata":{}},{"cell_type":"code","source":"cname[1].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:04.502063Z","iopub.execute_input":"2022-12-28T13:42:04.502842Z","iopub.status.idle":"2022-12-28T13:42:04.513343Z","shell.execute_reply.started":"2022-12-28T13:42:04.502796Z","shell.execute_reply":"2022-12-28T13:42:04.512302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"list of mobile phone used in testing","metadata":{}},{"cell_type":"code","source":"cname_ = glob('../input/smartphone-decimeter-2022/test/*')\ntmp = []\nfor i in cname_:\n    tmp.extend(glob(f'{i}/*'))\n\ncname=[]\n\nfor r in tmp:\n    cname.append([r.split('/')[4],r.split('/')[5]])\n    \ncname = pd.DataFrame(sorted(cname))\ncname[1].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:04.514880Z","iopub.execute_input":"2022-12-28T13:42:04.515976Z","iopub.status.idle":"2022-12-28T13:42:04.545521Z","shell.execute_reply.started":"2022-12-28T13:42:04.515939Z","shell.execute_reply":"2022-12-28T13:42:04.544642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\nraw = open('../input/smartphone-decimeter-2022/metadata/raw_state_bit_map.json', 'r')\njson.load(raw)","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:04.547226Z","iopub.execute_input":"2022-12-28T13:42:04.548036Z","iopub.status.idle":"2022-12-28T13:42:04.560729Z","shell.execute_reply.started":"2022-12-28T13:42:04.547989Z","shell.execute_reply":"2022-12-28T13:42:04.559686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\nbit = open('../input/smartphone-decimeter-2022/metadata/accumulated_delta_range_state_bit_map.json', 'r')\njson.load(bit)","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:04.562765Z","iopub.execute_input":"2022-12-28T13:42:04.563580Z","iopub.status.idle":"2022-12-28T13:42:04.575793Z","shell.execute_reply.started":"2022-12-28T13:42:04.563533Z","shell.execute_reply":"2022-12-28T13:42:04.574704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mapping = pd.read_csv('../input/smartphone-decimeter-2022/metadata/constellation_type_mapping.csv')\nmapping","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:04.577152Z","iopub.execute_input":"2022-12-28T13:42:04.578167Z","iopub.status.idle":"2022-12-28T13:42:04.593002Z","shell.execute_reply.started":"2022-12-28T13:42:04.578132Z","shell.execute_reply":"2022-12-28T13:42:04.591858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ground = pd.read_csv('../input/smartphone-decimeter-2022/train/2020-05-15-US-MTV-1/GooglePixel4XL/ground_truth.csv')\nground","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:04.594430Z","iopub.execute_input":"2022-12-28T13:42:04.594816Z","iopub.status.idle":"2022-12-28T13:42:04.634352Z","shell.execute_reply.started":"2022-12-28T13:42:04.594784Z","shell.execute_reply":"2022-12-28T13:42:04.633483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imu= pd.read_csv('../input/smartphone-decimeter-2022/train/2020-05-15-US-MTV-1/GooglePixel4XL/device_imu.csv')\nimu","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:04.635664Z","iopub.execute_input":"2022-12-28T13:42:04.636167Z","iopub.status.idle":"2022-12-28T13:42:05.837534Z","shell.execute_reply.started":"2022-12-28T13:42:04.636135Z","shell.execute_reply":"2022-12-28T13:42:05.836274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gnss = pd.read_csv('../input/smartphone-decimeter-2022/train/2020-05-15-US-MTV-1/GooglePixel4XL/device_gnss.csv')\ngnss","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:05.838665Z","iopub.execute_input":"2022-12-28T13:42:05.839008Z","iopub.status.idle":"2022-12-28T13:42:07.490431Z","shell.execute_reply.started":"2022-12-28T13:42:05.838979Z","shell.execute_reply":"2022-12-28T13:42:07.489077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from folium import plugins\ndf_locs = list(ground[['LatitudeDegrees','LongitudeDegrees']].values)\nfol_map = folium.Map([ground['LatitudeDegrees'].median(), ground['LongitudeDegrees'].median()],zoom_start=11)\nheat_map = plugins.HeatMap(df_locs)\nfol_map.add_child(heat_map)\nmarkers = plugins.MarkerCluster(locations = df_locs)\nfol_map.add_child(markers)","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:07.491954Z","iopub.execute_input":"2022-12-28T13:42:07.492365Z","iopub.status.idle":"2022-12-28T13:42:09.889099Z","shell.execute_reply.started":"2022-12-28T13:42:07.492329Z","shell.execute_reply":"2022-12-28T13:42:09.888074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f = open('../input/smartphone-decimeter-2022/train/2020-05-15-US-MTV-1/GooglePixel4XL/supplemental/gnss_log.txt', 'r')\nlog = f.read()\nf.close()\nlog[:500]","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:09.890691Z","iopub.execute_input":"2022-12-28T13:42:09.891019Z","iopub.status.idle":"2022-12-28T13:42:10.579019Z","shell.execute_reply.started":"2022-12-28T13:42:09.890987Z","shell.execute_reply":"2022-12-28T13:42:10.578140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"    path ='../input/smartphone-decimeter-2022/train/2020-05-15-US-MTV-1/GooglePixel4XL/supplemental/gnss_log.txt'\n    gnss_section_names = {'Raw','UncalAccel', 'UncalGyro', 'UncalMag', 'Fix', 'Status', 'OrientationDeg'}\n    with open(path) as f_open:\n        datalines = f_open.readlines()\n\n    datas = {k: [] for k in gnss_section_names}\n    gnss_map = {k: [] for k in gnss_section_names}\n    for dataline in datalines:\n      if dataline !='' and dataline[0] !='':\n        is_header = dataline.startswith('#')\n        dataline = dataline.strip('#').strip().split(',')\n        # skip over notes, version numbers, etc\n        if is_header and dataline[0] in gnss_section_names:\n            gnss_map[dataline[0]] = dataline[1:]\n        elif not is_header:\n            if dataline !='' and dataline[0] !='':\n                datas[dataline[0]].append(dataline[1:])\n\n    results = dict()\n    for k, v in datas.items():\n        results[k] = pd.DataFrame(v, columns=gnss_map[k])\n    for k, df in results.items():\n        for col in df.columns:\n            if col == 'CodeType':\n                continue\n            results[k][col] = pd.to_numeric(results[k][col])","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:10.580138Z","iopub.execute_input":"2022-12-28T13:42:10.581118Z","iopub.status.idle":"2022-12-28T13:42:20.845767Z","shell.execute_reply.started":"2022-12-28T13:42:10.581083Z","shell.execute_reply":"2022-12-28T13:42:20.844798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results['Raw']","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:20.847017Z","iopub.execute_input":"2022-12-28T13:42:20.847345Z","iopub.status.idle":"2022-12-28T13:42:20.908282Z","shell.execute_reply.started":"2022-12-28T13:42:20.847316Z","shell.execute_reply":"2022-12-28T13:42:20.907107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results['UncalAccel']","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:20.910132Z","iopub.execute_input":"2022-12-28T13:42:20.910903Z","iopub.status.idle":"2022-12-28T13:42:20.933313Z","shell.execute_reply.started":"2022-12-28T13:42:20.910842Z","shell.execute_reply":"2022-12-28T13:42:20.932112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results['UncalGyro']","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:20.935019Z","iopub.execute_input":"2022-12-28T13:42:20.936023Z","iopub.status.idle":"2022-12-28T13:42:20.961224Z","shell.execute_reply.started":"2022-12-28T13:42:20.935870Z","shell.execute_reply":"2022-12-28T13:42:20.959963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results[ 'UncalMag']","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:20.962728Z","iopub.execute_input":"2022-12-28T13:42:20.963158Z","iopub.status.idle":"2022-12-28T13:42:20.984700Z","shell.execute_reply.started":"2022-12-28T13:42:20.963117Z","shell.execute_reply":"2022-12-28T13:42:20.983601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results['Fix']","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:20.986195Z","iopub.execute_input":"2022-12-28T13:42:20.986557Z","iopub.status.idle":"2022-12-28T13:42:21.000269Z","shell.execute_reply.started":"2022-12-28T13:42:20.986526Z","shell.execute_reply":"2022-12-28T13:42:20.998989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results['Status']","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:21.008326Z","iopub.execute_input":"2022-12-28T13:42:21.009105Z","iopub.status.idle":"2022-12-28T13:42:21.022554Z","shell.execute_reply.started":"2022-12-28T13:42:21.009061Z","shell.execute_reply":"2022-12-28T13:42:21.021307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results['OrientationDeg']","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:21.025712Z","iopub.execute_input":"2022-12-28T13:42:21.026200Z","iopub.status.idle":"2022-12-28T13:42:21.038323Z","shell.execute_reply.started":"2022-12-28T13:42:21.026154Z","shell.execute_reply":"2022-12-28T13:42:21.037099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f = open('../input/smartphone-decimeter-2022/train/2020-05-15-US-MTV-1/GooglePixel4XL/supplemental/gnss_rinex.20o', 'r')\nrinex = f.read()\nf.close()\nrinex[:500]","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:21.040112Z","iopub.execute_input":"2022-12-28T13:42:21.040951Z","iopub.status.idle":"2022-12-28T13:42:21.131124Z","shell.execute_reply.started":"2022-12-28T13:42:21.040901Z","shell.execute_reply":"2022-12-28T13:42:21.130006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rinex =pd.read_csv('../input/smartphone-decimeter-2022/train/2020-05-15-US-MTV-1/GooglePixel4XL/supplemental/gnss_rinex.20o')\nrinex","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:21.133001Z","iopub.execute_input":"2022-12-28T13:42:21.133871Z","iopub.status.idle":"2022-12-28T13:42:21.247285Z","shell.execute_reply.started":"2022-12-28T13:42:21.133820Z","shell.execute_reply":"2022-12-28T13:42:21.246279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f = open('../input/smartphone-decimeter-2022/train/2020-05-15-US-MTV-1/GooglePixel4XL/supplemental/span_log.nmea', 'r')\nspan = f.read()\nf.close()\nspan[:500]","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:21.248553Z","iopub.execute_input":"2022-12-28T13:42:21.248875Z","iopub.status.idle":"2022-12-28T13:42:21.262598Z","shell.execute_reply.started":"2022-12-28T13:42:21.248845Z","shell.execute_reply":"2022-12-28T13:42:21.261535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"span = pd.read_csv('../input/smartphone-decimeter-2022/train/2020-05-15-US-MTV-1/GooglePixel4XL/supplemental/span_log.nmea')\nspan","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:21.266399Z","iopub.execute_input":"2022-12-28T13:42:21.267075Z","iopub.status.idle":"2022-12-28T13:42:21.312288Z","shell.execute_reply.started":"2022-12-28T13:42:21.267034Z","shell.execute_reply":"2022-12-28T13:42:21.311004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('../input/smartphone-decimeter-2022/sample_submission.csv')\nsub","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:21.315599Z","iopub.execute_input":"2022-12-28T13:42:21.315948Z","iopub.status.idle":"2022-12-28T13:42:21.387114Z","shell.execute_reply.started":"2022-12-28T13:42:21.315917Z","shell.execute_reply":"2022-12-28T13:42:21.386009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv('../input/smartphone-decimeter-2022/train/2020-08-06-US-MTV-2/GooglePixel4/ground_truth.csv')","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:21.388364Z","iopub.execute_input":"2022-12-28T13:42:21.389188Z","iopub.status.idle":"2022-12-28T13:42:21.418054Z","shell.execute_reply.started":"2022-12-28T13:42:21.389153Z","shell.execute_reply":"2022-12-28T13:42:21.416694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Download geojson file of US San Francisco Bay Area.\nr = requests.get(\"https://data.sfgov.org/api/views/wamw-vt4s/rows.json?accessType=DOWNLOAD\")\nr.raise_for_status()\n\n#get geojson from response\ndata = r.json()\n\n#get polygons that represents San Francisco Bay Area.\nshapes = []\nfor d in data[\"data\"]:\n    shapes.append(shapely.wkt.loads(d[8]))\n    \n#Convert list of porygons to geopandas dataframe.\ngdf_bayarea = pd.DataFrame()\n\n#I'll use only 6 and 7th object.\nfor shp in shapes[5:7]:\n    tmp = pd.DataFrame(shp, columns=[\"geometry\"])\n    gdf_bayarea = pd.concat([gdf_bayarea, tmp])\ngdf_bayarea = GeoDataFrame(gdf_bayarea)","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:21.419819Z","iopub.execute_input":"2022-12-28T13:42:21.420722Z","iopub.status.idle":"2022-12-28T13:42:22.116007Z","shell.execute_reply.started":"2022-12-28T13:42:21.420673Z","shell.execute_reply":"2022-12-28T13:42:22.114669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gdf_bayarea","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:22.117455Z","iopub.execute_input":"2022-12-28T13:42:22.118386Z","iopub.status.idle":"2022-12-28T13:42:22.169354Z","shell.execute_reply.started":"2022-12-28T13:42:22.118348Z","shell.execute_reply":"2022-12-28T13:42:22.168117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\ncollectionNames = [item.split(\"/\")[-1] for item in glob(\"../input/smartphone-decimeter-2022/train/*\")]\n\ngdfs = []\nfor collectionName in collectionNames:\n    gdfs_each_collectionName = []\n    csv_paths = glob(f\"../input/smartphone-decimeter-2022/train/{collectionName}/*/ground_truth.csv\")\n    for csv_path in csv_paths:\n        df_gt = pd.read_csv(csv_path)\n        df_gt[\"geometry\"] = [Point(lngDeg, latDeg) for lngDeg, latDeg in zip(df_gt[\"LatitudeDegrees\"], df_gt[\"LongitudeDegrees\"])]\n        gdfs_each_collectionName.append(GeoDataFrame(df_gt))\n    gdfs.append(gdfs_each_collectionName)\n    \ncolors = ['blue', 'green', 'purple', 'orange']","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:22.170631Z","iopub.execute_input":"2022-12-28T13:42:22.170972Z","iopub.status.idle":"2022-12-28T13:42:44.090738Z","shell.execute_reply.started":"2022-12-28T13:42:22.170942Z","shell.execute_reply":"2022-12-28T13:42:44.089628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gdfs_each_collectionName","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:44.092096Z","iopub.execute_input":"2022-12-28T13:42:44.092453Z","iopub.status.idle":"2022-12-28T13:42:44.144385Z","shell.execute_reply.started":"2022-12-28T13:42:44.092421Z","shell.execute_reply":"2022-12-28T13:42:44.143269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for collectionName, gdfs_each_collectionName in zip(collectionNames, gdfs):\n    fig, axs = plt.subplots(1, 2, figsize=(15, 5))\n    gdf_bayarea.plot(figsize=(10,10), color='none', edgecolor='gray', zorder=5, ax=axs[0])\n    for i, gdf in enumerate(gdfs_each_collectionName):\n        g2 = gdf.plot(color=colors[i], ax=axs[1])\n        g2.set_title(f\"Phone track of {collectionName}\")","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:42:44.146076Z","iopub.execute_input":"2022-12-28T13:42:44.146454Z","iopub.status.idle":"2022-12-28T13:44:11.060344Z","shell.execute_reply.started":"2022-12-28T13:42:44.146419Z","shell.execute_reply":"2022-12-28T13:44:11.059063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train,X_test,y_train,y_test=train_test_split(X,y,test_size=0.3,random_state=0)","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:44:11.061966Z","iopub.execute_input":"2022-12-28T13:44:11.062353Z","iopub.status.idle":"2022-12-28T13:44:11.086917Z","shell.execute_reply.started":"2022-12-28T13:44:11.062317Z","shell.execute_reply":"2022-12-28T13:44:11.085646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\nclf=lgb.LGBMClassifier()\nclf.fit(X_train,y_train)\n","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:44:11.088513Z","iopub.execute_input":"2022-12-28T13:44:11.088897Z","iopub.status.idle":"2022-12-28T13:44:11.186493Z","shell.execute_reply.started":"2022-12-28T13:44:11.088860Z","shell.execute_reply":"2022-12-28T13:44:11.184861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred=clf.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:44:11.187357Z","iopub.status.idle":"2022-12-28T13:44:11.187794Z","shell.execute_reply.started":"2022-12-28T13:44:11.187587Z","shell.execute_reply":"2022-12-28T13:44:11.187607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\naccuracy=accuracy_score(y_pred,y_test)\nprint('LightGBM Model Accuracy score:{0:0.4f}'.format(accuracy_score(y_test,y_pred)))","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:44:11.189308Z","iopub.status.idle":"2022-12-28T13:44:11.189715Z","shell.execute_reply.started":"2022-12-28T13:44:11.189519Z","shell.execute_reply":"2022-12-28T13:44:11.189537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_train=clf.predict(X_train)","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:44:11.191201Z","iopub.status.idle":"2022-12-28T13:44:11.191813Z","shell.execute_reply.started":"2022-12-28T13:44:11.191514Z","shell.execute_reply":"2022-12-28T13:44:11.191541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Training-set accuracy score:{0:0.4f}'.format(accuracy_score(y_train,y_pred_train)))","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:44:11.193340Z","iopub.status.idle":"2022-12-28T13:44:11.194092Z","shell.execute_reply.started":"2022-12-28T13:44:11.193858Z","shell.execute_reply":"2022-12-28T13:44:11.193879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Training set score:{:.4f}'.format(clf.score(X_train,y_train)))\nprint('Test set score:{:.4f}'.format(clf.score(X_test,y_test)))","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:44:11.195285Z","iopub.status.idle":"2022-12-28T13:44:11.195919Z","shell.execute_reply.started":"2022-12-28T13:44:11.195701Z","shell.execute_reply":"2022-12-28T13:44:11.195721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\ncm=confusion_matrix(y_test,y_pred)\nprint('Confusion matrix\\n\\n',cm)\nprint('\\n True Positives(TP)= ',cm[0,0])\nprint('\\n True Negatives(TN)= ',cm[1,1])\nprint('\\n False Positives(FP)= ',cm[0,1])\nprint('\\n False Negatives(FN)= ',cm[1,0])","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:44:11.197502Z","iopub.status.idle":"2022-12-28T13:44:11.197909Z","shell.execute_reply.started":"2022-12-28T13:44:11.197701Z","shell.execute_reply":"2022-12-28T13:44:11.197720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.utils.multiclass import unique_labels\nunique_labels(y_test)","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:44:11.199040Z","iopub.status.idle":"2022-12-28T13:44:11.199749Z","shell.execute_reply.started":"2022-12-28T13:44:11.199544Z","shell.execute_reply":"2022-12-28T13:44:11.199565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot(y_true,y_pred):\n    labels=unique_labels(y_test)\n    columns=[f'Predicted{label}' for label in labels]\n    index=[f'Actual{label}' for label in labels]\n    table=pd.DataFrame(confusion_matrix(y_true,y_pred),columns=columns,index=index)\n    return table","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:44:11.200828Z","iopub.status.idle":"2022-12-28T13:44:11.201208Z","shell.execute_reply.started":"2022-12-28T13:44:11.201020Z","shell.execute_reply":"2022-12-28T13:44:11.201038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot(y_test,y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:44:11.203112Z","iopub.status.idle":"2022-12-28T13:44:11.203731Z","shell.execute_reply.started":"2022-12-28T13:44:11.203423Z","shell.execute_reply":"2022-12-28T13:44:11.203452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:44:11.205079Z","iopub.status.idle":"2022-12-28T13:44:11.205682Z","shell.execute_reply.started":"2022-12-28T13:44:11.205380Z","shell.execute_reply":"2022-12-28T13:44:11.205409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot2(y_true,y_pred):\n    labels=unique_labels(y_test)\n    column=[f'Predicted{label}' for label in labels]\n    indices=[f'Actual{label}' for label in labels]\n    table=pd.DataFrame(confusion_matrix(y_true,y_pred),columns=column,index=indices)\n    return sns.heatmap(table,annot=True,fmt='d',cmap='viridis')","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:44:11.207700Z","iopub.status.idle":"2022-12-28T13:44:11.208399Z","shell.execute_reply.started":"2022-12-28T13:44:11.208038Z","shell.execute_reply":"2022-12-28T13:44:11.208072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot2(y_test,y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:44:11.210180Z","iopub.status.idle":"2022-12-28T13:44:11.210810Z","shell.execute_reply.started":"2022-12-28T13:44:11.210495Z","shell.execute_reply":"2022-12-28T13:44:11.210523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\nprint(classification_report(y_test,y_pred))","metadata":{"execution":{"iopub.status.busy":"2022-12-28T13:44:11.212499Z","iopub.status.idle":"2022-12-28T13:44:11.213079Z","shell.execute_reply.started":"2022-12-28T13:44:11.212776Z","shell.execute_reply":"2022-12-28T13:44:11.212813Z"},"trusted":true},"execution_count":null,"outputs":[]}]}