{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Google Smartphone Decimeter Challenge_MyFirstTry by Neural Net\n* This is my first try using Neural Net(Keras)\n* Input of NN is (_derived.csv, baseline)\n* I couldn't still understand the data about _derived.csv(https://www.kaggle.com/c/google-smartphone-decimeter-challenge/discussion/246424)\n* Getting fucking score like 21263.733\n  * I found some wrong point about creating submission file. Time and estimated values were set to wrong correspondence. Thats why I got fuckin score like  21263.733. See the NeuralNet().test()\n  * Fixed this wrong point then I get the score 7.345, this is the start line I think.\n* I will try to improve the score based on this code\n  * Many people are using Kalman Filter as a Post processing.","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom pathlib import Path\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\n\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.model_selection import train_test_split\nimport gc\nfrom contextlib import redirect_stdout","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:02:45.112152Z","iopub.execute_input":"2021-06-23T08:02:45.112520Z","iopub.status.idle":"2021-06-23T08:02:46.009336Z","shell.execute_reply.started":"2021-06-23T08:02:45.112443Z","shell.execute_reply":"2021-06-23T08:02:46.008529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Show Baseline Score","metadata":{}},{"cell_type":"code","source":"baseline_train = pd.read_csv(\"../input/google-smartphone-decimeter-challenge/baseline_locations_train.csv\")\nbaseline_test = pd.read_csv(\"../input/google-smartphone-decimeter-challenge/baseline_locations_test.csv\")\nsubmission = pd.read_csv(\"../input/google-smartphone-decimeter-challenge/sample_submission.csv\")\n# submission = submission.assign(latDeg=test.latDeg.round(6), lngDeg=test.lngDeg.round(6))","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:02:46.010822Z","iopub.execute_input":"2021-06-23T08:02:46.011192Z","iopub.status.idle":"2021-06-23T08:02:46.610290Z","shell.execute_reply.started":"2021-06-23T08:02:46.011152Z","shell.execute_reply":"2021-06-23T08:02:46.609440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\n# evaluation\ndef calc_haversine(lat1, lon1, lat2, lon2):\n    \"\"\"Calculates the great circle（円周） distance between two points\n    on the earth. Inputs are array-like and specified in decimal degrees.\n    緯度経度の2地点間の地球上の円周距離を計算（haversine formula）\n    \"\"\"\n    RADIUS = 6_367_000 # 赤道半径：6_378_100[m], 極半径： 6_356_775[m], avg. = 6_367_437.5[m]\n    lat1, lon1, lat2, lon2 = map(np.radians, [lat1, lon1, lat2, lon2])\n    dlat = lat2 - lat1\n    dlon = lon2 - lon1\n    a = np.sin(dlat/2)**2 + \\\n        np.cos(lat1) * np.cos(lat2) * np.sin(dlon/2)**2\n    dist = 2 * RADIUS * np.arcsin(a**0.5)\n    return dist\n\n# percentile: 中央値や4分数の一般化、50パーセンタイル→中央値、95パーセンタイル→上位5%の区切りの値\ndef percentile50(x):\n    return np.percentile(x, 50)\n\ndef percentile95(x):\n    return np.percentile(x, 95)\n\ndef get_train_score(df, gt):\n    gt = gt.rename(columns={'latDeg':'latDeg_gt', 'lngDeg':'lngDeg_gt'})\n    df = df.merge(gt, on=['collectionName', 'phoneName', 'millisSinceGpsEpoch'], how='inner')\n    # calc_distance_error\n    df['err'] = calc_haversine(df['latDeg_gt'], df['lngDeg_gt'], df['latDeg'], df['lngDeg'])\n    # calc_evaluate_score\n    df['phone'] = df['collectionName'] + '_' + df['phoneName']\n    res = df.groupby('phone')['err'].agg([percentile50, percentile95])\n    res['p50_p90_mean'] = (res['percentile50'] + res['percentile95']) / 2 \n    score = res['p50_p90_mean'].mean()\n    return score","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:02:46.612078Z","iopub.execute_input":"2021-06-23T08:02:46.612370Z","iopub.status.idle":"2021-06-23T08:02:46.621757Z","shell.execute_reply.started":"2021-06-23T08:02:46.612344Z","shell.execute_reply":"2021-06-23T08:02:46.620975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ground_truth\nimport pathlib\nfrom tqdm.notebook import tqdm\nINPUT = '../input/google-smartphone-decimeter-challenge'\n\np = pathlib.Path(INPUT)\ngt_files = list(p.glob('train/*/*/ground_truth.csv'))\nprint('ground_truth.csv count : ', len(gt_files))\n\ngts = []\nfor gt_file in tqdm(gt_files):\n    gts.append(pd.read_csv(gt_file))\nground_truth = pd.concat(gts)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:02:46.623408Z","iopub.execute_input":"2021-06-23T08:02:46.623899Z","iopub.status.idle":"2021-06-23T08:02:47.665442Z","shell.execute_reply.started":"2021-06-23T08:02:46.623864Z","shell.execute_reply":"2021-06-23T08:02:47.664674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(get_train_score(baseline_train, ground_truth))","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:02:47.666679Z","iopub.execute_input":"2021-06-23T08:02:47.667026Z","iopub.status.idle":"2021-06-23T08:02:47.901170Z","shell.execute_reply.started":"2021-06-23T08:02:47.666989Z","shell.execute_reply":"2021-06-23T08:02:47.900327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ansolute_error(df, gt):\n    # calc_distance_error\n    err_latDeg = abs(df[\"latDeg\"] - gt[\"latDeg\"])\n    err_lngDeg = abs(df[\"lngDeg\"] - gt[\"lngDeg\"])\n    return err_latDeg, err_lngDeg\n#err_latDeg, err_lngDeg = ansolute_error(baseline_train, ground_truth)\n#print(err_latDeg.mean(), err_lngDeg.mean())","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:02:47.902341Z","iopub.execute_input":"2021-06-23T08:02:47.902681Z","iopub.status.idle":"2021-06-23T08:02:47.909086Z","shell.execute_reply.started":"2021-06-23T08:02:47.902650Z","shell.execute_reply":"2021-06-23T08:02:47.906128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del baseline_train, baseline_test, submission\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:02:47.910499Z","iopub.execute_input":"2021-06-23T08:02:47.910864Z","iopub.status.idle":"2021-06-23T08:02:48.005592Z","shell.execute_reply.started":"2021-06-23T08:02:47.910828Z","shell.execute_reply":"2021-06-23T08:02:48.004837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Enginerrling","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport copy\nimport numpy as np\nimport pathlib\nfrom tqdm.notebook import tqdm\nimport matplotlib.pyplot as plt\nfrom multiprocessing import Pool\nimport multiprocessing as multi\n\n\nINPUT = '../input/google-smartphone-decimeter-challenge'\np = pathlib.Path(INPUT)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:02:48.009128Z","iopub.execute_input":"2021-06-23T08:02:48.009472Z","iopub.status.idle":"2021-06-23T08:02:48.015777Z","shell.execute_reply.started":"2021-06-23T08:02:48.009436Z","shell.execute_reply":"2021-06-23T08:02:48.015020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"def ×××(self, df)　は思い付きの特徴量計算（大きすぎる値を小さくするなど）\n\ndef ×××2(self, df)は👆の結果を見て、ちゃんと整形したもの（忘れないように、以下にその過程を示す）","metadata":{}},{"cell_type":"code","source":"def file_load_multi_pricessing(filename):\n    df = pd.read_csv(filename)\n    return df","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:02:48.017927Z","iopub.execute_input":"2021-06-23T08:02:48.018337Z","iopub.status.idle":"2021-06-23T08:02:48.024703Z","shell.execute_reply.started":"2021-06-23T08:02:48.018303Z","shell.execute_reply":"2021-06-23T08:02:48.023904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class FeatureEngineering:\n    def __init__(self, debug_flag=True):\n        if debug_flag:\n            self.train_files = list(p.glob('train/*/*/*_derived.csv'))\n            self.test_files = list(p.glob('test/*/*/*_derived.csv')) \n            self.raw_data = self.load_raw_data()\n        else:\n            pass\n        \n    def load_raw_data(self):\n        # load all data, return merged train/test data\n        result_train = []\n        thread_num = multi.cpu_count()\n        with Pool(thread_num) as pool:\n            imap = pool.imap(file_load_multi_pricessing, self.train_files)\n            result_train = list(tqdm(imap, total=len(self.train_files), desc=\"load train data\"))\n        train_data = pd.concat(result_train, ignore_index=True)\n\n        result_test = []\n        with Pool(4) as pool:\n            imap = pool.imap(file_load_multi_pricessing, self.test_files)\n            result_test = list(tqdm(imap, total=len(self.test_files), desc=\"load test data\"))\n        test_data = pd.concat(result_test, ignore_index=True)\n\n        return pd.concat([train_data, test_data])\n\n    def test_all(self):\n        df = self.load_raw_data()        \n        self.df = self.data_process(df)\n    \n    def data_process(self, df):\n        assert len(df) == 6357741, \"train, testに同様の変換が行われることを保証するために、２つのデータを１つのDataFrameにまとめてください。len(df) = 6357741行になるはずです。\\n To guarantee train/test are same transformation, it requires combine train and test data to one DataFrame. It should be len(df)=6357741.\" \n        \n        df = self.received_sv_time_nanos2(df)        \n        df = self.raw_prm2(df)\n        df = self.signal_type2(df)\n        df = self.sat_pos2(df)\n        df = self.sat_vel2(df)\n\n        df = self.sat_clk_bias_m2(df)\n        df = self.sat_clk_drift_mps2(df)\n        df = self.raw_pr_unc_m2(df)\n        df = self.isrb_m2(df)\n        df = self.iono_delay_m2(df)\n        df = self.tropo_delay_m2(df)\n        df = self.constellation_type2(df)\n        df = self.svid2(df)\n\n        gc.collect()\n        return df\n    \n    def received_sv_time_nanos(self, df):\n        print(\"received_sv_time_nanos ... it takes a bit time\")\n        # 'receivedSvTimeInGpsNanos' - 'millisSinceGpsEpoch' = 衛星送信時間からスマホでの記録時間の差？millisSinceGpsEpoch ←こいつは何だ\n        df['receivedSvTimeInGpsNanos_processed'] = 1000_000*df['millisSinceGpsEpoch'] - df['receivedSvTimeInGpsNanos'] # floatで計算できない大きな値なので、intで計算した後にfloatにする\n        df['receivedSvTimeInGpsNanos_processed'] = 0.001*0.001*0.001*df['receivedSvTimeInGpsNanos_processed']  # [sec]にする\n\n        return df\n    \n    def received_sv_time_nanos2(self, df):\n        print(\"received_sv_time_nanos2\")\n        df = self.received_sv_time_nanos(df)\n        target = df['receivedSvTimeInGpsNanos_processed']\n        \n        target = (target-1.075)/0.025\n        target[-1 > target] = -1\n        target[1 < target] = 1\n        \n        df['receivedSvTimeInGpsNanos_processed']  = target\n        return df\n    \n    def raw_prm(self, df):\n        print(\"raw_prm\")\n        param = 0.00000001\n        df['rawPrM_processed'] = df['rawPrM'] * param \n        return df.drop(['rawPrM'], axis=1) \n    \n    def raw_prm2(self, df):\n        print(\"raw_prm_2\")\n        df = self.raw_prm(df)\n        target = df['rawPrM_processed']\n        \n        target[target > 0.3] = 0.3\n        target = (target - np.min(target))\n        target = target / np.max(target)\n        target = (target - 0.5) * 2\n        \n        df['rawPrM_processed'] = target\n        return df\n      \n    def signal_type(self, df):\n        print(\"signal_type\")\n        ret = []\n        signal_list = [\"GPS_L1\", \"GPS_L5\", \"GAL_E1\", \"GAL_E5A\", \"GLO_G1\", \"BDS_B1I\", \"BDS_B1C\", \"BDS_B2A\", \"QZS_J1\", \"QZS_J5\"]\n        for data in tqdm(df['signalType']):\n            \n            if \"GPS_L1\":\n                if data in signal_list:\n                    ret.append(signal_list.index(data)+1)\n            else:\n                ret.append(0)\n            \n        df['signalType_processed'] = ret\n        return df.drop(['signalType'], axis=1)\n    \n    def signal_type2(self, df):\n        # 0~1にする\n        print(\"signal_type2\")\n        df = self.signal_type(df)\n        df['signalType_processed'] = df['signalType_processed'] * 0.1\n        return df\n\n    \n    def sat_pos(self, df):\n        print(\"sat_pos\")\n        param = 10e-7\n        df['xSatPosM_processed']  =  df['xSatPosM'] * param\n        df['ySatPosM_processed']  =  df['ySatPosM'] * param\n        df['zSatPosM_processed']  =  df['zSatPosM'] * param\n        return df.drop(['xSatPosM', 'ySatPosM', 'zSatPosM'], axis=1)\n    \n    def sat_pos2(self, df):\n        print(\"sat_pos2\")\n        #df = self.sat_pos(df)\n        def func(target):\n            target = (target - np.min(target))\n            target = target / np.max(target)\n            target = (target - 0.5) * 2\n            return target\n        \n        df['xSatPosM_processed'] = func(df['xSatPosM'])\n        df['ySatPosM_processed'] = func(df['ySatPosM'])\n        df['zSatPosM_processed'] = func(df['zSatPosM'])\n        return df.drop(['xSatPosM', 'ySatPosM', 'zSatPosM'], axis=1)\n\n    def sat_vel(self, df):\n        print(\"sat_pos\")\n        param = 10e-4\n        df['xSatVelMps_processed']  =  df['xSatVelMps'] * param\n        df['ySatVelMps_processed']  =  df['ySatVelMps'] * param\n        df['zSatVelMps_processed']  =  df['zSatVelMps'] * param\n        return df.drop(['xSatVelMps', 'ySatVelMps', 'zSatVelMps'], axis=1)\n    \n    def sat_vel2(self, df):\n        print(\"sat_vel2\")\n        def func(target):\n            target = (target - np.min(target))\n            target = target / np.max(target)\n            target = (target - 0.5) * 2\n            return target\n        \n        df['xSatVelMps_processed'] = func(df['xSatVelMps'])\n        df['ySatVelMps_processed'] = func(df['ySatVelMps'])\n        df['zSatVelMps_processed'] = func(df['zSatVelMps'])\n        return df.drop(['xSatVelMps', 'ySatVelMps', 'zSatVelMps'], axis=1)    \n\n    def sat_clk_bias_m(self, df):\n        print(\"sat_clk_bias_m\")\n        param = 10e-6\n        df['satClkBiasM_processed'] = df['satClkBiasM'] * param \n        return df.drop(['satClkBiasM'], axis=1)\n    \n    def sat_clk_bias_m2(self, df):\n        print(\"sat_clk_bias_m2\")\n        df = self.sat_clk_bias_m(df)\n        target = df['satClkBiasM_processed']\n        \n        target[target > 5] = 5\n        target[target < -5] = -5\n\n        target = (target - np.min(target))\n        target = target / np.max(target)\n        target = (target - 0.5) * 2\n        \n        df['satClkBiasM_processed'] = target\n        return df\n    \n    def sat_clk_drift_mps(self, df):\n        print(\"sat_clk_drift_mps\")\n        param = 1.0\n        df['satClkDriftMps_processed'] = df['satClkDriftMps'] * param \n        return df.drop(['satClkDriftMps'], axis=1)\n    \n    def sat_clk_drift_mps2(self, df):\n        print(\"sat_clk_drift_mps2\")\n        df = self.sat_clk_drift_mps(df)\n        target = df['satClkDriftMps_processed']\n        \n        target[target > 0.015] = 0.015\n        target[target < -0.015] = -0.015\n\n        target = (target - np.min(target))\n        target = target / np.max(target)\n        target = (target - 0.5) * 2\n        \n        df['satClkDriftMps_processed'] = target\n        return df\n\n    \n    def raw_pr_unc_m(self, df):\n        print(\"raw_pr_unc_m\")\n        param = 1.0\n        df['rawPrUncM_processed'] = df['rawPrUncM'] * param \n        return df.drop(['rawPrUncM'], axis=1)\n\n    def raw_pr_unc_m2(self, df):\n        print(\"raw_pr_unc_m2\")\n        df = self.raw_pr_unc_m(df)\n        target = df['rawPrUncM_processed']\n        \n        target[target > 26] = 25\n        target[target < 0] = 0\n\n        target = (target - np.min(target))\n        target = target / np.max(target)\n        target = (target - 0.5) * 2\n        \n        df['rawPrUncM_processed'] = target\n        return df\n    \n    def isrb_m(self, df):\n        print(\"isrb_m\")\n        param = 1.0\n        df['isrbM_processed'] = df['isrbM'] * param \n        return df.drop(['isrbM'], axis=1)\n    \n    def isrb_m2(self, df):\n        print(\"isrb_m2: don't use this value\")\n        return df.drop(['isrbM'], axis=1)\n    \n    def iono_delay_m(self, df):\n        print(\"iono_delay_m\")\n        param = 1.0\n        df['ionoDelayM_processed'] = df['ionoDelayM'] * param \n        return df.drop(['ionoDelayM'], axis=1)\n    \n    def iono_delay_m2(self, df):\n        print(\"iono_delay_m2\")\n        df = self.iono_delay_m(df)\n        target = df['ionoDelayM_processed']\n        \n        target[target > 21] = 21\n        target[target < 2] = 2\n\n        target = (target - np.min(target))\n        target = target / np.max(target)\n        target = (target - 0.5) * 2\n        \n        df['ionoDelayM_processed'] = target\n        return df\n\n    def tropo_delay_m(self, df):\n        print(\"tropo_delay_m\")\n        param = 1.0\n        df['tropoDelayM_processed'] = df['tropoDelayM'] * param \n        return df.drop(['tropoDelayM'], axis=1)\n    \n    def tropo_delay_m2(self, df):\n        print(\"tropo_delay_m2\")\n        df = self.tropo_delay_m(df)\n        target = df['tropoDelayM_processed']\n        \n        target[target > 30] = 30\n        target[target < 2] = 2\n\n        target = (target - np.min(target))\n        target = target / np.max(target)\n        target = (target - 0.5) * 2\n        \n        df['tropoDelayM_processed'] = target\n        return df\n    \n    def constellation_type(self, df):\n        print(\"constellation_type\")\n        return df\n        \n    def constellation_type2(self, df):\n        print(\"constellation_type2\")\n        df['constellationType_processed'] = df['constellationType'] / max(df['constellationType'])\n        return df.drop(['constellationType'], axis=1)\n    \n    def svid(self, df):\n        print(\"svid\")\n        return df\n    \n    def svid2(self, df):\n        print(\"svid2\")\n        df['svid_processed'] = df['svid'] / max(df['svid'])\n        return df.drop(['svid'], axis=1)\n","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:02:48.025905Z","iopub.execute_input":"2021-06-23T08:02:48.026246Z","iopub.status.idle":"2021-06-23T08:02:48.070531Z","shell.execute_reply.started":"2021-06-23T08:02:48.026208Z","shell.execute_reply":"2021-06-23T08:02:48.069721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Analysis with Feature Engineering for Neural Net","metadata":{}},{"cell_type":"code","source":"feature_engineering = FeatureEngineering()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:02:48.073218Z","iopub.execute_input":"2021-06-23T08:02:48.073775Z","iopub.status.idle":"2021-06-23T08:03:10.717767Z","shell.execute_reply.started":"2021-06-23T08:02:48.073738Z","shell.execute_reply":"2021-06-23T08:03:10.716853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### receivedSvTimeInGpsNanos(and millisSinceGpsEpoch)\nチップセットが受信した信号の送信時間で、GPSエポック以降のナノ秒数で表します。ReceivedSvTimeNanosから変換すると、この派生値はすべての星座で統一された時間スケールになりますが、ReceivedSvTimeNanosはGLONASSでは一日の時間、GLONASS以外の星座では週の時間を参照します。\n\nThe signal transmission time received by the chipset, in the numbers of nanoseconds since the GPS epoch. Converted from ReceivedSvTimeNanos, this derived value is in a unified time scale for all constellations, while ReceivedSvTimeNanos refers to the time of day for GLONASS and the time of week for non-GLONASS constellations.\n\nmillisSinceGpsEpoch(GPSエポック(1980/1/6 midnight UTC)からのミリ秒単位の整数。)との差分をとって値を小さくする。どちらも大きい値なので、適当に小さくしたかった。この差分が何を意味するのかはいまいちわかってない（millisSinceGpsEpochが何を意味しているのかがいまいち掴めていない）\n\nreceived_sv_time_nanos_result = param1\\*receivedSvTimeInGpsNanos - param2\\*millisSinceGpsEpoch","metadata":{}},{"cell_type":"code","source":"data = feature_engineering.received_sv_time_nanos(feature_engineering.raw_data)[\"receivedSvTimeInGpsNanos_processed\"]\n\nplt.hist(data)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:03:10.719212Z","iopub.execute_input":"2021-06-23T08:03:10.719535Z","iopub.status.idle":"2021-06-23T08:03:11.057847Z","shell.execute_reply.started":"2021-06-23T08:03:10.719503Z","shell.execute_reply":"2021-06-23T08:03:11.057052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"ヒストグラムがずいぶん偏っているので範囲を1.05~1.10に狭めてみてみる(この範囲は何回か試して決めた)","metadata":{}},{"cell_type":"code","source":"target = data[(1.05 < data) & (data < 1.1)]\nplt.hist(target, 50)\nprint(len(data))\nprint(len(target))\nprint(len(target)/len(data))","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:03:11.059127Z","iopub.execute_input":"2021-06-23T08:03:11.059475Z","iopub.status.idle":"2021-06-23T08:03:11.403217Z","shell.execute_reply.started":"2021-06-23T08:03:11.059435Z","shell.execute_reply":"2021-06-23T08:03:11.402429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"全データのうち約95%が1.05 ~ 1.10にある。\n\n1.75が中央になるように変更、1.05未満を0、1.10以上を1にする","metadata":{}},{"cell_type":"code","source":"plt.hist((target-1.075)/0.025, 50)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:03:11.404425Z","iopub.execute_input":"2021-06-23T08:03:11.404800Z","iopub.status.idle":"2021-06-23T08:03:11.723817Z","shell.execute_reply.started":"2021-06-23T08:03:11.404763Z","shell.execute_reply":"2021-06-23T08:03:11.723065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"p_data = data\np_data = (p_data-1.075)/0.025\np_data[-1 > p_data] = -1\np_data[1 < p_data] = 1\n\nplt.hist(p_data, 50)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:03:11.726510Z","iopub.execute_input":"2021-06-23T08:03:11.726777Z","iopub.status.idle":"2021-06-23T08:03:12.089803Z","shell.execute_reply.started":"2021-06-23T08:03:11.726752Z","shell.execute_reply":"2021-06-23T08:03:12.089017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"-> この結果からreceived_sv_time_nanos2を決めた","metadata":{}},{"cell_type":"code","source":"data = feature_engineering.received_sv_time_nanos2(feature_engineering.raw_data)[\"receivedSvTimeInGpsNanos_processed\"]\nplt.hist(data, 50)\ndel data, p_data, target # to reduce using RAM\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:03:12.091059Z","iopub.execute_input":"2021-06-23T08:03:12.091390Z","iopub.status.idle":"2021-06-23T08:03:12.839849Z","shell.execute_reply.started":"2021-06-23T08:03:12.091354Z","shell.execute_reply":"2021-06-23T08:03:12.839082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### signalType\n\nGNSS信号タイプは、コンステレーション名と周波数帯の組み合わせです。スマートフォンで測定される一般的な信号タイプは以下の通りです。GPS_L1、GPS_L5、GAL_E1、GAL_E5A、GLO_G1、BDS_B1I、BDS_B1C、BDS_B2A、QZS_J1、QZS_J5。\n\nThe GNSS signal type is a combination of the constellation name and the frequency band. Common signal types measured by smartphones include: GPS_L1, GPS_L5, GAL_E1, GAL_E5A, GLO_G1, BDS_B1I, BDS_B1C, BDS_B2A, QZS_J1, and QZS_J5.\n\n**cf.    def signal_type(self, df):**    \n\n| 0 | 1 | 2 | 3 | 4 | 5 | 6 | 7 | 8 | 9 | 10 |\n| ---- | ---- | ---- | ---- | ---- | ---- | ---- | ---- | ---- | ---- | ---- |\n| Other | GPS_L1 | GPS_L5 | GAL_E1 | GAL_E5A | GLO_G1 | BDS_B1I | BDS_B1C | BDS_B2A | QZS_J1 | QZS_J5 |","metadata":{}},{"cell_type":"code","source":"feature_engineering.signal_type(feature_engineering.raw_data)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:03:12.841074Z","iopub.execute_input":"2021-06-23T08:03:12.841395Z","iopub.status.idle":"2021-06-23T08:03:23.410465Z","shell.execute_reply.started":"2021-06-23T08:03:12.841360Z","shell.execute_reply":"2021-06-23T08:03:23.409592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"signal_type_data = feature_engineering.signal_type(feature_engineering.raw_data)[\"signalType_processed\"].value_counts()\nprint(signal_type_data)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:03:23.414277Z","iopub.execute_input":"2021-06-23T08:03:23.416172Z","iopub.status.idle":"2021-06-23T08:03:31.386048Z","shell.execute_reply.started":"2021-06-23T08:03:23.416134Z","shell.execute_reply":"2021-06-23T08:03:31.384674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"OK\n\n1が多くて10が少ない\n\n0~1にする **<span style=\"color: red; \">本当はone-hot vectorにすべきだと思う</span>**","metadata":{}},{"cell_type":"code","source":"signal_type_data = feature_engineering.signal_type2(feature_engineering.raw_data)[\"signalType_processed\"].value_counts()\nprint(signal_type_data)\n\ndel signal_type_data # to reduce using RAM\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:03:31.388431Z","iopub.execute_input":"2021-06-23T08:03:31.388698Z","iopub.status.idle":"2021-06-23T08:03:39.655524Z","shell.execute_reply.started":"2021-06-23T08:03:31.388671Z","shell.execute_reply":"2021-06-23T08:03:39.654509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### rawPrM\nメートル単位の生の疑似距離。これは、光速と、信号送信時刻(receivedSvTimeInGpsNanos)から信号到着時刻(Raw::TimeNanos - Raw::FullBiasNanos - Raw::BiasNanos)までの時間差の積である。\n\nRaw pseudorange in meters. It is the product between the speed of light and the time difference from the signal transmission time (receivedSvTimeInGpsNanos) to the signal arrival time (Raw::TimeNanos - Raw::FullBiasNanos - Raw::BiasNanos).\n\n**※ここではヘンテコ処理をした後なので値は小さくなっているが、ヒストグラムの形状は変わらないはず(This result is processed so the value is different from the original rawPrM but the shape of histgram is same.)**\n\n**cf.     def raw_prm(self, df):**","metadata":{}},{"cell_type":"code","source":"rawPrM_processed = feature_engineering.raw_prm(feature_engineering.raw_data)[\"rawPrM_processed\"]\nplt.hist(rawPrM_processed, 50)\nprint(len(rawPrM_processed))\nprint(len(rawPrM_processed[rawPrM_processed<0.3]))\nprint(len(rawPrM_processed[rawPrM_processed<0.3])/len(rawPrM_processed))","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:03:39.656995Z","iopub.execute_input":"2021-06-23T08:03:39.657361Z","iopub.status.idle":"2021-06-23T08:03:41.142521Z","shell.execute_reply.started":"2021-06-23T08:03:39.657322Z","shell.execute_reply":"2021-06-23T08:03:41.141754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"0.3未満に97%のデータが含まれているので、0.3以上はoutlierとして扱う","metadata":{}},{"cell_type":"code","source":"target = rawPrM_processed\ntarget[target > 0.3] = 0.3\nplt.hist(target, 100)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:03:41.143881Z","iopub.execute_input":"2021-06-23T08:03:41.144212Z","iopub.status.idle":"2021-06-23T08:03:41.543952Z","shell.execute_reply.started":"2021-06-23T08:03:41.144177Z","shell.execute_reply":"2021-06-23T08:03:41.542979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"-1~1に整形する","metadata":{}},{"cell_type":"code","source":"target2 = (target - np.min(target))\ntarget2 = target2 / np.max(target2)\ntarget2 = (target2 - 0.5) * 2\nplt.hist(target2, 100)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:03:41.545551Z","iopub.execute_input":"2021-06-23T08:03:41.545920Z","iopub.status.idle":"2021-06-23T08:03:42.066467Z","shell.execute_reply.started":"2021-06-23T08:03:41.545882Z","shell.execute_reply":"2021-06-23T08:03:42.065693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"これをraw_prm2とする","metadata":{}},{"cell_type":"code","source":"rawPrM_processed = feature_engineering.raw_prm2(feature_engineering.raw_data)[\"rawPrM_processed\"]\nplt.hist(rawPrM_processed, 50)\n\ndel rawPrM_processed, target2, target\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:03:42.070951Z","iopub.execute_input":"2021-06-23T08:03:42.071231Z","iopub.status.idle":"2021-06-23T08:03:43.154453Z","shell.execute_reply.started":"2021-06-23T08:03:42.071205Z","shell.execute_reply":"2021-06-23T08:03:43.153549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## [x/y/z]SatPosM\nttx = receivedSvTimeInGpsNanos - satClkBiasNanos (以下に定義)で定義される \"真の信号送信時間 \"の最良の推定値におけるECEF座標フレーム内の衛星位置(メートル)。これらは、衛星放送のエフェメリスを用いて計算され、真の衛星位置に対して約1mの誤差があります。\n\nThe satellite position (meters) in an ECEF coordinate frame at best estimate of “true signal transmission time” defined as ttx = receivedSvTimeInGpsNanos - satClkBiasNanos (defined below). They are computed with the satellite broadcast ephemeris, and have ~1-meter error with respect to the true satellite position.","metadata":{}},{"cell_type":"code","source":"position_processed = feature_engineering.sat_pos(feature_engineering.raw_data)\n\nx_position = position_processed[\"xSatPosM_processed\"]\ny_position = position_processed[\"ySatPosM_processed\"]\nz_position = position_processed[\"zSatPosM_processed\"]\n\nplt.figure(figsize=(34, 7), dpi=50)\nplt.subplot(1,3,1)\nplt.hist(x_position, 50) \nplt.subplot(1,3,2)\nplt.hist(y_position, 50)\nplt.subplot(1,3,3)\nplt.hist(z_position, 50)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:03:43.156663Z","iopub.execute_input":"2021-06-23T08:03:43.156996Z","iopub.status.idle":"2021-06-23T08:03:45.117292Z","shell.execute_reply.started":"2021-06-23T08:03:43.156960Z","shell.execute_reply":"2021-06-23T08:03:45.116509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"outlierはなさそうなので、このまま1~-1に整形する(sat_pos2を参照)","metadata":{}},{"cell_type":"code","source":"position_processed = feature_engineering.sat_pos2(feature_engineering.raw_data)\n\nx_position = position_processed[\"xSatPosM_processed\"]\ny_position = position_processed[\"ySatPosM_processed\"]\nz_position = position_processed[\"zSatPosM_processed\"]\n\nplt.figure(figsize=(34, 7), dpi=50)\nplt.subplot(1,3,1)\nplt.hist(x_position, 50) \nplt.subplot(1,3,2)\nplt.hist(y_position, 50)\nplt.subplot(1,3,3)\nplt.hist(z_position, 50)\nplt.show()\n\ndel position_processed, x_position, y_position, z_position\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:03:45.118523Z","iopub.execute_input":"2021-06-23T08:03:45.119032Z","iopub.status.idle":"2021-06-23T08:03:47.209037Z","shell.execute_reply.started":"2021-06-23T08:03:45.118992Z","shell.execute_reply":"2021-06-23T08:03:47.208258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## sat vel","metadata":{}},{"cell_type":"code","source":"vel_processed = feature_engineering.sat_vel(feature_engineering.raw_data)\n\nx_vel = vel_processed[\"xSatVelMps_processed\"]\ny_vel = vel_processed[\"ySatVelMps_processed\"]\nz_vel = vel_processed[\"zSatVelMps_processed\"]\n\nplt.figure(figsize=(34, 7), dpi=50)\nplt.subplot(1,3,1)\nplt.hist(x_vel, 50) \nplt.subplot(1,3,2)\nplt.hist(y_vel, 50)\nplt.subplot(1,3,3)\nplt.hist(z_vel, 50)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:03:47.210235Z","iopub.execute_input":"2021-06-23T08:03:47.210572Z","iopub.status.idle":"2021-06-23T08:03:49.366287Z","shell.execute_reply.started":"2021-06-23T08:03:47.210535Z","shell.execute_reply":"2021-06-23T08:03:49.365396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"outlierはなさそうなので、このまま1~-1に整形する(sat vel2を参照)","metadata":{}},{"cell_type":"markdown","source":"sat vel2","metadata":{}},{"cell_type":"code","source":"vel_processed = feature_engineering.sat_vel2(feature_engineering.raw_data)\n\nx_vel = vel_processed[\"xSatVelMps_processed\"]\ny_vel = vel_processed[\"ySatVelMps_processed\"]\nz_vel = vel_processed[\"zSatVelMps_processed\"]\n\nplt.figure(figsize=(34, 7), dpi=50)\nplt.subplot(1,3,1)\nplt.hist(x_vel, 50) \nplt.subplot(1,3,2)\nplt.hist(y_vel, 50)\nplt.subplot(1,3,3)\nplt.hist(z_vel, 50)\nplt.show()\n\ndel vel_processed, x_vel, y_vel, z_vel\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:03:49.367556Z","iopub.execute_input":"2021-06-23T08:03:49.367912Z","iopub.status.idle":"2021-06-23T08:03:51.386338Z","shell.execute_reply.started":"2021-06-23T08:03:49.367875Z","shell.execute_reply":"2021-06-23T08:03:51.385599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Others - satClkBiasM\tsatClkDriftMps\trawPrUncM\tisrbM\tionoDelayM\ttropoDelayM\t","metadata":{}},{"cell_type":"code","source":"sat_clk_bias_m = feature_engineering.sat_clk_bias_m(feature_engineering.raw_data)['satClkBiasM_processed']\nsat_clk_drift_mps = feature_engineering.sat_clk_drift_mps(feature_engineering.raw_data)['satClkDriftMps_processed']\nraw_pr_unc_m = feature_engineering.raw_pr_unc_m(feature_engineering.raw_data)['rawPrUncM_processed']\nisrb_m = feature_engineering.isrb_m(feature_engineering.raw_data)['isrbM_processed']\niono_delay_m = feature_engineering.iono_delay_m(feature_engineering.raw_data)['ionoDelayM_processed']\ntropo_delay_m = feature_engineering.tropo_delay_m(feature_engineering.raw_data)['tropoDelayM_processed']\n\nplt.figure(figsize=(34, 14), dpi=50)\nplt.subplot(2,3,1)\nplt.title(\"sat_clk_bias_m\")\nplt.hist(sat_clk_bias_m, 50) \n\nplt.subplot(2,3,2)\nplt.title(\"sat_clk_drift_mps\")\nplt.hist(sat_clk_drift_mps, 50)\n\nplt.subplot(2,3,3)\nplt.title(\"raw_pr_unc_m\")\nplt.hist(raw_pr_unc_m, 50)\n\nplt.subplot(2,3,4)\nplt.title(\"isrb_m\")\nplt.hist(isrb_m, 50)\n\nplt.subplot(2,3,5)\nplt.title(\"iono_delay_m\")\nplt.hist(iono_delay_m, 50)\n\nplt.subplot(2,3,6)\nplt.title(\"tropo_delay_m\")\nplt.hist(tropo_delay_m, 50)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:03:51.389033Z","iopub.execute_input":"2021-06-23T08:03:51.389298Z","iopub.status.idle":"2021-06-23T08:04:03.350513Z","shell.execute_reply.started":"2021-06-23T08:03:51.389271Z","shell.execute_reply":"2021-06-23T08:04:03.349727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"outlierと偏りがそれぞれあるので、１つづつ調整する","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(34, 14), dpi=50)\nplt.subplot(2,3,1)\nplt.title(\"sat_clk_bias_m(-5, 5)\")\nplt.hist(sat_clk_bias_m[(-5 < sat_clk_bias_m)&(sat_clk_bias_m<5)], 50) \nprint(\"sat_clk_bias_mの-5 ~ 5に含まれる割合\")\nprint(len(sat_clk_bias_m[(-5 < sat_clk_bias_m)&(sat_clk_bias_m<5)])/len(sat_clk_bias_m))\n\nplt.subplot(2,3,2)\nplt.title(\"sat_clk_drift_mps(-0.015, 0.015)\")\nplt.hist(sat_clk_drift_mps[(-0.015 < sat_clk_drift_mps)&(sat_clk_drift_mps<0.015)], 50)\nprint(\"sat_clk_drift_mps -0.015 ~ 0.015に含まれる割合\")\nprint(len(sat_clk_drift_mps[(-0.015 < sat_clk_drift_mps)&(sat_clk_drift_mps<0.015)])/len(sat_clk_drift_mps))\n\nplt.subplot(2,3,3)\nplt.title(\"raw_pr_unc_m(0,25)\")\nplt.hist(raw_pr_unc_m[(0 < raw_pr_unc_m)&(raw_pr_unc_m<25)], 50)\nprint(\"raw_pr_unc_m 0 ~ 25に含まれる割合\")\nprint(len(raw_pr_unc_m[(0 < raw_pr_unc_m)&(raw_pr_unc_m<25)])/len(raw_pr_unc_m))\n\nplt.subplot(2,3,4)\nplt.title(\"isrb_m\")\nplt.hist(isrb_m[(-0.001 < isrb_m)&(isrb_m<0.001)], 50)\nprint(\"isrb_m: 傾向が見えないので、後で別処理\")\n\nplt.subplot(2,3,5)\nplt.title(\"iono_delay_m\")\nplt.hist(iono_delay_m[(2 < iono_delay_m)&(iono_delay_m<21)], 100)\nprint(\"iono_delay_m 2 ~ 21に含まれる割合\")\nprint(len(iono_delay_m[(1 < iono_delay_m)&(iono_delay_m<21)])/len(iono_delay_m))\n\nplt.subplot(2,3,6)\nplt.title(\"tropo_delay_m\")\nplt.hist(tropo_delay_m[(2 < tropo_delay_m)&(tropo_delay_m<30)], 100)\nprint(\"tropo_delay_m 2 ~ 30に含まれる割合\")\nprint(len(tropo_delay_m[(2 < tropo_delay_m)&(tropo_delay_m<30)])/len(tropo_delay_m))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:04:03.351843Z","iopub.execute_input":"2021-06-23T08:04:03.352171Z","iopub.status.idle":"2021-06-23T08:04:06.015932Z","shell.execute_reply.started":"2021-06-23T08:04:03.352134Z","shell.execute_reply":"2021-06-23T08:04:06.015171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"isrb_m.value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:04:06.017155Z","iopub.execute_input":"2021-06-23T08:04:06.017475Z","iopub.status.idle":"2021-06-23T08:04:06.211874Z","shell.execute_reply.started":"2021-06-23T08:04:06.017439Z","shell.execute_reply":"2021-06-23T08:04:06.210889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"isrb_mはほとんど0\n\nノイズにしかならなそうなので、削除？","metadata":{}},{"cell_type":"code","source":"sat_clk_bias_m = feature_engineering.sat_clk_bias_m2(feature_engineering.raw_data)['satClkBiasM_processed']\nsat_clk_drift_mps = feature_engineering.sat_clk_drift_mps2(feature_engineering.raw_data)['satClkDriftMps_processed']\nraw_pr_unc_m = feature_engineering.raw_pr_unc_m2(feature_engineering.raw_data)['rawPrUncM_processed']\n#isrb_m = feature_engineering.isrb_m2(feature_engineering.raw_data)['isrbM_processed']\niono_delay_m = feature_engineering.iono_delay_m2(feature_engineering.raw_data)['ionoDelayM_processed']\ntropo_delay_m = feature_engineering.tropo_delay_m2(feature_engineering.raw_data)['tropoDelayM_processed']\n\nplt.figure(figsize=(34, 14), dpi=50)\nplt.subplot(2,3,1)\nplt.title(\"sat_clk_bias_m\")\nplt.hist(sat_clk_bias_m, 50) \n\nplt.subplot(2,3,2)\nplt.title(\"sat_clk_drift_mps\")\nplt.hist(sat_clk_drift_mps, 50)\n\nplt.subplot(2,3,3)\nplt.title(\"raw_pr_unc_m\")\nplt.hist(raw_pr_unc_m, 50)\n\n#plt.subplot(2,3,4)\n#plt.title(\"isrb_m\")\n#plt.hist(isrb_m, 50)\n\nplt.subplot(2,3,5)\nplt.title(\"iono_delay_m\")\nplt.hist(iono_delay_m, 50)\n\nplt.subplot(2,3,6)\nplt.title(\"tropo_delay_m\")\nplt.hist(tropo_delay_m, 50)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:04:06.213303Z","iopub.execute_input":"2021-06-23T08:04:06.213653Z","iopub.status.idle":"2021-06-23T08:04:13.470272Z","shell.execute_reply.started":"2021-06-23T08:04:06.213600Z","shell.execute_reply":"2021-06-23T08:04:13.469400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del sat_clk_bias_m, sat_clk_drift_mps, raw_pr_unc_m, isrb_m, iono_delay_m, tropo_delay_m\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:04:13.471608Z","iopub.execute_input":"2021-06-23T08:04:13.472113Z","iopub.status.idle":"2021-06-23T08:04:13.640939Z","shell.execute_reply.started":"2021-06-23T08:04:13.472074Z","shell.execute_reply":"2021-06-23T08:04:13.639861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### constellationType, svid","metadata":{}},{"cell_type":"code","source":"constellationType = feature_engineering.constellation_type2(feature_engineering.raw_data)['constellationType_processed']\nsvid = feature_engineering.svid2(feature_engineering.raw_data)['svid_processed']\n\nprint(constellationType.value_counts())\nprint(svid.value_counts())\n\ndel constellationType, svid\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:04:13.642538Z","iopub.execute_input":"2021-06-23T08:04:13.642984Z","iopub.status.idle":"2021-06-23T08:04:18.906265Z","shell.execute_reply.started":"2021-06-23T08:04:13.642944Z","shell.execute_reply":"2021-06-23T08:04:18.905431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## test all","metadata":{}},{"cell_type":"code","source":"feature_engineering.test_all()\ndisplay(feature_engineering.df)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:04:18.907584Z","iopub.execute_input":"2021-06-23T08:04:18.907949Z","iopub.status.idle":"2021-06-23T08:05:02.686166Z","shell.execute_reply.started":"2021-06-23T08:04:18.907911Z","shell.execute_reply":"2021-06-23T08:05:02.685336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del feature_engineering\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:05:02.690052Z","iopub.execute_input":"2021-06-23T08:05:02.691935Z","iopub.status.idle":"2021-06-23T08:05:02.844270Z","shell.execute_reply.started":"2021-06-23T08:05:02.691896Z","shell.execute_reply":"2021-06-23T08:05:02.843439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check memory: https://qiita.com/AnchorBlues/items/883790e43417640140aa\nimport sys\n\nprint(\"{}{: >25}{}{: >10}{}\".format('|','Variable Name','|','Memory','|'))\nprint(\" ------------------------------------ \")\nfor var_name in dir():\n    if not var_name.startswith(\"_\"):\n        print(\"{}{: >25}{}{: >10}{}\".format('|',var_name,'|',sys.getsizeof(eval(var_name)),'|'))","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:05:02.845532Z","iopub.execute_input":"2021-06-23T08:05:02.845880Z","iopub.status.idle":"2021-06-23T08:05:02.903487Z","shell.execute_reply.started":"2021-06-23T08:05:02.845843Z","shell.execute_reply":"2021-06-23T08:05:02.902658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Neural Net","metadata":{}},{"cell_type":"markdown","source":"## data loader","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport copy\nimport numpy as np\nimport pathlib\nfrom tqdm.notebook import tqdm\nimport matplotlib.pyplot as plt\nimport os\nimport glob\nfrom multiprocessing import Pool\nimport multiprocessing as multi\n\n\nINPUT = '../input/google-smartphone-decimeter-challenge'\np = pathlib.Path(INPUT)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:05:02.905517Z","iopub.execute_input":"2021-06-23T08:05:02.906167Z","iopub.status.idle":"2021-06-23T08:05:02.914450Z","shell.execute_reply.started":"2021-06-23T08:05:02.906131Z","shell.execute_reply":"2021-06-23T08:05:02.913672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def file_load_multi_pricessing(filename):\n    df = pd.read_csv(filename)\n    return df","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:05:02.915697Z","iopub.execute_input":"2021-06-23T08:05:02.916043Z","iopub.status.idle":"2021-06-23T08:05:02.928242Z","shell.execute_reply.started":"2021-06-23T08:05:02.916006Z","shell.execute_reply":"2021-06-23T08:05:02.927396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DataLoader:\n    def __init__(self):\n        self.train_files = list(p.glob('train/*/*/*_derived.csv'))\n        self.test_files = list(p.glob('test/*/*/*_derived.csv'))\n    \n        self.feature_engineering = FeatureEngineering(debug_flag=False)\n        \n        raw_data = self.load_raw_data() \n        \n        raw_data = self.feature_engineering.data_process(raw_data)\n        \n        # train/test split\n        train_data = raw_data[raw_data[\"train_flag\"]==1]\n        test_data = raw_data[raw_data[\"train_flag\"]==0]\n        \n        del raw_data\n        gc.collect()\n        \n        # add ground truth to train data\n        gt_files = list(p.glob('train/*/*/ground_truth.csv'))\n        data=[]\n        for filename in tqdm(gt_files, desc=\"Loading ground truth\"):\n            df = pd.read_csv(filename)\n            data.append(df)\n        train_gt = pd.concat(data, ignore_index=True)\n\n        train_gt[\"receivedSvTimeInGpsNanos\"] = train_gt.millisSinceGpsEpoch*int(1e6)\n        train_data = train_data.drop(\"millisSinceGpsEpoch\", axis=1)\n        \n        train_gt_data = pd.merge_asof(train_data.sort_values('receivedSvTimeInGpsNanos'), train_gt.sort_values('receivedSvTimeInGpsNanos'), \n                                                   on=\"receivedSvTimeInGpsNanos\", by=[\"collectionName\", \"phoneName\"], direction='nearest',tolerance=int(1e9))\n        self.train_gt_data = train_gt_data.sort_values(by=[\"collectionName\", \"phoneName\", \"millisSinceGpsEpoch\"], ignore_index=True)\n        \n        # add baseline value\n        # merge with baseline data\n        baseline_train = pd.read_csv(\"../input/google-smartphone-decimeter-challenge/baseline_locations_train.csv\")\n        baseline_test = pd.read_csv(\"../input/google-smartphone-decimeter-challenge/baseline_locations_test.csv\")\n\n\n        self.train_gt_data = pd.merge_asof(self.train_gt_data.sort_values('millisSinceGpsEpoch'), baseline_train.sort_values('millisSinceGpsEpoch'), \n                                                           on=\"millisSinceGpsEpoch\", suffixes=(\"_gt\", \"_base\"),  by=[\"collectionName\", \"phoneName\"], direction='nearest',tolerance=int(1e9))\n        self.test_data = pd.merge_asof(test_data.sort_values('millisSinceGpsEpoch'), baseline_test.sort_values('millisSinceGpsEpoch'), \n                                                           on=\"millisSinceGpsEpoch\", suffixes=(\"_gt\", \"_base\"),  by=[\"collectionName\", \"phoneName\"], direction='nearest',tolerance=int(1e9))\n\n        self.train_gt_data = self.gt_ajust(self.train_gt_data)\n        #self.train_gt_data, self.test_data, self.lat_bias, self.lng_bias = self.lat_lng_ajust(self.train_gt_data, self.test_data)\n\n    def load_raw_data(self):\n        # load all data, return merged train/test data\n        result_train = []\n        thread_num = multi.cpu_count()\n        with Pool(thread_num) as pool:\n            imap = pool.imap(file_load_multi_pricessing, self.train_files)\n            result_train = list(tqdm(imap, total=len(self.train_files), desc=\"load train data\"))\n        train_data_ = pd.concat(result_train, ignore_index=True)\n        train_data = pd.DataFrame([{\"train_flag\":1}]*len(train_data_)).join([train_data_])\n\n        result_test = []\n        with Pool(4) as pool:\n            imap = pool.imap(file_load_multi_pricessing, self.test_files)\n            result_test = list(tqdm(imap, total=len(self.test_files), desc=\"load test data\"))\n        test_data = pd.concat(result_test, ignore_index=True)\n        test_data = pd.DataFrame([{\"train_flag\":0}]*len(test_data)).join([test_data])\n        \n        return pd.concat([train_data, test_data])\n        \n    def load_one_data(self, df):\n        ret_data = []\n        unique_time = sorted(df[\"millisSinceGpsEpoch\"].unique())\n        for target_time in unique_time:\n            tmp_df = df[df[\"millisSinceGpsEpoch\"]==target_time]\n            ret_data.append(copy.deepcopy(tmp_df))\n            break # debug\n        return ret_data\n    \n    def lat_lng_ajust(self, df_train, df_test):\n        lat_bias = df_train['latDeg_gt'].mean()\n        lng_bias = df_train['lngDeg_gt'].mean()\n        \n        df_train['latDeg_gt'] = df_train['latDeg_gt'] - lat_bias\n        df_train['lngDeg_gt'] = df_train['lngDeg_gt'] - lng_bias\n        df_train['latDeg_base'] = df_train['latDeg_base'] - lat_bias\n        df_train['lngDeg_base'] = df_train['lngDeg_base'] - lng_bias\n\n        # same transformation with train\n        df_test['latDeg'] = df_test['latDeg'] - lat_bias\n        df_test['lngDeg'] = df_test['lngDeg'] - lng_bias\n        return df_train, df_test, lat_bias, lng_bias\n    \n    def gt_ajust(self, df_train):\n        df_train['latDeg_gt2'] = df_train['latDeg_gt'] - df_train['latDeg_base']\n        df_train['lngDeg_gt2'] = df_train['lngDeg_gt'] - df_train['lngDeg_base']\n        return df_train","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:05:02.929676Z","iopub.execute_input":"2021-06-23T08:05:02.930025Z","iopub.status.idle":"2021-06-23T08:05:02.950881Z","shell.execute_reply.started":"2021-06-23T08:05:02.929990Z","shell.execute_reply":"2021-06-23T08:05:02.950151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:05:02.952178Z","iopub.execute_input":"2021-06-23T08:05:02.952521Z","iopub.status.idle":"2021-06-23T08:05:03.060454Z","shell.execute_reply.started":"2021-06-23T08:05:02.952487Z","shell.execute_reply":"2021-06-23T08:05:03.059638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_loader = DataLoader()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-06-23T08:05:03.061860Z","iopub.execute_input":"2021-06-23T08:05:03.062362Z","iopub.status.idle":"2021-06-23T08:06:17.538669Z","shell.execute_reply.started":"2021-06-23T08:05:03.062323Z","shell.execute_reply":"2021-06-23T08:06:17.537779Z"},"_kg_hide-input":false,"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = data_loader.train_gt_data[['latDeg_gt', 'lngDeg_gt', 'latDeg_base', 'lngDeg_base']]\nprint(df.max())\nprint(df.min())","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:06:17.540287Z","iopub.execute_input":"2021-06-23T08:06:17.540648Z","iopub.status.idle":"2021-06-23T08:06:18.674198Z","shell.execute_reply.started":"2021-06-23T08:06:17.540600Z","shell.execute_reply":"2021-06-23T08:06:18.673337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = data_loader.test_data[['latDeg', 'lngDeg']]\nprint(df.max())\nprint(df.min())","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:06:18.675382Z","iopub.execute_input":"2021-06-23T08:06:18.675713Z","iopub.status.idle":"2021-06-23T08:06:19.330267Z","shell.execute_reply.started":"2021-06-23T08:06:18.675684Z","shell.execute_reply":"2021-06-23T08:06:19.329304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_loader.train_gt_data","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:06:19.334261Z","iopub.execute_input":"2021-06-23T08:06:19.336830Z","iopub.status.idle":"2021-06-23T08:06:20.771013Z","shell.execute_reply.started":"2021-06-23T08:06:19.336761Z","shell.execute_reply":"2021-06-23T08:06:20.770194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_loader.test_data","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:06:20.772219Z","iopub.execute_input":"2021-06-23T08:06:20.772733Z","iopub.status.idle":"2021-06-23T08:06:21.462780Z","shell.execute_reply.started":"2021-06-23T08:06:20.772693Z","shell.execute_reply":"2021-06-23T08:06:21.461806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"data_loader.train_gt_data.columns","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:06:21.464109Z","iopub.execute_input":"2021-06-23T08:06:21.464453Z","iopub.status.idle":"2021-06-23T08:06:21.470788Z","shell.execute_reply.started":"2021-06-23T08:06:21.464415Z","shell.execute_reply":"2021-06-23T08:06:21.469837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del data_loader\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:07:10.437674Z","iopub.execute_input":"2021-06-23T08:07:10.438001Z","iopub.status.idle":"2021-06-23T08:07:10.453181Z","shell.execute_reply.started":"2021-06-23T08:07:10.437973Z","shell.execute_reply":"2021-06-23T08:07:10.451764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training and Test(submission)","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import scale\n\nimport os\nfrom contextlib import redirect_stdout\n\n# 評価指標\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import f1_score\nfrom sklearn.metrics import precision_score\nfrom sklearn.metrics import recall_score\nfrom sklearn.metrics import log_loss\n\n# keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Activation\nfrom keras.layers.core import Dropout\nfrom keras.layers.normalization import BatchNormalization\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\n\nfrom keras.wrappers.scikit_learn import KerasClassifier\nfrom keras.callbacks import ModelCheckpoint","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:07:17.298093Z","iopub.execute_input":"2021-06-23T08:07:17.298422Z","iopub.status.idle":"2021-06-23T08:07:21.937604Z","shell.execute_reply.started":"2021-06-23T08:07:17.298392Z","shell.execute_reply":"2021-06-23T08:07:21.936757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nall_feature_train = ['train_flag', 'collectionName', 'phoneName', 'receivedSvTimeInGpsNanos',\n                   'receivedSvTimeInGpsNanos_processed', 'rawPrM_processed',\n                   'signalType_processed', 'xSatPosM_processed', 'ySatPosM_processed',\n                   'zSatPosM_processed', 'xSatVelMps_processed', 'ySatVelMps_processed',\n                   'zSatVelMps_processed', 'satClkBiasM_processed',\n                   'satClkDriftMps_processed', 'rawPrUncM_processed',\n                   'ionoDelayM_processed', 'tropoDelayM_processed',\n                   'constellationType_processed', 'svid_processed', 'millisSinceGpsEpoch',\n                   'latDeg_gt', 'lngDeg_gt', 'heightAboveWgs84EllipsoidM_gt',\n                   'timeSinceFirstFixSeconds', 'hDop', 'vDop', 'speedMps', 'courseDegree',\n                   'latDeg_base', 'lngDeg_base', 'heightAboveWgs84EllipsoidM_base',\n                   'phone']\n'''\n\nuse_feature_train = ['receivedSvTimeInGpsNanos_processed', 'rawPrM_processed',\n           'signalType_processed', 'xSatPosM_processed', 'ySatPosM_processed',\n           'zSatPosM_processed', 'xSatVelMps_processed', 'ySatVelMps_processed',\n           'zSatVelMps_processed', 'satClkBiasM_processed',\n           'satClkDriftMps_processed', 'rawPrUncM_processed',\n           'ionoDelayM_processed', 'tropoDelayM_processed',\n           'constellationType_processed', 'svid_processed',\n           'latDeg_base', 'lngDeg_base']\n\nuse_feature_test = ['receivedSvTimeInGpsNanos_processed', 'rawPrM_processed',\n   'signalType_processed', 'xSatPosM_processed', 'ySatPosM_processed',\n   'zSatPosM_processed', 'xSatVelMps_processed', 'ySatVelMps_processed',\n   'zSatVelMps_processed', 'satClkBiasM_processed',\n   'satClkDriftMps_processed', 'rawPrUncM_processed',\n   'ionoDelayM_processed', 'tropoDelayM_processed',\n   'constellationType_processed', 'svid_processed',\n   'latDeg', 'lngDeg']\n\nnn_data_param_length = len(use_feature_train)\n\n\nassert_error_str = \"must be same length with use_feature_train=\"+str(len(use_feature_train))+\" and use_feature_test=\"+str(len(use_feature_test))\nassert nn_data_param_length == len(use_feature_test), assert_error_str\n                                                                                                                          ","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:07:21.939033Z","iopub.execute_input":"2021-06-23T08:07:21.939379Z","iopub.status.idle":"2021-06-23T08:07:21.945795Z","shell.execute_reply.started":"2021-06-23T08:07:21.939344Z","shell.execute_reply":"2021-06-23T08:07:21.944995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class NeuralNet:\n    def __init__(self, data_loader):\n        \n        print(\"split to x/y\")\n        self.train_data_x = data_loader.train_gt_data[use_feature_train]\n        self.train_data_y = data_loader.train_gt_data[['latDeg_gt2', 'lngDeg_gt2']]\n        \n        self.test_data_x = data_loader.test_data[use_feature_test]\n        \n        self.x_train, self.x_val, self.y_train, self.y_val = train_test_split(self.train_data_x, self.train_data_y, test_size=0.1, random_state=40, shuffle=False)        \n\n\n        \n    \n    ###########################################################################################################\n    #\n    #    model \n    #\n    #########################################################################################################    \n    \n    \n    def create_model_simple(self):\n        activation_fun = \"swish\"\n        \n        inputs = keras.Input(shape=(nn_data_param_length,))\n        x = inputs\n                \n        x = layers.Dense(512, activation=activation_fun, kernel_initializer='he_normal')(x)\n        x = BatchNormalization()(x)\n\n        x = layers.Dense(512, activation=activation_fun, kernel_initializer='he_normal')(x)\n        x = BatchNormalization()(x)\n        \n        x = layers.Dense(512, activation=activation_fun, kernel_initializer='he_normal')(x)\n        x = BatchNormalization()(x)\n        \n        x = layers.Dense(512, activation=activation_fun, kernel_initializer='he_normal')(x)\n        x = BatchNormalization()(x)\n        x = Dropout(0.5)(x)\n        \n        outputs = layers.Dense(2, activation=\"linear\")(x)\n        return keras.Model(inputs, outputs)\n    \n    def create_model_self_attention(self):\n        activation_fun = \"swish\"\n        \n        inputs = keras.Input(shape=(nn_data_param_length,))\n        x = inputs\n        \n        query_encoding = x\n        query_value_attention = tf.keras.layers.Attention()([x, x])\n        x = tf.keras.layers.Concatenate()([query_encoding, query_value_attention])\n        x = layers.Dense(256, activation=activation_fun, kernel_initializer='he_normal')(x)\n        x = BatchNormalization()(x)\n        \n        x = layers.Dense(512, activation=activation_fun, kernel_initializer='he_normal')(x)\n        x = BatchNormalization()(x)\n        \n        x_ = x\n        \n        x = layers.Dense(512, activation=activation_fun, kernel_initializer='he_normal')(x)\n        x = BatchNormalization()(x)\n        \n        query_encoding = x\n        query_value_attention = tf.keras.layers.Attention()([x, x])\n        x = tf.keras.layers.Concatenate()([query_encoding, query_value_attention])\n        x = layers.Dense(512, activation=activation_fun, kernel_initializer='he_normal')(x)\n        x = BatchNormalization()(x)\n        \n        x = layers.Dense(512, activation=activation_fun, kernel_initializer='he_normal')(x)\n        x = BatchNormalization()(x)\n        \n        x = keras.layers.Concatenate()([x, x_])\n        x = layers.Dense(512, activation=activation_fun, kernel_initializer='he_normal')(x)\n        x = BatchNormalization()(x)\n        \n        x = layers.Dense(512, activation=activation_fun, kernel_initializer='he_normal')(x)\n        x = BatchNormalization()(x)\n        \n        x = layers.Dense(256, activation=activation_fun, kernel_initializer='he_normal')(x)\n        x = BatchNormalization()(x)\n        x = Dropout(0.5)(x)\n        \n        outputs = layers.Dense(2, activation=\"linear\")(x)\n        return keras.Model(inputs, outputs)\n    \n    ###########################################################################################################\n    #\n    #    training & test\n    #\n    #########################################################################################################   \n    \n    def calculate_loss(self, ground_truth, pred):\n        loss = pd.DataFrame({'LogLoss': log_loss(ground_truth, pred)},\n                               index = ['scores'])\n        return loss\n    \n    def calculate_scores(self, ground_truth, pred):\n        scores = pd.DataFrame({'Accuracy': accuracy_score(ground_truth, pred),\n                              'F1': f1_score(ground_truth, pred),\n                              'Precision': precision_score(ground_truth, pred),\n                              'Recall': recall_score(ground_truth, pred)},\n                               index = ['scores'])\n        return scores\n    \n    def haversine_my_metrics(self, y_true, y_pred):\n        \n        lat1, lon1, lat2, lon2 = y_true[0], y_true[1], y_pred[0], y_pred[1]\n\n        RADIUS = 6_367_000 # 赤道半径：6_378_100[m], 極半径： 6_356_775[m], avg. = 6_367_437.5[m]\n        def deg2rad(deg):\n            pi_on_180 = 0.017453292519943295\n            return deg * pi_on_180\n        lat1, lon1, lat2, lon2 = map(deg2rad, [lat1, lon1, lat2, lon2])\n        dlat = lat2 - lat1\n        dlon = lon2 - lon1\n        a = tf.sin(dlat/2)**2 + \\\n            tf.cos(lat1) * tf.cos(lat2) * tf.sin(dlon/2)**2\n        dist = 2 * RADIUS * tf.asin(a**0.5)\n        return dist\n        \n    def test(self):\n        self.model_compile()\n        self.model.load_weights('./best_weights.hdf5')\n        self.result = self.model.predict(self.test_data_x, batch_size=2048)\n        \n        result = pd.DataFrame(self.result, columns=[\"latDeg\", \"lngDeg\"])\n        result[\"latDeg\"] = result[\"latDeg\"] + self.test_data_x[\"latDeg\"]\n        result[\"lngDeg\"] = result[\"lngDeg\"] + self.test_data_x[\"lngDeg\"]\n\n        tmp = data_loader.test_data[['phone', 'millisSinceGpsEpoch']]\n        test_result = tmp.join(result)        \n        test_result = test_result.groupby([\"phone\", \"millisSinceGpsEpoch\"]).mean().reset_index()\n\n        sample_df = pd.read_csv(\"../input/google-smartphone-decimeter-challenge/sample_submission.csv\")\n        sample_df = sample_df.drop([\"latDeg\", \"lngDeg\"], axis=1)\n        self.submission = pd.merge_asof(sample_df.sort_values('millisSinceGpsEpoch'), \n                                    test_result.sort_values('millisSinceGpsEpoch'), \n                                    on=\"millisSinceGpsEpoch\", by=[\"phone\"], direction='nearest', tolerance=100000)\n        \n        self.submission = sample_df.merge(self.submission) # sort to the submission order\n        self.submission.to_csv(\"my_submission.csv\", index=False)\n        display(self.submission)\n        \n    def model_compile(self):\n        #self.model = self.create_model_self_attention()\n        self.model = self.create_model_simple()\n        lr_schedule = keras.optimizers.schedules.ExponentialDecay(\n            initial_learning_rate=1e-2,\n            decay_steps=10000,\n            decay_rate=0.9)\n        self.optimizer = keras.optimizers.SGD(learning_rate=lr_schedule)\n        self.model.compile(loss='huber', optimizer=self.optimizer, metrics=[\"mae\", self.haversine_my_metrics])\n        \n    def train(self, print_result_flag=False):\n        self.model_compile()\n        cp = ModelCheckpoint(\"best_weights.hdf5\", monitor=\"val_mae\", verbose=1,\n             save_best_only=True, save_weights_only=True, mode=\"min\")\n        self.model.fit(self.x_train, self.y_train, epochs=10, batch_size=2048, verbose=2, callbacks=[cp], validation_data=(self.x_val, self.y_val))\n\n    def evaluate_train(self):\n        self.model_compile()\n        self.model.load_weights('./best_weights.hdf5')\n        \n        train_score = self.model.evaluate(self.x_train, self.y_train, batch_size=2048)\n        val_score = self.model.evaluate(self.x_val, self.y_val, batch_size=2048)\n        \n        print(f\"train_score: {train_score}, val_score: {val_score}\")\n        \n    def evaluate_train2(self):\n        self.model_compile()\n        self.model.load_weights('./best_weights.hdf5')\n        result = self.model.predict(self.train_data_x, batch_size=2048)\n        \n        self.train_result = pd.DataFrame(result, columns=['latDeg', 'lngDeg'])\n        \n        self.train_result = self.train_result.rename(columns={\"latDeg\": \"latDeg_original\", \"lngDeg\": \"lngDeg_original\"})\n        self.train_result[\"latDeg\"] = self.train_result[\"latDeg_original\"] + self.train_data_x[\"latDeg_base\"]\n        self.train_result[\"lngDeg\"] = self.train_result[\"lngDeg_original\"] + self.train_data_x[\"lngDeg_base\"]\n        err_latDeg, err_lngDeg = ansolute_error(self.train_result, ground_truth)\n        print(err_latDeg.mean(), err_lngDeg.mean())        \n        \n        tmp = data_loader.train_gt_data[['collectionName', 'phoneName', 'millisSinceGpsEpoch']]\n        train_result = tmp.join(self.train_result)\n        print(get_train_score(train_result, ground_truth))","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:07:21.947800Z","iopub.execute_input":"2021-06-23T08:07:21.948268Z","iopub.status.idle":"2021-06-23T08:07:22.141705Z","shell.execute_reply.started":"2021-06-23T08:07:21.948232Z","shell.execute_reply":"2021-06-23T08:07:22.140814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_loader = DataLoader()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:07:22.143260Z","iopub.execute_input":"2021-06-23T08:07:22.143655Z","iopub.status.idle":"2021-06-23T08:08:35.679380Z","shell.execute_reply.started":"2021-06-23T08:07:22.143586Z","shell.execute_reply":"2021-06-23T08:08:35.678427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\nmy_nn = NeuralNet(data_loader)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:08:35.680810Z","iopub.execute_input":"2021-06-23T08:08:35.681125Z","iopub.status.idle":"2021-06-23T08:08:37.771910Z","shell.execute_reply.started":"2021-06-23T08:08:35.681092Z","shell.execute_reply":"2021-06-23T08:08:37.771129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"baseline absolute error: lat = 0.10239727214110347, lng = 0.1432227609733908","metadata":{}},{"cell_type":"code","source":"my_nn.train()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:08:37.773174Z","iopub.execute_input":"2021-06-23T08:08:37.773503Z","iopub.status.idle":"2021-06-23T08:10:24.560981Z","shell.execute_reply.started":"2021-06-23T08:08:37.773467Z","shell.execute_reply":"2021-06-23T08:10:24.559861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_nn.evaluate_train()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:10:24.562480Z","iopub.execute_input":"2021-06-23T08:10:24.562851Z","iopub.status.idle":"2021-06-23T08:10:31.113858Z","shell.execute_reply.started":"2021-06-23T08:10:24.562812Z","shell.execute_reply":"2021-06-23T08:10:31.113024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_nn.evaluate_train2()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:10:31.116211Z","iopub.execute_input":"2021-06-23T08:10:31.116574Z","iopub.status.idle":"2021-06-23T08:10:42.871401Z","shell.execute_reply.started":"2021-06-23T08:10:31.116534Z","shell.execute_reply":"2021-06-23T08:10:42.870260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_train_score(df, gt):\n    gt = gt.rename(columns={'latDeg':'latDeg_gt', 'lngDeg':'lngDeg_gt'})\n    df = df.merge(gt, on=['collectionName', 'phoneName', 'millisSinceGpsEpoch'], how='inner')\n    # calc_distance_error\n    df['err'] = calc_haversine(df['latDeg_gt'], df['lngDeg_gt'], df['latDeg'], df['lngDeg'])\n    # calc_evaluate_score\n    df['phone'] = df['collectionName'] + '_' + df['phoneName']\n    res = df.groupby('phone')['err'].agg([percentile50, percentile95])\n    res['p50_p90_mean'] = (res['percentile50'] + res['percentile95']) / 2 \n    score = res['p50_p90_mean'].mean()\n    return score","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:10:42.873390Z","iopub.execute_input":"2021-06-23T08:10:42.873741Z","iopub.status.idle":"2021-06-23T08:10:42.882097Z","shell.execute_reply.started":"2021-06-23T08:10:42.873701Z","shell.execute_reply":"2021-06-23T08:10:42.880957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"baseline absolute error: lat = 0.10239727214110347, lng = 0.1432227609733908\n\n5.287970649084159","metadata":{}},{"cell_type":"code","source":"my_nn.test()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T08:10:42.883420Z","iopub.execute_input":"2021-06-23T08:10:42.884117Z","iopub.status.idle":"2021-06-23T08:10:47.642109Z","shell.execute_reply.started":"2021-06-23T08:10:42.884078Z","shell.execute_reply":"2021-06-23T08:10:47.641228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}