{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# https://www.kaggle.com/t88take/gsdc-phones-mean-prediction\n# import library\nimport os\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom matplotlib_venn import venn2, venn2_circles   #ベン図を作れる\nimport seaborn as sns\nfrom tqdm.notebook import tqdm\nimport pathlib\nimport plotly  # 動的なグラフを作ることができる https://qiita.com/inoory/items/12028af62018bf367722\nimport plotly.express as px  #　上記の進化版","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-18T06:27:21.296033Z","iopub.execute_input":"2021-06-18T06:27:21.296357Z","iopub.status.idle":"2021-06-18T06:27:21.301429Z","shell.execute_reply.started":"2021-06-18T06:27:21.296327Z","shell.execute_reply":"2021-06-18T06:27:21.300475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Utils","metadata":{}},{"cell_type":"code","source":"def calc_vincenty_formula(lat1, lon1, lat2, lon2):  # 経度:longitude, 緯度:latitude\n    \"\"\"Calculates the great circle distance between two points\n    on the earth. Inputs are array-like and specified in decimal degrees.\n    \"\"\"\n    #球面上の2点間の長さが最短となる距離(大遠距離を量る)\n    RADIUS = 6_367_000\n    lat1, lon1, lat2, lon2 = map(np.radians, [lat1, lon1, lat2, lon2])\n    dlat = lat2 - lat1\n    dlon = lon2 - lon1\n    a = (np.cos(lat2)*np.sin(dlon))**2 + (np.cos(lat1)*np.sin(lat2) - np.sin(lat1)*np.cos(lat2)*np.cos(dlon))**2\n    b = (a**0.5)/(np.sin(lat1)*np.sin(lat2) + np.cos(lat1)*np.cos(lat2)*np.cos(dlon))\n    dist = np.arctan(b)\n    return dist","metadata":{"execution":{"iopub.status.busy":"2021-06-18T06:27:21.302617Z","iopub.execute_input":"2021-06-18T06:27:21.303055Z","iopub.status.idle":"2021-06-18T06:27:21.312545Z","shell.execute_reply.started":"2021-06-18T06:27:21.303019Z","shell.execute_reply":"2021-06-18T06:27:21.311601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def visualize_trafic(df, center, zoom=9):\n    # px.scatter_mapbox: 動的な地図の作成に有効  https://plotly.com/python/scattermapbox/\n    fig = px.scatter_mapbox(df,    \n                            \n                            # Here, plotly gets, (x,y) coordinates\n                            lat=\"latDeg\",\n                            lon=\"lngDeg\",\n                            \n                            #Here, plotly detects color of series\n                            color='phoneName',\n                            labels='phoneName',\n                            \n                            zoom=zoom,\n                            center=center,\n                            height=600,\n                            width=800)\n    fig.update_layout(mapbox_style='stamen-terrain')    # 表示する地図の種類を指定\n    fig.update_layout(margin={\"r\":0, \"t\":0, \"l\":0, \"b\":0})\n    fig.update_layout(title_text='GPS trafic')\n    fig.show()\n    \ndef visualize_collection(df, collection):\n    target_df = df[df['collectionName']==collection].copy()\n    lat_center = target_df['latDeg'].mean()\n    lng_center = target_df['lngDeg'].mean()\n    center = {'lat':lat_center, 'lon':lng_center}\n    \n    visualize_trafic(target_df, center)","metadata":{"execution":{"iopub.status.busy":"2021-06-18T06:27:21.313898Z","iopub.execute_input":"2021-06-18T06:27:21.314320Z","iopub.status.idle":"2021-06-18T06:27:21.327949Z","shell.execute_reply.started":"2021-06-18T06:27:21.314294Z","shell.execute_reply":"2021-06-18T06:27:21.327106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# data prep","metadata":{}},{"cell_type":"code","source":"INPUT = '../input/google-smartphone-decimeter-challenge'","metadata":{"execution":{"iopub.status.busy":"2021-06-18T06:27:21.329239Z","iopub.execute_input":"2021-06-18T06:27:21.329740Z","iopub.status.idle":"2021-06-18T06:27:21.338020Z","shell.execute_reply.started":"2021-06-18T06:27:21.329702Z","shell.execute_reply":"2021-06-18T06:27:21.337144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_train = pd.read_csv(INPUT + '/' + 'baseline_locations_train.csv')\nbase_test = pd.read_csv(INPUT + '/' + 'baseline_locations_test.csv')\nsample_sub = pd.read_csv(INPUT + '/' + 'sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2021-06-18T06:27:21.339149Z","iopub.execute_input":"2021-06-18T06:27:21.339590Z","iopub.status.idle":"2021-06-18T06:27:21.716021Z","shell.execute_reply.started":"2021-06-18T06:27:21.339552Z","shell.execute_reply":"2021-06-18T06:27:21.715366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ground_truth\np = pathlib.Path(INPUT)\ngt_files = list(p.glob('train/*/*/ground_truth.csv'))\nprint('ground_truth.csv count : ', len(gt_files))\n\ngts = []\nfor gt_file in tqdm(gt_files):   # tqdm：　プログレスバー表示\n    gts.append(pd.read_csv(gt_file))\nground_truth = pd.concat(gts)    # dataFrameを一つにまとめる\n\ndisplay(ground_truth.head())","metadata":{"execution":{"iopub.status.busy":"2021-06-18T06:27:21.717066Z","iopub.execute_input":"2021-06-18T06:27:21.717501Z","iopub.status.idle":"2021-06-18T06:27:22.302357Z","shell.execute_reply.started":"2021-06-18T06:27:21.717454Z","shell.execute_reply":"2021-06-18T06:27:22.301672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# reject outlier","metadata":{}},{"cell_type":"code","source":"def add_distance_diff(df):\n    df['latDeg_prev'] = df['latDeg'].shift(1)\n    df['latDeg_next'] = df['latDeg'].shift(-1)\n    df['lngDeg_prev'] = df['lngDeg'].shift(1)\n    df['lngDeg_next'] = df['lngDeg'].shift(-1)\n    df['phone_prev'] = df['phone'].shift(1)\n    df['phone_next'] = df['phone'].shift(-1)\n    \n    df['dist_prev'] = calc_vincenty_formula(df['latDeg'], df['lngDeg'], df['latDeg_prev'], df['lngDeg_prev'])  #現在地と前の地点で距離を計算\n    df['dist_next'] = calc_vincenty_formula(df['latDeg'], df['lngDeg'], df['latDeg_next'], df['lngDeg_next'])  #     と次の\n    \n    df.loc[df['phone'] != df['phone_prev'], ['latDeg_prev', 'lngDeg_prev', 'dist_prev']] = np.nan    #現在と前の携帯が違うものに変わっているので、ひとつ前のデータは別の携帯のものなので、そのデータを消す。\n    df.loc[df['phone']!=df['phone_next'], ['latDeg_next', 'lngDeg_next', 'dist_next']] = np.nan      #現在と次の                                   後の\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2021-06-18T06:27:22.304926Z","iopub.execute_input":"2021-06-18T06:27:22.305466Z","iopub.status.idle":"2021-06-18T06:27:22.314412Z","shell.execute_reply.started":"2021-06-18T06:27:22.305427Z","shell.execute_reply":"2021-06-18T06:27:22.313659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# reject outlier\ntrain_ro = add_distance_diff(base_train)\nth = 50\ntrain_ro.loc[((train_ro['dist_prev'] > th) & (train_ro['dist_next'] > th)), ['latDeg', 'lngDeg']] = np.nan   #50を基準として、欠損値を除いている(行は減らしていない、値のみ)","metadata":{"execution":{"iopub.status.busy":"2021-06-18T06:27:22.316062Z","iopub.execute_input":"2021-06-18T06:27:22.316712Z","iopub.status.idle":"2021-06-18T06:27:22.493671Z","shell.execute_reply.started":"2021-06-18T06:27:22.316672Z","shell.execute_reply":"2021-06-18T06:27:22.492565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ro","metadata":{"execution":{"iopub.status.busy":"2021-06-18T06:27:22.495097Z","iopub.execute_input":"2021-06-18T06:27:22.495416Z","iopub.status.idle":"2021-06-18T06:27:22.527950Z","shell.execute_reply.started":"2021-06-18T06:27:22.495388Z","shell.execute_reply":"2021-06-18T06:27:22.527076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# kalman filter\nEN: https://www.kaggle.com/emaerthin/demonstration-of-the-kalman-filter\nJP: https://logics-of-blue.com/kalman-filter-concept/","metadata":{}},{"cell_type":"markdown","source":"カルマンフィルタ：　観測値や前の状態から現在の状態を推測するためのもの","metadata":{}},{"cell_type":"code","source":"!pip install simdkalman","metadata":{"execution":{"iopub.status.busy":"2021-06-18T06:27:22.529069Z","iopub.execute_input":"2021-06-18T06:27:22.529348Z","iopub.status.idle":"2021-06-18T06:27:30.731984Z","shell.execute_reply.started":"2021-06-18T06:27:22.529322Z","shell.execute_reply":"2021-06-18T06:27:30.731073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import simdkalman","metadata":{"execution":{"iopub.status.busy":"2021-06-18T06:27:30.733504Z","iopub.execute_input":"2021-06-18T06:27:30.733941Z","iopub.status.idle":"2021-06-18T06:27:30.744127Z","shell.execute_reply.started":"2021-06-18T06:27:30.733888Z","shell.execute_reply":"2021-06-18T06:27:30.743024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"T = 1.0\nstate_transition = np.array([[1, 0, T, 0, 0.5 * T ** 2, 0], [0, 1, 0, T, 0, 0.5 * T ** 2], [0, 0, 1, 0, T, 0],\n                             [0, 0, 0, 1, 0, T], [0, 0, 0, 0, 1, 0], [0, 0, 0, 0, 0, 1]])\nprocess_noise = np.diag([1e-5, 1e-5, 5e-6, 5e-6, 1e-6, 1e-6]) + np.ones((6, 6)) * 1e-9  # np.diag(): ①行列の対角成分を一次元配列で取得する ②一次元配列をを引数に渡すとその引数の対角行列を生成\n\nobservation_model = np.array([[1, 0, 0, 0, 0, 0], [0, 1, 0, 0, 0, 0]])\nobservation_noise = np.diag([5e-5, 5e-5]) + np.ones((2, 2)) * 1e-9    #np.ones((a,b)): [a,b]のすべて1で構成されている行列を生成する。 \n\nkf = simdkalman.KalmanFilter(    # https://simdkalman.readthedocs.io/en/latest/\n        state_transition = state_transition,   # A\n        process_noise = process_noise,      # Q\n        observation_model = observation_model,  # H\n        observation_noise = observation_noise)  # R\n\ndef apply_kf_smoothing(df, kf_=kf):\n    unique_paths = df[['collectionName', 'phoneName']].drop_duplicates().to_numpy()   #drop_duplicates(): 重複したところを落とす  collectionameとphonenameのペアの種類を保存\n    for collection, phone in unique_paths:\n        cond = np.logical_and(df['collectionName'] == collection, df['phoneName'] == phone)  #np.logical_and: 論理積  ようは&\n        data = df[cond][['latDeg', 'lngDeg']].to_numpy()\n        data = data.reshape(1, len(data), 2)  # (テンソル、行、列)\n        smoothed = kf_.smooth(data)   #カルマンフィルタにかける\n        df.loc[cond, 'latDeg'] = smoothed.states.mean[0, :, 0]\n        df.loc[cond, 'lngDeg'] = smoothed.states.mean[0, :, 1]\n    return df","metadata":{"execution":{"iopub.status.busy":"2021-06-18T06:27:30.745598Z","iopub.execute_input":"2021-06-18T06:27:30.745887Z","iopub.status.idle":"2021-06-18T06:27:30.759146Z","shell.execute_reply.started":"2021-06-18T06:27:30.745858Z","shell.execute_reply":"2021-06-18T06:27:30.757960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = ['collectionName', 'phoneName', 'millisSinceGpsEpoch', 'latDeg', 'lngDeg']\ntrain_ro_kf = apply_kf_smoothing(train_ro[cols])","metadata":{"execution":{"iopub.status.busy":"2021-06-18T06:27:30.760599Z","iopub.execute_input":"2021-06-18T06:27:30.760870Z","iopub.status.idle":"2021-06-18T06:28:18.255397Z","shell.execute_reply.started":"2021-06-18T06:27:30.760844Z","shell.execute_reply":"2021-06-18T06:28:18.254317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# phones mean prediction","metadata":{}},{"cell_type":"code","source":"def make_lerp_data(df):    #lerpとは、2点間の数値を近似的に求めること。\n    '''\n    Generate interpolated lat,lng values for different phone times in the same collection.\n    '''\n    org_columns = df.columns\n    \n    # Generate a combination of time x collection x phone and combine it with the original data (generate records to be interpolated)\n    time_list = df[['collectionName', 'millisSinceGpsEpoch']].drop_duplicates()\n    phone_list =df[['collectionName', 'phoneName']].drop_duplicates()\n    tmp = time_list.merge(phone_list, on='collectionName', how='outer')\n    \n    lerp_df = tmp.merge(df, on=['collectionName', 'millisSinceGpsEpoch', 'phoneName'], how='left')\n    lerp_df['phone'] = lerp_df['collectionName'] + '_' + lerp_df['phoneName']\n    lerp_df = lerp_df.sort_values(['phone', 'millisSinceGpsEpoch'])\n    \n    # linear interpolation\n    lerp_df['latDeg_prev'] = lerp_df['latDeg'].shift(1)\n    lerp_df['latDeg_next'] = lerp_df['latDeg'].shift(-1)\n    lerp_df['lngDeg_prev'] = lerp_df['lngDeg'].shift(1)\n    lerp_df['lngDeg_next'] = lerp_df['lngDeg'].shift(-1)\n    lerp_df['phone_prev'] = lerp_df['phone'].shift(1)\n    lerp_df['phone_next'] = lerp_df['phone'].shift(-1)\n    lerp_df['time_prev'] = lerp_df['millisSinceGpsEpoch'].shift(1)\n    lerp_df['time_next'] = lerp_df['millisSinceGpsEpoch'].shift(-1)\n    # Leave only records to be interpolated\n    lerp_df = lerp_df[(lerp_df['latDeg'].isnull())&(lerp_df['phone']==lerp_df['phone_prev'])&(lerp_df['phone']==lerp_df['phone_next'])].copy()\n    # calc lerp\n    lerp_df['latDeg'] = lerp_df['latDeg_prev'] + ((lerp_df['latDeg_next'] - lerp_df['latDeg_prev']) * ((lerp_df['millisSinceGpsEpoch'] - lerp_df['time_prev']) / (lerp_df['time_next'] - lerp_df['time_prev']))) \n    lerp_df['lngDeg'] = lerp_df['lngDeg_prev'] + ((lerp_df['lngDeg_next'] - lerp_df['lngDeg_prev']) * ((lerp_df['millisSinceGpsEpoch'] - lerp_df['time_prev']) / (lerp_df['time_next'] - lerp_df['time_prev']))) \n    \n    # Leave only the data that has a complete set of previous and next data.\n    lerp_df = lerp_df[~lerp_df['latDeg'].isnull()]     #欠損値を含む行を除外\n    \n    return lerp_df[org_columns]","metadata":{"execution":{"iopub.status.busy":"2021-06-18T06:28:18.256946Z","iopub.execute_input":"2021-06-18T06:28:18.257551Z","iopub.status.idle":"2021-06-18T06:28:18.276507Z","shell.execute_reply.started":"2021-06-18T06:28:18.257493Z","shell.execute_reply":"2021-06-18T06:28:18.275477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calc_mean_pred(df, lerp_df):\n    '''\n    Make a prediction based on the average of the predictions of phones in the same collection.\n    '''\n    add_lerp = pd.concat([df, lerp_df])\n    mean_pred_result = add_lerp.groupby(['collectionName', 'millisSinceGpsEpoch'])[['latDeg', 'lngDeg']].mean().reset_index()\n    mean_pred_df = df[['collectionName', 'phoneName', 'millisSinceGpsEpoch']].copy()\n    mean_pred_df = mean_pred_df.merge(mean_pred_result[['collectionName', 'millisSinceGpsEpoch', 'latDeg', 'lngDeg']], on=['collectionName', 'millisSinceGpsEpoch'], how='left')\n    return mean_pred_df","metadata":{"execution":{"iopub.status.busy":"2021-06-18T06:28:18.278054Z","iopub.execute_input":"2021-06-18T06:28:18.278765Z","iopub.status.idle":"2021-06-18T06:28:18.293974Z","shell.execute_reply.started":"2021-06-18T06:28:18.278717Z","shell.execute_reply":"2021-06-18T06:28:18.293043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_lerp = make_lerp_data(train_ro_kf)\ntrain_mean_pred = calc_mean_pred(train_ro_kf, train_lerp)    # 時間がダブっているところをひとまとめにしようとしていたが、もともと整理整頓されていたデータだったので、結果はtrain_ro_kfと一緒","metadata":{"execution":{"iopub.status.busy":"2021-06-18T06:28:18.295623Z","iopub.execute_input":"2021-06-18T06:28:18.296310Z","iopub.status.idle":"2021-06-18T06:28:19.350941Z","shell.execute_reply.started":"2021-06-18T06:28:18.296265Z","shell.execute_reply":"2021-06-18T06:28:19.349847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ro_kf","metadata":{"execution":{"iopub.status.busy":"2021-06-18T07:14:32.655667Z","iopub.execute_input":"2021-06-18T07:14:32.656007Z","iopub.status.idle":"2021-06-18T07:14:32.673885Z","shell.execute_reply.started":"2021-06-18T07:14:32.655979Z","shell.execute_reply":"2021-06-18T07:14:32.672868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_lerp","metadata":{"execution":{"iopub.status.busy":"2021-06-18T07:02:33.805158Z","iopub.execute_input":"2021-06-18T07:02:33.805508Z","iopub.status.idle":"2021-06-18T07:02:33.821151Z","shell.execute_reply.started":"2021-06-18T07:02:33.805477Z","shell.execute_reply":"2021-06-18T07:02:33.820256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_mean_pred","metadata":{"execution":{"iopub.status.busy":"2021-06-18T07:02:18.066870Z","iopub.execute_input":"2021-06-18T07:02:18.067182Z","iopub.status.idle":"2021-06-18T07:02:18.082820Z","shell.execute_reply.started":"2021-06-18T07:02:18.067157Z","shell.execute_reply":"2021-06-18T07:02:18.081944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ground_truth","metadata":{"execution":{"iopub.status.busy":"2021-06-18T07:17:57.330297Z","iopub.execute_input":"2021-06-18T07:17:57.330667Z","iopub.status.idle":"2021-06-18T07:17:57.359885Z","shell.execute_reply.started":"2021-06-18T07:17:57.330637Z","shell.execute_reply":"2021-06-18T07:17:57.358959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp1 = train_ro_kf.copy()\ntmp2 = train_mean_pred.copy()\ntmp2['phoneName'] = tmp2['phoneName'] + '_MEAN'\ntmp3 = ground_truth.copy()\ntmp3['phoneName'] = tmp3['phoneName'] + '_GT'\ntmp = pd.concat([tmp1, tmp2, tmp3])\nvisualize_collection(tmp, '2020-05-14-US-MTV-1')","metadata":{"execution":{"iopub.status.busy":"2021-06-18T06:28:19.352143Z","iopub.execute_input":"2021-06-18T06:28:19.352465Z","iopub.status.idle":"2021-06-18T06:28:20.526560Z","shell.execute_reply.started":"2021-06-18T06:28:19.352436Z","shell.execute_reply":"2021-06-18T06:28:20.525842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# evaluate train score","metadata":{}},{"cell_type":"code","source":"def percentile50(x):\n    return np.percentile(x, 50)\ndef percentile95(x):\n    return np.percentile(x, 95)","metadata":{"execution":{"iopub.status.busy":"2021-06-18T06:28:20.528834Z","iopub.execute_input":"2021-06-18T06:28:20.529285Z","iopub.status.idle":"2021-06-18T06:28:20.533587Z","shell.execute_reply.started":"2021-06-18T06:28:20.529257Z","shell.execute_reply":"2021-06-18T06:28:20.532575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_train_score(df, gt):\n    gt = gt.rename(columns={'latDeg':'latDeg_gt', 'lngDeg':'lngDeg_gt'})\n    df = df.merge(gt, on=['collectionName', 'phoneName', 'millisSinceGpsEpoch'], how='inner')\n    # calc_distance_error\n    df['err'] = calc_vincenty_formula(df['latDeg_gt'], df['lngDeg_gt'], df['latDeg'], df['lngDeg'])\n    # calc_evaluate_score\n    df['phone'] = df['collectionName'] + '_' + df['phoneName']\n    res = df.groupby('phone')['err'].agg([percentile50, percentile95])\n    res['p50_p90_mean'] = (res['percentile50'] + res['percentile95']) / 2\n    score = res['p50_p90_mean'].mean()\n    return score","metadata":{"execution":{"iopub.status.busy":"2021-06-18T06:28:20.534854Z","iopub.execute_input":"2021-06-18T06:28:20.535142Z","iopub.status.idle":"2021-06-18T06:28:20.544933Z","shell.execute_reply.started":"2021-06-18T06:28:20.535115Z","shell.execute_reply":"2021-06-18T06:28:20.544296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('kf + reject_outlier : ', get_train_score(train_ro_kf, ground_truth))\nprint('+ phones_mean_pred : ', get_train_score(train_mean_pred, ground_truth))","metadata":{"execution":{"iopub.status.busy":"2021-06-18T06:28:20.546089Z","iopub.execute_input":"2021-06-18T06:28:20.546405Z","iopub.status.idle":"2021-06-18T06:28:21.023955Z","shell.execute_reply.started":"2021-06-18T06:28:20.546342Z","shell.execute_reply":"2021-06-18T06:28:21.018908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# make submission","metadata":{}},{"cell_type":"code","source":"base_test = add_distance_diff(base_test)\nth = 50\nbase_test.loc[((base_test['dist_prev'] > th) & (base_test['dist_next'] > th)), ['latDeg', 'lngDeg']] = np.nan\n\ntest_kf = apply_kf_smoothing(base_test)\n\ntest_lerp = make_lerp_data(test_kf)\ntest_mean_pred = calc_mean_pred(test_kf, test_lerp)\n\nsample_sub['latDeg'] = test_mean_pred['latDeg']\nsample_sub['lngDeg'] = test_mean_pred['lngDeg']\nsample_sub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-06-18T06:28:21.025590Z","iopub.execute_input":"2021-06-18T06:28:21.025966Z","iopub.status.idle":"2021-06-18T06:28:55.146931Z","shell.execute_reply.started":"2021-06-18T06:28:21.025927Z","shell.execute_reply":"2021-06-18T06:28:55.145744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}