{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Average top public notebooks\n\nWeighted mean in cartesian coordinates at equal altitude.\n\n**All credit to the authors of those notebooks; please look at their work:**\n* [GSDC phones mean prediction](https://www.kaggle.com/t88take/gsdc-phones-mean-prediction)\n* [device EDA & Interpolate by removing device[en,ja]](https://www.kaggle.com/columbia2131/device-eda-interpolate-by-removing-device-en-ja)\n* [GSDC: Position shift](https://www.kaggle.com/wrrosa/gsdc-position-shift)","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy\nimport pandas as pd\nimport wgs_ecef","metadata":{"_uuid":"9955fe5e-58da-4178-8680-c0acec0bbd93","_cell_guid":"50aef91c-6606-489e-9b1f-a1d28f6c2659","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-07-28T05:17:36.287308Z","iopub.execute_input":"2021-07-28T05:17:36.287917Z","iopub.status.idle":"2021-07-28T05:17:38.421245Z","shell.execute_reply.started":"2021-07-28T05:17:36.287831Z","shell.execute_reply":"2021-07-28T05:17:38.420377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls ../input/gsdc-smart-ensembling","metadata":{"execution":{"iopub.status.busy":"2021-07-28T05:17:59.099556Z","iopub.execute_input":"2021-07-28T05:17:59.099908Z","iopub.status.idle":"2021-07-28T05:17:59.888003Z","shell.execute_reply.started":"2021-07-28T05:17:59.099874Z","shell.execute_reply":"2021-07-28T05:17:59.88708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gsdc_position_shift = pd.read_csv('../input/gsdc-position-shift/submission.csv')\ngsdc_phones_mean_prediction = pd.read_csv('../input/gsdc-phones-mean-prediction/submission.csv')\ngoogle_smartphone_decimeter_challenge = pd.read_csv('../input/device-eda-interpolate-by-removing-device-en-ja/submission.csv')\nadaptive_gauss_phone_mean = pd.read_csv('../input/adaptive-gauss-phone-mean/submission.csv')\ngsdc_smart_ensembling3 = pd.read_csv('../input/gsdc-smart-ensembling/submission3.csv')\ngsdc_smart_ensembling2 = pd.read_csv('../input/gsdc-smart-ensembling/submission2.csv')\ngsdc_smart_ensembling1 = pd.read_csv('../input/gsdc-smart-ensembling/submission1.csv')","metadata":{"execution":{"iopub.status.busy":"2021-07-28T05:18:56.648869Z","iopub.execute_input":"2021-07-28T05:18:56.649229Z","iopub.status.idle":"2021-07-28T05:18:57.117267Z","shell.execute_reply.started":"2021-07-28T05:18:56.649194Z","shell.execute_reply":"2021-07-28T05:18:57.116566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"7.6484375 + (2.25 + 2.5) / 2","metadata":{"execution":{"iopub.status.busy":"2021-07-26T06:48:50.683756Z","iopub.execute_input":"2021-07-26T06:48:50.684325Z","iopub.status.idle":"2021-07-26T06:48:50.692235Z","shell.execute_reply.started":"2021-07-26T06:48:50.684289Z","shell.execute_reply":"2021-07-26T06:48:50.691468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\nblend_weights_lat = {\n    \"gsdc-phones-mean-prediction\": (7.5 + 7.625) / 2 + 0.0625 + 0.0625/2 - 0.0625/4 + 2*0.0625/8,\n    \"device-eda-interpolate-by-removing-device-en-ja\":  (2.25 + 2.5) / 2 + (2.25 + 2.5) / 4,\n    \"gsdc-position-shift\":  0.75 - 0.125,\n}\n\n\nblend_weights_long = {\n    \"gsdc-phones-mean-prediction\": (7.5 + 7.625) / 2 + 0.0625  + 0.0625/2 - 0.0625/4 + 0.0625/8,\n    \"device-eda-interpolate-by-removing-device-en-ja\":(2.25 + 2.5) / 2 + (2.25 + 2.5) / 4,\n    \"gsdc-position-shift\": 0.75 - 0.125,\n}\n\n\nlat_norm = sum(blend_weights_lat.values())\nlong_norm = sum(blend_weights_long.values())\n\nblend = gsdc_phones_mean_prediction\n# blend['millisSinceGpsEpoch'] = (\n#     gsdc_position_shift['millisSinceGpsEpoch'] * blend_weights[\"gsdc-position-shift\"]/norm +\n#     gsdc_phones_mean_prediction['millisSinceGpsEpoch'] * blend_weights[\"gsdc-phones-mean-prediction\"]/norm +\n#     google_smartphone_decimeter_challenge['millisSinceGpsEpoch'] * blend_weights[\"device-eda-interpolate-by-removing-device-en-ja\"]/norm\n# )\n\n\nblend['latDeg'] = (\n    gsdc_position_shift['latDeg'] **2 * blend_weights_lat[\"gsdc-position-shift\"]/lat_norm +\n    gsdc_phones_mean_prediction['latDeg'] **2 * blend_weights_lat[\"gsdc-phones-mean-prediction\"]/lat_norm +\n    google_smartphone_decimeter_challenge['latDeg']**2  * blend_weights_lat[\"device-eda-interpolate-by-removing-device-en-ja\"]/lat_norm \n)**0.5\n\nblend['lngDeg'] = (\n    gsdc_position_shift['lngDeg'] * blend_weights_long[\"gsdc-position-shift\"]/long_norm +\n    gsdc_phones_mean_prediction['lngDeg'] * blend_weights_long[\"gsdc-phones-mean-prediction\"]/long_norm +\n    google_smartphone_decimeter_challenge['lngDeg'] * blend_weights_long[\"device-eda-interpolate-by-removing-device-en-ja\"]/long_norm \n)\n\n","metadata":{"execution":{"iopub.status.busy":"2021-07-26T06:51:28.823061Z","iopub.execute_input":"2021-07-26T06:51:28.82359Z","iopub.status.idle":"2021-07-26T06:51:28.8775Z","shell.execute_reply.started":"2021-07-26T06:51:28.823557Z","shell.execute_reply":"2021-07-26T06:51:28.876563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"blend['millisSinceGpsEpoch'] = blend['millisSinceGpsEpoch'].astype(np.int64)","metadata":{"execution":{"iopub.status.busy":"2021-07-26T06:51:36.277573Z","iopub.execute_input":"2021-07-26T06:51:36.2779Z","iopub.status.idle":"2021-07-26T06:51:36.283654Z","shell.execute_reply.started":"2021-07-26T06:51:36.277871Z","shell.execute_reply":"2021-07-26T06:51:36.282507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"blend","metadata":{"execution":{"iopub.status.busy":"2021-07-26T06:52:50.453104Z","iopub.execute_input":"2021-07-26T06:52:50.453464Z","iopub.status.idle":"2021-07-26T06:52:50.468409Z","shell.execute_reply.started":"2021-07-26T06:52:50.453434Z","shell.execute_reply":"2021-07-26T06:52:50.467706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"blend['latDeg'] = 0.35 * blend['latDeg'] +  0.65 * adaptive_gauss_phone_mean['latDeg']\nblend['lngDeg'] = 0.35 * blend['lngDeg'] +  0.65 * adaptive_gauss_phone_mean['lngDeg']","metadata":{"execution":{"iopub.status.busy":"2021-07-26T06:53:07.663679Z","iopub.execute_input":"2021-07-26T06:53:07.664253Z","iopub.status.idle":"2021-07-26T06:53:07.674026Z","shell.execute_reply.started":"2021-07-26T06:53:07.664206Z","shell.execute_reply":"2021-07-26T06:53:07.673143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"blend['latDeg'] = 0.3 * blend['latDeg'] +  0.3*gsdc_smart_ensembling3['latDeg'] +  0.4*gsdc_smart_ensembling2['latDeg'] # + 0.175*gsdc_smart_ensembling1['latDeg']\nblend['lngDeg'] = 0.3 * blend['lngDeg'] +  0.3*gsdc_smart_ensembling3['lngDeg'] +  0.4*gsdc_smart_ensembling2['lngDeg'] # + 0.175*gsdc_smart_ensembling1['lngDeg']","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"blend.to_csv(\"submission1.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2021-07-26T06:53:09.931841Z","iopub.execute_input":"2021-07-26T06:53:09.932285Z","iopub.status.idle":"2021-07-26T06:53:10.360706Z","shell.execute_reply.started":"2021-07-26T06:53:09.932243Z","shell.execute_reply":"2021-07-26T06:53:10.359581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom scipy.ndimage import gaussian_filter1d\nfrom scipy.interpolate import interp1d","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def apply_gauss_smoothing(df, params):\n    SZ_1 = params['sz_1']\n    SZ_2 = params['sz_2']\n    SZ_CRIT = params['sz_crit']    \n    \n    unique_paths = df[['collectionName', 'phoneName']].drop_duplicates().to_numpy()\n    for collection, phone in unique_paths:\n        cond = np.logical_and(df['collectionName'] == collection, df['phoneName'] == phone)\n        data = df[cond][['latDeg', 'lngDeg']].to_numpy()\n                \n        lat_g1 = gaussian_filter1d(data[:, 0], np.sqrt(SZ_1))\n        lon_g1 = gaussian_filter1d(data[:, 1], np.sqrt(SZ_1))\n        lat_g2 = gaussian_filter1d(data[:, 0], np.sqrt(SZ_2))\n        lon_g2 = gaussian_filter1d(data[:, 1], np.sqrt(SZ_2))\n\n        lat_dif = data[1:,0] - data[:-1,0]\n        lon_dif = data[1:,1] - data[:-1,1]\n\n        lat_crit = np.append(np.abs(gaussian_filter1d(lat_dif, np.sqrt(SZ_CRIT)) / (1e-9 + gaussian_filter1d(np.abs(lat_dif), np.sqrt(SZ_CRIT)))),[0])\n        lon_crit = np.append(np.abs(gaussian_filter1d(lon_dif, np.sqrt(SZ_CRIT)) / (1e-9 + gaussian_filter1d(np.abs(lon_dif), np.sqrt(SZ_CRIT)))),[0])           \n            \n        df.loc[cond, 'latDeg'] = lat_g1 * lat_crit + lat_g2 * (1.0 - lat_crit)\n        df.loc[cond, 'lngDeg'] = lon_g1 * lon_crit + lon_g2 * (1.0 - lon_crit)    \n                       \n    return df","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mean_with_other_phones(df):\n    collections_list = df[['collectionName']].drop_duplicates().to_numpy()\n\n    for collection in collections_list:\n        phone_list = df[df['collectionName'].to_list() == collection][['phoneName']].drop_duplicates().to_numpy()\n\n        phone_data = {}\n        corrections = {}\n        for phone in phone_list:\n            cond = np.logical_and(df['collectionName'] == collection[0], df['phoneName'] == phone[0]).to_list()\n            phone_data[phone[0]] = df[cond][['millisSinceGpsEpoch', 'latDeg', 'lngDeg']].to_numpy()\n\n        for current in phone_data:\n            correction = np.ones(phone_data[current].shape, dtype=np.float)\n            correction[:,1:] = phone_data[current][:,1:]\n            \n            # Telephones data don't complitely match by time, so - interpolate.\n            for other in phone_data:\n                if other == current:\n                    continue\n\n                loc = interp1d(phone_data[other][:,0], \n                               phone_data[other][:,1:], \n                               axis=0, \n                               kind='linear', \n                               copy=False, \n                               bounds_error=None, \n                               fill_value='extrapolate', \n                               assume_sorted=True)\n                \n                start_idx = 0\n                stop_idx = 0\n                for idx, val in enumerate(phone_data[current][:,0]):\n                    if val < phone_data[other][0,0]:\n                        start_idx = idx\n                    if val < phone_data[other][-1,0]:\n                        stop_idx = idx\n\n                if stop_idx - start_idx > 0:\n                    correction[start_idx:stop_idx,0] += 1\n                    correction[start_idx:stop_idx,1:] += loc(phone_data[current][start_idx:stop_idx,0])                    \n\n            correction[:,1] /= correction[:,0]\n            correction[:,2] /= correction[:,0]\n            \n            corrections[current] = correction.copy()\n        \n        for phone in phone_list:\n            cond = np.logical_and(df['collectionName'] == collection[0], df['phoneName'] == phone[0]).to_list()\n            \n            df.loc[cond, ['latDeg', 'lngDeg']] = corrections[phone[0]][:,1:]            \n            \n    return df","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_base = pd.read_csv('../input/google-smartphone-decimeter-challenge/baseline_locations_test.csv')\nsub = pd.read_csv('../input/google-smartphone-decimeter-challenge/sample_submission.csv')\n\nsmoothed_baseline = apply_gauss_smoothing(test_base, {'sz_1' : 0.85, 'sz_2' : 5.65, 'sz_crit' : 0.8})\nsmoothed_baseline = mean_with_other_phones(smoothed_baseline)\n\nsub = sub.assign( latDeg=smoothed_baseline.latDeg, lngDeg=smoothed_baseline.lngDeg )\nsub.to_csv('submission2.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport seaborn as sns\n\nimport matplotlib.pyplot as plt\nimport plotly.figure_factory as ff\nimport plotly.express as px","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ensembling(main, support, coeff1, coeff2): \n    \n    suba  = main.copy() \n    subav = suba.values\n       \n    subb  = support.copy()\n    subbv = subb.values    \n           \n    ense  = main.copy()    \n    ensev = ense.values  \n \n    for i in range (len(main)):\n        \n        pera1 = subav[i, 2]\n        pera2 = subav[i, 3]\n        \n        perb1 = subbv[i, 2]\n        perb2 = subbv[i, 3]\n\n        per1 = (pera1 * coeff1) + (perb1 * (1.0 - coeff1))\n        per2 = (pera2 * coeff2) + (perb2 * (1.0 - coeff2))\n        \n        ensev[i, 2] = per1\n        ensev[i, 3] = per2\n        \n    ense.iloc[:, 2:] = ensev[:, 2:]  \n  \n    return ense ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub1 = ensembling(blend, sub, 0.55, 0.50)\nsub1.to_csv(\"submission.csv\",index=False)","metadata":{},"execution_count":null,"outputs":[]}]}