{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-08T11:01:13.373532Z","iopub.execute_input":"2023-09-08T11:01:13.373901Z","iopub.status.idle":"2023-09-08T11:01:14.640852Z","shell.execute_reply.started":"2023-09-08T11:01:13.373870Z","shell.execute_reply":"2023-09-08T11:01:14.639705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom scipy.ndimage import gaussian_filter1d\nfrom scipy.interpolate import interp1d\n","metadata":{"execution":{"iopub.status.busy":"2023-09-08T11:01:41.564819Z","iopub.execute_input":"2023-09-08T11:01:41.565401Z","iopub.status.idle":"2023-09-08T11:01:41.785517Z","shell.execute_reply.started":"2023-09-08T11:01:41.565340Z","shell.execute_reply":"2023-09-08T11:01:41.784411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def A_G_S(df, params):\n    SZ_1 = params['sz_1']\n    SZ_2 = params['sz_2']\n    SZ_CRIT = params['sz_crit']    \n    \n    unique_paths = df[['collectionName', 'phoneName']].drop_duplicates().to_numpy()\n    for collection_name, phone_name in unique_paths:\n        cond = np.logical_and(df['collectionName'] == collection_name, df['phoneName'] == phone_name)\n        data = df[cond][['latDeg', 'lngDeg']].to_numpy()\n                \n        lat_g1 = gaussian_filter1d(data[:, 0], np.sqrt(SZ_1))\n        lon_g1 = gaussian_filter1d(data[:, 1], np.sqrt(SZ_1))\n        lat_g2 = gaussian_filter1d(data[:, 0], np.sqrt(SZ_2))\n        lon_g2 = gaussian_filter1d(data[:, 1], np.sqrt(SZ_2))\n\n        lat_dif = data[1:,0] - data[:-1,0]\n        lon_dif = data[1:,1] - data[:-1,1]\n\n        lat_crit = np.append(np.abs(gaussian_filter1d(lat_dif, np.sqrt(SZ_CRIT)) / (1e-9 + gaussian_filter1d(np.abs(lat_dif), np.sqrt(SZ_CRIT)))),[0])\n        lon_crit = np.append(np.abs(gaussian_filter1d(lon_dif, np.sqrt(SZ_CRIT)) / (1e-9 + gaussian_filter1d(np.abs(lon_dif), np.sqrt(SZ_CRIT)))),[0])           \n            \n        df.loc[cond, 'latDeg'] = lat_g1 * lat_crit + lat_g2 * (1.0 - lat_crit)\n        df.loc[cond, 'lngDeg'] = lon_g1 * lon_crit + lon_g2 * (1.0 - lon_crit)    \n                       \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-09-08T11:09:00.223429Z","iopub.execute_input":"2023-09-08T11:09:00.224317Z","iopub.status.idle":"2023-09-08T11:09:00.237965Z","shell.execute_reply.started":"2023-09-08T11:09:00.224264Z","shell.execute_reply":"2023-09-08T11:09:00.236989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def M_W_Other_P(df):\n    collections_list = df[['collectionName']].drop_duplicates().to_numpy()\n\n    for collection in collections_list:\n        phone_list = df[df['collectionName'].to_list() == collection][['phoneName']].drop_duplicates().to_numpy()\n\n        phone_data = {}\n        corrections = {}\n        for phone in phone_list:\n            cond = np.logical_and(df['collectionName'] == collection[0], df['phoneName'] == phone[0]).to_list()\n            phone_data[phone[0]] = df[cond][['millisSinceGpsEpoch', 'latDeg', 'lngDeg']].to_numpy()\n\n        for current in phone_data:\n            correction = np.ones(phone_data[current].shape, dtype=np.float)\n            correction[:,1:] = phone_data[current][:,1:]\n            \n            # Telephones data don't complitely match by time, so - interpolate.\n            for other in phone_data:\n                if other == current:\n                    continue\n\n                loc = interp1d(phone_data[other][:,0], \n                               phone_data[other][:,1:], \n                               axis=0, \n                               kind='linear', \n                               copy=False, \n                               bounds_error=None, \n                               fill_value='extrapolate', \n                               assume_sorted=True)\n                \n                start_idx = 0\n                stop_idx = 0\n                for idx, val in enumerate(phone_data[current][:,0]):\n                    if val < phone_data[other][0,0]:\n                        start_idx = idx\n                    if val < phone_data[other][-1,0]:\n                        stop_idx = idx\n\n                if stop_idx - start_idx > 0:\n                    correction[start_idx:stop_idx,0] += 1\n                    correction[start_idx:stop_idx,1:] += loc(phone_data[current][start_idx:stop_idx,0])                    \n\n            correction[:,1] /= correction[:,0]\n            correction[:,2] /= correction[:,0]\n            \n            corrections[current] = correction.copy()\n        \n        for phone in phone_list:\n            cond = np.logical_and(df['collectionName'] == collection[0], df['phoneName'] == phone[0]).to_list()\n            \n            df.loc[cond, ['latDeg', 'lngDeg']] = corrections[phone[0]][:,1:]            \n            \n    return df\n","metadata":{"execution":{"iopub.status.busy":"2023-09-08T11:09:07.039509Z","iopub.execute_input":"2023-09-08T11:09:07.039967Z","iopub.status.idle":"2023-09-08T11:09:07.056137Z","shell.execute_reply.started":"2023-09-08T11:09:07.039921Z","shell.execute_reply":"2023-09-08T11:09:07.054937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test_base = pd.read_csv('/kaggle/input/google-smartphone-decimeter-challenge/baseline_locations_test.csv')\ndf_sub = pd.read_csv('/kaggle/input/google-smartphone-decimeter-challenge/sample_submission.csv')\n","metadata":{"execution":{"iopub.status.busy":"2023-09-08T11:10:00.226857Z","iopub.execute_input":"2023-09-08T11:10:00.227277Z","iopub.status.idle":"2023-09-08T11:10:00.684446Z","shell.execute_reply.started":"2023-09-08T11:10:00.227242Z","shell.execute_reply":"2023-09-08T11:10:00.683256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"smoothed_baseline_aply_gause_smoting = A_G_S(df_test_base, {'sz_1' : 0.85, 'sz_2' : 5.65, 'sz_crit' : 1.5})\nsmoothed_baseline_for_men_with_othrt_phones = M_W_Other_P(smoothed_baseline_aply_gause_smoting)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-08T11:10:08.239774Z","iopub.execute_input":"2023-09-08T11:10:08.240177Z","iopub.status.idle":"2023-09-08T11:10:15.297275Z","shell.execute_reply.started":"2023-09-08T11:10:08.240143Z","shell.execute_reply":"2023-09-08T11:10:15.296441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = df_sub.assign( latDeg=smoothed_baseline_aply_gause_smoting.latDeg, lngDeg=smoothed_baseline_for_men_with_othrt_phones.lngDeg )\nsubmission.to_csv('submission.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-08T11:10:22.031278Z","iopub.execute_input":"2023-09-08T11:10:22.031703Z","iopub.status.idle":"2023-09-08T11:10:22.798936Z","shell.execute_reply.started":"2023-09-08T11:10:22.031670Z","shell.execute_reply":"2023-09-08T11:10:22.798069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-08T11:11:10.382767Z","iopub.execute_input":"2023-09-08T11:11:10.383196Z","iopub.status.idle":"2023-09-08T11:11:10.401449Z","shell.execute_reply.started":"2023-09-08T11:11:10.383162Z","shell.execute_reply":"2023-09-08T11:11:10.400241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('Submition.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-09-08T11:12:37.476626Z","iopub.execute_input":"2023-09-08T11:12:37.477066Z","iopub.status.idle":"2023-09-08T11:12:38.234587Z","shell.execute_reply.started":"2023-09-08T11:12:37.477030Z","shell.execute_reply":"2023-09-08T11:12:38.233645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}