{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Simple lightgbm notebook\n\n## If you find this notebook useful, please upvote it.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-05-02T17:35:44.253262Z","iopub.execute_input":"2022-05-02T17:35:44.253875Z","iopub.status.idle":"2022-05-02T17:35:44.258239Z","shell.execute_reply.started":"2022-05-02T17:35:44.253822Z","shell.execute_reply":"2022-05-02T17:35:44.257597Z"}}},{"cell_type":"markdown","source":"### Import library","metadata":{}},{"cell_type":"code","source":"import os\nfrom glob import glob\nfrom pathlib import Path\n\nimport pandas as pd\nfrom tqdm.auto import tqdm\nfrom sklearn.model_selection import KFold","metadata":{"execution":{"iopub.status.busy":"2022-05-02T19:36:42.742412Z","iopub.execute_input":"2022-05-02T19:36:42.742748Z","iopub.status.idle":"2022-05-02T19:36:42.748565Z","shell.execute_reply.started":"2022-05-02T19:36:42.742716Z","shell.execute_reply":"2022-05-02T19:36:42.747807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://github.com/nyk510/vivid/blob/master/vivid/utils.py\nfrom contextlib import contextmanager\nfrom time import time\n\nclass Timer:\n    def __init__(self, logger=None, format_str=\"{:.3f}[s]\", prefix=None, suffix=None, sep=\" \"):\n\n        if prefix: format_str = str(prefix) + sep + format_str\n        if suffix: format_str = format_str + sep + str(suffix)\n        self.format_str = format_str\n        self.logger = logger\n        self.start = None\n        self.end = None\n\n    @property\n    def duration(self):\n        if self.end is None:\n            return 0\n        return self.end - self.start\n\n    def __enter__(self):\n        self.start = time()\n\n    def __exit__(self, exc_type, exc_val, exc_tb):\n        self.end = time()\n        out_str = self.format_str.format(self.duration)\n        if self.logger:\n            self.logger.info(out_str)\n        else:\n            print(out_str)","metadata":{"execution":{"iopub.status.busy":"2022-05-02T19:36:43.100759Z","iopub.execute_input":"2022-05-02T19:36:43.101290Z","iopub.status.idle":"2022-05-02T19:36:43.111320Z","shell.execute_reply.started":"2022-05-02T19:36:43.101242Z","shell.execute_reply":"2022-05-02T19:36:43.110653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Read sample csv","metadata":{}},{"cell_type":"code","source":"BASE_DIR = Path('../input/smartphone-decimeter-2022')\ndf_sample_submission = pd.read_csv(BASE_DIR / \"sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-05-02T19:36:43.449624Z","iopub.execute_input":"2022-05-02T19:36:43.450042Z","iopub.status.idle":"2022-05-02T19:36:43.587714Z","shell.execute_reply.started":"2022-05-02T19:36:43.450013Z","shell.execute_reply":"2022-05-02T19:36:43.587004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Make dataset","metadata":{}},{"cell_type":"code","source":"train_folders = glob(str(BASE_DIR / \"train/*/*\"))\n\nX = []\ny_LatitudeDegrees = []\ny_LongitudeDegrees = []\n\nfor train_folder in tqdm(train_folders):\n    df_device_gnss = pd.read_csv( f\"{train_folder}/device_gnss.csv\").rename(columns={'utcTimeMillis': 'UnixTimeMillis'})\n    df_device_gnss = df_device_gnss.groupby(\"UnixTimeMillis\").mean()\n    \n    df_ground_truth = pd.read_csv( f\"{train_folder}/ground_truth.csv\", usecols=['UnixTimeMillis', 'LatitudeDegrees', 'LongitudeDegrees'])\n    \n    df_merged = pd.merge(df_device_gnss, df_ground_truth, on=\"UnixTimeMillis\", how=\"left\")\n    \n    X.append(df_merged.drop(columns=['LatitudeDegrees', 'LongitudeDegrees']))\n    y_LatitudeDegrees.append(df_merged['LatitudeDegrees'])\n    y_LongitudeDegrees.append(df_merged['LongitudeDegrees'])\n    \n# Concat df\nX = pd.concat(X).values\ny_LatitudeDegrees = pd.concat(y_LatitudeDegrees).values\ny_LongitudeDegrees = pd.concat(y_LongitudeDegrees).values","metadata":{"execution":{"iopub.status.busy":"2022-05-02T19:36:43.975928Z","iopub.execute_input":"2022-05-02T19:36:43.976337Z","iopub.status.idle":"2022-05-02T19:39:41.512021Z","shell.execute_reply.started":"2022-05-02T19:36:43.976307Z","shell.execute_reply":"2022-05-02T19:39:41.510846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_folders = glob(str(BASE_DIR / \"test/*/*\"))\n\ntest_index = []\ntest_X = []\ntest_y_LatitudeDegrees = []\ntest_y_LongitudeDegrees = []\n\nfor test_folder in tqdm(test_folders):\n    df_device_gnss = pd.read_csv( f\"{test_folder}/device_gnss.csv\").rename(columns={'utcTimeMillis': 'UnixTimeMillis'})\n    df_device_gnss = df_device_gnss.groupby(\"UnixTimeMillis\").mean()\n    \n    dir_name, device_name = os.path.split(test_folder)\n    _, id = os.path.split(dir_name)\n    \n    df_sample_by_tripId = df_sample_submission[df_sample_submission[\"tripId\"] == f\"{id}/{device_name}\"]\n    df_merged = pd.merge(df_device_gnss, df_sample_by_tripId.drop(columns=[\"tripId\"]), on=\"UnixTimeMillis\", how=\"left\")\n    \n    df_index = df_merged[[\"UnixTimeMillis\"]]\n    df_index[\"tripId\"] = f\"{id}/{device_name}\"\n    \n    test_index.append(df_index)\n    test_X.append(df_merged.drop(columns=['LatitudeDegrees', 'LongitudeDegrees']))\n    test_y_LatitudeDegrees.append(df_merged['LatitudeDegrees'])\n    test_y_LongitudeDegrees.append(df_merged['LongitudeDegrees'])\n    \n# Concat df\ntest_index = pd.concat(test_index).reset_index(drop=True)\ntest_X = pd.concat(test_X).values\ntest_y_LatitudeDegrees = pd.concat(test_y_LatitudeDegrees).values\ntest_y_LongitudeDegrees = pd.concat(test_y_LongitudeDegrees).values","metadata":{"execution":{"iopub.status.busy":"2022-05-02T19:39:41.517959Z","iopub.execute_input":"2022-05-02T19:39:41.518279Z","iopub.status.idle":"2022-05-02T19:40:08.208732Z","shell.execute_reply.started":"2022-05-02T19:39:41.518244Z","shell.execute_reply":"2022-05-02T19:40:08.207866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data split","metadata":{}},{"cell_type":"code","source":"fold = KFold(n_splits=5, shuffle=True, random_state=510)\ncv = fold.split(X, y_LatitudeDegrees)\ncv = list(cv)","metadata":{"execution":{"iopub.status.busy":"2022-05-02T19:40:08.210162Z","iopub.execute_input":"2022-05-02T19:40:08.210501Z","iopub.status.idle":"2022-05-02T19:40:08.248187Z","shell.execute_reply.started":"2022-05-02T19:40:08.210461Z","shell.execute_reply":"2022-05-02T19:40:08.247342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### lightgbm","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error\nimport lightgbm as lgbm\nimport numpy as np\n\ndef fit_lgbm(X, \n             y, \n             cv, \n             params: dict=None, \n             verbose: int=50):\n\n    if params is None:\n        params = {}\n\n    models = []\n    n_records = len(X)\n    oof_pred = np.zeros((n_records,), dtype=np.float32)\n\n    for i, (idx_train, idx_valid) in enumerate(cv): \n        x_train, y_train = X[idx_train], y[idx_train]\n        x_valid, y_valid = X[idx_valid], y[idx_valid]\n\n        clf = lgbm.LGBMRegressor(**params)\n\n        with Timer(prefix=\"fit fold={} \".format(i)):\n            clf.fit(x_train, y_train, \n                    eval_set=[(x_valid, y_valid)],  \n                    early_stopping_rounds=100,\n                    verbose=verbose)\n            \n        pred_i = clf.predict(x_valid)\n        oof_pred[idx_valid] = pred_i\n        models.append(clf)\n        score = mean_squared_error(y_valid, pred_i)\n        print(f\" - fold{i + 1} - {score:.10f}\")\n\n    score = mean_squared_error(y, oof_pred)\n    print(f\"{score:.10f}\")\n    return oof_pred, models","metadata":{"execution":{"iopub.status.busy":"2022-05-02T19:40:08.249932Z","iopub.execute_input":"2022-05-02T19:40:08.250143Z","iopub.status.idle":"2022-05-02T19:40:08.261612Z","shell.execute_reply.started":"2022-05-02T19:40:08.250118Z","shell.execute_reply":"2022-05-02T19:40:08.260800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_LatitudeDegrees, models_LatitudeDegrees = fit_lgbm(X=X, y=y_LatitudeDegrees, cv=cv)\noof_LongitudeDegrees, models_LongitudeDegrees = fit_lgbm(X=X, y=y_LongitudeDegrees, cv=cv)","metadata":{"execution":{"iopub.status.busy":"2022-05-02T19:40:48.379934Z","iopub.execute_input":"2022-05-02T19:40:48.380447Z","iopub.status.idle":"2022-05-02T19:49:26.681069Z","shell.execute_reply.started":"2022-05-02T19:40:48.380370Z","shell.execute_reply":"2022-05-02T19:49:26.680321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Predict","metadata":{"execution":{"iopub.status.busy":"2022-05-02T19:07:20.885185Z","iopub.execute_input":"2022-05-02T19:07:20.885891Z","iopub.status.idle":"2022-05-02T19:07:20.890476Z","shell.execute_reply.started":"2022-05-02T19:07:20.885843Z","shell.execute_reply":"2022-05-02T19:07:20.889350Z"}}},{"cell_type":"code","source":"pred_LatitudeDegrees = np.mean(np.array([model.predict(test_X) for model in models_LatitudeDegrees]), axis=0)\npred_LongitudeDegrees = np.mean(np.array([model.predict(test_X) for model in models_LongitudeDegrees]), axis=0)\ndf_res = pd.DataFrame({\n    \"LatitudeDegrees\": pred_LatitudeDegrees,\n    \"LongitudeDegrees\": pred_LongitudeDegrees\n})\ndf_res = pd.concat([test_index, df_res], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-05-02T19:49:26.685620Z","iopub.execute_input":"2022-05-02T19:49:26.687625Z","iopub.status.idle":"2022-05-02T19:50:32.255495Z","shell.execute_reply.started":"2022-05-02T19:49:26.687561Z","shell.execute_reply":"2022-05-02T19:50:32.254736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Save submission.csv","metadata":{"execution":{"iopub.status.busy":"2022-05-02T19:33:12.588231Z","iopub.execute_input":"2022-05-02T19:33:12.588567Z","iopub.status.idle":"2022-05-02T19:33:12.596660Z","shell.execute_reply.started":"2022-05-02T19:33:12.588530Z","shell.execute_reply":"2022-05-02T19:33:12.595480Z"}}},{"cell_type":"code","source":"df_res = df_res.reindex(columns=['tripId', 'UnixTimeMillis', 'LatitudeDegrees', 'LongitudeDegrees']).sort_values([\"tripId\", \"UnixTimeMillis\"]).reset_index(drop=True).to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-05-02T19:50:32.256872Z","iopub.execute_input":"2022-05-02T19:50:32.257614Z","iopub.status.idle":"2022-05-02T19:50:32.765641Z","shell.execute_reply.started":"2022-05-02T19:50:32.257575Z","shell.execute_reply":"2022-05-02T19:50:32.764722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Thanks for reading!","metadata":{}}]}