{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-17T03:00:58.65307Z","iopub.execute_input":"2022-03-17T03:00:58.653428Z","iopub.status.idle":"2022-03-17T03:00:58.68Z","shell.execute_reply.started":"2022-03-17T03:00:58.653329Z","shell.execute_reply":"2022-03-17T03:00:58.679288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nimport gc\nimport holidays\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_absolute_error\n\nfrom catboost import CatBoostRegressor\n\n#warnings.filterwarnings(\"ignore\")\nseed = 256","metadata":{"execution":{"iopub.status.busy":"2022-03-17T03:00:58.681847Z","iopub.execute_input":"2022-03-17T03:00:58.682334Z","iopub.status.idle":"2022-03-17T03:01:00.1341Z","shell.execute_reply.started":"2022-03-17T03:00:58.682292Z","shell.execute_reply":"2022-03-17T03:01:00.133338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('../input/tabular-playground-series-mar-2022/train.csv')\nreal_test_df = pd.read_csv('../input/tabular-playground-series-mar-2022/test.csv')\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-17T03:01:00.135269Z","iopub.execute_input":"2022-03-17T03:01:00.135499Z","iopub.status.idle":"2022-03-17T03:01:00.979094Z","shell.execute_reply.started":"2022-03-17T03:01:00.135472Z","shell.execute_reply":"2022-03-17T03:01:00.97849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-03-17T03:01:00.980632Z","iopub.execute_input":"2022-03-17T03:01:00.981276Z","iopub.status.idle":"2022-03-17T03:01:00.987978Z","shell.execute_reply.started":"2022-03-17T03:01:00.981241Z","shell.execute_reply":"2022-03-17T03:01:00.987178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define Model Evaluation functions\ndef evaluate_model(model, x, y):\n    y_pred = model.predict(x)\n    result = mean_absolute_error(y, y_pred)\n    return result","metadata":{"execution":{"iopub.status.busy":"2022-03-17T03:01:00.992253Z","iopub.execute_input":"2022-03-17T03:01:00.992543Z","iopub.status.idle":"2022-03-17T03:01:01.000348Z","shell.execute_reply.started":"2022-03-17T03:01:00.992496Z","shell.execute_reply":"2022-03-17T03:01:00.999689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Define data pre-processing functions\ndef label_encoder(df):\n    # Create coodinate codes\n    dir_mapper = {'EB': [1,0], 'NB': [0,1], 'SB': [0,-1], 'WB': [-1,0], \n                  'NE': [1,1], 'SW': [-1,-1], 'NW': [-1,1], 'SE': [1,-1]}\n    # Encode lables\n    direction = {d : i for i, d in enumerate(df['direction'].unique())}\n    df = df.copy()\n    df['direction_coord_0'] = df['direction'].map(lambda x: dir_mapper[x][0])\n    df['direction_coord_1'] = df['direction'].map(lambda x: dir_mapper[x][1])\n    df['direction'] = df['direction'].replace(direction)\n    return df\n\ndef preprocess_dates(df):\n    df = df.copy()\n    df['time'] = pd.to_datetime(df['time'])\n    df['minute'] = df['time'].dt.minute\n    df['hour'] = df['time'].dt.hour\n    df['weekday'] = df['time'].dt.weekday\n    df['day_of_month']=df['time'].dt.day\n    #df['week']=df['time'].dt.isocalendar().week     \n    #df['week'][df['week']>52]=52                    \n    #df['week']=df['week'].astype('int')\n    df['month']=df['time'].dt.month\n    #df['quarter'] = df['time'].dt.quarter\n    #df['year']=df['time'].dt.year\n    #df['day_of_year'] = df['time'].dt.day_of_year\n    df['is_month_start'] = df['time'].dt.is_month_start.astype('int')\n    df['is_month_end'] = df['time'].dt.is_month_end.astype('int')\n    df['is_weekend']=(df['weekday']//5 == 1).astype('int') \n    df['hour+minute'] = df['time'].dt.hour * 60 + df['time'].dt.minute\n    df['is_afternoon'] = (df['time'].dt.hour > 12).astype('int')\n    df['x+y'] = df['x'].astype('str') + df['y'].astype('str')\n    df['x+y+direction'] = df['x'].astype('str') + df['y'].astype('str') + df['direction'].astype('str')\n    #df['x+y+direction0'] = df['x'].astype('str') + df['y'].astype('str') + df['direction_coord_0'].astype('str')\n    #df['x+y+direction1'] = df['x'].astype('str') + df['y'].astype('str') + df['direction_coord_1'].astype('str')\n    df['hour+direction'] = df['hour'].astype('str') + df['direction'].astype('str')\n    df['hour+x+y'] = df['hour'].astype('str') + df['x'].astype('str') + df['y'].astype('str')\n    df['hour+direction+x'] = df['hour'].astype('str') + df['direction'].astype('str') + df['x'].astype('str')\n    df['hour+direction+y'] = df['hour'].astype('str') + df['direction'].astype('str') + df['y'].astype('str')\n    df['hour+direction+x+y'] = df['hour'].astype('str') + df['direction'].astype('str') + df['x'].astype('str') + df['y'].astype('str')\n    df['hour+x'] = df['hour'].astype('str') + df['x'].astype('str')\n    df['hour+y'] = df['hour'].astype('str') + df['y'].astype('str')\n    return df\n\ndef preprocess_holidays(df):\n    holidays_usa = holidays.CountryHoliday(country='US', years=[1991])\n    dates = list(holidays_usa.keys())\n    dates = sorted(pd.to_datetime(dates))\n    df = df.copy()\n    df['is_holiday'] = df['time'].apply(lambda x : 1 if x in dates else 0)\n    return df\n\ndef preprocess_timeseries(df):\n    df = df.copy()\n    # Sin of date values\n    df['sin_minute'] = np.sin(df['minute'])\n    df['sin_hour'] = np.sin(df['hour'])\n    df['sin_weekday'] = np.sin(df['weekday'])\n    df['sin_day_of_month'] = np.sin(df['day_of_month'])\n    # Cos of date values\n    df['cos_minute'] = np.cos(df['minute'])\n    df['cos_hour'] = np.cos(df['hour'])\n    df['cos_weekday'] = np.cos(df['weekday'])\n    df['cos_day_of_month'] = np.cos(df['day_of_month'])\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-03-17T03:01:01.001768Z","iopub.execute_input":"2022-03-17T03:01:01.002146Z","iopub.status.idle":"2022-03-17T03:01:01.027493Z","shell.execute_reply.started":"2022-03-17T03:01:01.0021Z","shell.execute_reply":"2022-03-17T03:01:01.026809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = label_encoder(train_df)\ntrain_df = preprocess_dates(train_df)\ntrain_df = preprocess_holidays(train_df)\ntrain_df = preprocess_timeseries(train_df)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2022-03-17T03:01:01.028794Z","iopub.execute_input":"2022-03-17T03:01:01.029477Z","iopub.status.idle":"2022-03-17T03:01:34.91099Z","shell.execute_reply.started":"2022-03-17T03:01:01.029441Z","shell.execute_reply":"2022-03-17T03:01:34.910085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train data split\ntarget_name = 'congestion'\n\nX_train = train_df.drop(['row_id', 'time', target_name], axis=1)\ny_train = train_df[target_name]\nX_train, X_test, y_train, y_test = train_test_split(X_train, y_train, test_size=0.2, random_state=seed, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-17T03:01:34.912613Z","iopub.execute_input":"2022-03-17T03:01:34.913077Z","iopub.status.idle":"2022-03-17T03:01:35.371865Z","shell.execute_reply.started":"2022-03-17T03:01:34.913033Z","shell.execute_reply":"2022-03-17T03:01:35.371232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model: CatBoost Regressor\nhttps://catboost.ai/en/docs/concepts/python-quickstart","metadata":{}},{"cell_type":"code","source":"params = {'iterations': 200, \n          'depth': 8, \n          'learning_rate': 0.1, \n          'loss_function': 'RMSE',  # 'RMSE', 'MAE', 'Poisson'\n         }\n\nmodel =  CatBoostRegressor(**params,\n                           random_seed=seed,\n                           #n_jobs = -1,\n                           verbose=0)\n\nmodel.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-03-17T03:01:35.374272Z","iopub.execute_input":"2022-03-17T03:01:35.374668Z","iopub.status.idle":"2022-03-17T03:01:54.694899Z","shell.execute_reply.started":"2022-03-17T03:01:35.374624Z","shell.execute_reply":"2022-03-17T03:01:54.694052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Evaluate model\nscore = evaluate_model(model, X_test, y_test)\nprint(score)\n#6.919078746957684","metadata":{"execution":{"iopub.status.busy":"2022-03-17T03:01:54.696405Z","iopub.execute_input":"2022-03-17T03:01:54.696928Z","iopub.status.idle":"2022-03-17T03:01:55.105926Z","shell.execute_reply.started":"2022-03-17T03:01:54.696871Z","shell.execute_reply":"2022-03-17T03:01:55.105043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Optuna Optimization","metadata":{}},{"cell_type":"code","source":"import optuna","metadata":{"execution":{"iopub.status.busy":"2022-03-17T03:01:55.107337Z","iopub.execute_input":"2022-03-17T03:01:55.107739Z","iopub.status.idle":"2022-03-17T03:01:55.781899Z","shell.execute_reply.started":"2022-03-17T03:01:55.107693Z","shell.execute_reply":"2022-03-17T03:01:55.7812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def objective(trial):\n   \n    iters = trial.suggest_int('iterations', 100, 800)\n    depth = trial.suggest_int('depth', 7, 12)\n    lr = trial.suggest_float('learning_rate', 0.05, 0.3, step = 0.01)\n    #loss = trial.suggest_categorical('loss_function', ['RMSE', 'MAE', 'Poisson'])    \n    \n    params = {'iterations': iters, \n              'depth': depth, \n              'learning_rate': lr, \n              'loss_function': 'MAE',  \n             }\n    \n    model =  CatBoostRegressor(**params,\n                               random_seed=seed,\n                               #n_jobs = -1,\n                               verbose=0)\n    \n    model.fit(X_train, y_train)\n    score = evaluate_model(model, X_test, y_test)\n    \n    return score","metadata":{"execution":{"iopub.status.busy":"2022-03-17T03:01:55.782942Z","iopub.execute_input":"2022-03-17T03:01:55.783293Z","iopub.status.idle":"2022-03-17T03:01:55.790547Z","shell.execute_reply.started":"2022-03-17T03:01:55.783263Z","shell.execute_reply":"2022-03-17T03:01:55.789785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# treat all python warnings as lower-level \"ignore\" events\nwarnings.filterwarnings(\"ignore\")\n\n# Create Optuna Trial\nstudy = optuna.create_study(direction=\"minimize\", sampler=optuna.samplers.RandomSampler(seed=seed))\n\n# Run trials\nstudy.optimize(objective , n_trials = 125)\n#study.optimize(objective, timeout = int(3600*9))    # an hour * X","metadata":{"execution":{"iopub.status.busy":"2022-03-17T03:01:55.79157Z","iopub.execute_input":"2022-03-17T03:01:55.79181Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# See optimization history\nfig = optuna.visualization.plot_optimization_history(study)\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#See hyper-parameters importances\nfig = optuna.visualization.plot_param_importances(study)\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#See slice\nfig = optuna.visualization.plot_slice(study)\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Best trial\nprint('Best trial score:', study.best_trial.value)\nstudy.best_trial.params","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create model with best trial parameters\nparams = {'iterations': study.best_trial.params['iterations'], \n          'depth': study.best_trial.params['depth'], \n          'learning_rate': study.best_trial.params['learning_rate'],\n          'loss_function': 'MAE',\n         }\n\nbest_model =  CatBoostRegressor(**params,\n                                random_seed=seed,\n                                #n_jobs = -1,\n                                verbose=0)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"# Train best model with all train data\nX_train = train_df.drop(['row_id', 'time', target_name], axis=1)\ny_train = train_df[target_name]\n\nbest_model.fit(X_train, y_train, verbose=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"real_test_df = label_encoder(real_test_df)\nreal_test_df = preprocess_dates(real_test_df)\nreal_test_df = preprocess_holidays(real_test_df)\nreal_test_df = preprocess_timeseries(real_test_df)\nX_real_test = real_test_df.drop(['row_id', 'time'], axis=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction = best_model.predict(X_real_test).squeeze()\nrow_id =  real_test_df['row_id'].values\nsubmission = pd.DataFrame({'row_id' : row_id, target_name : prediction})\nsubmission.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Since target values are integer, and model output is float, let's round the predicted values\nsubmission[target_name] = submission[target_name].round().astype(int)\nsubmission.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}