{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Simple LightGBM model\n\n**Private LB rank 500 (1st update)**","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nfrom pathlib import Path\n\nfrom datetime import datetime, date, time\n\nimport gc\nimport copy\n\nimport pyarrow.parquet as pq\nimport pyarrow as pa","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-22T07:42:32.024611Z","iopub.execute_input":"2022-07-22T07:42:32.024992Z","iopub.status.idle":"2022-07-22T07:42:32.036726Z","shell.execute_reply.started":"2022-07-22T07:42:32.024932Z","shell.execute_reply":"2022-07-22T07:42:32.035921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.options.display.max_rows = 100\npd.options.display.max_columns = 100\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport pytorch_lightning as pl\nrandom_seed=8968\npl.seed_everything(random_seed)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:42:32.040584Z","iopub.execute_input":"2022-07-22T07:42:32.040966Z","iopub.status.idle":"2022-07-22T07:42:34.744593Z","shell.execute_reply.started":"2022-07-22T07:42:32.040934Z","shell.execute_reply":"2022-07-22T07:42:34.743287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ndf=pd.read_parquet(r'/kaggle/input/train-parquet/train_lowmem.parquet', engine='pyarrow')","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:42:34.746340Z","iopub.execute_input":"2022-07-22T07:42:34.747099Z","iopub.status.idle":"2022-07-22T07:43:01.416484Z","shell.execute_reply.started":"2022-07-22T07:42:34.747055Z","shell.execute_reply":"2022-07-22T07:43:01.415661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.sort_values(by=['time_id', 'investment_id'], ascending=[True, True], inplace=True)\ndf.set_index(keys=['row_id'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:43:01.418893Z","iopub.execute_input":"2022-07-22T07:43:01.421306Z","iopub.status.idle":"2022-07-22T07:43:12.115870Z","shell.execute_reply.started":"2022-07-22T07:43:01.421257Z","shell.execute_reply":"2022-07-22T07:43:12.115006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"time_ids=df['time_id'].unique().tolist()\ntime_ids.sort()\nprint(len(time_ids))\nprint(time_ids[:2], time_ids[-2:])","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:43:12.117428Z","iopub.execute_input":"2022-07-22T07:43:12.118062Z","iopub.status.idle":"2022-07-22T07:43:12.142574Z","shell.execute_reply.started":"2022-07-22T07:43:12.118001Z","shell.execute_reply":"2022-07-22T07:43:12.141098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_cols=['target']\nexcl_cols=['time_id',  'target']\nx_cols=list(set(df.columns.tolist())-set(excl_cols))\nx_cols.sort()\nprint(len(x_cols))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:43:12.144198Z","iopub.execute_input":"2022-07-22T07:43:12.144477Z","iopub.status.idle":"2022-07-22T07:43:12.152008Z","shell.execute_reply.started":"2022-07-22T07:43:12.144446Z","shell.execute_reply":"2022-07-22T07:43:12.150853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_test_list=[]\n\n\nX_train=df.iloc[-1650000:, ][x_cols].copy(deep=True)\ny_train=df.iloc[-1650000:, ][y_cols].copy(deep=True)\nX_test=df.iloc[-30:, ][x_cols].copy(deep=True)\ny_test=df.iloc[-30:, ][y_cols].copy(deep=True)\n\n\nprint(X_train.shape, y_train.shape, X_test.shape, y_test.shape)\n\ntrain_test_list.append([X_train, y_train, X_test, y_test])","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:43:12.153563Z","iopub.execute_input":"2022-07-22T07:43:12.154058Z","iopub.status.idle":"2022-07-22T07:43:14.987802Z","shell.execute_reply.started":"2022-07-22T07:43:12.154000Z","shell.execute_reply":"2022-07-22T07:43:14.986077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:43:14.989150Z","iopub.execute_input":"2022-07-22T07:43:14.989398Z","iopub.status.idle":"2022-07-22T07:43:15.203509Z","shell.execute_reply.started":"2022-07-22T07:43:14.989366Z","shell.execute_reply":"2022-07-22T07:43:15.202527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### LightGBM","metadata":{}},{"cell_type":"code","source":"import lightgbm as lgb\ndef train_trees(X_train, y_train, num_round=100, params={}, verbosity=-1, cat_feats=[] ):\n    \n    dtrain = lgb.Dataset(X_train, y_train, categorical_feature=cat_feats)\n    \n    params['verbosity'] = verbosity\n    \n    tree_model = lgb.train(params,\n                dtrain,\n                num_boost_round=num_round)\n    \n    del dtrain\n    gc.collect()\n\n    return tree_model","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:43:15.205001Z","iopub.execute_input":"2022-07-22T07:43:15.205596Z","iopub.status.idle":"2022-07-22T07:43:16.153053Z","shell.execute_reply.started":"2022-07-22T07:43:15.205549Z","shell.execute_reply":"2022-07-22T07:43:16.152088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params1={'boosting_type': 'gbdt', 'colsample_bytree': 0.3, 'learning_rate': 0.003, 'max_bin': 5000, 'max_depth': 7, 'metric': 'regression_l2', 'min_child_samples': 80, 'min_data_in_bin': 800, 'n_estimators': 2500, 'num_leaves': 255, 'objective': 'regression_l2', 'random_state': 5566, 'reg_alpha': 1, 'reg_lambda': 1, 'subsample': 0.3, 'subsample_freq': 8}\n\n\nparams2={'boosting_type': 'gbdt', 'colsample_bytree': 0.5, 'learning_rate': 0.003, 'max_bin': 2500, 'max_depth': 31, 'metric': 'regression_l2', 'min_child_samples': 100, 'min_data_in_bin': 400, 'n_estimators': 1500, 'num_leaves': 255, 'objective': 'regression_l2', 'random_state': 5566, 'reg_alpha': 1, 'reg_lambda': 3, 'subsample': 0.85, 'subsample_freq': 4}\n\n\nparams3={'boosting_type': 'gbdt', 'colsample_bytree': 0.35, 'learning_rate': 0.003, 'max_bin': 1500, 'max_depth': 13, 'metric': 'regression_l2', 'min_child_samples': 80, 'min_data_in_bin': 1000, 'n_estimators': 1500, 'num_leaves': 63, 'objective': 'regression_l2', 'random_state': 5566, 'reg_alpha': 0.05, 'reg_lambda': 0.1, 'subsample': 0.7, 'subsample_freq': 8}\n\nparams4={'boosting_type': 'gbdt', 'colsample_bytree': 0.6, 'learning_rate': 0.005, 'max_bin': 5000, 'max_depth': 11, 'metric': 'regression_l2', 'min_child_samples': 80, 'min_data_in_bin': 400, 'n_estimators': 1500, 'num_leaves': 255, 'objective': 'regression_l2', 'random_state': 5566, 'reg_alpha': 10, 'reg_lambda': 0.1, 'subsample': 0.35, 'subsample_freq': 13}\n\nparams5={'boosting_type': 'gbdt', 'colsample_bytree': 0.55, 'learning_rate': 0.005, 'max_bin': 2500, 'max_depth': 25, 'metric': 'regression_l2', 'min_child_samples': 350, 'min_data_in_bin': 300, 'n_estimators': 1250, 'num_leaves': 255, 'objective': 'regression_l2', 'random_state': 5566, 'reg_alpha': 3, 'reg_lambda': 0.5, 'subsample': 0.7, 'subsample_freq': 7}\n\n\n\n\nparam_list=[params1, params2 , params3 , params4 , params5 ]","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:43:16.154424Z","iopub.execute_input":"2022-07-22T07:43:16.154774Z","iopub.status.idle":"2022-07-22T07:43:16.168916Z","shell.execute_reply.started":"2022-07-22T07:43:16.154728Z","shell.execute_reply":"2022-07-22T07:43:16.167864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tree_models=[]\nfor params in param_list:\n\n    num_boost_round = params['n_estimators']\n    params_ = copy.deepcopy(params)\n    del params_['n_estimators']\n    \n    for X_train, y_train, _, _ in train_test_list:\n        \n        \n        tree_model = train_trees(X_train, y_train['target'].values,\n                                num_round=num_boost_round, \n                                params=params_, verbosity=-1, \n                                cat_feats=['investment_id'])\n        \n  \n        tree_models.append(tree_model)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:43:16.170543Z","iopub.execute_input":"2022-07-22T07:43:16.170816Z","iopub.status.idle":"2022-07-22T09:58:54.153884Z","shell.execute_reply.started":"2022-07-22T07:43:16.170780Z","shell.execute_reply":"2022-07-22T09:58:54.150210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submit","metadata":{}},{"cell_type":"code","source":"import ubiquant\nenv = ubiquant.make_env()   # initialize the environment\niter_test = env.iter_test()    # an iterator which loops over the test set and sample submission\nfor (test_df, sample_prediction_df) in iter_test:\n    pred_list=[]\n    for tree_model in tree_models:\n        y_preds = tree_model.predict(test_df[x_cols], num_iteration=tree_model.best_iteration)\n        pred_list.append(y_preds)\n    \n    \n    sample_prediction_df['target'] = np.mean(pred_list, axis=0)   # make your predictions here\n    env.predict(sample_prediction_df)   # register your predictions\n    display(sample_prediction_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:58:54.159495Z","iopub.execute_input":"2022-07-22T09:58:54.160077Z","iopub.status.idle":"2022-07-22T09:58:54.581555Z","shell.execute_reply.started":"2022-07-22T09:58:54.159977Z","shell.execute_reply":"2022-07-22T09:58:54.577827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}