{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**hello,this is my notebook.and I want to share my idea :use lightgbm with bayes-opt to optimize prediction.However,I wonder is this method feasible?\nAnd I will continue my study based on https://www.kaggle.com/code/jirkaborovec/icecube-neutrino-baseline-xgboost-regression Thank you all for your**","metadata":{}},{"cell_type":"code","source":"%matplotlib inline\n\nimport os\nimport glob\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nPATH_DATASET = \"/kaggle/input/icecube-neutrinos-in-deep-ice\"\nmeta_test = pd.read_parquet(os.path.join(PATH_DATASET, \"test_meta.parquet\"))\ndel meta_test","metadata":{"execution":{"iopub.status.busy":"2023-01-30T03:58:45.669869Z","iopub.execute_input":"2023-01-30T03:58:45.670275Z","iopub.status.idle":"2023-01-30T03:58:45.802074Z","shell.execute_reply.started":"2023-01-30T03:58:45.670242Z","shell.execute_reply":"2023-01-30T03:58:45.801106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_train = pd.read_parquet(os.path.join(PATH_DATASET, \"train_meta.parquet\"))\nmeta_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-30T03:58:47.303052Z","iopub.execute_input":"2023-01-30T03:58:47.303518Z","iopub.status.idle":"2023-01-30T03:59:35.545913Z","shell.execute_reply.started":"2023-01-30T03:58:47.303472Z","shell.execute_reply":"2023-01-30T03:59:35.544710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm.auto import tqdm\ndef transform_batch(path_batch_parquet, nb_sensors=5160):\n    df = pd.read_parquet(path_batch_parquet)\n    data = np.zeros((len(df.index.unique()), nb_sensors), dtype=np.float16)\n    event_ids = []\n    for i, (idx, dfg) in tqdm(enumerate(df.groupby(level=0))):\n        event_ids.append(idx)\n        data[i, dfg['sensor_id']] = dfg['charge']\n    return event_ids, data","metadata":{"execution":{"iopub.status.busy":"2023-01-30T03:59:35.548041Z","iopub.execute_input":"2023-01-30T03:59:35.549126Z","iopub.status.idle":"2023-01-30T03:59:35.670995Z","shell.execute_reply.started":"2023-01-30T03:59:35.549076Z","shell.execute_reply":"2023-01-30T03:59:35.669710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"event_ids, data = transform_batch(os.path.join(PATH_DATASET, \"train/batch_10.parquet\"))\nprint(f\"data size: {data.shape}\")\nmeta_train_ = meta_train[meta_train['event_id'].isin(event_ids)]\nmeta_train_ = dict(zip(meta_train_['event_id'].values, meta_train_[[\"azimuth\", \"zenith\"]].values.tolist()))\nprint(f\"LUT size: {len(meta_train_)}\")\nangles = np.array([meta_train_[eid] for eid in event_ids], dtype=np.float16)\nprint(f\"angles size: {angles.shape}\")","metadata":{"execution":{"iopub.status.busy":"2023-01-30T03:59:35.672266Z","iopub.execute_input":"2023-01-30T03:59:35.672657Z","iopub.status.idle":"2023-01-30T04:00:19.843425Z","shell.execute_reply.started":"2023-01-30T03:59:35.672624Z","shell.execute_reply":"2023-01-30T04:00:19.842052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.model_selection import train_test_split\n# X_train, X_test, y_train, y_test = train_test_split(data, angles, train_size=0.8)\n# del data, angles","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**In terms of data pre-processing, I also perform dimensionality reduction based on the PCA method**","metadata":{}},{"cell_type":"code","source":"from sklearn.decomposition import PCA\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import StandardScaler\nfrom xgboost import XGBRegressor\npreprocess = Pipeline([\n    ('scaler', StandardScaler()),\n     (\"PCA\", PCA(\n        n_components=510,\n        copy=False,\n    ))])\n# X_train = preprocess.fit_transform(X_train)\n# X_test = preprocess.transform(X_test)\ndata=preprocess.fit_transform(data)","metadata":{"execution":{"iopub.status.busy":"2023-01-30T04:00:19.845784Z","iopub.execute_input":"2023-01-30T04:00:19.846165Z","iopub.status.idle":"2023-01-30T04:07:38.918810Z","shell.execute_reply.started":"2023-01-30T04:00:19.846130Z","shell.execute_reply":"2023-01-30T04:07:38.916400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**And now you can see the data shape and angles**","metadata":{}},{"cell_type":"code","source":"data","metadata":{"execution":{"iopub.status.busy":"2023-01-30T04:07:38.926388Z","iopub.execute_input":"2023-01-30T04:07:38.926939Z","iopub.status.idle":"2023-01-30T04:07:38.975483Z","shell.execute_reply.started":"2023-01-30T04:07:38.926892Z","shell.execute_reply":"2023-01-30T04:07:38.974028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"angles","metadata":{"execution":{"iopub.status.busy":"2023-01-30T04:07:38.979499Z","iopub.execute_input":"2023-01-30T04:07:38.979929Z","iopub.status.idle":"2023-01-30T04:07:38.990028Z","shell.execute_reply.started":"2023-01-30T04:07:38.979892Z","shell.execute_reply":"2023-01-30T04:07:38.988474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**With the purpose of optimizing hyperparameters, I continue to study the Bayesian optimization method, which can be found in my notebook if you are interested https://www.kaggle.com/code/oldjerry/happy-new-year-9-kinds-of-models-ensemble**","metadata":{}},{"cell_type":"code","source":"#!pip install bayesian-optimization","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As for modeling,I use lightgbm+linearRegession,for lightgbm I use bayes_opt for hyperparameters \n\n**include:(n_estimators,max_depth,num_leaves\n         ,subsample,colsample_bytree,reg_lambda,reg_alpha,learning_rate)**\n         \nalso,I use 5*5-kfold method to make sure that the model will not be overfitted.However it may spend a long time \nwhat's more,I use data to predict angles[:,0],and data to predict angles[:,1]\nand I hope the best model would be as follows:\n\n$ best model= ensemble(a*rmse(lgb)+(1-a)*rmse(lgb1),a*rmse(lr)+(1-a)*rmse(lr1))$\n","metadata":{}},{"cell_type":"code","source":"from lightgbm import LGBMRegressor\nfrom sklearn.multioutput import MultiOutputRegressor\nfrom sklearn.multioutput import RegressorChain\nfrom sklearn.model_selection import KFold\n# from bayes_opt import BayesianOptimization\nfrom sklearn.metrics import mean_squared_error\n# def lgb_evaluate(n_estimators,max_depth,num_leaves\n#                 ,subsample,colsample_bytree,\n#                 reg_lambda,reg_alpha,learning_rate):\n#         rmse_scores=[]\n#         cv_scores=[]\n#         model=RegressorChain(LGBMRegressor(\n#                            n_estimators=int(n_estimators),\n#                            max_depth=int(max_depth),\n#                            num_leaves=int(num_leaves),\n#                            subsample=subsample,\n#                            colsample_bytree=colsample_bytree,\n#                            reg_alpha=reg_alpha,\n#                            reg_lambda=reg_lambda,\n#                            learning_rate=learning_rate))\n#         for i in tqdm(range(5)):\n#             kf = KFold(n_splits = 5, shuffle=True)\n#             for train_ix, test_ix in tqdm(kf.split(data)): \n#                 X_train, X_test = data[train_ix], data[test_ix]\n#                 Y_train, Y_test = angles[train_ix], angles[test_ix]\n#                 model.fit(X_train,Y_train)\n#                 lgb_pred = model.predict(X_test)\n#                 rmse_scores.append(-mean_squared_error(Y_test,lgb_pred)**0.5)\n#             cv_scores.append(np.mean(rmse_scores))\n#         lgb_score=np.mean(cv_scores)\n#         return lgb_score\n# pbounds = {'max_depth': (3,5),\n#            \"num_leaves\":(5,31),\n#            'n_estimators': (5, 200),\n#            \"subsample\":(0.8, 1),\n#            \"colsample_bytree\":(0.8,1),\n#            \"learning_rate\":(0.0001,0.5),\n#            \"reg_lambda\":(0.001,1000),\n#            \"reg_alpha\":(0.001,1000)}\n# lgb_bo = BayesianOptimization(\n#         f=lgb_evaluate,   \n#         pbounds=pbounds,  \n#         verbose=2,  \n#         random_state=1,\n# )\n# lgb_bo.maximize(init_points=1,   \n#                    n_iter=2)\n# print(lgb_bo.max)\n# res_lgb = lgb_bo.max\n# params_lgb = res_lgb['params']\n# for key in params_lgb:\n#     if key in [\"max_depth\",\"num_leaves\",\"n_estimators\"]:\n#         params_lgb[key]=int(params_lgb[key])\n# lgb=LGBMRegressor(**params_lgb)","metadata":{"execution":{"iopub.status.busy":"2023-01-30T04:07:38.992666Z","iopub.execute_input":"2023-01-30T04:07:38.993275Z","iopub.status.idle":"2023-01-30T04:52:10.445830Z","shell.execute_reply.started":"2023-01-30T04:07:38.993220Z","shell.execute_reply":"2023-01-30T04:52:10.443670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from lightgbm import LGBMRegressor\nparams_lgb={'colsample_bytree':0.8359697722366822, \n            'learning_rate': 0.37022800614783546, \n            'max_depth': 3, \n            'n_estimators': 66,\n            'num_leaves': 29, \n            'reg_alpha': 912.8321623783283,\n            'reg_lambda':370.64843969655743 , 'subsample':0.8239338654645402}\nlgb=RegressorChain(LGBMRegressor(**params_lgb))","metadata":{"execution":{"iopub.status.busy":"2023-01-30T05:07:19.336581Z","iopub.execute_input":"2023-01-30T05:07:19.337072Z","iopub.status.idle":"2023-01-30T05:07:19.345267Z","shell.execute_reply.started":"2023-01-30T05:07:19.337030Z","shell.execute_reply":"2023-01-30T05:07:19.343608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb.fit(\n    data,angles\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-30T05:07:21.851036Z","iopub.execute_input":"2023-01-30T05:07:21.851697Z","iopub.status.idle":"2023-01-30T05:08:00.441578Z","shell.execute_reply.started":"2023-01-30T05:07:21.851640Z","shell.execute_reply":"2023-01-30T05:08:00.440310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.multioutput import MultiOutputRegressor\nfrom xgboost import XGBRegressor\nxgb = MultiOutputRegressor(XGBRegressor())\nxgb.fit(data,angles)","metadata":{"execution":{"iopub.status.busy":"2023-01-30T05:09:25.077263Z","iopub.execute_input":"2023-01-30T05:09:25.078787Z","iopub.status.idle":"2023-01-30T05:47:35.833888Z","shell.execute_reply.started":"2023-01-30T05:09:25.078729Z","shell.execute_reply":"2023-01-30T05:47:35.832894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression \nlr=LinearRegression()\nlr.fit(data,angles)","metadata":{"execution":{"iopub.status.busy":"2023-01-30T05:47:35.942847Z","iopub.execute_input":"2023-01-30T05:47:35.943304Z","iopub.status.idle":"2023-01-30T05:47:46.369787Z","shell.execute_reply.started":"2023-01-30T05:47:35.943264Z","shell.execute_reply":"2023-01-30T05:47:46.368092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**And here is the ensemble model I build,just combine with two kinds of models,because of the requirements for high computing power**","metadata":{}},{"cell_type":"code","source":"models=[(\"lgb\",lgb),\n        (\"lr\",lr),\n        (\"xgb\",xgb)\n         ]","metadata":{"execution":{"iopub.status.busy":"2023-01-30T06:12:26.262085Z","iopub.execute_input":"2023-01-30T06:12:26.262635Z","iopub.status.idle":"2023-01-30T06:12:26.269882Z","shell.execute_reply.started":"2023-01-30T06:12:26.262594Z","shell.execute_reply":"2023-01-30T06:12:26.268324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**And here is the same for prediction,just be careful for the ensemble models'predict,just a little bit different**","metadata":{}},{"cell_type":"code","source":"import gc, time\ngc.collect()\ntime.sleep(9)\nssub = pd.read_parquet(os.path.join(PATH_DATASET, \"sample_submission.parquet\"))\nssub.set_index(\"event_id\", inplace=True)\nssub = ssub.apply(np.float16)","metadata":{"execution":{"iopub.status.busy":"2023-01-30T06:12:28.941436Z","iopub.execute_input":"2023-01-30T06:12:28.942817Z","iopub.status.idle":"2023-01-30T06:12:38.346808Z","shell.execute_reply.started":"2023-01-30T06:12:28.942751Z","shell.execute_reply":"2023-01-30T06:12:38.345749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ssub['zenith'] = meta_train['zenith'].mean()","metadata":{"execution":{"iopub.status.busy":"2023-01-30T06:12:38.348991Z","iopub.execute_input":"2023-01-30T06:12:38.349364Z","iopub.status.idle":"2023-01-30T06:12:39.696898Z","shell.execute_reply.started":"2023-01-30T06:12:38.349331Z","shell.execute_reply":"2023-01-30T06:12:39.695570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ls = glob.glob(os.path.join(PATH_DATASET, \"test\", \"*.parquet\"))\n\nfor batch_file in ls:\n    print(f\"processing: {batch_file}\")\n    event_ids, data = transform_batch(batch_file)\n    preds=0\n    for model in models:\n        pred=np.array(list(model[1].predict(preprocess.transform(data))))\n        preds+=pred\n        del pred\n    preds=preds/3\n    del data\n    for eid, (a, z) in zip(event_ids, preds):\n        ssub.at[eid, \"azimuth\"] = a\n        ssub.at[eid, \"zenith\"] = z\n    del event_ids, preds\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-30T06:12:39.698330Z","iopub.execute_input":"2023-01-30T06:12:39.698740Z","iopub.status.idle":"2023-01-30T06:12:49.304976Z","shell.execute_reply.started":"2023-01-30T06:12:39.698705Z","shell.execute_reply":"2023-01-30T06:12:49.303217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ssub.to_csv('submission.csv', index=True)\n!head submission.csv","metadata":{"execution":{"iopub.status.busy":"2023-01-30T06:12:49.307690Z","iopub.execute_input":"2023-01-30T06:12:49.308099Z","iopub.status.idle":"2023-01-30T06:12:50.612993Z","shell.execute_reply.started":"2023-01-30T06:12:49.308064Z","shell.execute_reply":"2023-01-30T06:12:50.611317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Last but not the least,hope my method could help you,and I also want to know if it is feasible to predict neutrino direction by machine learning algorithm only? Feel free to ask me questions, I appreciate it!**","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}