{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Feature importance graph.","metadata":{}},{"cell_type":"code","source":"import os \nimport glob\n\nimport pandas as pd\nimport numpy as np\nimport pyarrow.parquet as pq\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\n\nimport time\n\nimport xgboost\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import f1_score\n\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pyarrow.parquet import ParquetFile\nimport pyarrow as pa \n\npf = ParquetFile('/kaggle/input/icecube-neutrinos-in-deep-ice/train_meta.parquet') \nfirst_ten_rows = next(pf.iter_batches(batch_size = 1000000)) \ndf = pa.Table.from_batches([first_ten_rows]).to_pandas() ","metadata":{"execution":{"iopub.status.busy":"2023-01-21T18:25:26.274033Z","iopub.execute_input":"2023-01-21T18:25:26.274557Z","iopub.status.idle":"2023-01-21T18:25:42.081042Z","shell.execute_reply.started":"2023-01-21T18:25:26.274507Z","shell.execute_reply":"2023-01-21T18:25:42.079943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-21T18:25:42.082334Z","iopub.execute_input":"2023-01-21T18:25:42.082698Z","iopub.status.idle":"2023-01-21T18:25:42.104481Z","shell.execute_reply.started":"2023-01-21T18:25:42.082665Z","shell.execute_reply":"2023-01-21T18:25:42.103400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sensor = pd.read_csv('/kaggle/input/icecube-neutrinos-in-deep-ice/sensor_geometry.csv')\nfig = px.scatter_3d(sensor, x='x', y='y', z='z', color='sensor_id', size='sensor_id',\n                   title=\"IceCube Observatory\")\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-21T18:25:42.106850Z","iopub.execute_input":"2023-01-21T18:25:42.107178Z","iopub.status.idle":"2023-01-21T18:25:43.249634Z","shell.execute_reply.started":"2023-01-21T18:25:42.107151Z","shell.execute_reply":"2023-01-21T18:25:43.248317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = df[['first_pulse_index','last_pulse_index']]\ny = df[['azimuth','zenith']]\n\nmodel = xgboost.XGBRegressor(n_estimators=100, max_depth=10)\n\nx_train, x_test, y_train, y_test = train_test_split(x,y, test_size=0.2)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-01-21T18:30:04.310567Z","iopub.execute_input":"2023-01-21T18:30:04.310984Z","iopub.status.idle":"2023-01-21T18:30:04.455486Z","shell.execute_reply.started":"2023-01-21T18:30:04.310951Z","shell.execute_reply":"2023-01-21T18:30:04.454521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(x_train, y_train)\npred = model.predict(x_test)","metadata":{"execution":{"iopub.status.busy":"2023-01-21T18:30:04.943999Z","iopub.execute_input":"2023-01-21T18:30:04.944441Z","iopub.status.idle":"2023-01-21T18:32:13.396148Z","shell.execute_reply.started":"2023-01-21T18:30:04.944404Z","shell.execute_reply":"2023-01-21T18:32:13.395099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pq.read_pandas('/kaggle/input/icecube-neutrinos-in-deep-ice/test_meta.parquet').to_pandas()","metadata":{"execution":{"iopub.status.busy":"2023-01-21T18:32:44.400585Z","iopub.execute_input":"2023-01-21T18:32:44.400986Z","iopub.status.idle":"2023-01-21T18:32:44.448771Z","shell.execute_reply.started":"2023-01-21T18:32:44.400955Z","shell.execute_reply":"2023-01-21T18:32:44.447916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2023-01-21T18:32:45.697981Z","iopub.execute_input":"2023-01-21T18:32:45.698435Z","iopub.status.idle":"2023-01-21T18:32:45.710128Z","shell.execute_reply.started":"2023-01-21T18:32:45.698396Z","shell.execute_reply":"2023-01-21T18:32:45.708889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## XGBRegressor generally classifies the order of importance of each feature used for the prediction. A benefit of using gradient boosting is that after the boosted trees are constructed, it is relatively straightforward to retrieve importance scores for each attribute","metadata":{}},{"cell_type":"code","source":"pred = model.predict(test[['first_pulse_index','last_pulse_index']])\nsub = pd.DataFrame({'event_id':test.index,'azimuth':pred[:,0],'zenith':pred[:,1]})\nsub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-21T18:32:46.154226Z","iopub.execute_input":"2023-01-21T18:32:46.154679Z","iopub.status.idle":"2023-01-21T18:32:46.170402Z","shell.execute_reply.started":"2023-01-21T18:32:46.154638Z","shell.execute_reply":"2023-01-21T18:32:46.169235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"XGBOOST\")","metadata":{"execution":{"iopub.status.busy":"2023-01-21T18:32:47.761894Z","iopub.execute_input":"2023-01-21T18:32:47.762953Z","iopub.status.idle":"2023-01-21T18:32:47.768362Z","shell.execute_reply.started":"2023-01-21T18:32:47.762913Z","shell.execute_reply":"2023-01-21T18:32:47.767435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"References: \n1. https://www.kaggle.com/code/faelk8/neutrinos-in-deep-ice-xgbregressor \n2. https://www.datatechnotes.com/2019/06/regression-example-with-xgbregressor-in.html","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}