{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Kernel description","metadata":{}},{"cell_type":"markdown","source":"This is just a trial submission that uses a basic LinearRegression model and the simplest features.","metadata":{}},{"cell_type":"markdown","source":"## Import libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\n\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.model_selection import train_test_split","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-09T09:19:29.160048Z","iopub.execute_input":"2023-04-09T09:19:29.160837Z","iopub.status.idle":"2023-04-09T09:19:30.452649Z","shell.execute_reply.started":"2023-04-09T09:19:29.160797Z","shell.execute_reply":"2023-04-09T09:19:30.451562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading data","metadata":{}},{"cell_type":"code","source":"INPUT_DIR = '/kaggle/input/icecube-neutrinos-in-deep-ice'\n\nsample_meta = pd.read_parquet(f'{INPUT_DIR}/sample_submission.parquet')\nsensor_geometry = pd.read_csv(f'{INPUT_DIR}/sensor_geometry.csv')\ntrain_meta = pd.read_parquet(f'{INPUT_DIR}/train_meta.parquet')\ntest_meta = pd.read_parquet(f'{INPUT_DIR}/test_meta.parquet')\n\n\ntrain_data = pd.concat([pd.read_parquet(f'{INPUT_DIR}/train/batch_{i}.parquet') for i in tqdm(train_meta['batch_id'].unique()[:1])])\ntest_data = pd.concat([pd.read_parquet(f'{INPUT_DIR}/test/batch_{i}.parquet') for i in tqdm(test_meta['batch_id'].unique()[:1])])","metadata":{"execution":{"iopub.status.busy":"2023-04-09T09:19:40.534193Z","iopub.execute_input":"2023-04-09T09:19:40.534692Z","iopub.status.idle":"2023-04-09T09:20:19.112465Z","shell.execute_reply.started":"2023-04-09T09:19:40.534646Z","shell.execute_reply":"2023-04-09T09:20:19.110951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_meta.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T09:20:38.456240Z","iopub.execute_input":"2023-04-09T09:20:38.457258Z","iopub.status.idle":"2023-04-09T09:20:38.482162Z","shell.execute_reply.started":"2023-04-09T09:20:38.457198Z","shell.execute_reply":"2023-04-09T09:20:38.480268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Simple feature engineering","metadata":{}},{"cell_type":"code","source":"train_data = train_data.merge(sensor_geometry, left_on='sensor_id', right_index=True)\n\ntrain_features = train_data.groupby('event_id').agg({'charge': ['mean', 'std', 'sum'], \n                                                     'time': ['min', 'max']})\n\ntrain_features.columns = ['_'.join(col) for col in train_features.columns]\n\ntrain_features = train_features.merge(train_meta, left_index=True, right_on='event_id')","metadata":{"execution":{"iopub.status.busy":"2023-04-09T09:20:46.132670Z","iopub.execute_input":"2023-04-09T09:20:46.133108Z","iopub.status.idle":"2023-04-09T09:22:06.301859Z","shell.execute_reply.started":"2023-04-09T09:20:46.133073Z","shell.execute_reply":"2023-04-09T09:22:06.299543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = test_data.merge(sensor_geometry, left_on='sensor_id', right_index=True)\n\ntest_features = test_data.groupby('event_id').agg({'charge': ['mean', 'std', 'sum'], \n                                                   'time': ['min', 'max']})\n\ntest_features.columns = [\"_\".join(col) for col in test_features.columns]\n\ntest_features = test_features.merge(test_meta, left_index=True, right_on='event_id')","metadata":{"execution":{"iopub.status.busy":"2023-04-09T09:23:09.473108Z","iopub.execute_input":"2023-04-09T09:23:09.473540Z","iopub.status.idle":"2023-04-09T09:23:09.492667Z","shell.execute_reply.started":"2023-04-09T09:23:09.473503Z","shell.execute_reply":"2023-04-09T09:23:09.490389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train test split","metadata":{"execution":{"iopub.status.busy":"2023-04-08T23:09:20.332950Z","iopub.execute_input":"2023-04-08T23:09:20.333364Z","iopub.status.idle":"2023-04-08T23:09:20.341943Z","shell.execute_reply.started":"2023-04-08T23:09:20.333328Z","shell.execute_reply":"2023-04-08T23:09:20.340024Z"}}},{"cell_type":"code","source":"train_X, val_X, train_y, val_y = train_test_split(train_features.drop(['azimuth', 'zenith'], axis=1),\n                                                  train_features[['azimuth', 'zenith']],\n                                                  test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-04-09T09:23:24.300311Z","iopub.execute_input":"2023-04-09T09:23:24.301890Z","iopub.status.idle":"2023-04-09T09:23:24.357626Z","shell.execute_reply.started":"2023-04-09T09:23:24.301813Z","shell.execute_reply":"2023-04-09T09:23:24.355906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## LinearRegression","metadata":{}},{"cell_type":"code","source":"clf = LinearRegression()\nclf.fit(train_X.values, train_y.values)\n\nval_pred = clf.predict(val_X.values)","metadata":{"execution":{"iopub.status.busy":"2023-04-09T09:23:26.830239Z","iopub.execute_input":"2023-04-09T09:23:26.830634Z","iopub.status.idle":"2023-04-09T09:23:27.037894Z","shell.execute_reply.started":"2023-04-09T09:23:26.830600Z","shell.execute_reply":"2023-04-09T09:23:27.036960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predictions saving","metadata":{}},{"cell_type":"code","source":"test_pred = clf.predict(test_features.values)\n\npd.DataFrame({\n    'event_id': test_features.event_id.values,\n    'azimuth': test_pred[:, 0],\n    'zenith': test_pred[:, 1]\n}).to_csv('submission.csv', index=False)\n\npd.read_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-04-09T09:25:21.722925Z","iopub.execute_input":"2023-04-09T09:25:21.723569Z","iopub.status.idle":"2023-04-09T09:25:21.744568Z","shell.execute_reply.started":"2023-04-09T09:25:21.723523Z","shell.execute_reply":"2023-04-09T09:25:21.743391Z"},"trusted":true},"execution_count":null,"outputs":[]}]}