{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pyarrow.parquet as pq","metadata":{"execution":{"iopub.status.busy":"2023-01-22T13:20:34.204927Z","iopub.execute_input":"2023-01-22T13:20:34.206027Z","iopub.status.idle":"2023-01-22T13:20:35.246678Z","shell.execute_reply.started":"2023-01-22T13:20:34.205909Z","shell.execute_reply":"2023-01-22T13:20:35.245516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read the meta data\ntrain_meta = pq.read_pandas('/kaggle/input/icecube-neutrinos-in-deep-ice/train_meta.parquet').to_pandas()\ndata = '/kaggle/input/icecube-neutrinos-in-deep-ice/train/'\n# Read the batch data in chunks\nbatch_files = [data+'batch_1.parquet', data+'batch_2.parquet',data+'batch_10.parquet']\nbatch_data = pd.concat([pq.read_pandas(file, columns=['event_id', 'time', 'sensor_id', 'charge', 'auxiliary']).to_pandas() for file in batch_files])","metadata":{"execution":{"iopub.status.busy":"2023-01-22T13:20:35.249050Z","iopub.execute_input":"2023-01-22T13:20:35.250223Z","iopub.status.idle":"2023-01-22T13:20:57.064961Z","shell.execute_reply.started":"2023-01-22T13:20:35.250173Z","shell.execute_reply":"2023-01-22T13:20:57.063644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# read sensor_geometry.csv using pandas\nsensor_geometry = pd.read_csv('/kaggle/input/icecube-neutrinos-in-deep-ice/sensor_geometry.csv')\nsub = pq.read_pandas('/kaggle/input/icecube-neutrinos-in-deep-ice/sample_submission.parquet').to_pandas()\ntest = pq.read_pandas('/kaggle/input/icecube-neutrinos-in-deep-ice/test/batch_661.parquet').to_pandas()\ntest_meta = pq.read_pandas('/kaggle/input/icecube-neutrinos-in-deep-ice/test_meta.parquet').to_pandas()\n# read in the test data\ntest_data = pd.read_parquet('/kaggle/input/icecube-neutrinos-in-deep-ice/test_meta.parquet')","metadata":{"execution":{"iopub.status.busy":"2023-01-22T13:20:57.066533Z","iopub.execute_input":"2023-01-22T13:20:57.066907Z","iopub.status.idle":"2023-01-22T13:20:57.113912Z","shell.execute_reply.started":"2023-01-22T13:20:57.066872Z","shell.execute_reply":"2023-01-22T13:20:57.112872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import mean_squared_error, r2_score","metadata":{"execution":{"iopub.status.busy":"2023-01-22T13:20:57.116872Z","iopub.execute_input":"2023-01-22T13:20:57.117571Z","iopub.status.idle":"2023-01-22T13:20:57.394660Z","shell.execute_reply.started":"2023-01-22T13:20:57.117531Z","shell.execute_reply":"2023-01-22T13:20:57.393129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# select a subset of the data to train the model on\nsubset = train_meta.sample(frac=0.02, random_state=42)\n\n# split the data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(subset.drop(['azimuth', 'zenith'], axis=1), subset[['azimuth', 'zenith']], test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-01-22T13:20:57.396681Z","iopub.execute_input":"2023-01-22T13:20:57.397168Z","iopub.status.idle":"2023-01-22T13:21:06.572756Z","shell.execute_reply.started":"2023-01-22T13:20:57.397100Z","shell.execute_reply":"2023-01-22T13:21:06.571084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import PolynomialFeatures\n\n# Define the polynomial features\npoly = PolynomialFeatures(degree=2)\nX_train_poly = poly.fit_transform(X_train)\nX_test_poly = poly.fit_transform(X_test)\n\n# Define the model\nmodel = LinearRegression()\n\n# Train the model\nmodel.fit(X_train_poly, y_train)\n\n# Evaluate the model\ny_pred = model.predict(X_test_poly)\n","metadata":{"execution":{"iopub.status.busy":"2023-01-22T13:21:07.098556Z","iopub.execute_input":"2023-01-22T13:21:07.099690Z","iopub.status.idle":"2023-01-22T13:21:09.182582Z","shell.execute_reply.started":"2023-01-22T13:21:07.099624Z","shell.execute_reply":"2023-01-22T13:21:09.180861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test_poly = poly.transform(test_data)\n\n# Make predictions on the test data\ny_pred = model.predict(X_test_poly)","metadata":{"execution":{"iopub.status.busy":"2023-01-22T13:24:03.050420Z","iopub.execute_input":"2023-01-22T13:24:03.051432Z","iopub.status.idle":"2023-01-22T13:24:03.060429Z","shell.execute_reply.started":"2023-01-22T13:24:03.051382Z","shell.execute_reply":"2023-01-22T13:24:03.059364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# create the submission dataframe\nsubmission = pd.DataFrame({'event_id': test_data.index, 'azimuth': y_pred[:,0], 'zenith': y_pred[:,1]})\n","metadata":{"execution":{"iopub.status.busy":"2023-01-22T13:25:38.316451Z","iopub.execute_input":"2023-01-22T13:25:38.316902Z","iopub.status.idle":"2023-01-22T13:25:38.323051Z","shell.execute_reply.started":"2023-01-22T13:25:38.316869Z","shell.execute_reply":"2023-01-22T13:25:38.321981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df = pd.read_parquet('/kaggle/input/icecube-neutrinos-in-deep-ice/sample_submission.parquet')","metadata":{"execution":{"iopub.status.busy":"2023-01-22T13:25:38.647266Z","iopub.execute_input":"2023-01-22T13:25:38.647943Z","iopub.status.idle":"2023-01-22T13:25:38.676879Z","shell.execute_reply.started":"2023-01-22T13:25:38.647896Z","shell.execute_reply":"2023-01-22T13:25:38.675673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df['azimuth']=submission['azimuth']\nsub_df['zenith']=submission['zenith']","metadata":{"execution":{"iopub.status.busy":"2023-01-22T13:25:41.272370Z","iopub.execute_input":"2023-01-22T13:25:41.272813Z","iopub.status.idle":"2023-01-22T13:25:41.279983Z","shell.execute_reply.started":"2023-01-22T13:25:41.272777Z","shell.execute_reply":"2023-01-22T13:25:41.278643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# write the submission dataframe to a CSV file\nsub_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-22T13:25:41.546688Z","iopub.execute_input":"2023-01-22T13:25:41.547104Z","iopub.status.idle":"2023-01-22T13:25:41.554064Z","shell.execute_reply.started":"2023-01-22T13:25:41.547071Z","shell.execute_reply":"2023-01-22T13:25:41.552874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df","metadata":{"execution":{"iopub.status.busy":"2023-01-22T13:25:43.190070Z","iopub.execute_input":"2023-01-22T13:25:43.191408Z","iopub.status.idle":"2023-01-22T13:25:43.203767Z","shell.execute_reply.started":"2023-01-22T13:25:43.191346Z","shell.execute_reply":"2023-01-22T13:25:43.202329Z"},"trusted":true},"execution_count":null,"outputs":[]}]}