{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pytorch_lightning as pl\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2023-04-17T03:31:15.571205Z","iopub.execute_input":"2023-04-17T03:31:15.571966Z","iopub.status.idle":"2023-04-17T03:31:25.585365Z","shell.execute_reply.started":"2023-04-17T03:31:15.571926Z","shell.execute_reply":"2023-04-17T03:31:25.584018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def normalize_targets(y):\n    y_min = np.array([0.0, 0.0], dtype=np.float32)  # azimuth: [0, 2*pi], zenith: [0, pi]\n    y_max = np.array([2 * np.pi, np.pi], dtype=np.float32)\n    return (y - y_min) / (y_max - y_min)\n\ndef denormalize_targets(y_norm):\n    y_min = np.array([0.0, 0.0], dtype=np.float32)\n    y_max = np.array([2 * np.pi, np.pi], dtype=np.float32)\n    return y_norm * (y_max - y_min) + y_min\n","metadata":{"execution":{"iopub.status.busy":"2023-04-17T03:31:25.590358Z","iopub.execute_input":"2023-04-17T03:31:25.590843Z","iopub.status.idle":"2023-04-17T03:31:25.600252Z","shell.execute_reply.started":"2023-04-17T03:31:25.590795Z","shell.execute_reply":"2023-04-17T03:31:25.598631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class NeutrinoDataset(Dataset):\n    def __init__(self, metadata, batch_file, mode=\"train\"):\n        self.metadata = metadata\n        self.batch_data = pd.read_parquet(batch_file)\n        self.mode = mode\n        # You can add some function to process the whole batch_file here\n        # Then just add a line in combine_features as shown bellow\n        # for example:\n        # self.very_good_batch_features = self.great_batch_processing(batch_file)\n        \n\n    def __len__(self):\n        return len(self.metadata)\n\n    def __getitem__(self, idx):\n        event = self.metadata.iloc[idx]\n        event_data = self.batch_data.loc[event.event_id]\n        features = self.combine_features(event_data, idx)\n        if self.mode == \"train\":\n            y = np.array([event['azimuth'], event['zenith']], dtype=np.float32)\n            y_norm = normalize_targets(y)\n            return features, y_norm\n        else:\n            return event.event_id, features\n\n    def extract_event_features_1(self, event_data):\n        num_sensors = len(event_data['sensor_id'].unique()) \n        return np.array([num_sensors])\n\n    def extract_event_features_2(self, event_data):\n        sum_time = event_data['time'].sum()\n        return np.array([sum_time])\n    \n#     def great_batch_processing(self, batch_file):\n#         pass\n\n    def combine_features(self, event_data, idx):\n        features_1 = self.extract_event_features_1(event_data)\n        features_2 = self.extract_event_features_2(event_data)\n        # batch_features = self.very_good_batch_features.iloc[idx].values\n\n        combined_features = np.hstack([\n            features_1,\n            features_2,\n            # batch_features\n        ])\n        return combined_features.astype(np.float32)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-17T03:31:25.601835Z","iopub.execute_input":"2023-04-17T03:31:25.602198Z","iopub.status.idle":"2023-04-17T03:31:25.631850Z","shell.execute_reply.started":"2023-04-17T03:31:25.602164Z","shell.execute_reply":"2023-04-17T03:31:25.630925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def angular_dist_score(az_true, zen_true, az_pred, zen_pred):\n    '''\n    calculate the MAE of the angular distance between two directions.\n    The two vectors are first converted to cartesian unit vectors,\n    and then their scalar product is computed, which is equal to\n    the cosine of the angle between the two vectors. The inverse \n    cosine (arccos) thereof is then the angle between the two input vectors\n    \n    Parameters:\n    -----------\n    \n    az_true : float (or array thereof)\n        true azimuth value(s) in radian\n    zen_true : float (or array thereof)\n        true zenith value(s) in radian\n    az_pred : float (or array thereof)\n        predicted azimuth value(s) in radian\n    zen_pred : float (or array thereof)\n        predicted zenith value(s) in radian\n    \n    Returns:\n    --------\n    \n    dist : float\n        mean over the angular distance(s) in radian\n    '''\n    \n    if not (np.all(np.isfinite(az_true)) and\n            np.all(np.isfinite(zen_true)) and\n            np.all(np.isfinite(az_pred)) and\n            np.all(np.isfinite(zen_pred))):\n        raise ValueError(\"All arguments must be finite\")\n    \n    # pre-compute all sine and cosine values\n    sa1 = np.sin(az_true)\n    ca1 = np.cos(az_true)\n    sz1 = np.sin(zen_true)\n    cz1 = np.cos(zen_true)\n    \n    sa2 = np.sin(az_pred)\n    ca2 = np.cos(az_pred)\n    sz2 = np.sin(zen_pred)\n    cz2 = np.cos(zen_pred)\n    \n    # scalar product of the two cartesian vectors (x = sz*ca, y = sz*sa, z = cz)\n    scalar_prod = sz1*sz2*(ca1*ca2 + sa1*sa2) + (cz1*cz2)\n    \n    # scalar product of two unit vectors is always between -1 and 1, this is against nummerical instability\n    # that might otherwise occure from the finite precision of the sine and cosine functions\n    scalar_prod =  np.clip(scalar_prod, -1, 1)\n    \n    # convert back to an angle (in radian)\n    return np.average(np.abs(np.arccos(scalar_prod)))","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-04-17T03:31:25.633727Z","iopub.execute_input":"2023-04-17T03:31:25.634483Z","iopub.status.idle":"2023-04-17T03:31:25.647834Z","shell.execute_reply.started":"2023-04-17T03:31:25.634442Z","shell.execute_reply":"2023-04-17T03:31:25.646471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class NeutrinoModel(pl.LightningModule):\n    def __init__(self, input_dims):\n        super().__init__()\n        self.batch_norm = nn.BatchNorm1d(input_dims)\n        self.linear1 = nn.Linear(input_dims, input_dims*2)\n        self.relu = nn.ReLU()\n        self.linear2 = nn.Linear(input_dims*2, 2)\n        self.sigmoid = nn.Sigmoid()\n\n    def forward(self, x):\n        x = self.batch_norm(x)\n        x = self.linear1(x)\n        x = self.relu(x)\n        x = self.linear2(x)\n        x = self.sigmoid(x)\n        return x\n\n    def training_step(self, batch, batch_idx):\n        x, y = batch\n        y_hat = self(x)\n        loss = F.mse_loss(y_hat, y, reduction='mean')\n        self.log(\"train_loss\", loss, on_step=True, on_epoch=True, prog_bar=True, logger=True)\n        return loss\n\n    def validation_step(self, batch, batch_idx):\n        x, y_norm = batch\n        y_hat_norm = self(x)\n        loss = F.mse_loss(y_hat_norm, y_norm, reduction='mean')\n        \n        # Denormalize target values and predictions\n        y = denormalize_targets(y_norm.numpy())\n        y_hat = denormalize_targets(y_hat_norm.detach().cpu().numpy())\n\n        # Calculate angular distance score\n        az_true, zen_true = y[:, 0], y[:, 1]\n        az_pred, zen_pred = y_hat[:, 0], y_hat[:, 1]\n        ang_dist = angular_dist_score(az_true, zen_true, az_pred, zen_pred)\n        \n        self.log(\"val_loss\", loss, on_step=False, on_epoch=True, prog_bar=True, logger=True)\n        self.log(\"angular_dist_score\", ang_dist, on_step=False, on_epoch=True, prog_bar=True, logger=True)\n        \n        return loss\n\n    def configure_optimizers(self):\n        return torch.optim.Adam(self.parameters(), lr=0.001)","metadata":{"execution":{"iopub.status.busy":"2023-04-17T03:31:25.651053Z","iopub.execute_input":"2023-04-17T03:31:25.651491Z","iopub.status.idle":"2023-04-17T03:31:25.667362Z","shell.execute_reply.started":"2023-04-17T03:31:25.651453Z","shell.execute_reply":"2023-04-17T03:31:25.665825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_one_epoch_on_batch(model, metadata, batch_file, val_split=0.2):\n    batch_id = int(batch_file.split('_')[-1].split('.')[0])\n    batch_metadata = metadata[metadata['batch_id'] == batch_id]\n    \n    train_metadata, val_metadata = train_test_split(batch_metadata, test_size=val_split, random_state=42)\n    \n    train_dataset = NeutrinoDataset(train_metadata, batch_file, mode=\"train\")\n    train_dataloader = DataLoader(train_dataset, batch_size=64, shuffle=True, num_workers=4)\n    \n    val_dataset = NeutrinoDataset(val_metadata, batch_file, mode=\"train\")\n    val_dataloader = DataLoader(val_dataset, batch_size=64, shuffle=False, num_workers=4)\n\n    trainer = pl.Trainer(max_epochs=1)\n    trainer.fit(model, train_dataloader, val_dataloader)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-17T03:31:25.669088Z","iopub.execute_input":"2023-04-17T03:31:25.669535Z","iopub.status.idle":"2023-04-17T03:31:25.683966Z","shell.execute_reply.started":"2023-04-17T03:31:25.669495Z","shell.execute_reply":"2023-04-17T03:31:25.682864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = '/kaggle/input/icecube-neutrinos-in-deep-ice/'\nmetadata_file = f'{data_dir}/train_meta.parquet'\nbatch_files = [f'{data_dir}train/batch_{i}.parquet' for i in range(1, 2)] # You can set up to 660 files\n\nmetadata = pd.read_parquet(metadata_file)","metadata":{"execution":{"iopub.status.busy":"2023-04-17T03:31:25.685860Z","iopub.execute_input":"2023-04-17T03:31:25.686749Z","iopub.status.idle":"2023-04-17T03:32:06.835062Z","shell.execute_reply.started":"2023-04-17T03:31:25.686687Z","shell.execute_reply":"2023-04-17T03:32:06.833717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = NeutrinoModel(input_dims=2) # Don't forget to change this if you add new features\n\nfor epoch in range(1): # You can set any number of epochs but first use all of 660 files\n    for batch_file in batch_files:\n        train_one_epoch_on_batch(model, metadata, batch_file)","metadata":{"execution":{"iopub.status.busy":"2023-04-17T03:32:06.836967Z","iopub.execute_input":"2023-04-17T03:32:06.837371Z","iopub.status.idle":"2023-04-17T03:34:01.089759Z","shell.execute_reply.started":"2023-04-17T03:32:06.837334Z","shell.execute_reply":"2023-04-17T03:34:01.088627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict(model, metadata, batch_files, output_file=\"submission.csv\"):\n    model.eval()\n    \n    with open(output_file, 'w') as f:\n        f.write('event_id,azimuth,zenith\\n')\n\n    for batch_file in batch_files:\n        batch_id = int(batch_file.split('_')[-1].split('.')[0])\n        batch_metadata = metadata[metadata['batch_id'] == batch_id]\n        test_dataset = NeutrinoDataset(batch_metadata, batch_file, mode=\"test\")\n        test_dataloader = DataLoader(test_dataset, batch_size=64, num_workers=4)\n        \n        for event_ids, features in test_dataloader:\n            with torch.no_grad():\n                pred_norm = model(features)\n                pred = denormalize_targets(pred_norm.detach().numpy())\n                \n                with open(output_file, 'a') as f:\n                    for event_id, prediction in zip(event_ids, pred):\n                        f.write(f'{event_id},{prediction[0]},{prediction[1]}\\n')","metadata":{"execution":{"iopub.status.busy":"2023-04-17T03:44:28.229934Z","iopub.execute_input":"2023-04-17T03:44:28.232079Z","iopub.status.idle":"2023-04-17T03:44:28.246260Z","shell.execute_reply.started":"2023-04-17T03:44:28.231998Z","shell.execute_reply":"2023-04-17T03:44:28.244650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_metadata_file = f'{data_dir}/test_meta.parquet'\ntest_metadata = pd.read_parquet(test_metadata_file)\n\n\ntest_dir = '/kaggle/input/icecube-neutrinos-in-deep-ice/test/'\ntest_batch_files = [f'{test_dir}/{file}' for file in os.listdir(test_dir)]\n\npredict(model, test_metadata, test_batch_files, \"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-04-17T03:44:37.929045Z","iopub.execute_input":"2023-04-17T03:44:37.929522Z","iopub.status.idle":"2023-04-17T03:44:38.722204Z","shell.execute_reply.started":"2023-04-17T03:44:37.929483Z","shell.execute_reply":"2023-04-17T03:44:38.719584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-04-17T03:44:39.455017Z","iopub.execute_input":"2023-04-17T03:44:39.455553Z","iopub.status.idle":"2023-04-17T03:44:39.500304Z","shell.execute_reply.started":"2023-04-17T03:44:39.455502Z","shell.execute_reply":"2023-04-17T03:44:39.499213Z"},"trusted":true},"execution_count":null,"outputs":[]}]}