{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"def reduce_memory_usage(df):\n    \n    start_mem = df.memory_usage().sum() / 1024**2\n    #print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))  \n    for col in df.columns:\n        col_type = df[col].dtype.name\n        if ((col_type != 'datetime64[ns]') & (col_type != 'category')):\n            if (col_type != 'object'):\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                if str(col_type)[:3] == 'int':\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n\n                else:\n                    if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                        df[col] = df[col].astype(np.float16)\n                    elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        pass\n            else:\n                df[col] = df[col].astype('category')\n    mem_usg = df.memory_usage().sum() / 1024**2 \n    #print(\"Memory usage became: \",mem_usg,\" MB\")\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-10-26T09:21:47.021186Z","iopub.execute_input":"2023-10-26T09:21:47.021456Z","iopub.status.idle":"2023-10-26T09:21:47.038481Z","shell.execute_reply.started":"2023-10-26T09:21:47.021430Z","shell.execute_reply":"2023-10-26T09:21:47.037681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport random\nfrom functools import reduce\nimport os\nimport torch\nimport torchmetrics\nfrom torch.utils.data import DataLoader, Dataset\nimport pytorch_lightning as pl\nfrom pytorch_lightning.loggers import TensorBoardLogger\nfrom pytorch_lightning.callbacks import TQDMProgressBar\nimport logging\nfrom sklearn.metrics import average_precision_score\nimport torchvision \nfrom typing import List, Set, Dict, Tuple","metadata":{"execution":{"iopub.status.busy":"2023-10-26T09:22:23.708517Z","iopub.execute_input":"2023-10-26T09:22:23.709070Z","iopub.status.idle":"2023-10-26T09:22:38.196533Z","shell.execute_reply.started":"2023-10-26T09:22:23.709031Z","shell.execute_reply":"2023-10-26T09:22:38.195553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def normalize(dataframe):\n    dataframe_norm = dataframe.astype('float32')\n    eps = 1e-9\n    mean = dataframe_norm.mean(axis=0)\n    std = dataframe_norm.std(axis=0)\n    dataframe = (dataframe - mean) / (std + eps)\n    return dataframe","metadata":{"execution":{"iopub.status.busy":"2023-10-26T09:29:16.868681Z","iopub.execute_input":"2023-10-26T09:29:16.869076Z","iopub.status.idle":"2023-10-26T09:29:16.874221Z","shell.execute_reply.started":"2023-10-26T09:29:16.869046Z","shell.execute_reply":"2023-10-26T09:29:16.873295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"configure = {\n    'seq_length': 12096,\n    'sample_stride': 12096//16,\n    'patch_embed_size': 300,\n    'patch_size': 14,\n    'feature_size': 3,\n    'label_size': 4,\n    'dropout_rate': 0.1,\n    'num_encoder_layers': 5,\n    'encoder_embed_size': 300,\n    'encoder_num_heads': 6,\n    'num_lstm_layers': 2,\n    'batch_size': 16,\n}","metadata":{"execution":{"iopub.status.busy":"2023-10-26T09:29:15.481361Z","iopub.execute_input":"2023-10-26T09:29:15.482313Z","iopub.status.idle":"2023-10-26T09:29:15.487753Z","shell.execute_reply.started":"2023-10-26T09:29:15.482270Z","shell.execute_reply":"2023-10-26T09:29:15.486847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train'\ntest_data_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test'\ndata_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/'\ndef get_dataframes_by_path(data_path, specific_ids=None):\n    if specific_ids is None:\n        filenames = os.listdir(data_path)\n    else:\n        filenames = specific_ids\n        filenames = list(map(lambda filename: filename + '.csv', filenames))\n \n    #filenames = list(filter(lambda x: os.path.exists(x), filenames))\n    print(f\"len of files: {len(filenames)}\")\n    dataframes = []\n    for filename in filenames:\n        try:\n            dataframe = pd.read_csv(os.path.join(data_path, filename))\n            if 'Task' not in dataframe.columns:            \n                dataframe[['Task', 'Valid']] = 1\n            dataframe[['Task', 'Valid']].astype('int')\n            dataframe['Event'] = dataframe[['StartHesitation', 'Turn', 'Walking']].aggregate('max', axis=1)\n\n            dataframe['StartHesitation_mask'] = \\\n            dataframe['Turn_mask'] = \\\n            dataframe['Walking_mask'] = \\\n            dataframe['Event_mask'] = (dataframe['Task'] & dataframe['Valid'])\n            dataframes.append(dataframe)\n        \n        except FileNotFoundError: \n            dataframe = pd.read_csv(f'/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/notype/{filename}')\n            dataframe['StartHesitation'] = \\\n            dataframe['Turn'] = \\\n            dataframe['Walking'] = \\\n            dataframe['StartHesitation_mask'] = \\\n            dataframe['Turn_mask'] = \\\n            dataframe['Walking_mask'] = 0\n            dataframe['Event_mask'] = (dataframe['Task'] & dataframe['Valid']).astype('int')\n            dataframes.append(dataframe)\n    \n    # dataframes = list(map(lambda x: pd.read_csv(x), filenames))\n    \n    '''\n    for dataframe, filename in zip(dataframes, filenames):\n        dataframe['Id'] = os.path.basename(filename).split('.')[0]\n    '''\n    return dataframes\n\ndef get_test_dataframes_by_path(data_path, specific_ids=None):\n    if specific_ids is None:\n        filenames = os.listdir(data_path)\n    else:\n        filenames = specific_ids\n        filenames = list(map(lambda filename: filename + '.csv', filenames))\n    \n    #filenames = list(filter(lambda x: os.path.exists(x), filenames))\n\n    dataframes = []\n    for filename in filenames:\n        try:\n            dataframe = pd.read_csv(os.path.join(data_path, filename))\n            dataframes.append(dataframe)\n        \n        except FileNotFoundError: \n            raise(FileNotFoundError)\n    # dataframes = list(map(lambda x: pd.read_csv(x), filenames))\n    \n    '''\n    for dataframe, filename in zip(dataframes, filenames):\n        dataframe['Id'] = os.path.basename(filename).split('.')[0]\n    '''\n    return dataframes\n\ndef get_dataset(name, data_dir, ids=None, mode=\"train\"):\n    data_path = os.path.join(data_dir, name)\n    if mode == \"train\" or mode == \"val\":\n        dataset = get_dataframes_by_path(data_path, ids)\n    elif mode == \"test\":\n        dataset = get_test_dataframes_by_path(data_path, ids)\n    dataset = list(map(reduce_memory_usage, dataset))\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2023-10-26T09:29:20.161007Z","iopub.execute_input":"2023-10-26T09:29:20.161616Z","iopub.status.idle":"2023-10-26T09:29:20.180191Z","shell.execute_reply.started":"2023-10-26T09:29:20.161577Z","shell.execute_reply":"2023-10-26T09:29:20.179209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_dataframe_blocks(dataframe, block_size, columns=None):\n    dataframe = dataframe.copy()\n    if columns != None:\n        dataframe = dataframe[columns]\n        \n    dataframe = dataframe.values\n    dataframe = dataframe.astype(np.float32) # TODO: check\n    \n    if dataframe.shape[0] % block_size != 0:\n        pad_size = block_size - dataframe.shape[0] % block_size\n        dataframe = np.concatenate((dataframe, np.zeros((pad_size, dataframe.shape[1]))))\n    blocks = []\n    block_start_indices = list(range(0, dataframe.shape[0]-block_size+1, configure['sample_stride']))\n    #block_start_indices = list(range(0, len(dataframe), configure['sample_stride']))\n    #block_begins = [x for x in block_start_indices if x+configure['seq_length'] <= len(dataframe)]\n    \n    for i in block_start_indices:\n        block_info = {\n            'start_index': i,\n            'end_index': i + block_size - 1,\n            'values': dataframe[i : i + block_size].copy(),\n        }\n        blocks.append(block_info)\n    return blocks","metadata":{"execution":{"iopub.status.busy":"2023-10-26T09:29:22.806884Z","iopub.execute_input":"2023-10-26T09:29:22.807271Z","iopub.status.idle":"2023-10-26T09:29:22.815574Z","shell.execute_reply.started":"2023-10-26T09:29:22.807239Z","shell.execute_reply":"2023-10-26T09:29:22.814437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CustomDataset(Dataset):\n    \n    def __init__(self, dataframes, seq_length, patch_size):\n        self.window = seq_length\n        self.dataframes = dataframes\n        self.seq_length = seq_length\n        self.patch_size = patch_size\n        self.feature_size = 3\n        self.label_size = configure['label_size']\n        self.data = train_data # []\n        #self.preprocessing()\n        #self.dataset_size = reduce(lambda x, y: x + (y.shape[0] - 1 - (self.window - 1)), self.dataframes, 0)\n        \n    def preprocessing(self):\n        dataframe_blocks = []\n        dataframe_index = 0\n        for dataframe in self.dataframes:\n            print('processing:', dataframe_index)\n            blocks = get_dataframe_blocks(dataframe, configure['seq_length'])\n            for block in blocks:\n                sub_dataframe = dataframe[block['start_index']: block['end_index']]\n                if (sub_dataframe['Valid'].astype('int') & sub_dataframe['Task'].astype('int')).max() > 0 \\\n                    and sub_dataframe[['StartHesitation', 'Turn', 'Walking']].max(axis=1).max() > 0:\n                    dataframe_blocks.append((dataframe_index, block))\n            dataframe_index += 1\n \n        for offset_index in range(configure['seq_length'] // configure['sample_stride']):\n            for index, dataframe_block in dataframe_blocks:\n                sample_blocks = dataframe_block[offset_index::configure['sample_stride']]\n                for block in sample_blocks:\n                    self.data.append(block.values)\n                    \n    def __len__(self):\n        # 256 * 16: steps_per_epoch_times * batch_size\n        return  len(self.data) # 256*16  #self.dataset_size\n        \n    def __getitem__(self, i):\n        i = i % len(self.data)\n        dataframe = self.data[i]\n        #start_timestep = 126479 # REMOVE\n        data = dataframe[:, :3] # dataframe[['AccV', 'AccML', 'AccAP']]\n        # data = data.to_numpy()\n        #data = np.concatenate((data, np.zeros((self.seq_length - self.window, self.feature_size))), axis=0)\n        #print(data.shape)\n        data_tensor = torch.from_numpy(data)\n        data_tensor = torch.reshape(data_tensor, (self.seq_length//self.patch_size, self.patch_size, self.feature_size))\n        data_tensor = torch.reshape(data_tensor, (-1, self.patch_size * self.feature_size))\n        \n        target = dataframe[:, 3:] # dataframe[['StartHesitation', 'Turn', 'Walking', 'Event']].copy()\n \n        # target_tensor: [timesteps, 3 + 3]\n        target = np.concatenate((target, np.zeros((self.seq_length - self.window, self.label_size * 2))), axis=0)\n        target_tensor = torch.from_numpy(target)\n        # target_tensor: [timestep//patch_size, patch_size, 4 * 2(mask)]\n        target_tensor = torch.reshape(target_tensor, (self.seq_length//self.patch_size, self.patch_size, self.label_size * 2))\n        # target_tensor: [timestep//patch_size, 4]\n        target_tensor = torch.max(target_tensor, axis = 1)[0]\n        return data_tensor, target_tensor\n\n\nclass DataModule(pl.LightningDataModule):\n    \n    def __init__(self, train_dataset):\n        super().__init__()\n        self.train_dataset = train_dataset\n        \n    def prepare_data(self):\n        pass\n    \n    def setup(self, stage = None):\n        if stage == 'fit' or stage == None:\n            self.train = CustomDataset(self.train_dataset, configure['seq_length'], configure['patch_size'])\n\n    def train_dataloader(self):\n        return DataLoader(self.train, batch_size=configure['batch_size'])\n\n    def val_dataloader(self):\n        return DataLoader(self.train, batch_size=configure['batch_size'])","metadata":{"execution":{"iopub.status.busy":"2023-10-26T09:29:24.432173Z","iopub.execute_input":"2023-10-26T09:29:24.432522Z","iopub.status.idle":"2023-10-26T09:29:24.449098Z","shell.execute_reply.started":"2023-10-26T09:29:24.432493Z","shell.execute_reply":"2023-10-26T09:29:24.448059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def custom_loss(pred, ground_truth):\n    #print(\"ground_truth:\", ground_truth.shape)\n    #loss_func = nn.CrossEntropyLoss(reduction='none')\n    loss_func = nn.BCEWithLogitsLoss(reduction='none')\n    loss_value = loss_func(pred, ground_truth[:, :, :4])\n    #loss_value = torchvision.ops.sigmoid_focal_loss(pred, ground_truth[:, :, :3])\n    # if grouth_truth in shape of [Batch, Seq_Length, Label_Size+Invalid_Size]\n    \n    mask = ground_truth[:, :, 4:]\n    loss_value = loss_value * mask\n    return torch.sum(loss_value) / torch.sum(mask)","metadata":{"execution":{"iopub.status.busy":"2023-10-26T09:29:26.765639Z","iopub.execute_input":"2023-10-26T09:29:26.766103Z","iopub.status.idle":"2023-10-26T09:29:26.771782Z","shell.execute_reply.started":"2023-10-26T09:29:26.766068Z","shell.execute_reply":"2023-10-26T09:29:26.770868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch.nn as nn\nimport torch\n\nclass Encoder_Layer(nn.Module):\n    def __init__(self, configure):\n        super().__init__()\n        # MHA takes 'total' dimensions as parameter\n        self.mha = nn.MultiheadAttention(configure['encoder_embed_size'], configure['encoder_num_heads'], configure['dropout_rate'])\n        self.mha_linear = nn.Linear(configure['encoder_embed_size'], configure['encoder_embed_size'])\n        self.layer_norm = nn.LayerNorm(configure['encoder_embed_size'])\n        self.last_layer = nn.Sequential(\n            nn.Linear(configure['encoder_embed_size'], configure['encoder_embed_size']),\n            nn.ReLU(),\n            nn.Dropout(configure['dropout_rate']),\n            nn.Linear(configure['encoder_embed_size'], configure['encoder_embed_size']),\n            nn.Dropout(configure['dropout_rate']),\n        )\n        \n    def forward(self, input):\n        feature = input\n        #feature = torch.tile(input, (1, 1, configure['encoder_num_heads']))\n        feature = self.mha(feature, feature, feature)[0]\n        feature = self.mha_linear(feature)\n        feature = feature + input\n        feature = self.layer_norm(feature)\n        feature = feature + self.last_layer(feature)\n        feature = self.layer_norm(feature)\n        return feature\n    \nclass Encoder(nn.Module):\n    def __init__(self, configure):\n        super().__init__()\n        self.patch_embedding = nn.Linear(configure['patch_size']*configure['feature_size'], configure['patch_embed_size'])\n        self.pos_embedding = nn.Parameter(torch.normal(0, 0.02, (1, configure['seq_length']//configure['patch_size'], configure['patch_embed_size'])))\n        self.dropout = nn.Dropout(configure['dropout_rate'])\n        self.encoder_layers = nn.ModuleList()\n        self.lstm = nn.LSTM(configure['encoder_embed_size'], configure['encoder_embed_size'], num_layers=configure['num_lstm_layers'], bidirectional=True, batch_first=True)\n        \n        for i in range(configure['num_encoder_layers']):\n            self.encoder_layers.append(\n                Encoder_Layer(configure)\n            )\n            \n        \n    def forward(self, input, training):\n        input = input / 50 # Normalization, does it help?\n        #input: [Batch, Seq_Length, Patch_Size * Feature_size]\n        #Feature after linear transform: [Batch, Seq_Length, embed_size]\n        feature = self.patch_embedding(input)\n        if training:\n            shifts = torch.empty((input.shape[0],)).uniform_(-configure['seq_length'], 0).to(torch.int).tolist()\n            random_pos_encoding = torch.roll(torch.tile(self.pos_embedding, (input.shape[0], 1, 1)), \n                                             shifts=shifts,\n                                             dims=input.shape[0]*[1])\n            feature = feature + random_pos_encoding\n        else:\n            feature = feature + torch.tile(self.pos_embedding, (input.shape[0], 1, 1))  \n        feature = self.dropout(feature)\n        for layer in self.encoder_layers:\n            feature = layer(feature)\n        feature, _ = self.lstm(feature)    \n        return feature\n    \nclass Model(pl.LightningModule):\n    \n    def __init__(self, configure):\n        super().__init__()\n        self.training_losses = []\n        self.encoder = Encoder(configure)\n        self.linear = nn.Linear(configure['encoder_embed_size']*2, configure['label_size'])\n        self.accuracy = torchmetrics.Accuracy(task=\"multiclass\", num_classes=configure['label_size'])\n        self.learning_rate = 0.01/62\n        self.name = \"\"\n        \n    def set_model_name(self, name):\n        self.name = name\n    \n    def configure_optimizers(self):\n        \"\"\"\n        def lr_foo(epoch):\n            if epoch < self.hparams.warm_up_step:\n                # warm up lr\n                lr_scale = 0.1 ** (self.hparams.warm_up_step - epoch)\n            else:\n                lr_scale = 0.95 ** epoch\n\n            return lr_scale\n\n        scheduler = LambdaLR(\n            optimizer,\n            lr_lambda=lr_foo\n        )\n        \"\"\"\n\n        optimizer = torch.optim.AdamW(self.parameters(), lr=self.learning_rate, betas=(0.9, 0.98), eps=1e-9)\n        # optimizer = torch.optim.Adam(self.parameters(), lr=self.learning_rate, betas=(0.9, 0.98), eps=1e-9)\n        return optimizer\n    \n\n    def optimizer_step(self, epoch, batch_idx, optimizer, optimizer_closure):\n        print(f\"epoch - {epoch}\")\n        STEPS_PER_EPOCH = 256\n        if self.trainer.global_step < STEPS_PER_EPOCH:\n            lr_scale = min(1., float(self.trainer.global_step + 1) / STEPS_PER_EPOCH)\n            for pg in optimizer.param_groups:\n                pg['lr'] = lr_scale * self.learning_rate\n\n        optimizer.step(closure=optimizer_closure)\n        optimizer.zero_grad()\n\n    \n    def training_step(self, train_batch, batch_idx):\n        # target: [Batch, Seq_Length, Patch_Size, Feature]\n        x, y = train_batch\n        x = x.to(self.linear.weight.dtype)\n        y = y.to(self.linear.weight.dtype)\n        x = self(x, training=True)\n        loss = custom_loss(x, y)\n        \n        self.accuracy(x.reshape(-1, configure['label_size']), y[:, :, :4].reshape(-1, configure['label_size']))\n        self.log('train_acc_epoch', self.accuracy, on_step=True, on_epoch=False, prog_bar=True)\n        if batch_idx % 50 == 0:\n            \"\"\"\n            y = pd.DataFrame(y[0].clone().detach().numpy())\n            x = pd.DataFrame(x[0].clone().detach().numpy())\n            for idx, i in enumerate((y.max(axis=1)!=0)):\n                if i == True:\n                    print(idx, \" pred, gt:\", x.iloc[idx].values, \" \", y.iloc[idx].values)\n            \"\"\"\n            print(loss)\n        #self.training_losses.append(loss)\n        return loss\n    \n    def on_train_epoch_end(self):\n        if val_defog_data == None:\n            return\n        print(self.training_losses)\n        self.training_losses.clear()\n        total_turn_score = []\n        total_walking_score = []\n        total_StartHesitation_score = []\n        total_event_score = []\n        all_prediction_values = np.empty((0, configure['label_size']+1))\n        all_targets = pd.DataFrame()\n        \n        for id in range(0, len(val_defog_data)):\n            if 'Task' not in val_defog_data[id].columns:            \n                val_defog_data[id][['Task', 'Valid']] = 1\n            blocks = get_dataframe_blocks(val_defog_data[id], configure['seq_length'], ['AccV', 'AccML', 'AccAP'])\n            prediction_values = np.zeros((blocks[-1]['end_index']+1, configure['label_size']+1))\n\n            for block in blocks:\n                with torch.no_grad():\n                    block_tensor = torch.from_numpy(block['values'])\n                    block_tensor = torch.reshape(block_tensor, (configure['seq_length']//configure['patch_size'], configure['patch_size']*configure['feature_size']))\n                    block_tensor = torch.unsqueeze(block_tensor, axis=0) # add batch dimension\n                    block_tensor = block_tensor.to(torch.float32).to(self.device)\n                    \n                    predictions = self(block_tensor, training=False).detach()\n                    predictions = torch.unsqueeze(predictions, axis=2)\n                    predictions = torch.tile(predictions, (1, 1, configure['patch_size'], 1))\n                    predictions = torch.reshape(predictions, (predictions.shape[0], predictions.shape[1]*predictions.shape[2], predictions.shape[3]))\n                    # TODO: output by softmax of statistical predictions\n                    #print(\"prediction_values size: \", prediction_values.shape)\n                    #print(\"prediction size: \", predictions.shape)\n                    # predictions[0]: get first of the batch as prediction result, because we haven't implement batch test now\n                    prediction_values[block['start_index']: block['end_index']+1, 0:4] += predictions[0].detach().cpu().numpy()\n                    prediction_values[block['start_index']: block['end_index']+1, 4] += 1\n            prediction_values = prediction_values[:val_defog_data[id].shape[0]]\n            \n            all_prediction_values = np.concatenate((all_prediction_values, prediction_values), axis=0)\n            all_targets = pd.concat([all_targets, val_defog_data[id][['StartHesitation', 'Turn', 'Walking', 'Event', 'Task', 'Valid']]], axis=0).reset_index(drop=True)\n             \n        # display(all_targets)\n\n        all_prediction_values[:, 0] = all_prediction_values[:, 0] / all_prediction_values[:, 4]\n        all_prediction_values[:, 1] = all_prediction_values[:, 1] / all_prediction_values[:, 4]\n        all_prediction_values[:, 2] = all_prediction_values[:, 2] / all_prediction_values[:, 4]\n        all_prediction_values[:, 3] = all_prediction_values[:, 3] / all_prediction_values[:, 4]\n        all_targets['StartHesitation_Prediction'] = all_prediction_values[:, 0]\n        all_targets['Turn_Prediction'] = all_prediction_values[:, 1]\n        all_targets['Walking_Prediction'] = all_prediction_values[:, 2]\n        all_targets['Event_Prediction'] = all_prediction_values[:, 3]\n        all_targets = all_targets.reset_index(drop=True) # TODO: check\n        \n        all_targets = all_targets[(all_targets['Task'].astype('bool') & all_targets['Valid'].astype('bool'))]\n        all_targets = all_targets[~all_targets['Turn'].isna()]\n        StartHesitation_score = average_precision_score(all_targets['StartHesitation'], all_targets['StartHesitation_Prediction'])\n        turn_score = average_precision_score(all_targets['Turn'], all_targets['Turn_Prediction'])\n        walking_score = average_precision_score(all_targets['Walking'], all_targets['Walking_Prediction'])\n        event_score = average_precision_score(all_targets['Event'], all_targets['Event_Prediction'])\n\n        print(\"StartHesitation_score:\", StartHesitation_score)\n        print(\"turn_score:\", turn_score)\n        print(\"walking_score:\", walking_score)\n        print(\"event_score:\", event_score)\n        save_path = f\"{self.name}/checkpoint{str(self.current_epoch)}.pth\"\n        if not os.path.exists(self.name):\n            os.makedirs(self.name)\n        torch.save(self.state_dict(), save_path)\n        \n    def forward(self, input, training):\n        # input: [Batch, Seq_Length, Patch_Size * Feature_size]\n        input = input.to(self.linear.weight.dtype)\n        feature = self.encoder(input, training)\n        feature = self.linear(feature)\n        # feature = torch.sigmoid(feature)\n        return feature","metadata":{"execution":{"iopub.status.busy":"2023-10-26T09:29:28.026365Z","iopub.execute_input":"2023-10-26T09:29:28.026734Z","iopub.status.idle":"2023-10-26T09:29:28.067450Z","shell.execute_reply.started":"2023-10-26T09:29:28.026704Z","shell.execute_reply":"2023-10-26T09:29:28.066514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_defog_data = None\n\n\ndef get_blocks(data: pd.DataFrame) -> List:\n    train_data = []\n    dataframe_blocks = [] # [[blocks of file 1: block_1, block_2...], [blocks of file 2:...]]\n    dataframe_index = 0\n    print(\"len(data): \", len(data))\n    total_blocks = 0\n    for dataframe in data:\n        blocks = get_dataframe_blocks(dataframe, configure['seq_length'], ['AccV', 'AccML', 'AccAP', 'StartHesitation', 'Turn', 'Walking', 'Event', 'StartHesitation_mask', 'Turn_mask', 'Walking_mask', 'Event_mask'])\n             \n        dataframe_blocks.append([])\n        for block in blocks:\n            sub_dataframe = dataframe[block['start_index']: block['end_index']]\n            \"\"\"\n            if (sub_dataframe['Valid'].astype('int') & sub_dataframe['Task'].astype('int')).max() > 0 \\\n                and sub_dataframe[['StartHesitation', 'Turn', 'Walking']].max(axis=1).max() > 0:\n                dataframe_blocks[-1].append((dataframe_index, block))\n            \"\"\"\n            if (sub_dataframe['Valid'].astype('int') & sub_dataframe['Task'].astype('int')).max() > 0:\n                dataframe_blocks[-1].append((dataframe_index, block))\n            total_blocks += 1\n        dataframe_index += 1\n    \n    print(\"total_blocks:\", total_blocks)\n    for offset_index in range(configure['seq_length'] // configure['sample_stride']):\n        for dataframe_block in dataframe_blocks:\n            # print(\"dataframe_block size:\", len(dataframe_block))\n            sample_blocks = dataframe_block[offset_index::configure['seq_length'] // configure['sample_stride']]\n            # print(\"sample_blocks size:\", len(sample_blocks))\n            for index, block in sample_blocks:\n                train_data.append(block['values'])\n    print(\"len(train_data):\", len(train_data))\n    return train_data\n\ndef prepare_data(dataset_name, mode=\"train\"):\n    metadata = pd.read_csv(f'/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/{dataset_name}_metadata.csv').set_index('Id')\n    if dataset_name == 'defog':\n        val_subjects = ['00f674', '8d43d9', '107712', '7b2e84', '575c60', '7f8949', '2874c5', '72e2c7']\n    elif dataset_name == 'tdcsfog':\n        val_subjects = ['4dc2f8', '242a3e', '301ada', '66341b', '575c60', '7fcee9']\n    train_defog_ids = [i not in val_subjects for i in metadata['Subject']]\n    train_defog_ids = metadata[train_defog_ids].index.to_list()\n    val_defog_ids = [i in val_subjects for i in metadata['Subject']]\n    val_defog_ids = metadata[val_defog_ids].index.to_list()\n     \n    \n    # tdcsfog_data = get_dataset(\"tdcsfog\", train_data_path)\n    # defog_submission_filenames = os.listdir(os.path.join(test_data_path, \"defog\"))\n    # tdcsfog_submission_filenames = os.listdir(os.path.join(test_data_path, \"tdcsfog\"))\n\n    # test_data = get_blocks(test_defog_data, ['AccV', 'AccML', 'AccAP'])\n    data = None\n    if mode == \"train\":\n        data = get_dataset(dataset_name, train_data_path, train_defog_ids, mode=\"train\")\n    elif mode == \"val\":\n        data = get_dataset(dataset_name, train_data_path, val_defog_ids, mode=\"val\")\n    elif mode == \"test\":\n        data = get_dataset(dataset_name, test_data_path, mode=\"test\")\n     \n   \n    # Normalization\n    for dataframe in data:\n        dataframe[['AccV', 'AccML', 'AccAP']] = normalize(dataframe[['AccV', 'AccML', 'AccAP']])\n    if mode == \"train\":\n \n        data = get_blocks(data)\n    return data","metadata":{"execution":{"iopub.status.busy":"2023-10-26T09:33:39.868595Z","iopub.execute_input":"2023-10-26T09:33:39.868950Z","iopub.status.idle":"2023-10-26T09:33:39.885733Z","shell.execute_reply.started":"2023-10-26T09:33:39.868921Z","shell.execute_reply":"2023-10-26T09:33:39.884866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train(train_data):\n    '''\n    for name, param in model.named_parameters():\n        if param.requires_grad:\n            print(name)\n    '''\n    # train_loader = DataModule()\n    train_dataset = CustomDataset(train_data, configure['seq_length'], configure['patch_size'])\n    print(len(train_dataset))\n \n    # val_dataset = CustomDataset(val_data, configure['seq_length'], configure['patch_size'])\n    train_dataloader = DataLoader(train_dataset, batch_size=configure['batch_size'])\n    # val_dataloader = DataLoader(train_dataset, batch_size=configure['batch_size'])\n    #tb_logger = pl_loggers.TensorBoardLogger(save_dir=\"logs/\")\n    #logger = TensorBoardLogger(\"tb_logs\", name=\"my_model\")\n    precision = 16 if torch.cuda.is_available() else 32\n    loger = logging.getLogger('lightning')\n    trainer = pl.Trainer(max_epochs=15, \n                         log_every_n_steps=1, \n                         precision=precision, \n                         accelerator=\"auto\", \n                         devices=\"auto\", \n                         strategy=\"auto\", \n                         callbacks=[TQDMProgressBar(refresh_rate=10)])\n    #trainer = pl.Trainer(max_epochs=3, log_every_n_steps=1, accelerator=\"auto\", devices=\"auto\", strategy=\"auto\")\n    trainer.fit(model, train_dataloader)","metadata":{"execution":{"iopub.status.busy":"2023-10-26T09:33:44.173542Z","iopub.execute_input":"2023-10-26T09:33:44.173885Z","iopub.status.idle":"2023-10-26T09:33:44.181353Z","shell.execute_reply.started":"2023-10-26T09:33:44.173858Z","shell.execute_reply":"2023-10-26T09:33:44.180322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_names = [\"defog\"] # [\"tdcsfog\", \"defog\"]\nval_defog_data = None\ntest_data = None\nfor dataset_name in dataset_names:\n    train_data = prepare_data(dataset_name, \"train\")\n    val_defog_data = prepare_data(dataset_name, \"val\")\n    test_data = prepare_data(dataset_name, \"test\")\n    model = Model(configure)\n    model.set_model_name(dataset_name)\n    train(train_data)","metadata":{"execution":{"iopub.status.busy":"2023-10-26T09:31:31.582976Z","iopub.execute_input":"2023-10-26T09:31:31.583589Z","iopub.status.idle":"2023-10-26T09:32:47.906045Z","shell.execute_reply.started":"2023-10-26T09:31:31.583557Z","shell.execute_reply":"2023-10-26T09:32:47.905154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def test():\n    all_prediction_values = np.empty((0, configure['label_size']+1))\n    all_targets = pd.DataFrame()\n    output_prediction_df_list = []\n    for id in range(0, len(test_data)):\n        if 'Task' not in test_data[id].columns:            \n            test_data[id][['Task', 'Valid']] = 1\n        blocks = get_dataframe_blocks(test_data[id], configure['seq_length'], ['AccV', 'AccML', 'AccAP'])\n        prediction_values = np.zeros((blocks[-1]['end_index']+1, configure['label_size']+1))\n        \n        for block in blocks:\n            with torch.no_grad():\n                block_tensor = torch.from_numpy(block['values'])\n                block_tensor = torch.reshape(block_tensor, (configure['seq_length']//configure['patch_size'], configure['patch_size']*configure['feature_size']))\n                block_tensor = torch.unsqueeze(block_tensor, axis=0) # add batch dimension\n                block_tensor = block_tensor.to(torch.float32).to(model.device)\n\n                predictions = model(block_tensor, training=False).detach()\n                predictions = torch.unsqueeze(predictions, axis=2)\n                predictions = torch.tile(predictions, (1, 1, configure['patch_size'], 1))\n                predictions = torch.reshape(predictions, (predictions.shape[0], predictions.shape[1]*predictions.shape[2], predictions.shape[3]))\n                # TODO: output by softmax of statistical predictions\n                # print(\"prediction_values size: \", prediction_values.shape)\n                # print(\"prediction size: \", predictions.shape)\n                # predictions[0]: get first of the batch as prediction result, because we haven't implement batch test now\n                prediction_values[block['start_index']: block['end_index']+1, 0:4] += predictions[0].detach().cpu().numpy()\n                prediction_values[block['start_index']: block['end_index']+1, 4] += 1\n        prediction_values = prediction_values[:test_data[id].shape[0]]\n        output_prediction_df = pd.DataFrame(index=range(prediction_values.shape[0]))\n        output_prediction_df[\"AccV\"] = prediction_values[:, 0] / prediction_values[:, 4]\n        output_prediction_df[\"AccML\"] = prediction_values[:, 1] / prediction_values[:, 4]\n        output_prediction_df[\"AccAP\"] = prediction_values[:, 2] / prediction_values[:, 4]\n        output_prediction_df_list.append(output_prediction_df)\n        print(output_prediction_df.shape)\n    # display(all_targets)","metadata":{"execution":{"iopub.status.busy":"2023-10-20T05:26:51.965791Z","iopub.execute_input":"2023-10-20T05:26:51.966552Z","iopub.status.idle":"2023-10-20T05:26:51.979011Z","shell.execute_reply.started":"2023-10-20T05:26:51.966513Z","shell.execute_reply":"2023-10-20T05:26:51.977058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test()","metadata":{"execution":{"iopub.status.busy":"2023-10-20T05:26:53.104728Z","iopub.execute_input":"2023-10-20T05:26:53.105087Z","iopub.status.idle":"2023-10-20T05:26:53.290291Z","shell.execute_reply.started":"2023-10-20T05:26:53.105060Z","shell.execute_reply":"2023-10-20T05:26:53.289222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nfrom torch.nn.utils import (\n  parameters_to_vector as Params2Vec,\n  vector_to_parameters as Vec2Params\n)\nimport matplotlib.pyplot as plt\nimport gc\ngc.collect()\ntorch.cuda.empty_cache()\ndef tau_2d(alpha, beta, theta_ast):\n    a = (alpha * theta_ast[:,None,None]).detach()\n    b = (beta * alpha * theta_ast[:,None,None]).detach()\n    return a + b\n\ndataloader = DataLoader(CustomDataset(defog_data, configure['seq_length'], configure['patch_size']), batch_size=configure['batch_size'])\n\nPATH = f\"checkpoint8.pth\"\ninfer_net = Model(configure)\n#infer_net.load_state_dict(torch.load(PATH))\ninfer_net.eval()\nwith torch.no_grad():\n    x = torch.linspace(-40, 40, 18).detach()\n    y = torch.linspace(-40, 40, 18).detach()\n    alpha, beta = torch.meshgrid(x, y)\n\n    theta_ast = Params2Vec(infer_net.parameters()).detach()\n    space = tau_2d(alpha, beta, theta_ast)\n\n    losses = torch.empty_like(space[0, :, :])\n    '''\n    for a, _ in enumerate(x):\n        print(f'a = {a}')\n        for b, _ in enumerate(y):\n            Vec2Params(space[:, a, b], infer_net.parameters())\n            for _, (data, label) in enumerate(dataloader):\n                \n                losses[a][b] = custom_loss(infer_net(data, training=False), label).detach().cpu()\n                break\n    '''\n\n    fig = plt.figure()\n    ax = fig.add_subplot(111, projection='3d')\n    ax.plot_surface(x, y, losses)\n    plt.show()\n    plt.savefig('plot.png')       \ninfer_net.cpu()\ndel infer_net\ngc.collect()\ntorch.cuda.empty_cache()\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-08-05T17:25:24.597753Z","iopub.execute_input":"2023-08-05T17:25:24.598301Z","iopub.status.idle":"2023-08-05T17:25:25.123328Z","shell.execute_reply.started":"2023-08-05T17:25:24.598253Z","shell.execute_reply":"2023-08-05T17:25:25.120766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nfrom torch.nn.utils import (\n  parameters_to_vector as Params2Vec,\n  vector_to_parameters as Vec2Params\n)\ntest = torch.from_numpy(np.random.rand(2, 3, 2))\ninfer_net = Model(configure)\ntheta_ast = Params2Vec(infer_net.parameters()).detach()\nprint(theta_ast.shape)\n'''","metadata":{"execution":{"iopub.status.busy":"2023-08-05T17:24:48.554655Z","iopub.execute_input":"2023-08-05T17:24:48.555069Z","iopub.status.idle":"2023-08-05T17:24:48.682121Z","shell.execute_reply.started":"2023-08-05T17:24:48.555035Z","shell.execute_reply":"2023-08-05T17:24:48.681068Z"},"trusted":true},"execution_count":null,"outputs":[]}]}