{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2023-06-09T10:29:35.854208Z","iopub.execute_input":"2023-06-09T10:29:35.854623Z","iopub.status.idle":"2023-06-09T10:29:35.861672Z","shell.execute_reply.started":"2023-06-09T10:29:35.854594Z","shell.execute_reply":"2023-06-09T10:29:35.859727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport itertools\nfrom sklearn.model_selection import train_test_split\nimport torch\nfrom torch.utils.data import Dataset, DataLoader\nimport pytorch_lightning as pl\nfrom torch import nn\nfrom torch.nn import functional as F","metadata":{"execution":{"iopub.status.busy":"2023-06-09T10:29:35.864999Z","iopub.execute_input":"2023-06-09T10:29:35.867014Z","iopub.status.idle":"2023-06-09T10:29:35.878505Z","shell.execute_reply.started":"2023-06-09T10:29:35.866946Z","shell.execute_reply":"2023-06-09T10:29:35.876833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class SegmentedDataset(Dataset):\n    def __init__(self, folder_path, files, is_train=True, segment_length=256):\n        self.folder_path = folder_path\n        self.is_train = is_train\n        self.segment_length = segment_length\n        self.data = []\n        self.time = []\n        self.ids = []\n        self.targets = []\n\n        # Read and segment all data at initialization\n        for file_name in files:\n            file_path = os.path.join(folder_path, file_name)\n            data = pd.read_csv(file_path)\n            file_id = file_name.replace('.csv', '')\n\n            time = data['Time'].values\n            features = data[['AccV', 'AccML', 'AccAP']].values\n            # add here more features\n            \n            if self.is_train:\n                targets = data[['StartHesitation', 'Turn', 'Walking']].values\n                # add extra target column\n                targets = np.concatenate([targets, (1 - targets.sum(axis=1)).reshape(-1,1)], axis=1)\n                \n                self.segment_and_pad_data(file_id, features, time, targets)\n            else:\n                self.segment_and_pad_data(file_id, features, time)\n\n    def segment_and_pad_data(self, file_id, features, time, targets=None):\n        num_segments = len(features) // self.segment_length\n        remainder = len(features) % self.segment_length\n\n        # First, handle the full segments\n        for i in range(num_segments):\n            feature_segment = features[i*self.segment_length:(i+1)*self.segment_length]\n            self.data.append(feature_segment)\n            \n            time_segment = time[i*self.segment_length:(i+1)*self.segment_length]\n            self.time.append(time_segment)\n            \n            if self.is_train:\n                target_segment = targets[i*self.segment_length:(i+1)*self.segment_length]\n                self.targets.append(target_segment)\n                \n            self.ids.append(file_id)\n\n        # Now handle the last segment if it's shorter than segment_length\n        if remainder > 0:\n            padding_length = self.segment_length - remainder\n\n            feature_segment = np.pad(features[-remainder:], ((0, padding_length), (0, 0)), mode='constant')\n            self.data.append(feature_segment)\n            \n            time_segment = np.pad(time[-remainder:], (0, padding_length), mode='constant', constant_values=-1)\n\n            self.time.append(time_segment)\n            \n            if self.is_train:\n                target_segment = np.pad(targets[-remainder:], ((0, padding_length), (0, 0)), mode='constant')\n                self.targets.append(target_segment)\n                \n            self.ids.append(file_id)\n\n\n    def get_segment_indices(self):\n        return self.segment_indices\n\n    def get_segment_padding(self):\n        return self.segment_padding\n\n    def __len__(self):\n        return len(self.data)\n\n    def __getitem__(self, idx):\n        if self.is_train:\n            return torch.Tensor(self.data[idx]), torch.Tensor(np.argmax(self.targets[idx],axis=1)).to(torch.int64)\n        else:\n            return torch.Tensor(self.data[idx]), self.ids[idx], self.time[idx]","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-09T10:29:35.880513Z","iopub.execute_input":"2023-06-09T10:29:35.881478Z","iopub.status.idle":"2023-06-09T10:29:35.906627Z","shell.execute_reply.started":"2023-06-09T10:29:35.881436Z","shell.execute_reply":"2023-06-09T10:29:35.905166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ConvBlock(nn.Module):\n    def __init__(self, in_channels, out_channels, kernel_sizes, dropout_rate, dilations):\n        super(ConvBlock, self).__init__()\n\n        self.convs = nn.ModuleList([\n            nn.Conv1d(in_channels, out_channels, kernel_size, padding=(kernel_size+(kernel_size -1)*(dilation-1))//2, dilation=dilation)\n            for kernel_size, dilation in itertools.product(kernel_sizes, dilations)\n        ])\n\n        self.batch_norm = nn.BatchNorm1d(out_channels * len(kernel_sizes)*len(dilations))\n        self.dropout = nn.Dropout(dropout_rate)\n\n        self.skip_connection = nn.Conv1d(in_channels, out_channels * len(kernel_sizes)*len(dilations), kernel_size=1) \\\n            if in_channels != out_channels * len(kernel_sizes)*len(dilations) else nn.Identity()\n\n\n    def forward(self, x):\n        x = x.transpose(1, 2)\n        skip = self.skip_connection(x)\n\n        x = torch.cat([conv(x) for conv in self.convs], dim=1)\n\n        x = self.batch_norm(x + skip)\n        x = F.relu(x)\n        x = self.dropout(x)\n        x = x.transpose(1, 2)\n\n        return x\n        \n\nclass Model(pl.LightningModule):\n    def __init__(self, in_channels, out_channels, kernel_sizes, dilations, dropout_rate, num_blocks, lr=0.001):\n        super(Model, self).__init__()\n        self.train_loss_history = []\n        self.val_loss_history = []\n        self.val_score_history = []\n        self.lr = lr\n        self.blocks = nn.Sequential(*[\n            ConvBlock(in_channels if i == 0 else out_channels * len(kernel_sizes)*len(dilations), \n                      out_channels, \n                      kernel_sizes, \n                      dropout_rate,\n                      dilations)\n            for i in range(num_blocks)\n        ])\n\n        num_classes = 4\n        self.linear = nn.Linear(out_channels * len(kernel_sizes)*len(dilations), num_classes)\n        weights = torch.tensor([1.0, 1.0, 1.0, 0.1])\n        self.loss = torch.nn.NLLLoss(weight=weights)\n        self.logsoftmax = nn.LogSoftmax(dim=2)\n        self.output = []\n\n    def forward(self, x):\n        x = self.blocks(x)\n        x = self.linear(x)\n        return x\n\n    def training_step(self, batch, batch_idx):\n        x, y = batch\n        y_hat = self(x)\n        y_hat = self.logsoftmax(y_hat)\n        \n        # Reshape the outputs and labels\n        loss = self.loss(y_hat.view(-1, y_hat.size(-1)), y.view(-1))\n        \n        self.log('train_loss', loss, on_step=False, on_epoch=True, prog_bar=True, logger=True)\n        self.train_loss_history.append(loss.item())\n\n        return loss\n    \n    def validation_step(self, batch, batch_idx):\n        x, y = batch\n        y_hat = self(x)\n        \n        # Reshape the outputs and labels\n        loss = self.loss(y_hat.view(-1, y_hat.size(-1)), y.view(-1))\n        self.log('val_loss', loss)\n        return loss\n    \n    def validation_step(self, batch, batch_idx):\n        x, y = batch\n        y_hat = self(x)\n        y_hat = self.logsoftmax(y_hat)\n        \n        # Reshape the outputs and labels\n        loss = self.loss(y_hat.view(-1, y_hat.size(-1)), y.view(-1))\n        self.log('val_loss', loss)\n        return loss\n    \n    def test_step(self, batch, batch_idx):\n        x, file_id, time = batch\n        y_hat = self.forward(x)\n        y_hat = F.softmax(y_hat, dim=2) \n        self.output.append((y_hat, file_id, time))\n        \n    def configure_optimizers(self):\n        optimizer = torch.optim.Adam(self.parameters(), lr=self.lr)\n        return optimizer","metadata":{"execution":{"iopub.status.busy":"2023-06-09T10:29:35.910040Z","iopub.execute_input":"2023-06-09T10:29:35.911694Z","iopub.status.idle":"2023-06-09T10:29:35.941559Z","shell.execute_reply.started":"2023-06-09T10:29:35.911579Z","shell.execute_reply":"2023-06-09T10:29:35.939923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"folder_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/'\nall_files = [file for file in os.listdir(folder_path)]\ntrain_files, val_files = train_test_split(all_files, random_state=13)\n\n\ntrain_dataset = SegmentedDataset(folder_path, train_files)\nval_dataset = SegmentedDataset(folder_path, val_files)\n\ntrain_loader = DataLoader(train_dataset, batch_size=64)\nval_loader = DataLoader(val_dataset, batch_size=64)","metadata":{"execution":{"iopub.status.busy":"2023-06-09T10:29:35.943324Z","iopub.execute_input":"2023-06-09T10:29:35.944011Z","iopub.status.idle":"2023-06-09T10:29:43.088297Z","shell.execute_reply.started":"2023-06-09T10:29:35.943974Z","shell.execute_reply":"2023-06-09T10:29:43.085956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Model(\n    in_channels=3,\n    out_channels=3,\n    kernel_sizes=[3,5,7],\n    dilations=[2,4,8],\n    dropout_rate=0.1,\n    num_blocks=1,\n)\n\ntrainer = pl.Trainer(max_epochs=1)\ntrainer.fit(model, train_loader, val_loader)","metadata":{"execution":{"iopub.status.busy":"2023-06-09T10:29:43.089554Z","iopub.status.idle":"2023-06-09T10:29:43.089992Z","shell.execute_reply.started":"2023-06-09T10:29:43.089782Z","shell.execute_reply":"2023-06-09T10:29:43.089801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-06-09T10:29:43.092894Z","iopub.status.idle":"2023-06-09T10:29:43.093637Z","shell.execute_reply.started":"2023-06-09T10:29:43.093295Z","shell.execute_reply":"2023-06-09T10:29:43.093327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_folder_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog/'\ntest_files = [file for file in os.listdir(test_folder_path)]\ntest_dataset = SegmentedDataset(test_folder_path, test_files, is_train=False)\ntest_loader = DataLoader(test_dataset, batch_size=64, shuffle=False)\n\ntrainer.test(model, test_loader)","metadata":{"execution":{"iopub.status.busy":"2023-06-09T10:29:43.095933Z","iopub.status.idle":"2023-06-09T10:29:43.096632Z","shell.execute_reply.started":"2023-06-09T10:29:43.096315Z","shell.execute_reply":"2023-06-09T10:29:43.096344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result = pd.DataFrame()\nfor batch in model.output:\n    preds = batch[0].cpu().numpy()\n    ids = batch[1]\n    times = batch[2].cpu().numpy()\n    for i, time, pred in zip(ids, times, preds):\n        id_time = [f'{i}_{t}' for t in time]\n        tmp = pd.DataFrame(pred[:,:3], index=id_time).loc[[t != -1 for t in time]]\n        result = pd.concat([result, tmp])\n        \nresult.columns = ['StartHesitation', 'Turn', 'Walking']\n\nsample_submission = sample_submission.set_index('Id')\nsample_submission.update(result)\nsample_submission.reset_index().to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-06-09T10:29:43.099113Z","iopub.status.idle":"2023-06-09T10:29:43.099830Z","shell.execute_reply.started":"2023-06-09T10:29:43.099493Z","shell.execute_reply":"2023-06-09T10:29:43.099524Z"},"trusted":true},"execution_count":null,"outputs":[]}]}