{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":41880,"databundleVersionId":5677426,"sourceType":"competition"},{"sourceId":10861506,"sourceType":"datasetVersion","datasetId":6747296}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Multimodal Deep Learning for Freezing of Gait Detection in Parkinson’s Disease Using  Wearable Sensors\n\nThis notebook implements a deep learning pipeline for detecting Freezing of Gait in Parkinson’s Disease patients using wearable sensor data. The pipeline includes:\n- Dataset loading and preprocessing (using sliding windows and z-score normalization)\n- data augmentation (Gaussian noise injection)\n- A deep learning model that combines CNN-based spatial feature extraction, BiLSTM-based temporal modeling, and an attention mechanism\n- Training, validation, and testing routines\n\nThe code is based on the research paper *\"Multimodal Deep Learning for Freezing of Gait Detection in Parkinson’s Disease Using Wearable Sensors\"*.\n","metadata":{}},{"cell_type":"markdown","source":"# Import Libraries (Code)\n This cell imports all the required libraries for data handling, model building, training, and evaluation.\n","metadata":{}},{"cell_type":"code","source":"import os\nimport glob\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader, Subset\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import train_test_split\nimport torch.optim as optim\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, roc_auc_score\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-26T14:54:15.371230Z","iopub.execute_input":"2025-02-26T14:54:15.371573Z","iopub.status.idle":"2025-02-26T14:54:18.245846Z","shell.execute_reply.started":"2025-02-26T14:54:15.371538Z","shell.execute_reply":"2025-02-26T14:54:18.244943Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# List Competition Files","metadata":{}},{"cell_type":"code","source":"# List all files in the competition directory\ncompetition_files = os.listdir('../input/tlvmc-parkinsons-freezing-gait-prediction')\nprint(\"Files available in the competition dataset:\")\nprint(competition_files)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-26T14:54:18.247005Z","iopub.execute_input":"2025-02-26T14:54:18.247451Z","iopub.status.idle":"2025-02-26T14:54:18.252845Z","shell.execute_reply.started":"2025-02-26T14:54:18.247426Z","shell.execute_reply":"2025-02-26T14:54:18.251926Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Utility Functions and Dataset Class\n This cell defines a sliding window function and a custom PyTorch Dataset class that works from a DataFrame.\n It uses z-score normalization and segments sensor data into overlapping windows.\n\n","metadata":{}},{"cell_type":"code","source":"def sliding_window(data, window_size, overlap):\n    \"\"\"\n    Splits the data into overlapping windows.\n    data: numpy array with shape (num_samples, num_features)\n    window_size: number of samples per window.\n    overlap: fraction overlap between consecutive windows (e.g., 0.5 for 50% overlap).\n    \"\"\"\n    step = int(window_size * (1 - overlap))\n    windows = []\n    for start in range(0, len(data) - window_size + 1, step):\n        windows.append(data[start:start + window_size])\n    return np.array(windows)\nclass FoGDatasetFromDF(Dataset):\n    def __init__(self, dataframe, window_size, overlap, sensor_columns, label_column, transform=None):\n        # Drop rows with missing label values\n        self.data = dataframe.dropna(subset=[label_column])\n        self.window_size = window_size\n        self.overlap = overlap\n        self.transform = transform\n        self.sensor_columns = sensor_columns\n        self.label_column = label_column\n        \n        sensor_data = self.data[self.sensor_columns].values\n        labels = self.data[self.label_column].values\n        \n        scaler = StandardScaler()\n        sensor_data = scaler.fit_transform(sensor_data)\n        \n        self.windows = sliding_window(sensor_data, window_size, overlap)\n        label_windows = sliding_window(labels, window_size, overlap)\n        \n        window_labels = []\n        for lw in label_windows:\n            unique, counts = np.unique(lw, return_counts=True)\n            majority_label = unique[np.argmax(counts)]\n            window_labels.append(majority_label)\n        window_labels = np.array(window_labels)\n        \n        # Filter out any windows where label is NaN\n        valid_idx = ~np.isnan(window_labels)\n        self.windows = self.windows[valid_idx]\n        self.window_labels = window_labels[valid_idx]\n        \n    def __len__(self):\n        return len(self.windows)\n    \n    def __getitem__(self, idx):\n        x = self.windows[idx]\n        y = int(self.window_labels[idx])  # Now guaranteed not to be NaN\n        x = torch.tensor(x, dtype=torch.float32).permute(1, 0)\n        y = torch.tensor(y, dtype=torch.long)\n        if self.transform:\n            x = self.transform(x)\n        return x, y\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-26T14:54:18.254695Z","iopub.execute_input":"2025-02-26T14:54:18.254919Z","iopub.status.idle":"2025-02-26T14:54:18.275196Z","shell.execute_reply.started":"2025-02-26T14:54:18.254900Z","shell.execute_reply":"2025-02-26T14:54:18.274375Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Augmentation Transform \n This cell defines an optional data augmentation transform that adds Gaussian noise to the sensor data.","metadata":{}},{"cell_type":"code","source":"class AddGaussianNoise(object):\n    def __init__(self, mean=0.0, std=0.05):\n        self.mean = mean\n        self.std = std\n        \n    def __call__(self, tensor):\n        noise = torch.randn(tensor.size()) * self.std + self.mean\n        return tensor + noise\n\n# Example transform; you can modify the std value as needed.\ntransform = AddGaussianNoise(mean=0.0, std=0.05)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-26T14:54:18.276962Z","iopub.execute_input":"2025-02-26T14:54:18.277367Z","iopub.status.idle":"2025-02-26T14:54:18.300527Z","shell.execute_reply.started":"2025-02-26T14:54:18.277336Z","shell.execute_reply":"2025-02-26T14:54:18.299288Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":" #  Load Dataset and Create DataLoaders\nIn this cell, we load the dataset using our custom FoGDataset class.\n We specify the CSV file path, sensor columns, label column, window size, and overlap.\n We then split the dataset into training, validation, and testing subsets using sklearn's train_test_split,\n and create DataLoaders for each subset.","metadata":{}},{"cell_type":"code","source":"# Set Kaggle dataset paths\nimport glob\nimport pandas as pd\n\n# Define file patterns for the training data\nDEF_TRAIN_PATH = \"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog/*.csv\"\nTDCS_TRAIN_PATH = \"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/*.csv\"\n\ndef load_files(pattern, label):\n    \"\"\"\n    Loads all CSV files matching the given pattern, assigns a fixed label,\n    and concatenates them into one DataFrame.\n    \n    Parameters:\n      pattern (str): File pattern to search for (e.g., using glob)\n      label (int): The label to assign to all rows in these files.\n      \n    Returns:\n      DataFrame: Combined data with an added \"label\" column.\n    \"\"\"\n    files = glob.glob(pattern)\n    df_list = []\n    for file in files:\n        df = pd.read_csv(file)\n        df[\"label\"] = label  # assign the provided label\n        df_list.append(df)\n    if df_list:\n        return pd.concat(df_list, ignore_index=True)\n    else:\n        return pd.DataFrame()\n\n# For example, assume that:\n# - Files from the \"defog\" folder represent FoG events (label = 1)\n# - Files from the \"tdcsfog\" folder represent non-FoG events (label = 0)\ndf_defog = load_files(DEF_TRAIN_PATH, label=1)\ndf_tdcs  = load_files(TDCS_TRAIN_PATH, label=0)\n\n# Combine the two DataFrames into one training DataFrame\ntrain_df = pd.concat([df_defog, df_tdcs], ignore_index=True)\nprint(\"Combined training data shape:\", train_df.shape)\nprint(train_df.head())\nMODEL_PATH = '/kaggle/working/defog_detection_model.h5'\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-26T14:54:18.301448Z","iopub.execute_input":"2025-02-26T14:54:18.301752Z","iopub.status.idle":"2025-02-26T14:54:41.437446Z","shell.execute_reply.started":"2025-02-26T14:54:18.301728Z","shell.execute_reply":"2025-02-26T14:54:41.436108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Update sensor column names based on your dataset structure.\nsensor_columns = [\"AccV\",\"AccML\",\"AccAP\",'StartHesitation',\n            'Turn', 'Walking']\nlabel_column = 'Valid'\n\nwindow_size = 100\noverlap = 0.5\n\ndataset = FoGDatasetFromDF(train_df, window_size, overlap, sensor_columns, label_column, transform=transform)\n\nprint(\"Dataset sample:\")\nprint(dataset.data.head())\n\n# Split dataset indices (70% train, 15% validation, 15% test)\nindices = np.arange(len(dataset))\ntrain_idx, test_idx = train_test_split(indices, test_size=0.3, random_state=42)\ntrain_idx, val_idx = train_test_split(train_idx, test_size=0.2, random_state=42)\n\ntrain_dataset = Subset(dataset, train_idx)\nval_dataset = Subset(dataset, val_idx)\ntest_dataset = Subset(dataset, test_idx)\n\nbatch_size = 32\ntrain_loader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=batch_size, shuffle=False)\ntest_loader = DataLoader(test_dataset, batch_size=batch_size, shuffle=False)\n\nprint(f\"Total windows: {len(dataset)}\")\nprint(f\"Training windows: {len(train_dataset)}\")\nprint(f\"Validation windows: {len(val_dataset)}\")\nprint(f\"Testing windows: {len(test_dataset)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-26T14:54:41.438384Z","iopub.execute_input":"2025-02-26T14:54:41.438673Z","iopub.status.idle":"2025-02-26T14:54:51.180757Z","shell.execute_reply.started":"2025-02-26T14:54:41.438648Z","shell.execute_reply":"2025-02-26T14:54:51.179575Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#  HTSAN Model Architecture\n In this cell, we define the model architecture.\n It includes:\n - A CNNBlock for spatial feature extraction with residual connections.\n - A TemporalModule using BiLSTM for capturing bidirectional temporal dependencies.\n - An Attention mechanism to compute a context vector from the BiLSTM outputs.\n - A ClassificationLayer to map the context vector to the final class logits.\n \n All these components are integrated into the complete HTSAN model.","metadata":{}},{"cell_type":"code","source":"# CNN Block for spatial feature extraction\nclass CNNBlock(nn.Module):\n    def __init__(self, in_channels, out_channels, kernel_size, stride=1, padding=0):\n        super(CNNBlock, self).__init__()\n        self.conv = nn.Conv1d(in_channels, out_channels, kernel_size, stride=stride, padding=padding)\n        self.bn = nn.BatchNorm1d(out_channels)\n        self.relu = nn.ReLU(inplace=False)  # ensure not in-place\n        self.res_conv = nn.Conv1d(in_channels, out_channels, kernel_size=1) if in_channels != out_channels else None\n        \n    def forward(self, x):\n        residual = x\n        out = self.conv(x)\n        out = self.bn(out)\n        out = self.relu(out)\n        if self.res_conv:\n            residual = self.res_conv(residual)\n        out = out + residual  # use out-of-place addition\n        out = self.relu(out)\n        return out\n\n\n# Temporal Module using BiLSTM\nclass TemporalModule(nn.Module):\n    def __init__(self, input_size, hidden_size, num_layers, dropout=0.2):\n        super(TemporalModule, self).__init__()\n        self.bilstm = nn.LSTM(input_size, hidden_size, num_layers=num_layers, \n                              batch_first=True, dropout=dropout, bidirectional=True)\n        \n    def forward(self, x):\n        out, _ = self.bilstm(x)\n        return out  # Shape: (batch, seq_len, 2*hidden_size)\n\n# Attention Mechanism\nclass Attention(nn.Module):\n    def __init__(self, hidden_dim):\n        super(Attention, self).__init__()\n        self.attn = nn.Linear(hidden_dim * 2, hidden_dim)\n        self.v = nn.Linear(hidden_dim, 1, bias=False)\n        \n    def forward(self, lstm_outputs):\n        attn_weights = torch.tanh(self.attn(lstm_outputs))\n        attn_weights = self.v(attn_weights).squeeze(-1)\n        attn_weights = F.softmax(attn_weights, dim=1)\n        context = torch.bmm(attn_weights.unsqueeze(1), lstm_outputs).squeeze(1)\n        return context, attn_weights\n\n# Classification Layer\nclass ClassificationLayer(nn.Module):\n    def __init__(self, input_dim, num_classes):\n        super(ClassificationLayer, self).__init__()\n        self.fc = nn.Linear(input_dim, num_classes)\n        \n    def forward(self, x):\n        return self.fc(x)\n\n# Complete HTSAN Model integrating all components\nclass HTSAN(nn.Module):\n    def __init__(self, cnn_in_channels, cnn_out_channels, cnn_kernel_size,\n                 lstm_hidden_size, lstm_num_layers, num_classes):\n        super(HTSAN, self).__init__()\n        self.cnn = CNNBlock(cnn_in_channels, cnn_out_channels, cnn_kernel_size, padding=cnn_kernel_size//2)\n        self.temporal = TemporalModule(input_size=cnn_out_channels, hidden_size=lstm_hidden_size, num_layers=lstm_num_layers)\n        self.attention = Attention(lstm_hidden_size)\n        self.classifier = ClassificationLayer(input_dim=2*lstm_hidden_size, num_classes=num_classes)\n        \n    def forward(self, x):\n        # x shape: (batch, channels, seq_length)\n        spatial_features = self.cnn(x)\n        # Permute to (batch, seq_length, feature_dim) for LSTM\n        temporal_input = spatial_features.permute(0, 2, 1)\n        lstm_out = self.temporal(temporal_input)\n        context, attn_weights = self.attention(lstm_out)\n        logits = self.classifier(context)\n        return logits, attn_weights\n\n# Instantiate the model with appropriate parameters.\nmodel = HTSAN(cnn_in_channels=6,       # For example: 3-axis accelerometer + 3-axis gyroscope\n              cnn_out_channels=64,\n              cnn_kernel_size=3,\n              lstm_hidden_size=128,\n              lstm_num_layers=2,\n              num_classes=2)           # FoG vs. non-FoG\n\nprint(model)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-26T14:54:51.181979Z","iopub.execute_input":"2025-02-26T14:54:51.182281Z","iopub.status.idle":"2025-02-26T14:54:51.202203Z","shell.execute_reply.started":"2025-02-26T14:54:51.182250Z","shell.execute_reply":"2025-02-26T14:54:51.201052Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training the Model\n This cell implements the training loop.\nWe define the loss function (CrossEntropyLoss) and the optimizer (Adam).\n The training loop iterates over the training DataLoader, computes the loss, and updates model parameters.\n Validation loss is computed after each epoch and early stopping is applied if validation loss does not improve.\n","metadata":{}},{"cell_type":"code","source":"torch.autograd.set_detect_anomaly(True)\ncriterion = nn.CrossEntropyLoss()\noptimizer = optim.Adam(model.parameters(), lr=0.001)\n\nnum_epochs = 50\nbest_val_loss = float('inf')\npatience = 5\ntrigger_times = 0\n\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nmodel.to(device)\n\nfor epoch in range(num_epochs):\n    model.train()\n    running_loss = 0.0\n    for inputs, labels in train_loader:\n        inputs, labels = inputs.to(device), labels.to(device)\n        optimizer.zero_grad()\n        outputs, _ = model(inputs)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n        running_loss += loss.item() * inputs.size(0)\n    epoch_loss = running_loss / len(train_loader.dataset)\n    \n    model.eval()\n    val_loss = 0.0\n    with torch.no_grad():\n        for inputs, labels in val_loader:\n            inputs, labels = inputs.to(device), labels.to(device)\n            outputs, _ = model(inputs)\n            loss = criterion(outputs, labels)\n            val_loss += loss.item() * inputs.size(0)\n    val_loss /= len(val_loader.dataset)\n    \n    print(f\"Epoch {epoch+1}/{num_epochs} | Train Loss: {epoch_loss:.4f} | Val Loss: {val_loss:.4f}\")\n    \n    if val_loss < best_val_loss:\n        best_val_loss = val_loss\n        trigger_times = 0\n        torch.save(model.state_dict(), 'best_model.pth')\n    else:\n        trigger_times += 1\n        if trigger_times >= patience:\n            print(\"Early stopping triggered.\")\n            break\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-26T14:54:51.204645Z","iopub.execute_input":"2025-02-26T14:54:51.204955Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Evaluation on Test Set\n After training, this cell loads the best saved model weights and evaluates the model on the test set.\n We compute Accuracy, Precision, Recall, F1 Score, and AUC (Area Under the ROC Curve) as evaluation metrics.\n","metadata":{}},{"cell_type":"code","source":"model.load_state_dict(torch.load('best_model.pth'))\nmodel.eval()\n\nall_preds = []\nall_labels = []\n\nwith torch.no_grad():\n    for inputs, labels in test_loader:\n        inputs = inputs.to(device)\n        outputs, _ = model(inputs)\n        preds = torch.argmax(outputs, dim=1).cpu().numpy()\n        all_preds.extend(preds)\n        all_labels.extend(labels.numpy())\n\n\naccuracy = accuracy_score(all_labels, all_preds)\nprecision = precision_score(all_labels, all_preds, zero_division=0)\nrecall = recall_score(all_labels, all_preds, zero_division=0)\nf1 = f1_score(all_labels, all_preds, zero_division=0)\nauc = roc_auc_score(all_labels, all_probs)\n\nprint(\"Evaluation Metrics:\")\nprint(f\"Accuracy:  {accuracy:.4f}\")\nprint(f\"Precision: {precision:.4f}\")\nprint(f\"Recall:    {recall:.4f}\")\nprint(f\"F1 Score:  {f1:.4f}\")\nprint(f\"AUC:       {auc:.4f}\")\n\nall_probs = []\nwith torch.no_grad():\n    for inputs, _ in test_loader:\n        inputs = inputs.to(device)\n        outputs, _ = model(inputs)\n        probs = F.softmax(outputs, dim=1)[:, 1].cpu().numpy()\n        all_probs.extend(probs)\n\nauc = roc_auc_score(all_labels, all_probs)\nprint(f\"Test AUC: {auc:.4f}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}