{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":41880,"databundleVersionId":5677426},{"sourceType":"datasetVersion","sourceId":10861506,"datasetId":6747296,"databundleVersionId":11226158}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport glob\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader, Subset\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import train_test_split\nimport torch.optim as optim\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, roc_auc_score\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-27T04:58:55.993761Z","iopub.execute_input":"2026-02-27T04:58:55.994069Z","iopub.status.idle":"2026-02-27T04:58:55.998511Z","shell.execute_reply.started":"2026-02-27T04:58:55.994045Z","shell.execute_reply":"2026-02-27T04:58:55.997702Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# List all files in the competition directory\ncompetition_files = os.listdir('../input/tlvmc-parkinsons-freezing-gait-prediction')\nprint(\"Files available in the competition dataset:\")\nprint(competition_files)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-27T04:58:56.009533Z","iopub.execute_input":"2026-02-27T04:58:56.009912Z","iopub.status.idle":"2026-02-27T04:58:56.016021Z","shell.execute_reply.started":"2026-02-27T04:58:56.009888Z","shell.execute_reply":"2026-02-27T04:58:56.015119Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def sliding_window(data, window_size, overlap):\n    \"\"\"\n    Splits the data into overlapping windows.\n    data: numpy array with shape (num_samples, num_features)\n    window_size: number of samples per window.\n    overlap: fraction overlap between consecutive windows (e.g., 0.5 for 50% overlap).\n    \"\"\"\n    step = int(window_size * (1 - overlap))\n    windows = []\n    for start in range(0, len(data) - window_size + 1, step):\n        windows.append(data[start:start + window_size])\n    return np.array(windows)\nclass FoGDatasetFromDF(Dataset):\n    def __init__(self, dataframe, window_size, overlap, sensor_columns, label_column, transform=None):\n        # Drop rows with missing label values\n        self.data = dataframe.dropna(subset=[label_column])\n        self.window_size = window_size\n        self.overlap = overlap\n        self.transform = transform\n        self.sensor_columns = sensor_columns\n        self.label_column = label_column\n        \n        sensor_data = self.data[self.sensor_columns].values\n        labels = self.data[self.label_column].values\n        \n        scaler = StandardScaler()\n        sensor_data = scaler.fit_transform(sensor_data)\n        \n        self.windows = sliding_window(sensor_data, window_size, overlap)\n        label_windows = sliding_window(labels, window_size, overlap)\n        \n        window_labels = []\n        for lw in label_windows:\n            unique, counts = np.unique(lw, return_counts=True)\n            majority_label = unique[np.argmax(counts)]\n            window_labels.append(majority_label)\n        window_labels = np.array(window_labels)\n        \n        # Filter out any windows where label is NaN\n        valid_idx = ~np.isnan(window_labels)\n        self.windows = self.windows[valid_idx]\n        self.window_labels = window_labels[valid_idx]\n        \n    def __len__(self):\n        return len(self.windows)\n    \n    def __getitem__(self, idx):\n        x = self.windows[idx]\n        y = int(self.window_labels[idx])  # Now guaranteed not to be NaN\n        x = torch.tensor(x, dtype=torch.float32).permute(1, 0)\n        y = torch.tensor(y, dtype=torch.long)\n        if self.transform:\n            x = self.transform(x)\n        return x, y\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-27T04:58:56.018655Z","iopub.execute_input":"2026-02-27T04:58:56.018866Z","iopub.status.idle":"2026-02-27T04:58:56.030110Z","shell.execute_reply.started":"2026-02-27T04:58:56.018849Z","shell.execute_reply":"2026-02-27T04:58:56.029364Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class AddGaussianNoise(object):\n    def __init__(self, mean=0.0, std=0.05):\n        self.mean = mean\n        self.std = std\n        \n    def __call__(self, tensor):\n        noise = torch.randn(tensor.size()) * self.std + self.mean\n        return tensor + noise\n\n# Example transform; you can modify the std value as needed.\ntransform = AddGaussianNoise(mean=0.0, std=0.05)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-27T04:58:56.052126Z","iopub.execute_input":"2026-02-27T04:58:56.052359Z","iopub.status.idle":"2026-02-27T04:58:56.057103Z","shell.execute_reply.started":"2026-02-27T04:58:56.052340Z","shell.execute_reply":"2026-02-27T04:58:56.056263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set Kaggle dataset paths\nimport glob\nimport pandas as pd\n\n# Define file patterns for the training data\nDEF_TRAIN_PATH = \"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog/*.csv\"\nTDCS_TRAIN_PATH = \"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/*.csv\"\n\ndef load_files(pattern, label):\n    \"\"\"\n    Loads all CSV files matching the given pattern, assigns a fixed label,\n    and concatenates them into one DataFrame.\n    \n    Parameters:\n      pattern (str): File pattern to search for (e.g., using glob)\n      label (int): The label to assign to all rows in these files.\n      \n    Returns:\n      DataFrame: Combined data with an added \"label\" column.\n    \"\"\"\n    files = glob.glob(pattern)\n    df_list = []\n    for file in files:\n        df = pd.read_csv(file)\n        df[\"label\"] = label  # assign the provided label\n        df_list.append(df)\n    if df_list:\n        return pd.concat(df_list, ignore_index=True)\n    else:\n        return pd.DataFrame()\n\n# For example, assume that:\n# - Files from the \"defog\" folder represent FoG events (label = 1)\n# - Files from the \"tdcsfog\" folder represent non-FoG events (label = 0)\ndf_defog = load_files(DEF_TRAIN_PATH, label=1)\ndf_tdcs  = load_files(TDCS_TRAIN_PATH, label=0)\n\n# Combine the two DataFrames into one training DataFrame\ntrain_df = pd.concat([df_defog, df_tdcs], ignore_index=True)\nprint(\"Combined training data shape:\", train_df.shape)\nprint(train_df.head())\nMODEL_PATH = '/kaggle/working/defog_detection_model.h5'\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-27T04:58:56.058185Z","iopub.execute_input":"2026-02-27T04:58:56.058460Z","iopub.status.idle":"2026-02-27T04:59:16.019157Z","shell.execute_reply.started":"2026-02-27T04:58:56.058440Z","shell.execute_reply":"2026-02-27T04:59:16.018303Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Update sensor column names based on your dataset structure.\nsensor_columns = [\"AccV\",\"AccML\",\"AccAP\",'StartHesitation',\n            'Turn', 'Walking']\nlabel_column = 'Valid'\n\nwindow_size = 100\noverlap = 0.5\n\ndataset = FoGDatasetFromDF(train_df, window_size, overlap, sensor_columns, label_column, transform=transform)\n\nprint(\"Dataset sample:\")\nprint(dataset.data.head())\n\n# Split dataset indices (70% train, 15% validation, 15% test)\nindices = np.arange(len(dataset))\ntrain_idx, test_idx = train_test_split(indices, test_size=0.3, random_state=42)\ntrain_idx, val_idx = train_test_split(train_idx, test_size=0.2, random_state=42)\n\ntrain_dataset = Subset(dataset, train_idx)\nval_dataset = Subset(dataset, val_idx)\ntest_dataset = Subset(dataset, test_idx)\n\nbatch_size = 64\ntrain_loader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=batch_size, shuffle=False)\ntest_loader = DataLoader(test_dataset, batch_size=batch_size, shuffle=False)\n\nprint(f\"Total windows: {len(dataset)}\")\nprint(f\"Training windows: {len(train_dataset)}\")\nprint(f\"Validation windows: {len(val_dataset)}\")\nprint(f\"Testing windows: {len(test_dataset)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-27T04:59:16.020939Z","iopub.execute_input":"2026-02-27T04:59:16.021180Z","iopub.status.idle":"2026-02-27T04:59:28.586551Z","shell.execute_reply.started":"2026-02-27T04:59:16.021159Z","shell.execute_reply":"2026-02-27T04:59:28.585663Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CNN Block for spatial feature extraction\nclass CNNBlock(nn.Module):\n    def __init__(self, in_channels, out_channels, kernel_size, stride=1, padding=0):\n        super(CNNBlock, self).__init__()\n        self.conv = nn.Conv1d(in_channels, out_channels, kernel_size, stride=stride, padding=padding)\n        self.bn = nn.BatchNorm1d(out_channels)\n        self.relu = nn.ReLU(inplace=False)  # ensure not in-place\n        self.res_conv = nn.Conv1d(in_channels, out_channels, kernel_size=1) if in_channels != out_channels else None\n        \n    def forward(self, x):\n        residual = x\n        out = self.conv(x)\n        out = self.bn(out)\n        out = self.relu(out)\n        if self.res_conv:\n            residual = self.res_conv(residual)\n        out = out + residual  # use out-of-place addition\n        out = self.relu(out)\n        return out\n\n\n# Temporal Module using BiLSTM\nclass TemporalModule(nn.Module):\n    def __init__(self, input_size, hidden_size, num_layers, dropout=0.2):\n        super(TemporalModule, self).__init__()\n        self.bilstm = nn.LSTM(input_size, hidden_size, num_layers=num_layers, \n                              batch_first=True, dropout=dropout, bidirectional=True)\n        \n    def forward(self, x):\n        out, _ = self.bilstm(x)\n        return out  # Shape: (batch, seq_len, 2*hidden_size)\n\n# Attention Mechanism\nclass Attention(nn.Module):\n    def __init__(self, hidden_dim):\n        super(Attention, self).__init__()\n        self.attn = nn.Linear(hidden_dim * 2, hidden_dim)\n        self.v = nn.Linear(hidden_dim, 1, bias=False)\n        \n    def forward(self, lstm_outputs):\n        attn_weights = torch.tanh(self.attn(lstm_outputs))\n        attn_weights = self.v(attn_weights).squeeze(-1)\n        attn_weights = F.softmax(attn_weights, dim=1)\n        context = torch.bmm(attn_weights.unsqueeze(1), lstm_outputs).squeeze(1)\n        return context, attn_weights\n\n# Classification Layer\nclass ClassificationLayer(nn.Module):\n    def __init__(self, input_dim, num_classes):\n        super(ClassificationLayer, self).__init__()\n        self.fc = nn.Linear(input_dim, num_classes)\n        \n    def forward(self, x):\n        return self.fc(x)\n\n# Complete GAIT Model integrating all components\nclass GAIT(nn.Module):\n    def __init__(self, cnn_in_channels, cnn_out_channels, cnn_kernel_size,\n                 lstm_hidden_size, lstm_num_layers, num_classes):\n        super(GAIT, self).__init__()\n        self.cnn = CNNBlock(cnn_in_channels, cnn_out_channels, cnn_kernel_size, padding=cnn_kernel_size//2)\n        self.temporal = TemporalModule(input_size=cnn_out_channels, hidden_size=lstm_hidden_size, num_layers=lstm_num_layers)\n        self.attention = Attention(lstm_hidden_size)\n        self.classifier = ClassificationLayer(input_dim=2*lstm_hidden_size, num_classes=num_classes)\n        \n    def forward(self, x):\n        # x shape: (batch, channels, seq_length)\n        spatial_features = self.cnn(x)\n        # Permute to (batch, seq_length, feature_dim) for LSTM\n        temporal_input = spatial_features.permute(0, 2, 1)\n        lstm_out = self.temporal(temporal_input)\n        context, attn_weights = self.attention(lstm_out)\n        logits = self.classifier(context)\n        return logits, attn_weights\n\n# Instantiate the model with appropriate parameters.\nmodel = GAIT(cnn_in_channels=6,       # For example: 3-axis accelerometer + 3-axis gyroscope\n              cnn_out_channels=64,\n              cnn_kernel_size=3,\n              lstm_hidden_size=128,\n              lstm_num_layers=2,\n              num_classes=2)           # FoG vs. non-FoG\n\nprint(model)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-27T04:59:28.587940Z","iopub.execute_input":"2026-02-27T04:59:28.588294Z","iopub.status.idle":"2026-02-27T04:59:28.643598Z","shell.execute_reply.started":"2026-02-27T04:59:28.588261Z","shell.execute_reply":"2026-02-27T04:59:28.642912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel.to(device)\n\n\n# SAFE BASELINE CONFIG\n\ncriterion = nn.CrossEntropyLoss()\noptimizer = optim.Adam(model.parameters(), lr=0.001)\n\nnum_epochs = 35\npatience = 5\nmin_delta = 0.001\n\nbest_val_loss = float('inf')\ntrigger_times = 0\n\nfor epoch in range(num_epochs):\n\n    #  TRAIN \n    model.train()\n    train_loss = 0.0\n    train_correct = 0\n    train_total = 0\n\n    for inputs, labels in train_loader:\n        inputs, labels = inputs.to(device), labels.to(device)\n\n        optimizer.zero_grad()\n        outputs, _ = model(inputs)\n        loss = criterion(outputs, labels)\n\n        loss.backward()\n        optimizer.step()\n\n        train_loss += loss.item() * inputs.size(0)\n\n        preds = torch.argmax(outputs, dim=1)\n        train_correct += (preds == labels).sum().item()\n        train_total += labels.size(0)\n\n    train_loss /= len(train_loader.dataset)\n    train_acc = train_correct / train_total\n\n    #  VALIDATION \n    model.eval()\n    val_loss = 0.0\n    val_correct = 0\n    val_total = 0\n\n    with torch.no_grad():\n        for inputs, labels in val_loader:\n            inputs, labels = inputs.to(device), labels.to(device)\n\n            outputs, _ = model(inputs)\n            loss = criterion(outputs, labels)\n            val_loss += loss.item() * inputs.size(0)\n\n            preds = torch.argmax(outputs, dim=1)\n            val_correct += (preds == labels).sum().item()\n            val_total += labels.size(0)\n\n    val_loss /= len(val_loader.dataset)\n    val_acc = val_correct / val_total\n\n    print(f\"Epoch {epoch+1}/{num_epochs} | \"\n          f\"Train Loss: {train_loss:.4f} | Train Acc: {train_acc:.4f} | \"\n          f\"Val Loss: {val_loss:.4f} | Val Acc: {val_acc:.4f}\")\n\n    #  EARLY STOPPING \n    if val_loss < best_val_loss - min_delta:\n        best_val_loss = val_loss\n        trigger_times = 0\n        torch.save(model.state_dict(), \"best_model.pth\")\n    else:\n        trigger_times += 1\n        if trigger_times >= patience:\n            print(\"Early stopping triggered\")\n            break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-27T04:59:28.644428Z","iopub.execute_input":"2026-02-27T04:59:28.644665Z","iopub.status.idle":"2026-02-27T05:22:06.959437Z","shell.execute_reply.started":"2026-02-27T04:59:28.644646Z","shell.execute_reply":"2026-02-27T05:22:06.958656Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.load_state_dict(torch.load('best_model.pth'))\nmodel.eval()\n\nall_preds = []\nall_labels = []\n\nwith torch.no_grad():\n    for inputs, labels in test_loader:\n        inputs = inputs.to(device)\n        outputs, _ = model(inputs)\n        preds = torch.argmax(outputs, dim=1).cpu().numpy()\n        all_preds.extend(preds)\n        all_labels.extend(labels.numpy())\n\n\naccuracy = accuracy_score(all_labels, all_preds)\nprecision = precision_score(all_labels, all_preds, zero_division=0)\nrecall = recall_score(all_labels, all_preds, zero_division=0)\nf1 = f1_score(all_labels, all_preds, zero_division=0)\n#auc = roc_auc_score(all_labels, all_probs)\n\nprint(\"Evaluation Metrics:\")\nprint(f\"Accuracy:  {accuracy:.4f}\")\nprint(f\"Precision: {precision:.4f}\")\nprint(f\"Recall:    {recall:.4f}\")\nprint(f\"F1 Score:  {f1:.4f}\")\n#print(f\"AUC:       {auc:.4f}\")\n\nall_probs = []\nwith torch.no_grad():\n    for inputs, _ in test_loader:\n        inputs = inputs.to(device)\n        outputs, _ = model(inputs)\n        probs = F.softmax(outputs, dim=1)[:, 1].cpu().numpy()\n        all_probs.extend(probs)\n\nauc = roc_auc_score(all_labels, all_probs)\nprint(f\"Test AUC: {auc:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-27T05:22:06.960380Z","iopub.execute_input":"2026-02-27T05:22:06.960848Z","iopub.status.idle":"2026-02-27T05:22:32.166646Z","shell.execute_reply.started":"2026-02-27T05:22:06.960823Z","shell.execute_reply":"2026-02-27T05:22:32.165874Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n\ncm = confusion_matrix(all_labels, all_preds)\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nplt.figure(figsize=(5,4))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues')\n\nplt.xlabel(\"Predicted\")\nplt.ylabel(\"Actual\")\nplt.title(\"Confusion Matrix\")\n\n# Optional labels for FoG\nplt.xticks([0.5,1.5], [\"Normal\", \"FoG\"])\nplt.yticks([0.5,1.5], [\"Normal\", \"FoG\"])\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-27T05:22:32.167402Z","iopub.execute_input":"2026-02-27T05:22:32.167706Z","iopub.status.idle":"2026-02-27T05:22:32.752832Z","shell.execute_reply.started":"2026-02-27T05:22:32.167680Z","shell.execute_reply":"2026-02-27T05:22:32.751937Z"}},"outputs":[],"execution_count":null}]}