{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":41880,"databundleVersionId":5677426,"sourceType":"competition"}],"dockerImageVersionId":30747,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Project Organization:\n\nDue 01.09.2024\n","metadata":{}},{"cell_type":"markdown","source":"### Data Loading and Standartization","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom sklearn.preprocessing import StandardScaler\n\nnum_of_subjects = 30\n\ndef load_csv_files_from_directory(directory_path, size):\n    \"\"\"\n    Loads and concatenates all CSV files from a specified directory into a single DataFrame.\n\n    Args:\n    directory_path (str): Path to the directory containing CSV files.\n\n    Returns:\n    pd.DataFrame: Concatenated DataFrame containing data from all CSV files in the directory.\n    \"\"\"\n    csv_files = [os.path.join(directory_path, file) for file in os.listdir(directory_path) if file.endswith('.csv')]\n    data_frames = []\n    for csv_file in csv_files[:size]:\n        tmp_df = pd.read_csv(csv_file)\n        tmp_df['subject'] = csv_file.split('/')[-1]\n        data_frames.append(tmp_df)\n    \n    return pd.concat(data_frames, ignore_index=True)\n\ndef normalize_data(df, exclude_columns):\n    \"\"\"\n    Normalizes all numerical columns in a DataFrame except the excluded columns using StandardScaler.\n\n    Args:\n    df (pd.DataFrame): DataFrame with numerical features.\n    exclude_columns (list): List of column names to exclude from normalization.\n\n    Returns:\n    pd.DataFrame: DataFrame with normalized features.\n    \"\"\"\n    # Select columns that are numeric and not in the exclude list\n    numeric_columns = df.select_dtypes(include=['float64', 'int64']).columns\n    columns_to_normalize = [col for col in numeric_columns if col not in exclude_columns]\n    \n    # Apply normalization to the selected columns\n    scaler = StandardScaler()\n    df[columns_to_normalize] = scaler.fit_transform(df[columns_to_normalize])\n    \n    return df\n\n# Define paths\ntdcsfog_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog'\ndefog_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog'\n\n\n# Load data\ntdcsfog_data = load_csv_files_from_directory(tdcsfog_path, 20)\ndefog_data = load_csv_files_from_directory(defog_path, 2)\n\nprint(f\"{tdcsfog_data.shape[0]:,} timestamps in tdcsfog_data\")\nprint(f\"{defog_data.shape[0]:,} timestamps in defog_data\")\n\n\nprint(f\"{len(set(tdcsfog_data.subject.tolist())):,} unique Subjects in tdcsfog_data\")\nprint(f\"{len(set(defog_data.subject.tolist())):,} unique Subjects in defog_data\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add a new column to each DataFrame to indicate the source\ntdcsfog_data['source'] = 'tdcsfog'\ndefog_data['source'] = 'defog'\n\n# Combine both DataFrames into one\n#combined_data = pd.concat([tdcsfog_data, defog_data],axis=0, ignore_index=True)\n\n# Specify columns to exclude from normalization\nexclude_columns = ['subject', 'Time', 'Valid', 'Task', 'source', 'StartHesitation', 'Turn', 'Walking']\n\n# Normalize the features, excluding specific columns\nnormalized_combined_data = normalize_data(tdcsfog_data, exclude_columns)\n\nnormalized_combined_data.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### EDA","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nsubject = \"a171e61840.csv\"\nplot_df = tdcsfog_data[tdcsfog_data['subject']==subject]\n\nplt.plot(plot_df.Turn)\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#TimeStamps Statistics:\nimport numpy as np\n\ntdcsfog_counts = tdcsfog_data.groupby('subject')['subject'].count()\nprint(\"tdcsfog stats:\")\nprint(f\"{np.mean(tdcsfog_counts):.1f} mean timestamps for each subject in tdcsfog\")\nprint(f\"{np.min(tdcsfog_counts):.0f} min timestamps for each subject in tdcsfog\")\nprint(f\"{np.max(tdcsfog_counts):.0f} max timestamps for each subject in tdcsfog\")\nprint(f\"{np.median(tdcsfog_counts):.0f} median timestamps for each subject in tdcsfog\")\nprint(f\"{np.std(tdcsfog_counts):.1f} std of timestamps in tdcsfog\\n\")\n\nprint(\"defog stats:\")\ndefog_counts = defog_data.groupby('subject')['subject'].count()\nprint(f\"{np.mean(defog_counts):.1f} mean timestamps for each subject in defog\")\nprint(f\"{np.min(defog_counts):.0f} min timestamps for each subject in defog\")\nprint(f\"{np.max(defog_counts):.0f} max timestamps for each subject in defog\")\nprint(f\"{np.median(defog_counts):.0f} median timestamps for each subject in defog\")\nprint(f\"{np.std(defog_counts):.1f} std of timestamps in defog\\n\\n\")\n\nmax_seq = np.max(defog_counts) ## for seq padding!\n\nStartHesitation_ratio = list(normalized_combined_data['StartHesitation'].value_counts())[1] / len(normalized_combined_data)\nTurn_ratio = list(normalized_combined_data['Turn'].value_counts())[1] / len(normalized_combined_data)\nwalking_ratio = list(normalized_combined_data['Walking'].value_counts())[1] / len(normalized_combined_data)\nprint(f\"StartHesitation_ratio: {StartHesitation_ratio:.2%}\")\nprint(f\"Turn_ratio: {Turn_ratio:.2%}\")\nprint(f\"walking_ratio: {walking_ratio:.2%}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Preprocessing","metadata":{}},{"cell_type":"code","source":"window_size = 200\n\ndf = normalized_combined_data.copy()\nsubjects = df.subject.unique()\nfeatures = ['AccV','AccML','AccAP']\ntargets = ['Turn']\n\nX = []\nY = []\n\nfor idx, subject in enumerate(subjects):\n    \"\"\"\n    Filter each subject timestamps\n    \"\"\"\n    x = df[df['subject']==subject][features].to_numpy()\n    y = df[df['subject']==subject][targets].to_numpy()\n    \n    for idx in range(0, len(x) - window_size):\n        \"\"\"\n        Append windows\n        \"\"\"\n        tmp_x = x[idx:idx+window_size].tolist()\n        tmp_y = y[idx:idx+window_size].tolist()\n        \n        X.append(tmp_x)\n        Y.append(tmp_y)\n    print(idx)\nX = np.array(X)\ny = np.array(Y)\n\nprint(f\"X shape -> (sequences={X.shape[0]}, timestamp={X.shape[1]}, features={X.shape[2]})\")\nprint(f\"y shape -> (sequences={y.shape[0]}, timestamp={y.shape[1]}, classes={y.shape[2]})\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\nfrom sklearn.model_selection import train_test_split\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=.2, random_state=42)\nnum_classes = np.max(y_train) + 1\ny_train_one_hot = tf.keras.utils.to_categorical(y_train, num_classes=num_classes)\ny_test_one_hot = tf.keras.utils.to_categorical(y_test, num_classes=num_classes)\n\nprint(f\"X_train shape: {X_train.shape}\")\nprint(f\"y_train_one_hot shape: {y_train_one_hot.shape}\\n\")\n\nprint(f\"X_test shape: {X_test.shape}\")\nprint(f\"y_test_one_hot shape: {y_test_one_hot.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-08-28T15:16:14.046820Z","iopub.execute_input":"2024-08-28T15:16:14.047205Z","iopub.status.idle":"2024-08-28T15:16:28.112872Z","shell.execute_reply.started":"2024-08-28T15:16:14.047176Z","shell.execute_reply":"2024-08-28T15:16:28.111676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Turns Model","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, TensorDataset\nimport numpy as np\n\n# Use GPU if available\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# Convert to PyTorch tensors with float32 dtype\nX_train_tensor = torch.tensor(X_train, dtype=torch.float32).to(device)\ny_train_tensor = torch.tensor(y_train_one_hot, dtype=torch.float32).to(device)\nX_test_tensor = torch.tensor(X_test, dtype=torch.float32).to(device)\ny_test_tensor = torch.tensor(y_test_one_hot, dtype=torch.float32).to(device)\n\n# Create TensorDataset for training and testing\ntrain_dataset = TensorDataset(X_train_tensor, y_train_tensor)\ntest_dataset = TensorDataset(X_test_tensor, y_test_tensor)\n\n# Create DataLoaders with a batch size of 64 (can be adjusted)\ntrain_loader = DataLoader(train_dataset, batch_size=64, shuffle=True)\ntest_loader = DataLoader(test_dataset, batch_size=64, shuffle=False)\n\nclass UnidirectionalLSTM(nn.Module):\n    def __init__(self, input_size, hidden_size, output_size):\n        super(UnidirectionalLSTM, self).__init__()\n        self.lstm1 = nn.LSTM(input_size, hidden_size, batch_first=True)\n        self.dropout1 = nn.Dropout(0.5)\n        self.lstm2 = nn.LSTM(hidden_size, hidden_size // 2, batch_first=True)\n        self.dropout2 = nn.Dropout(0.5)\n        self.fc1 = nn.Linear(hidden_size // 2, 512)\n        self.dropout3 = nn.Dropout(0.25)\n        self.fc2 = nn.Linear(512, 256)\n        self.dropout4 = nn.Dropout(0.25)\n        self.fc3 = nn.Linear(256, 128)\n        self.dropout5 = nn.Dropout(0.25)\n        self.fc4 = nn.Linear(128, 64)\n        self.dropout6 = nn.Dropout(0.25)\n        self.fc5 = nn.Linear(64, output_size)  # Change output_size to 2 for 2 classes\n\n    def forward(self, x):\n        x, _ = self.lstm1(x)\n        x = self.dropout1(x)\n        x, _ = self.lstm2(x)\n        x = self.dropout2(x)\n        x = torch.relu(self.fc1(x))\n        x = self.dropout3(x)\n        x = torch.relu(self.fc2(x))\n        x = self.dropout4(x)\n        x = torch.relu(self.fc3(x))\n        x = self.dropout5(x)\n        x = torch.relu(self.fc4(x))\n        x = self.dropout6(x)\n        x = torch.softmax(self.fc5(x), dim=-1)\n        return x\n\n# Instantiate model\nmodel = UnidirectionalLSTM(input_size=3, hidden_size=256, output_size=2).to(device)  # Adjust output size to 2\n\n# Define Focal Loss\nclass FocalLoss(nn.Module):\n    def __init__(self, gamma=2.0, alpha=None):\n        super(FocalLoss, self).__init__()\n        self.gamma = gamma\n        self.alpha = torch.tensor(alpha).to(device) if alpha is not None else None\n\n    def forward(self, inputs, targets):\n        BCE_loss = nn.functional.binary_cross_entropy(inputs, targets, reduction='none')\n        pt = torch.exp(-BCE_loss)\n        if self.alpha is not None:\n            F_loss = (self.alpha * (1 - pt) ** self.gamma * BCE_loss).mean()\n        else:\n            F_loss = ((1 - pt) ** self.gamma * BCE_loss).mean()\n        return F_loss\n\n# Assuming alpha values for focal loss are provided\nalpha = [1-Turn_ratio, Turn_ratio]  # Example values for the two classes\nfocal_loss = FocalLoss(gamma=2.0, alpha=alpha)\n\n# Define optimizer\noptimizer = optim.Adam(model.parameters(), lr=0.0001)","metadata":{"execution":{"iopub.status.busy":"2024-08-28T15:15:57.384800Z","iopub.execute_input":"2024-08-28T15:15:57.385175Z","iopub.status.idle":"2024-08-28T15:16:00.973647Z","shell.execute_reply.started":"2024-08-28T15:15:57.385144Z","shell.execute_reply":"2024-08-28T15:16:00.972369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training the model\nnum_epochs = 20\nfor epoch in range(num_epochs):\n    # Training phase\n    model.train()\n    running_loss = 0.0\n    for X_batch, y_batch in train_loader:\n        X_batch, y_batch = X_batch.to(device), y_batch.to(device)\n        \n        optimizer.zero_grad()\n        outputs = model(X_batch)\n        loss = focal_loss(outputs, y_batch)\n        loss.backward()\n        optimizer.step()\n        \n        running_loss += loss.item()\n    \n    avg_train_loss = running_loss / len(train_loader)\n    \n    # Validation phase\n    model.eval()\n    val_loss = 0.0\n    with torch.no_grad():\n        for X_val, y_val in test_loader:  # Assuming test_loader is used as validation loader\n            X_val, y_val = X_val.to(device), y_val.to(device)\n            val_outputs = model(X_val)\n            val_loss += focal_loss(val_outputs, y_val).item()\n    \n    avg_val_loss = val_loss / len(test_loader)\n    \n    print(f\"Epoch [{epoch+1}/{num_epochs}], Train Loss: {avg_train_loss:.4f}, Val Loss: {avg_val_loss:.4f}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subject_idx = 50\npredictions = model(X_test_tensor[subject_idx:subject_idx+1])\nthreshold = 0.5\npredicted_turns = []\nfor prediction in predictions[0]:\n    if prediction[1] > threshold:\n        predicted_turns.append(1)\n    else:\n        predicted_turns.append(0)\n\ntrue_labels = y_test_tensor[subject_idx:subject_idx+1]   \ntrue_turns = []\nfor y in true_labels[0]:\n    true_turns.append(torch.argmax(prediction).item())\n\ny_pred, y_true = predicted_turns, true_turns\nplt.plot(y_pred, label='Predicted') \nplt.plot(y_true, label='True')   \nplt.legend()\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.eval()\nwith torch.no_grad():\n    prediction = model(X_test_tensor[subject_idx:subject_idx+1])\nprediction","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subject_idx = 55\npredictions = model(X_test_tensor[subject_idx:subject_idx+1])\nthreshold = 0.5\npredicted_turns = []\nfor prediction in predictions[0]:\n    if prediction[1] > threshold:\n        predicted_turns.append(1)\n    else:\n        predicted_turns.append(0)\n\ntrue_labels = y_test_tensor[subject_idx:subject_idx+1]   \ntrue_turns = []\nfor y in true_labels[0]:\n    true_turns.append(torch.argmax( y).item())\n\ny_pred, y_true = predicted_turns, true_turns\nplt.plot(y_pred, label='Predicted') \nplt.plot(y_true, label='True')   \nplt.legend()\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Inference","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import precision_score, recall_score, accuracy_score, f1_score, confusion_matrix\n\naccuracy = accuracy_score(predicted_turns, true_turns)\nprecision = precision_score(predicted_turns, true_turns)\nrecall = recall_score(predicted_turns, true_turns)\nf1 = f1_score(predicted_turns, true_turns)\nprint(f\"Accuracy: {accuracy:.2%}\")\nprint(f\"Precision: {precision:.2%}\")\nprint(f\"Recall: {recall:.2%}\")\nprint(f\"F1: {f1:.2%}\")\nconfusion_matrix(predicted_turns, true_turns)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted_turns = []\ntrue_turns = []\ntest_size = 1_000\ntest_indices = np.random.randint(0, len(X_test_tensor), test_size)\nfor idx in test_indices:\n    predictions = model(X_test_tensor[idx:idx+1])\n    threshold = 0.5\n    for prediction in predictions[0]:\n        if prediction[1] > threshold:\n            predicted_turns.append(1)\n        else:\n            predicted_turns.append(0)\n    true_labels = y_test_tensor[idx:idx+1]   \n    for y in true_labels[0]:\n        true_turns.append(torch.argmax(y).item())\n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}