{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":41880,"databundleVersionId":5677426,"sourceType":"competition"},{"sourceId":105488,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":88399,"modelId":112626},{"sourceId":105489,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":88400,"modelId":112627}],"dockerImageVersionId":30762,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Project 01.09.24:\n\nAsaf Delmedigo, Romi Zarchi, Ori Armel, Tal Bocbot, Chen Zusman\n","metadata":{}},{"cell_type":"markdown","source":"**Importing libraries**","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nimport os\nimport pandas as pd\nfrom sklearn.preprocessing import StandardScaler\nimport tensorflow as tf\nfrom sklearn.model_selection import train_test_split\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, TensorDataset\nfrom sklearn.metrics import precision_score, recall_score, accuracy_score, f1_score, confusion_matrix\nfrom sklearn.metrics import classification_report, confusion_matrix, roc_curve, auc, precision_recall_curve\nimport seaborn as sns\nfrom scipy.signal import butter, filtfilt","metadata":{"execution":{"iopub.status.busy":"2024-09-01T17:12:26.247873Z","iopub.execute_input":"2024-09-01T17:12:26.248266Z","iopub.status.idle":"2024-09-01T17:12:26.255249Z","shell.execute_reply.started":"2024-09-01T17:12:26.248229Z","shell.execute_reply":"2024-09-01T17:12:26.254234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BidirectionalLSTM(nn.Module):\n    def __init__(self, input_size, hidden_size, output_size):\n        super(BidirectionalLSTM, self).__init__()\n        self.lstm1 = nn.LSTM(input_size, hidden_size, batch_first=True)\n        self.dropout1 = nn.Dropout(0.5)\n        self.lstm2 = nn.LSTM(hidden_size, hidden_size // 2, batch_first=True)\n        self.dropout2 = nn.Dropout(0.5)\n        self.fc1 = nn.Linear(hidden_size // 2, 512)\n        self.dropout3 = nn.Dropout(0.25)\n        self.fc2 = nn.Linear(512, 256)\n        self.dropout4 = nn.Dropout(0.25)\n        self.fc3 = nn.Linear(256, 128)\n        self.dropout5 = nn.Dropout(0.25)\n        self.fc4 = nn.Linear(128, 64)\n        self.dropout6 = nn.Dropout(0.25)\n        self.fc5 = nn.Linear(64, output_size)\n\n    def forward(self, x):\n        x, _ = self.lstm1(x)\n        x = self.dropout1(x)\n        x, _ = self.lstm2(x)\n        x = self.dropout2(x)\n        x = torch.relu(self.fc1(x))\n        x = self.dropout3(x)\n        x = torch.relu(self.fc2(x))\n        x = self.dropout4(x)\n        x = torch.relu(self.fc3(x))\n        x = self.dropout5(x)\n        x = torch.relu(self.fc4(x))\n        x = self.dropout6(x)\n        x = self.fc5(x)\n        return x\n\n# Define Focal Loss\nclass FocalLoss(nn.Module):\n    def __init__(self, gamma=2.0, alpha=None):\n        super(FocalLoss, self).__init__()\n        self.gamma = gamma\n        self.alpha = torch.tensor(alpha).to(device) if alpha is not None else None\n\n    def forward(self, inputs, targets):\n        BCE_loss = nn.functional.binary_cross_entropy_with_logits(inputs, targets, reduction='none')\n        pt = torch.exp(-BCE_loss)\n        if self.alpha is not None:\n            F_loss = (self.alpha * (1 - pt) ** self.gamma * BCE_loss).mean()\n        else:\n            F_loss = ((1 - pt) ** self.gamma * BCE_loss).mean()\n        return F_loss\n\n","metadata":{"execution":{"iopub.status.busy":"2024-09-01T17:12:26.354837Z","iopub.execute_input":"2024-09-01T17:12:26.355294Z","iopub.status.idle":"2024-09-01T17:12:26.369335Z","shell.execute_reply.started":"2024-09-01T17:12:26.355258Z","shell.execute_reply":"2024-09-01T17:12:26.368438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"import pickle\nmodel = torch.load(\"/kaggle/input/bilstm/pytorch/default/1/BiLSTM_model.pth\")\nwith open(\"/kaggle/input/scaler/scikitlearn/default/1/scaler (1).pkl\", \"rb\") as file:\n    scaler = pickle.load(file)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T17:12:27.067030Z","iopub.execute_input":"2024-09-01T17:12:27.067426Z","iopub.status.idle":"2024-09-01T17:12:27.091172Z","shell.execute_reply.started":"2024-09-01T17:12:27.067389Z","shell.execute_reply":"2024-09-01T17:12:27.090202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_test_csv_files_from_directory(directory_path):\n    \"\"\"\n    Loads and concatenates all CSV files from a specified directory into a single DataFrame.\n\n    Args:\n    directory_path (str): Path to the directory containing CSV files.\n\n    Returns:\n    pd.DataFrame: Concatenated DataFrame containing data from all CSV files in the directory.\n    \"\"\"\n    csv_files = [os.path.join(directory_path, file) for file in os.listdir(directory_path) \n                 if file.endswith('.csv')]\n    data_frames = []\n    for csv_file in csv_files:\n        tmp_df = pd.read_csv(csv_file)\n        tmp_df['Id'] = os.path.splitext(os.path.basename(csv_file))[0]        \n        tmp_df['subj'] = os.path.splitext(os.path.basename(csv_file))[0]   \n        data_frames.append(tmp_df)\n    \n    return pd.concat(data_frames, ignore_index=True)\n\ndef normalize_test_data(df, exclude_columns, scaler):\n    \"\"\"\n    Normalizes all numerical columns in a DataFrame except the excluded columns using StandardScaler.\n\n    Args:\n    df (pd.DataFrame): DataFrame with numerical features.\n    exclude_columns (list): List of column names to exclude from normalization.\n\n    Returns:\n    pd.DataFrame: DataFrame with normalized features.\n    \"\"\"\n    # Select columns that are numeric and not in the exclude list\n    numeric_columns = df.select_dtypes(include=['float64', 'int64']).columns\n    columns_to_normalize = [col for col in numeric_columns if col not in exclude_columns]\n    \n    # Apply normalization to the selected columns\n    df[columns_to_normalize] = scaler.transform(df[columns_to_normalize])\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-09-01T17:12:27.441906Z","iopub.execute_input":"2024-09-01T17:12:27.442785Z","iopub.status.idle":"2024-09-01T17:12:27.451225Z","shell.execute_reply.started":"2024-09-01T17:12:27.442743Z","shell.execute_reply":"2024-09-01T17:12:27.450207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_test_df = load_test_csv_files_from_directory(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/defog\")\ndefog_test_df['Id'] = defog_test_df.apply(lambda row: row[\"Id\"] + \"_\" + str(row[\"Time\"]),axis=1)\n\ntdcsfog_test_df = load_test_csv_files_from_directory(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog\")\ntdcsfog_test_df['Id'] = tdcsfog_test_df.apply(lambda row: row[\"Id\"] + \"_\" + str(row[\"Time\"]),axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T17:12:27.808100Z","iopub.execute_input":"2024-09-01T17:12:27.808855Z","iopub.status.idle":"2024-09-01T17:12:32.354083Z","shell.execute_reply.started":"2024-09-01T17:12:27.808804Z","shell.execute_reply":"2024-09-01T17:12:32.353227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# normalize data\nexclude_columns = ['Id', 'Time', 'source', 'StartHesitation', 'Turn', 'Walking']\ndefog_test_df = normalize_test_data(defog_test_df, exclude_columns, scaler)\ntdcsfog_test_df = normalize_test_data(tdcsfog_test_df, exclude_columns, scaler)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T17:12:32.381967Z","iopub.execute_input":"2024-09-01T17:12:32.382304Z","iopub.status.idle":"2024-09-01T17:12:32.401688Z","shell.execute_reply.started":"2024-09-01T17:12:32.382271Z","shell.execute_reply":"2024-09-01T17:12:32.400749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_defog_big_df = pd.DataFrame()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T17:12:32.405246Z","iopub.execute_input":"2024-09-01T17:12:32.405527Z","iopub.status.idle":"2024-09-01T17:12:32.410321Z","shell.execute_reply.started":"2024-09-01T17:12:32.405496Z","shell.execute_reply":"2024-09-01T17:12:32.409380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = ['AccV', 'AccML', 'AccAP']\ntargets = ['StartHesitation', 'Turn', 'Walking']\ndata = defog_test_df[features].to_numpy()\nsubjects = defog_test_df.subj.unique()\n# Set window size\nwindow_size = 128\nfor subject in subjects:\n    X = []\n    tmp_df= defog_test_df[defog_test_df[\"subj\"]==subject]\n    for idx in range(0, len(tmp_df) - window_size, window_size):\n        X.append(data[idx:idx+window_size].tolist())\n    X.append(data[idx:idx+window_size].tolist())\n    X = torch.tensor(X)\n    X = X.to(torch.device(\"cuda\"))\n    test_outputs = model(X)\n    test_outputs = torch.round(torch.sigmoid(test_outputs), decimals=3)\n    sub_defog_df = pd.DataFrame(np.array(test_outputs.reshape((len(X)*128,3)).detach().to(torch.device(\"cpu\"))), columns = targets)\n    sub_defog_df.insert(0, column=\"Id\", value = defog_test_df['Id'])\n    sub_defog_big_df = pd.concat([sub_defog_big_df, sub_defog_df.iloc[:len(tmp_df)]], axis=0)\nsub_defog_big_df","metadata":{"execution":{"iopub.status.busy":"2024-09-01T17:12:32.422166Z","iopub.execute_input":"2024-09-01T17:12:32.422524Z","iopub.status.idle":"2024-09-01T17:12:33.709519Z","shell.execute_reply.started":"2024-09-01T17:12:32.422482Z","shell.execute_reply":"2024-09-01T17:12:33.708517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_test_big_df = pd.DataFrame()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T17:12:33.710906Z","iopub.execute_input":"2024-09-01T17:12:33.711235Z","iopub.status.idle":"2024-09-01T17:12:33.716278Z","shell.execute_reply.started":"2024-09-01T17:12:33.711202Z","shell.execute_reply":"2024-09-01T17:12:33.715343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = ['AccV', 'AccML', 'AccAP']\ntargets = ['StartHesitation', 'Turn', 'Walking']\ndata = tdcsfog_test_df[features].to_numpy()\nsubjects = tdcsfog_test_df.subj.unique()\n# Set window size\nwindow_size = 128\nfor subject in subjects:\n    X = []\n    tmp_df= tdcsfog_test_df[tdcsfog_test_df[\"subj\"]==subject]\n    for idx in range(0, len(tmp_df) - window_size, window_size):\n        X.append(data[idx:idx+window_size].tolist())\n    X.append(data[idx:idx+window_size].tolist())\n    X = torch.tensor(X)\n    X = X.to(torch.device(\"cuda\"))\n    test_outputs = model(X)\n    test_outputs = torch.round(torch.sigmoid(test_outputs), decimals=3)\n    sub_tdcsfog_test_df = pd.DataFrame(np.array(test_outputs.reshape((len(X)*128,3)).detach().to(torch.device(\"cpu\"))), columns = targets)\n    sub_tdcsfog_test_df.insert(0, column=\"Id\", value = tdcsfog_test_df['Id'])\n    tdcsfog_test_big_df = pd.concat([tdcsfog_test_big_df, sub_tdcsfog_test_df.iloc[:len(tmp_df)]], axis=0)\ntdcsfog_test_big_df","metadata":{"execution":{"iopub.status.busy":"2024-09-01T17:24:48.082019Z","iopub.execute_input":"2024-09-01T17:24:48.083076Z","iopub.status.idle":"2024-09-01T17:24:48.177399Z","shell.execute_reply.started":"2024-09-01T17:24:48.082988Z","shell.execute_reply":"2024-09-01T17:24:48.176012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torch.round(test_outputs,decimals=2)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T17:25:47.193610Z","iopub.execute_input":"2024-09-01T17:25:47.194351Z","iopub.status.idle":"2024-09-01T17:25:47.206869Z","shell.execute_reply.started":"2024-09-01T17:25:47.194312Z","shell.execute_reply":"2024-09-01T17:25:47.206007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = pd.concat([tdcsfog_test_big_df,sub_defog_big_df], axis=0, ignore_index = True)\n\nsubmission_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T17:13:10.729258Z","iopub.execute_input":"2024-09-01T17:13:10.730170Z","iopub.status.idle":"2024-09-01T17:13:12.218399Z","shell.execute_reply.started":"2024-09-01T17:13:10.730128Z","shell.execute_reply":"2024-09-01T17:13:12.217372Z"},"trusted":true},"execution_count":null,"outputs":[]}]}