{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np \nimport pandas as pd\nimport os\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset\nfrom torch.utils.data import DataLoader\nfrom torch.nn import CrossEntropyLoss\nimport torch.nn.functional as F\n\nMAIN_DIR = \"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/\"\nDEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\nFEATURES = [\"AccV\", \"AccML\", \"AccAP\"]\nTARGETS = [\"StartHesitation\", \"Turn\", \"Walking\"]\n\nN_EPOCHS = 1","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-05T07:19:22.292896Z","iopub.execute_input":"2023-04-05T07:19:22.293682Z","iopub.status.idle":"2023-04-05T07:19:26.375299Z","shell.execute_reply.started":"2023-04-05T07:19:22.293637Z","shell.execute_reply":"2023-04-05T07:19:26.373985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. About the Data / Loading","metadata":{}},{"cell_type":"markdown","source":"CONTEXT : *People with FOG are filmed while performing certain tasks that are likely to increase its occurrence. Experts then review the video to score each frame, indicating when FOG occurred. While scoring in this manner is relatively reliable and sensitive, it is extremely time-consuming and requires specific expertise. Another method involves augmenting FOG-provoking testing with wearable devices. With more sensors, the detection of FOG becomes easier, however, compliance and usability may be reduced. Therefore, a combination of these two methods may be the best approach.*","metadata":{}},{"cell_type":"markdown","source":"The data series include three datasets, collected under distinct circumstances:\n\n- **tDCS FOG (tdcsfog)** dataset: data series collected in the lab, as subjects completed a FOG-provoking protocol.\n- **DeFOG (defog)** dataset: data series collected in the subject's home, as subjects completed a FOG-provoking protocol\n- **Daily Living (daily)** dataset: comprising one week of continuous 24/7 recordings from sixty-five subjects. Forty-five subjects exhibit FOG symptoms and also have series in the defog dataset, while the other twenty subjects do not exhibit FOG symptoms and do not have series elsewhere in the data.","metadata":{}},{"cell_type":"markdown","source":"**tDCS FOG** & **DeFOG** are annotated => Supervised Learning\n**Daily Living** is not annotated => Unsupervised/Semi-supervised Learning","metadata":{}},{"cell_type":"code","source":"# Reduce Memory Usage\n# reference : https://www.kaggle.com/code/arjanso/reducing-dataframe-memory-size-by-65 @ARJANGROEN\n\ndef reduce_memory_usage(df):\n    \n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype.name\n        if ((col_type != 'datetime64[ns]') & (col_type != 'category')):\n            if (col_type != 'object'):\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                if str(col_type)[:3] == 'int':\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n\n                else:\n                    if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                        df[col] = df[col].astype(np.float16)\n                    elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        pass\n            else:\n                df[col] = df[col].astype('category')\n    mem_usg = df.memory_usage().sum() / 1024**2 \n    print(\"Memory usage became: \",mem_usg,\" MB\")\n    \n    return df","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-04-05T07:19:30.608324Z","iopub.execute_input":"2023-04-05T07:19:30.609653Z","iopub.status.idle":"2023-04-05T07:19:30.630026Z","shell.execute_reply.started":"2023-04-05T07:19:30.609585Z","shell.execute_reply":"2023-04-05T07:19:30.628348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_data(\n    dataset,\n    datatype,\n    subject_id = None):\n    \n    metadata = pd.read_csv(MAIN_DIR + dataset + \"_metadata.csv\")\n    \n    DATA_ROOT = MAIN_DIR + datatype + \"/\" + dataset\n    \n    if subject_id is not None:\n        files = [file for file in files if subject_id in file]\n    \n    df_res = pd.DataFrame()\n    for root, dirs, files in os.walk(DATA_ROOT):\n        for name in tqdm(files):\n            f = os.path.join(root, name)\n            query_datatype = pd.read_csv(f)\n            query_datatype[\"file\"] = name.replace(\".csv\", \"\")\n            df_res = pd.concat([df_res,query_datatype])\n    \n    df_res = metadata.merge(df_res,\n                          how = 'inner',\n                          left_on = 'Id',\n                          right_on = 'file')\n    df_res = df_res.drop([\"file\"], axis = 1)\n\n    df_res = reduce_memory_usage(df_res)\n        \n    return df_res\n    ","metadata":{"execution":{"iopub.status.busy":"2023-04-05T07:19:33.925471Z","iopub.execute_input":"2023-04-05T07:19:33.926452Z","iopub.status.idle":"2023-04-05T07:19:33.936993Z","shell.execute_reply.started":"2023-04-05T07:19:33.926391Z","shell.execute_reply":"2023-04-05T07:19:33.935856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. DataLoader","metadata":{}},{"cell_type":"code","source":"class FOGDataset(Dataset):\n         \n    @staticmethod\n    def encode_target(data, targets_list):\n        conditions = []\n        for target in targets_list:\n            conditions.append((data[target] == 1))\n\n        event = np.select(conditions, targets_list, default='Normal')\n        le = LabelEncoder()\n        return le.fit_transform(event)\n\n    @staticmethod\n    def get_features_target(data, features_list, datatype):\n        if datatype == \"train\":\n            features, target = data[features_list], data[\"target\"]\n            return features, target\n        else:\n            features = data[features_list]\n            return features\n    \n    def __init__(self, dataset, datatype, features_list, targets_list, lookback):\n        self.datatype = datatype\n        self.data = read_data(dataset = dataset, datatype = datatype)\n        self.features = features_list\n        self.targets = targets_list\n        self.data[\"Id_encoded\"], _ = pd.factorize(self.data[\"Id\"])\n        self.lookback = lookback\n        \n        if datatype == \"train\":\n            self.data = self.data[:1_000]\n            self.data[\"target\"] = FOGDataset.encode_target(self.data, self.targets)\n    \n    def __len__(self):\n        return len(self.data)\n    \n    def __getitem__(self, idx):\n        if self.datatype == \"train\":            \n            features, targets = FOGDataset.get_features_target(self.data,\n                                               self.features,\n                                               self.datatype\n                                              )\n            \n            if idx < self.lookback :\n                features = features[0: self.lookback]\n                targets = targets[self.lookback]\n            \n            else:\n                features = features[idx - self.lookback: idx]\n                targets = targets[idx]\n                \n            features = torch.tensor(features.to_numpy(), dtype=torch.float32)\n            targets = torch.tensor(targets, dtype=torch.float32)\n            \n            return features, targets\n        else:\n            features = FOGDataset.get_features_target(self.data,\n                                               self.features,\n                                               self.datatype\n                                              )\n            \n            if idx < self.lookback :\n                features = features[0: self.lookback]\n            \n            else:\n                features = features[idx - self.lookback: idx]\n                \n            features = torch.tensor(features.to_numpy(), dtype=torch.float32)\n            \n            return features","metadata":{"execution":{"iopub.status.busy":"2023-04-05T07:19:38.247480Z","iopub.execute_input":"2023-04-05T07:19:38.248633Z","iopub.status.idle":"2023-04-05T07:19:38.264208Z","shell.execute_reply.started":"2023-04-05T07:19:38.248584Z","shell.execute_reply":"2023-04-05T07:19:38.262916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_train = FOGDataset(\n    dataset = \"tdcsfog\",\n    datatype = \"train\",\n    features_list = FEATURES,\n    targets_list = TARGETS,\n    lookback = 2\n)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-05T07:20:11.204862Z","iopub.execute_input":"2023-04-05T07:20:11.205274Z","iopub.status.idle":"2023-04-05T07:22:43.207640Z","shell.execute_reply.started":"2023-04-05T07:20:11.205236Z","shell.execute_reply":"2023-04-05T07:22:43.206291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_test = FOGDataset(\n    dataset = \"tdcsfog\",\n    datatype = \"test\",\n    features_list = FEATURES,\n    targets_list = TARGETS,\n    lookback = 2\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T07:23:36.455375Z","iopub.execute_input":"2023-04-05T07:23:36.456254Z","iopub.status.idle":"2023-04-05T07:23:36.507342Z","shell.execute_reply.started":"2023-04-05T07:23:36.456198Z","shell.execute_reply":"2023-04-05T07:23:36.505748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataloader_train = DataLoader(dataset_train, batch_size = 8, shuffle = False)\ndataloader_test = DataLoader(dataset_test, batch_size = 1000, shuffle = False)\n\ncount = 0\n\nfor batch in dataloader_train:\n    features, target = batch\n    print(\"FEATURES EXAMPLES\")\n    print(features.shape)\n    print(features)\n    print(\"TARGET EXAMPLES\")\n    print(target)\n    print(\"\\n\")\n    if count > 1:  \n        break\n    count += 1","metadata":{"execution":{"iopub.status.busy":"2023-04-05T07:23:38.303192Z","iopub.execute_input":"2023-04-05T07:23:38.304420Z","iopub.status.idle":"2023-04-05T07:23:38.459968Z","shell.execute_reply.started":"2023-04-05T07:23:38.304371Z","shell.execute_reply":"2023-04-05T07:23:38.458725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class LSTMNet(nn.Module):\n    def __init__(self, input_size, hidden_size, num_layers, num_classes):\n        super().__init__()\n        self.hidden_size = hidden_size\n        self.num_layers = num_layers\n        self.lstm = nn.LSTM(input_size, hidden_size, num_layers, batch_first=True)\n        self.fc1 = nn.Linear(hidden_size, num_classes)\n\n    def forward(self, x):\n\n        hidden_state = torch.zeros((self.num_layers, x.size(0), self.hidden_size), dtype=torch.float32)\n        cell_state = torch.zeros((self.num_layers, x.size(0), self.hidden_size), dtype=torch.float32)\n\n        out, _ = self.lstm(x, (hidden_state, cell_state))\n        out = out[:, -1,:]\n        out = self.fc1(out)\n        return out\n","metadata":{"execution":{"iopub.status.busy":"2023-04-05T07:23:43.575200Z","iopub.execute_input":"2023-04-05T07:23:43.575616Z","iopub.status.idle":"2023-04-05T07:23:43.584899Z","shell.execute_reply.started":"2023-04-05T07:23:43.575579Z","shell.execute_reply":"2023-04-05T07:23:43.583641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Training","metadata":{}},{"cell_type":"code","source":"def train(model, dataloader, loss_fn, optimizer):\n    model.train()\n    total_loss = 0\n    for epoch in range(N_EPOCHS):\n        mean_precision = []\n        for (features, targets) in tqdm(dataloader):\n            optimizer.zero_grad()\n            preds = model(features)\n            loss = loss_fn(preds, targets.long())\n            mean_precision.append(loss.item())\n            loss.backward()\n            optimizer.step()\n        \n        print(\"Average Precision : \", np.mean(mean_precision))\n    \n    return model\n\ndef predict(model, dataloader): \n    model.eval()\n    predictions = np.empty(len(dataset_test))\n    count = 0\n    for features in tqdm(dataloader):\n        preds = model(features)\n        preds = torch.argmax(preds, dim = 1)\n        preds = preds.numpy()\n        predictions[count : count + len(preds)] = preds\n        count += len(preds)\n            \n    return predictions","metadata":{"execution":{"iopub.status.busy":"2023-04-05T07:29:39.054513Z","iopub.execute_input":"2023-04-05T07:29:39.055027Z","iopub.status.idle":"2023-04-05T07:29:39.068154Z","shell.execute_reply.started":"2023-04-05T07:29:39.054977Z","shell.execute_reply":"2023-04-05T07:29:39.066699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INPUT_SIZE = len(FEATURES)\nHIDDEN_SIZE = 10\nNUM_LAYERS = 1\nNUM_CLASSES = 4\nPARAMS = {\n    \"input_size\" : INPUT_SIZE,\n    \"hidden_size\" : HIDDEN_SIZE,\n    \"num_layers\" : NUM_LAYERS,\n    \"num_classes\" : NUM_CLASSES\n}\nmodel = LSTMNet(**PARAMS)\n\nloss_fn = CrossEntropyLoss()\n\noptimizer = optim.Adam(model.parameters(), lr=0.001)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T07:23:47.075608Z","iopub.execute_input":"2023-04-05T07:23:47.076041Z","iopub.status.idle":"2023-04-05T07:23:47.087691Z","shell.execute_reply.started":"2023-04-05T07:23:47.076000Z","shell.execute_reply":"2023-04-05T07:23:47.086162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = train(\n    model, \n    dataloader_train,\n    loss_fn,\n    optimizer\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T07:23:47.846381Z","iopub.execute_input":"2023-04-05T07:23:47.847091Z","iopub.status.idle":"2023-04-05T07:23:48.747070Z","shell.execute_reply.started":"2023-04-05T07:23:47.847032Z","shell.execute_reply":"2023-04-05T07:23:48.745865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_tdcsfog = predict(model, dataloader_test)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T07:28:53.899530Z","iopub.execute_input":"2023-04-05T07:28:53.900763Z","iopub.status.idle":"2023-04-05T07:28:56.601507Z","shell.execute_reply.started":"2023-04-05T07:28:53.900710Z","shell.execute_reply":"2023-04-05T07:28:56.600222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_tdcsfog","metadata":{"execution":{"iopub.status.busy":"2023-04-05T07:28:56.603423Z","iopub.execute_input":"2023-04-05T07:28:56.603774Z","iopub.status.idle":"2023-04-05T07:28:56.610975Z","shell.execute_reply.started":"2023-04-05T07:28:56.603743Z","shell.execute_reply":"2023-04-05T07:28:56.609764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. First Submission","metadata":{"execution":{"iopub.status.busy":"2023-03-28T09:27:08.999527Z","iopub.execute_input":"2023-03-28T09:27:09.000095Z","iopub.status.idle":"2023-03-28T09:27:09.013584Z","shell.execute_reply.started":"2023-03-28T09:27:09.000046Z","shell.execute_reply":"2023-03-28T09:27:09.011646Z"}}},{"cell_type":"code","source":"test_tdcsfog = read_data(dataset = \"tdcsfog\", datatype = \"test\")\ntest_defog = read_data(dataset = \"defog\", datatype = \"test\")\n\nlen_test_tdcsfog = len(test_tdcsfog)\nlen_test_defog = len(test_defog)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T07:28:59.555553Z","iopub.execute_input":"2023-04-05T07:28:59.556120Z","iopub.status.idle":"2023-04-05T07:29:00.355391Z","shell.execute_reply.started":"2023-04-05T07:28:59.556064Z","shell.execute_reply":"2023-04-05T07:29:00.354462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_tdcsfog[\"y_pred\"] = preds_tdcsfog\ntest_defog[\"y_pred\"] = np.zeros(len_test_defog)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T07:29:05.587879Z","iopub.execute_input":"2023-04-05T07:29:05.588846Z","iopub.status.idle":"2023-04-05T07:29:05.595981Z","shell.execute_reply.started":"2023-04-05T07:29:05.588801Z","shell.execute_reply":"2023-04-05T07:29:05.595083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_fmt = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-04-05T07:29:08.026048Z","iopub.execute_input":"2023-04-05T07:29:08.026870Z","iopub.status.idle":"2023-04-05T07:29:08.288176Z","shell.execute_reply.started":"2023-04-05T07:29:08.026824Z","shell.execute_reply":"2023-04-05T07:29:08.286594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_fmt.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T07:29:08.326980Z","iopub.execute_input":"2023-04-05T07:29:08.328235Z","iopub.status.idle":"2023-04-05T07:29:08.361865Z","shell.execute_reply.started":"2023-04-05T07:29:08.328189Z","shell.execute_reply":"2023-04-05T07:29:08.360336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame()\nfor data in [test_tdcsfog, test_defog]:\n    temp = data.copy()\n    temp[\"Id\"] = temp.apply(lambda x : str(x.Id) + \"_\" + str(x.Time), axis = 1)\n    temp['StartHesitation'] = np.where(temp['y_pred']==1, 1, 0)\n    temp['Turn'] = np.where(temp['y_pred']==2, 1, 0)\n    temp['Walking'] = np.where(temp['y_pred']==3, 1, 0)\n    temp = temp[[\"Id\"] + TARGETS]\n    sub = pd.concat([sub, temp])","metadata":{"execution":{"iopub.status.busy":"2023-04-05T07:29:09.678124Z","iopub.execute_input":"2023-04-05T07:29:09.678710Z","iopub.status.idle":"2023-04-05T07:29:16.442817Z","shell.execute_reply.started":"2023-04-05T07:29:09.678653Z","shell.execute_reply":"2023-04-05T07:29:16.441539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T07:29:16.445073Z","iopub.execute_input":"2023-04-05T07:29:16.445569Z","iopub.status.idle":"2023-04-05T07:29:16.468408Z","shell.execute_reply.started":"2023-04-05T07:29:16.445518Z","shell.execute_reply":"2023-04-05T07:29:16.467002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(sub)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T07:29:16.470355Z","iopub.execute_input":"2023-04-05T07:29:16.470741Z","iopub.status.idle":"2023-04-05T07:29:16.479450Z","shell.execute_reply.started":"2023-04-05T07:29:16.470706Z","shell.execute_reply":"2023-04-05T07:29:16.478380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T07:29:23.725533Z","iopub.execute_input":"2023-04-05T07:29:23.725926Z","iopub.status.idle":"2023-04-05T07:29:23.756367Z","shell.execute_reply.started":"2023-04-05T07:29:23.725892Z","shell.execute_reply":"2023-04-05T07:29:23.755123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_fmt.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T07:29:23.987316Z","iopub.execute_input":"2023-04-05T07:29:23.988037Z","iopub.status.idle":"2023-04-05T07:29:23.999993Z","shell.execute_reply.started":"2023-04-05T07:29:23.987996Z","shell.execute_reply":"2023-04-05T07:29:23.998628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(sub_fmt)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T07:29:24.451105Z","iopub.execute_input":"2023-04-05T07:29:24.451606Z","iopub.status.idle":"2023-04-05T07:29:24.460253Z","shell.execute_reply.started":"2023-04-05T07:29:24.451562Z","shell.execute_reply":"2023-04-05T07:29:24.458841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(\"/kaggle/working/submission.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T07:29:25.253928Z","iopub.execute_input":"2023-04-05T07:29:25.254388Z","iopub.status.idle":"2023-04-05T07:29:25.682073Z","shell.execute_reply.started":"2023-04-05T07:29:25.254349Z","shell.execute_reply":"2023-04-05T07:29:25.681067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}