{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":41880,"databundleVersionId":5677426,"sourceType":"competition"},{"sourceId":121731966,"sourceType":"kernelVersion"}],"dockerImageVersionId":30408,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# PyTorch Solution: Rolling Window\nHere we introduce a pytorch solution which takes an window of past and future Acc readings to predict the outcomes. The different segments of the notebook can be modified and improved as per your liking to better the whole pipeline. As a way, this works as a good starter baseline!\n\nVERSION 9: \ni. Added the defog data in training phase.\nii. Using FP16 mode.\n\n**Please leave an upvote if you found this notebook helpful!**","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nimport random\nimport time\n\nimport json\nfrom tqdm import tqdm\nimport glob\nimport numpy as np\nimport pandas as pd\n\nimport torch\nimport torch.nn as nn\n\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\n\nfrom sklearn.model_selection import train_test_split, StratifiedGroupKFold\nfrom sklearn.metrics import accuracy_score, average_precision_score\n\nimport warnings\nwarnings.filterwarnings(action='ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-11T10:29:23.798189Z","iopub.execute_input":"2024-04-11T10:29:23.799044Z","iopub.status.idle":"2024-04-11T10:29:23.806027Z","shell.execute_reply.started":"2024-04-11T10:29:23.799004Z","shell.execute_reply":"2024-04-11T10:29:23.804895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    train_dir1 = \"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog\"\n    train_dir2 = \"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog\"\n\n    batch_size = 1024\n    window_size = 32\n    window_future = 8\n    window_past = window_size - window_future\n    \n    wx = 8\n    \n    model_dropout = 0.2\n    model_hidden = 512\n    model_nblocks = 3\n    \n    lr = 0.00015\n    num_epochs = 15\n    device = 'cuda' if torch.cuda.is_available() else 'cpu'\n    \n    feature_list = ['AccV', 'AccML', 'AccAP']\n    label_list = ['StartHesitation', 'Turn', 'Walking']\n    \n    \ncfg = Config()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T10:29:23.808665Z","iopub.execute_input":"2024-04-11T10:29:23.809138Z","iopub.status.idle":"2024-04-11T10:29:23.821314Z","shell.execute_reply.started":"2024-04-11T10:29:23.809096Z","shell.execute_reply":"2024-04-11T10:29:23.820276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cfg.device","metadata":{"execution":{"iopub.status.busy":"2024-04-11T10:29:23.822565Z","iopub.execute_input":"2024-04-11T10:29:23.822850Z","iopub.status.idle":"2024-04-11T10:29:23.833930Z","shell.execute_reply.started":"2024-04-11T10:29:23.822802Z","shell.execute_reply":"2024-04-11T10:29:23.832933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Stratified Group K Fold\n\nIt's mentioned in the data that the subjects are different in the train and test set and even different between the public/private splits of the test data. So we need to use Stratified Group K Fold. But since the positive instances in the sequences are very scarce, we need to pick up the best fold which will give us the best balance of the positive/negative instances.","metadata":{}},{"cell_type":"markdown","source":"### tdcsfog","metadata":{}},{"cell_type":"code","source":"# Analysis of positive instances in each fold of our CV folds\n\nn1_sum = []\nn2_sum = []\nn3_sum = []\ncount = []\n\n# Here I am using the metadata file available during training. Since the code will run again during submission, if \n# I used the usual file from the competition folder, it would have been updated with the test files too.\nmetadata = pd.read_csv(\"/kaggle/input/copy-train-metadata/tdcsfog_metadata.csv\")\n\nfor f in tqdm(metadata['Id']):\n    fpath = f\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/{f}.csv\"\n    df = pd.read_csv(fpath)\n    \n    n1_sum.append(np.sum(df['StartHesitation']))\n    n2_sum.append(np.sum(df['Turn']))\n    n3_sum.append(np.sum(df['Walking']))\n    count.append(len(df))\n    \nprint(f\"32 files have positive values in all 3 classes\")\n\nmetadata['n1_sum'] = n1_sum\nmetadata['n2_sum'] = n2_sum\nmetadata['n3_sum'] = n3_sum\nmetadata['count'] = count\n\nsgkf = StratifiedGroupKFold(n_splits=5, random_state=42, shuffle=True)\nfor i, (train_index, valid_index) in enumerate(sgkf.split(X=metadata['Id'], y=[1]*len(metadata), groups=metadata['Subject'])):\n    print(f\"Fold = {i}\")\n    train_ids = metadata.loc[train_index, 'Id']\n    valid_ids = metadata.loc[valid_index, 'Id']\n    \n    print(f\"Length of Train = {len(train_index)}, Length of Valid = {len(valid_index)}\")\n    n1_sum = metadata.loc[train_index, 'n1_sum'].sum()\n    n2_sum = metadata.loc[train_index, 'n2_sum'].sum()\n    n3_sum = metadata.loc[train_index, 'n3_sum'].sum()\n    print(f\"Train classes: {n1_sum:,}, {n2_sum:,}, {n3_sum:,}\")\n    \n    n1_sum = metadata.loc[valid_index, 'n1_sum'].sum()\n    n2_sum = metadata.loc[valid_index, 'n2_sum'].sum()\n    n3_sum = metadata.loc[valid_index, 'n3_sum'].sum()\n    print(f\"Valid classes: {n1_sum:,}, {n2_sum:,}, {n3_sum:,}\")\n    \n# FOLD 2 is the most well balanced\n# The actual train-test split (based on Fold 2)\n\nmetadata = pd.read_csv(\"/kaggle/input/copy-train-metadata/tdcsfog_metadata.csv\")\nsgkf = StratifiedGroupKFold(n_splits=5, random_state=42, shuffle=True)\nfor i, (train_index, valid_index) in enumerate(sgkf.split(X=metadata['Id'], y=[1]*len(metadata), groups=metadata['Subject'])):\n    if i != 2:\n        continue\n    print(f\"Fold = {i}\")\n    train_ids = metadata.loc[train_index, 'Id']\n    valid_ids = metadata.loc[valid_index, 'Id']\n    print(f\"Length of Train = {len(train_ids)}, Length of Valid = {len(valid_ids)}\")\n    \n    if i == 2:\n        break\n        \ntrain_fpaths_tdcs = [f\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/{_id}.csv\" for _id in train_ids]\nvalid_fpaths_tdcs = [f\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/{_id}.csv\" for _id in valid_ids]","metadata":{"execution":{"iopub.status.busy":"2024-04-11T10:29:23.835712Z","iopub.execute_input":"2024-04-11T10:29:23.836134Z","iopub.status.idle":"2024-04-11T10:29:38.803794Z","shell.execute_reply.started":"2024-04-11T10:29:23.836101Z","shell.execute_reply":"2024-04-11T10:29:38.802722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### defog","metadata":{}},{"cell_type":"code","source":"# Analysis of positive instances in each fold of our CV folds\n\nn1_sum = []\nn2_sum = []\nn3_sum = []\ncount = []\n\n# Here I am using the metadata file available during training. Since the code will run again during submission, if \n# I used the usual file from the competition folder, it would have been updated with the test files too.\nmetadata = pd.read_csv(\"/kaggle/input/copy-train-metadata/defog_metadata.csv\")\nmetadata['n1_sum'] = 0\nmetadata['n2_sum'] = 0\nmetadata['n3_sum'] = 0\nmetadata['count'] = 0\n\nfor f in tqdm(metadata['Id']):\n    fpath = f\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog/{f}.csv\"\n    if os.path.exists(fpath) == False:\n        continue\n        \n    df = pd.read_csv(fpath)\n    metadata.loc[metadata['Id'] == f, 'n1_sum'] = np.sum(df['StartHesitation'])\n    metadata.loc[metadata['Id'] == f, 'n2_sum'] = np.sum(df['Turn'])\n    metadata.loc[metadata['Id'] == f, 'n3_sum'] = np.sum(df['Walking'])\n    metadata.loc[metadata['Id'] == f, 'count'] = len(df)\n    \nmetadata = metadata[metadata['count'] > 0].reset_index()\n\nsgkf = StratifiedGroupKFold(n_splits=5, random_state=42, shuffle=True)\nfor i, (train_index, valid_index) in enumerate(sgkf.split(X=metadata['Id'], y=[1]*len(metadata), groups=metadata['Subject'])):\n    print(f\"Fold = {i}\")\n    train_ids = metadata.loc[train_index, 'Id']\n    valid_ids = metadata.loc[valid_index, 'Id']\n    \n    print(f\"Length of Train = {len(train_index)}, Length of Valid = {len(valid_index)}\")\n    n1_sum = metadata.loc[train_index, 'n1_sum'].sum()\n    n2_sum = metadata.loc[train_index, 'n2_sum'].sum()\n    n3_sum = metadata.loc[train_index, 'n3_sum'].sum()\n    print(f\"Train classes: {n1_sum:,}, {n2_sum:,}, {n3_sum:,}\")\n    \n    n1_sum = metadata.loc[valid_index, 'n1_sum'].sum()\n    n2_sum = metadata.loc[valid_index, 'n2_sum'].sum()\n    n3_sum = metadata.loc[valid_index, 'n3_sum'].sum()\n    print(f\"Valid classes: {n1_sum:,}, {n2_sum:,}, {n3_sum:,}\")\n    \n# FOLD 2 is the most well balanced\n# The actual train-test split (based on Fold 2)\n\nsgkf = StratifiedGroupKFold(n_splits=5, random_state=42, shuffle=True)\nfor i, (train_index, valid_index) in enumerate(sgkf.split(X=metadata['Id'], y=[1]*len(metadata), groups=metadata['Subject'])):\n    if i != 1:\n        continue\n    print(f\"Fold = {i}\")\n    train_ids = metadata.loc[train_index, 'Id']\n    valid_ids = metadata.loc[valid_index, 'Id']\n    print(f\"Length of Train = {len(train_ids)}, Length of Valid = {len(valid_ids)}\")\n    \n    if i == 2:\n        break\n        \ntrain_fpaths_de = [f\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog/{_id}.csv\" for _id in train_ids]\nvalid_fpaths_de = [f\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog/{_id}.csv\" for _id in valid_ids]","metadata":{"execution":{"iopub.status.busy":"2024-04-11T10:29:38.806525Z","iopub.execute_input":"2024-04-11T10:29:38.806863Z","iopub.status.idle":"2024-04-11T10:30:04.555075Z","shell.execute_reply.started":"2024-04-11T10:29:38.806830Z","shell.execute_reply":"2024-04-11T10:30:04.553944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_fpaths = [(f, 'de') for f in train_fpaths_de] + [(f, 'tdcs') for f in train_fpaths_tdcs]\nvalid_fpaths = [(f, 'de') for f in valid_fpaths_de] + [(f, 'tdcs') for f in valid_fpaths_tdcs]","metadata":{"execution":{"iopub.status.busy":"2024-04-11T10:30:04.556257Z","iopub.execute_input":"2024-04-11T10:30:04.556594Z","iopub.status.idle":"2024-04-11T10:30:04.562654Z","shell.execute_reply.started":"2024-04-11T10:30:04.556562Z","shell.execute_reply":"2024-04-11T10:30:04.561450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataset\n\nWe use a window comprised of past and future time Acc readings to form our dataset for a particular time instance. In case some portion of the window data is not available, we pad them with zeros.","metadata":{}},{"cell_type":"code","source":"class FOGDataset(Dataset):\n    def __init__(self, fpaths, scale=9.806, split=\"train\"):\n        super(FOGDataset, self).__init__()\n        tm = time.time()\n        self.split = split\n        self.scale = scale\n        \n        self.fpaths = fpaths\n        self.dfs = [self.read(f[0], f[1]) for f in fpaths]\n        self.f_ids = [os.path.basename(f[0])[:-4] for f in self.fpaths]\n        \n        self.end_indices = []\n        self.shapes = []\n        _length = 0\n        for df in self.dfs:\n            self.shapes.append(df.shape[0])\n            _length += df.shape[0]\n            self.end_indices.append(_length)\n        \n        self.dfs = np.concatenate(self.dfs, axis=0).astype(np.float16)\n        self.length = self.dfs.shape[0]\n        \n        shape1 = self.dfs.shape[1]\n        \n        self.dfs = np.concatenate([np.zeros((cfg.wx*cfg.window_past, shape1)), self.dfs, np.zeros((cfg.wx*cfg.window_future, shape1))], axis=0)\n        print(f\"Dataset initialized in {time.time() - tm} secs!\")\n        gc.collect()\n        \n    def read(self, f, _type):\n        df = pd.read_csv(f)\n        if self.split == \"test\":\n            return np.array(df)\n        \n        if _type ==\"tdcs\":\n            df['Valid'] = 1\n            df['Task'] = 1\n            df['tdcs'] = 1\n        else:\n            df['tdcs'] = 0\n        \n        return np.array(df)\n            \n    def __getitem__(self, index):\n        if self.split == \"train\":\n            row_idx = random.randint(0, self.length-1) + cfg.wx*cfg.window_past\n        elif self.split == \"test\":\n            for i,e in enumerate(self.end_indices):\n                if index >= e:\n                    continue\n                df_idx = i\n                break\n\n            row_idx_true = self.shapes[df_idx] - (self.end_indices[df_idx] - index)\n            _id = self.f_ids[df_idx] + \"_\" + str(row_idx_true)\n            row_idx = index + cfg.wx*cfg.window_past\n        else:\n            row_idx = index + cfg.wx*cfg.window_past\n            \n        #scale = 9.806 if self.dfs[row_idx, -1] == 1 else 1.0\n        x = self.dfs[row_idx - cfg.wx*cfg.window_past : row_idx + cfg.wx*cfg.window_future, 1:4]\n        x = x[::cfg.wx, :][::-1, :]\n        x = torch.tensor(x.astype('float'))#/scale\n        \n        t = self.dfs[row_idx, -3]*self.dfs[row_idx, -2]\n        \n        if self.split == \"test\":\n            return _id, x, t\n        \n        y = self.dfs[row_idx, 4:7].astype('float')\n        y = torch.tensor(y)\n        \n        return x, y, t\n    \n    def __len__(self):\n        # return self.length\n        if self.split == \"train\":\n            return 5_000_000\n        return self.length","metadata":{"execution":{"iopub.status.busy":"2024-04-11T10:30:04.564407Z","iopub.execute_input":"2024-04-11T10:30:04.564848Z","iopub.status.idle":"2024-04-11T10:30:04.587170Z","shell.execute_reply.started":"2024-04-11T10:30:04.564811Z","shell.execute_reply":"2024-04-11T10:30:04.586225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T10:30:04.588517Z","iopub.execute_input":"2024-04-11T10:30:04.589315Z","iopub.status.idle":"2024-04-11T10:30:04.736313Z","shell.execute_reply.started":"2024-04-11T10:30:04.589274Z","shell.execute_reply":"2024-04-11T10:30:04.735090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"def _block(in_features, out_features, drop_rate):\n    return nn.Sequential(\n        nn.Linear(in_features, out_features),\n        nn.BatchNorm1d(out_features),\n        nn.Linear(out_features, out_features), # 全連接層，保持特徵維度不變\n        nn.BatchNorm1d(out_features), # Batch Normalization，對輸出進行標準化\n        nn.Linear(out_features, out_features), # 全連接層，保持特徵維度不變\n        nn.BatchNorm1d(out_features), # Batch Normalization，對輸出進行標準化\n        nn.ReLU(),\n        nn.Dropout(drop_rate)\n    )\n\nclass FOGModel(nn.Module):\n    def __init__(self, p=cfg.model_dropout, dim=cfg.model_hidden, nblocks=cfg.model_nblocks):\n        super(FOGModel, self).__init__()\n        self.dropout = nn.Dropout(p)\n        self.in_layer = nn.Linear(cfg.window_size*3, dim)\n        self.blocks = nn.Sequential(*[_block(dim, dim, p) for _ in range(nblocks)])\n        self.out_layer = nn.Linear(dim, 3)\n        \n    def forward(self, x):\n        x = x.view(-1, cfg.window_size*3)\n        x = self.in_layer(x)\n        for block in self.blocks:\n            x = block(x)\n        x = self.out_layer(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2024-04-11T10:30:04.737688Z","iopub.execute_input":"2024-04-11T10:30:04.738290Z","iopub.status.idle":"2024-04-11T10:30:04.749241Z","shell.execute_reply.started":"2024-04-11T10:30:04.738255Z","shell.execute_reply":"2024-04-11T10:30:04.748305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def count_parameters(model):\n    return sum(p.numel() for p in model.parameters() if p.requires_grad)","metadata":{"execution":{"iopub.status.busy":"2024-04-11T10:30:04.750561Z","iopub.execute_input":"2024-04-11T10:30:04.750930Z","iopub.status.idle":"2024-04-11T10:30:04.763082Z","shell.execute_reply.started":"2024-04-11T10:30:04.750888Z","shell.execute_reply":"2024-04-11T10:30:04.762116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"from torch.cuda.amp import GradScaler\n\ndef train_one_epoch(model, loader, optimizer, criterion):\n    loss_sum = 0.\n    scaler = GradScaler()\n    \n    model.train()\n    for x,y,t in tqdm(loader):\n        x = x.to(cfg.device).float()\n        y = y.to(cfg.device).float()\n        t = t.to(cfg.device).float()\n        \n        y_pred = model(x)\n        loss = criterion(y_pred, y)\n        loss = torch.mean(loss*t.unsqueeze(-1), dim=1)\n        \n        t_sum = torch.sum(t)\n        if t_sum > 0:\n            loss = torch.sum(loss)/t_sum\n        else:\n            loss = torch.sum(loss)*0.\n        \n        # loss.backward()\n        scaler.scale(loss).backward()\n        # optimizer.step()\n        scaler.step(optimizer)\n        scaler.update()\n        \n        optimizer.zero_grad()\n        \n        loss_sum += loss.item()\n    \n    print(f\"Train Loss: {(loss_sum/len(loader)):.04f}\")\n    \n\ndef validation_one_epoch(model, loader, criterion):\n    loss_sum = 0.\n    y_true_epoch = []\n    y_pred_epoch = []\n    t_valid_epoch = []\n    \n    model.eval()\n    for x,y,t in tqdm(loader):\n        x = x.to(cfg.device).float()\n        y = y.to(cfg.device).float()\n        t = t.to(cfg.device).float()\n        \n        with torch.no_grad():\n            y_pred = model(x)\n            loss = criterion(y_pred, y)\n            loss = torch.mean(loss*t.unsqueeze(-1), dim=1)\n            \n            t_sum = torch.sum(t)\n            if t_sum > 0:\n                loss = torch.sum(loss)/t_sum\n            else:\n                loss = torch.sum(loss)*0.\n        \n        loss_sum += loss.item()\n        y_true_epoch.append(y.cpu().numpy())\n        y_pred_epoch.append(y_pred.cpu().numpy())\n        t_valid_epoch.append(t.cpu().numpy())\n        \n    y_true_epoch = np.concatenate(y_true_epoch, axis=0)\n    y_pred_epoch = np.concatenate(y_pred_epoch, axis=0)\n    \n    t_valid_epoch = np.concatenate(t_valid_epoch, axis=0)\n    y_true_epoch = y_true_epoch[t_valid_epoch > 0, :]\n    y_pred_epoch = y_pred_epoch[t_valid_epoch > 0, :]\n    \n    scores = [average_precision_score(y_true_epoch[:,i], y_pred_epoch[:,i]) for i in range(3)]\n    mean_score = np.mean(scores)\n    print(f\"Validation Loss: {(loss_sum/len(loader)):.04f}, Validation Score: {mean_score:.03f}, ClassWise: {scores[0]:.03f},{scores[1]:.03f},{scores[2]:.03f}\")\n    \n    return mean_score","metadata":{"execution":{"iopub.status.busy":"2024-04-11T10:30:04.764453Z","iopub.execute_input":"2024-04-11T10:30:04.764842Z","iopub.status.idle":"2024-04-11T10:30:04.783766Z","shell.execute_reply.started":"2024-04-11T10:30:04.764813Z","shell.execute_reply":"2024-04-11T10:30:04.782702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = FOGModel().to(cfg.device)\nprint(f\"Number of parameters in model - {count_parameters(model):,}\")\n\ntrain_dataset = FOGDataset(train_fpaths, split=\"train\")\nvalid_dataset = FOGDataset(valid_fpaths, split=\"valid\")\nprint(f\"lengths of datasets: train - {len(train_dataset)}, valid - {len(valid_dataset)}\")\n\ntrain_loader = DataLoader(train_dataset, batch_size=cfg.batch_size, num_workers=5, shuffle=True)\nvalid_loader = DataLoader(valid_dataset, batch_size=cfg.batch_size, num_workers=5)\n\noptimizer = torch.optim.Adam(model.parameters(), lr=cfg.lr)\ncriterion = torch.nn.BCEWithLogitsLoss(reduction='none').to(cfg.device)\n# sched = torch.optim.lr_scheduler.StepLR(optimizer, step_size=1, gamma=0.85)\n\nmax_score = 0.0\n\nprint(\"=\"*50)\nfor epoch in range(cfg.num_epochs):\n    print(f\"Epoch: {epoch}\")\n    train_one_epoch(model, train_loader, optimizer, criterion)\n    score = validation_one_epoch(model, valid_loader, criterion)\n    # sched.step()\n\n    if score > max_score:\n        max_score = score\n        torch.save(model.state_dict(), \"best_model_state.h5\")\n        print(\"Saving Model ...\")\n\n    print(\"=\"*50)\n    \ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T10:30:04.785046Z","iopub.execute_input":"2024-04-11T10:30:04.785322Z","iopub.status.idle":"2024-04-11T11:13:14.937734Z","shell.execute_reply.started":"2024-04-11T10:30:04.785295Z","shell.execute_reply":"2024-04-11T11:13:14.936651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"model = FOGModel().to(cfg.device)\nmodel.load_state_dict(torch.load(\"/kaggle/working/best_model_state.h5\"))\nmodel.eval()\n\ntest_defog_paths = glob.glob(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/defog/*.csv\")\ntest_tdcsfog_paths = glob.glob(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog/*.csv\")\ntest_fpaths = [(f, 'de') for f in test_defog_paths] + [(f, 'tdcs') for f in test_tdcsfog_paths]\n\ntest_dataset = FOGDataset(test_fpaths, split=\"test\")\ntest_loader = DataLoader(test_dataset, batch_size=cfg.batch_size, num_workers=5)\n\nids = []\npreds = []\n\nfor _id, x, _ in tqdm(test_loader):\n    x = x.to(cfg.device).float()\n    with torch.no_grad():\n        y_pred = model(x)*0.1\n    \n    ids.extend(_id)\n    preds.extend(list(np.nan_to_num(y_pred.cpu().numpy())))","metadata":{"execution":{"iopub.status.busy":"2024-04-11T11:13:14.939166Z","iopub.execute_input":"2024-04-11T11:13:14.939488Z","iopub.status.idle":"2024-04-11T11:13:18.582746Z","shell.execute_reply.started":"2024-04-11T11:13:14.939440Z","shell.execute_reply":"2024-04-11T11:13:18.581604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/sample_submission.csv\")\nsample_submission.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-11T11:13:18.586799Z","iopub.execute_input":"2024-04-11T11:13:18.587490Z","iopub.status.idle":"2024-04-11T11:13:18.857424Z","shell.execute_reply.started":"2024-04-11T11:13:18.587440Z","shell.execute_reply":"2024-04-11T11:13:18.856232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = np.array(preds)\nsubmission = pd.DataFrame({'Id': ids, 'StartHesitation': np.round(preds[:,0],5), \\\n                           'Turn': np.round(preds[:,1],5), 'Walking': np.round(preds[:,2],5)})\n\nsubmission = pd.merge(sample_submission[['Id']], submission, how='left', on='Id').fillna(0.0)\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-11T11:13:18.858693Z","iopub.execute_input":"2024-04-11T11:13:18.859000Z","iopub.status.idle":"2024-04-11T11:13:20.658710Z","shell.execute_reply.started":"2024-04-11T11:13:18.858970Z","shell.execute_reply":"2024-04-11T11:13:20.657834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(submission.shape)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T11:13:20.659952Z","iopub.execute_input":"2024-04-11T11:13:20.660286Z","iopub.status.idle":"2024-04-11T11:13:20.677310Z","shell.execute_reply.started":"2024-04-11T11:13:20.660256Z","shell.execute_reply":"2024-04-11T11:13:20.676349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}