{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Copy from https://www.kaggle.com/code/mayukh18/pytorch-fog-end-to-end-baseline-lb-0-254","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nimport random\nimport time\nimport math\nimport json\nfrom tqdm import tqdm\nimport glob\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split, StratifiedGroupKFold\nfrom sklearn.metrics import accuracy_score, average_precision_score\n\nimport warnings\nwarnings.filterwarnings(action='ignore')\nprint(\"Env set ok ..\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-13T11:34:28.617200Z","iopub.execute_input":"2023-06-13T11:34:28.617588Z","iopub.status.idle":"2023-06-13T11:34:32.285482Z","shell.execute_reply.started":"2023-06-13T11:34:28.617558Z","shell.execute_reply":"2023-06-13T11:34:32.284500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Stratified Group K Fold\n\nIt's mentioned in the data that the subjects are different in the train and test set and even different between the public/private splits of the test data. So we need to use Stratified Group K Fold. But since the positive instances in the sequences are very scarce, we need to pick up the best fold which will give us the best balance of the positive/negative instances.","metadata":{}},{"cell_type":"markdown","source":"### tdcsfog preprocessing","metadata":{}},{"cell_type":"code","source":"# Analysis of positive instances in each fold of our CV folds\n\nn1_sum = []\nn2_sum = []\nn3_sum = []\ncount = []\n\n# Here I am using the metadata file available during training. Since the code \n# will run again during submission, if \n# I used the usual file from the competition folder, \n# it would have been updated with the test files too.\nmetadata = pd.read_csv(\"/kaggle/input/copy-train-metadata/tdcsfog_metadata.csv\")\n\nfor f in tqdm(metadata['Id']):\n    fpath = f\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/{f}.csv\"\n    df = pd.read_csv(fpath)\n    \n    n1_sum.append(np.sum(df['StartHesitation']))\n    n2_sum.append(np.sum(df['Turn']))\n    n3_sum.append(np.sum(df['Walking']))\n    count.append(len(df))\n    \nprint(f\"32 files have positive values in all 3 classes\")\n\nmetadata['n1_sum'] = n1_sum\nmetadata['n2_sum'] = n2_sum\nmetadata['n3_sum'] = n3_sum\nmetadata['count'] = count\n\nsgkf = StratifiedGroupKFold(n_splits=5, random_state=42, shuffle=True)\nfor i, (train_index, valid_index) in enumerate(sgkf.split(X=metadata['Id'], y=[1]*len(metadata), groups=metadata['Subject'])):\n    print(f\"Fold = {i}\")\n    train_ids = metadata.loc[train_index, 'Id']\n    valid_ids = metadata.loc[valid_index, 'Id']\n    \n    print(f\"Length of Train = {len(train_index)}, Length of Valid = {len(valid_index)}\")\n    n1_sum = metadata.loc[train_index, 'n1_sum'].sum()\n    n2_sum = metadata.loc[train_index, 'n2_sum'].sum()\n    n3_sum = metadata.loc[train_index, 'n3_sum'].sum()\n    print(f\"Train classes: {n1_sum:,}, {n2_sum:,}, {n3_sum:,}\")\n    \n    n1_sum = metadata.loc[valid_index, 'n1_sum'].sum()\n    n2_sum = metadata.loc[valid_index, 'n2_sum'].sum()\n    n3_sum = metadata.loc[valid_index, 'n3_sum'].sum()\n    print(f\"Valid classes: {n1_sum:,}, {n2_sum:,}, {n3_sum:,}\")\n    \n# FOLD 2 is the most well balanced\n# The actual train-test split (based on Fold 2)\n\nmetadata = pd.read_csv(\"/kaggle/input/copy-train-metadata/tdcsfog_metadata.csv\")\nsgkf = StratifiedGroupKFold(n_splits=5, random_state=42, shuffle=True)\nfor i, (train_index, valid_index) in enumerate(sgkf.split(X=metadata['Id'], y=[1]*len(metadata), groups=metadata['Subject'])):\n    if i != 2:\n        continue\n    print(f\"Fold = {i}\")\n    train_ids = metadata.loc[train_index, 'Id']\n    valid_ids = metadata.loc[valid_index, 'Id']\n    print(f\"Length of Train = {len(train_ids)}, Length of Valid = {len(valid_ids)}\")\n    \n    if i == 2:\n        break\n        \ntrain_fpaths_tdcs = [f\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/{_id}.csv\" for _id in train_ids]\nvalid_fpaths_tdcs = [f\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/{_id}.csv\" for _id in valid_ids]","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:43:03.505931Z","iopub.execute_input":"2023-06-13T09:43:03.506724Z","iopub.status.idle":"2023-06-13T09:43:22.381407Z","shell.execute_reply.started":"2023-06-13T09:43:03.506689Z","shell.execute_reply":"2023-06-13T09:43:22.380496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### defog preprocessing","metadata":{}},{"cell_type":"code","source":"# Analysis of positive instances in each fold of our CV folds\n\nn1_sum = []\nn2_sum = []\nn3_sum = []\ncount = []\n\n# Here I am using the metadata file available during training. Since the code will run again during submission, if \n# I used the usual file from the competition folder, it would have been updated with the test files too.\nmetadata = pd.read_csv(\"/kaggle/input/copy-train-metadata/defog_metadata.csv\")\nmetadata['n1_sum'] = 0\nmetadata['n2_sum'] = 0\nmetadata['n3_sum'] = 0\nmetadata['count'] = 0\n\nfor f in tqdm(metadata['Id']):\n    fpath = f\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog/{f}.csv\"\n    if os.path.exists(fpath) == False:\n        continue\n        \n    df = pd.read_csv(fpath)\n    metadata.loc[metadata['Id'] == f, 'n1_sum'] = np.sum(df['StartHesitation'])\n    metadata.loc[metadata['Id'] == f, 'n2_sum'] = np.sum(df['Turn'])\n    metadata.loc[metadata['Id'] == f, 'n3_sum'] = np.sum(df['Walking'])\n    metadata.loc[metadata['Id'] == f, 'count'] = len(df)\n    \nmetadata = metadata[metadata['count'] > 0].reset_index()\n\nsgkf = StratifiedGroupKFold(n_splits=5, random_state=42, shuffle=True)\nfor i, (train_index, valid_index) in enumerate(sgkf.split(X=metadata['Id'], y=[1]*len(metadata), groups=metadata['Subject'])):\n    print(f\"Fold = {i}\")\n    train_ids = metadata.loc[train_index, 'Id']\n    valid_ids = metadata.loc[valid_index, 'Id']\n    \n    print(f\"Length of Train = {len(train_index)}, Length of Valid = {len(valid_index)}\")\n    n1_sum = metadata.loc[train_index, 'n1_sum'].sum()\n    n2_sum = metadata.loc[train_index, 'n2_sum'].sum()\n    n3_sum = metadata.loc[train_index, 'n3_sum'].sum()\n    print(f\"Train classes: {n1_sum:,}, {n2_sum:,}, {n3_sum:,}\")\n    \n    n1_sum = metadata.loc[valid_index, 'n1_sum'].sum()\n    n2_sum = metadata.loc[valid_index, 'n2_sum'].sum()\n    n3_sum = metadata.loc[valid_index, 'n3_sum'].sum()\n    print(f\"Valid classes: {n1_sum:,}, {n2_sum:,}, {n3_sum:,}\")\n    \n# FOLD 2 is the most well balanced\n# The actual train-test split (based on Fold 2)\n\nsgkf = StratifiedGroupKFold(n_splits=5, random_state=42, shuffle=True)\nfor i, (train_index, valid_index) in enumerate(sgkf.split(X=metadata['Id'], y=[1]*len(metadata), groups=metadata['Subject'])):\n    if i != 1:\n        continue\n    print(f\"Fold = {i}\")\n    train_ids = metadata.loc[train_index, 'Id']\n    valid_ids = metadata.loc[valid_index, 'Id']\n    print(f\"Length of Train = {len(train_ids)}, Length of Valid = {len(valid_ids)}\")\n    \n    if i == 2:\n        break\n        \ntrain_fpaths_de = [f\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog/{_id}.csv\" for _id in train_ids]\nvalid_fpaths_de = [f\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog/{_id}.csv\" for _id in valid_ids]","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:43:22.386250Z","iopub.execute_input":"2023-06-13T09:43:22.386725Z","iopub.status.idle":"2023-06-13T09:43:45.891422Z","shell.execute_reply.started":"2023-06-13T09:43:22.386687Z","shell.execute_reply":"2023-06-13T09:43:45.890317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_fpaths = [(f, 'de') for f in train_fpaths_de] + [(f, 'tdcs') for f in train_fpaths_tdcs]\nvalid_fpaths = [(f, 'de') for f in valid_fpaths_de] + [(f, 'tdcs') for f in valid_fpaths_tdcs]","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:43:45.894781Z","iopub.execute_input":"2023-06-13T09:43:45.895578Z","iopub.status.idle":"2023-06-13T09:43:45.901770Z","shell.execute_reply.started":"2023-06-13T09:43:45.895540Z","shell.execute_reply":"2023-06-13T09:43:45.900804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# DataLoader\n\nWe use a window comprised of past and future time Acc readings to form our dataset for a particular time instance. In case some portion of the window data is not available, we pad them with zeros.","metadata":{}},{"cell_type":"code","source":"class FOGDataset1(Dataset):\n    def __init__(self, fpaths, datacfg, scale=9.806, split=\"train\"):\n        super(FOGDataset, self).__init__()\n        tm = time.time()\n        self.split = split\n        self.scale = scale\n        self.datacfg = datacfg\n        self.fpaths = fpaths\n        self.dfs = [self.read(f[0], f[1]) for f in fpaths]\n        self.f_ids = [os.path.basename(f[0])[:-4] for f in self.fpaths]\n        \n        self.end_indices = []\n        self.shapes = []\n        _length = 0\n        for df in self.dfs:\n            self.shapes.append(df.shape[0])\n            _length += df.shape[0]\n            self.end_indices.append(_length)\n        \n        self.dfs = np.concatenate(self.dfs, axis=0).astype(np.float16)\n        self.length = self.dfs.shape[0]\n        \n        shape1 = self.dfs.shape[1]\n        \n        self.dfs = np.concatenate([np.zeros((cfg.wx*cfg.window_past, shape1)), self.dfs, np.zeros((cfg.wx*cfg.window_future, shape1))], axis=0)\n        print(f\"Dataset initialized in {time.time() - tm} secs!\")\n        gc.collect()\n        \n    def read(self, f, _type):\n        df = pd.read_csv(f)\n        if self.split == \"test\":\n            return np.array(df)\n        \n        if _type ==\"tdcs\":\n            df['Valid'] = 1\n            df['Task'] = 1\n            df['tdcs'] = 1\n        else:\n            df['tdcs'] = 0\n        \n        return np.array(df)\n            \n    def __getitem__(self, index):\n        if self.split == \"train\":\n            row_idx = random.randint(0, self.length-1) + cfg.wx*cfg.window_past\n        elif self.split == \"test\":\n            for i,e in enumerate(self.end_indices):\n                if index >= e:\n                    continue\n                df_idx = i\n                break\n\n            row_idx_true = self.shapes[df_idx] - (self.end_indices[df_idx] - index)\n            _id = self.f_ids[df_idx] + \"_\" + str(row_idx_true)\n            row_idx = index + cfg.wx*cfg.window_past\n        else:\n            row_idx = index + cfg.wx*cfg.window_past\n            \n        #scale = 9.806 if self.dfs[row_idx, -1] == 1 else 1.0\n        x = self.dfs[\n            row_idx - self.datacfg.wx*self.datacfg.window_past : \\\n            row_idx + self.datacfg.wx*self.datacfg.window_future, 1:4\n        ]\n        x = x[::self.datacfg.wx, :][::-1, :]\n        x = torch.tensor(x.astype('float'))#/scale\n        \n        t = self.dfs[row_idx, -3]*self.dfs[row_idx, -2]\n        \n        if self.split == \"test\":\n            return _id, x, t\n        \n        y = self.dfs[row_idx, 4:7].astype('float')\n        y = torch.tensor(y)\n        \n        return x, y, t\n    \n    def __len__(self):\n        # return self.length\n        if self.split == \"train\":\n            return 5_000_000\n        return self.length","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:43:45.903314Z","iopub.execute_input":"2023-06-13T09:43:45.903715Z","iopub.status.idle":"2023-06-13T09:43:45.923981Z","shell.execute_reply.started":"2023-06-13T09:43:45.903658Z","shell.execute_reply":"2023-06-13T09:43:45.922945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class FOGDataset(Dataset):\n    def __init__(self, fpaths, datacfg, scale=9.806, split=\"train\"):\n        super(FOGDataset, self).__init__()\n        tm = time.time()\n        self.split = split\n        self.scale = scale\n        self.datacfg = datacfg\n        self.fpaths = fpaths\n        self.dfs = [self.read(f[0], f[1]) for f in fpaths]\n        self.f_ids = [os.path.basename(f[0])[:-4] for f in self.fpaths]\n        \n        self.end_indices = []\n        self.shapes = []\n        _length = 0\n        for df in self.dfs:\n            self.shapes.append(df.shape[0])\n            _length += df.shape[0]\n            self.end_indices.append(_length)\n        \n        self.dfs = np.concatenate(self.dfs, axis=0).astype(np.float16)\n        self.length = self.dfs.shape[0]\n        \n        shape1 = self.dfs.shape[1]\n        \n        self.dfs = np.concatenate(\n            [\n                np.zeros((datacfg.wx*datacfg.window_past, shape1)), \n                self.dfs, np.zeros((datacfg.wx*datacfg.window_future, shape1))\n            ], \n            axis=0\n        )\n        print(f\"Dataset initialized in {time.time() - tm} secs!\")\n        gc.collect()\n        \n    def read(self, f, _type):\n        df = pd.read_csv(f)\n        if self.split == \"test\":\n            return np.array(df)\n        \n        if _type ==\"tdcs\":\n            df['Valid'] = 1\n            df['Task'] = 1\n            df['tdcs'] = 1\n        else:\n            df['tdcs'] = 0\n        \n        return np.array(df)\n            \n    def __getitem__(self, index):\n        if self.split == \"train\":\n            row_idx = random.randint(0, self.length-1) + self.datacfg.wx*self.datacfg.window_past\n        elif self.split == \"test\":\n            for i,e in enumerate(self.end_indices):\n                if index >= e:\n                    continue\n                df_idx = i\n                break\n\n            row_idx_true = self.shapes[df_idx] - (self.end_indices[df_idx] - index)\n            _id = self.f_ids[df_idx] + \"_\" + str(row_idx_true)\n            row_idx = index + self.datacfg.wx*self.datacfg.window_past\n        else:\n            row_idx = index + self.datacfg.wx*self.datacfg.window_past\n            \n        #scale = 9.806 if self.dfs[row_idx, -1] == 1 else 1.0\n        x = self.dfs[\n            row_idx - self.datacfg.wx*self.datacfg.window_past : \\\n            row_idx + self.datacfg.wx*self.datacfg.window_future, 1:4\n        ]\n        x = x[::self.datacfg.wx, :][::-1, :]\n        x = torch.tensor(x.astype('float'))#/scale\n        \n        t = self.dfs[row_idx, -3]*self.dfs[row_idx, -2]\n        \n        if self.split == \"test\":\n            return _id, x, t\n        \n        y = self.dfs[row_idx, 4:7].astype('float')\n        y = torch.tensor(y)\n        \n        return x, y, t\n    \n    def __len__(self):\n        # return self.length\n        if self.split == \"train\":\n            return 5_000_000\n        return self.length","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:43:45.925711Z","iopub.execute_input":"2023-06-13T09:43:45.926146Z","iopub.status.idle":"2023-06-13T09:43:45.945770Z","shell.execute_reply.started":"2023-06-13T09:43:45.926112Z","shell.execute_reply":"2023-06-13T09:43:45.944707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nclass DataConfig:\n    train_dir1 = \"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog\"\n    train_dir2 = \"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog\"\n\n    batch_size = 1024                              #batch size大小\n    window_size = 32                               #每次Input的總時序資料長度\n    window_future = 8                              #取目標時間點後多少為Input\n    window_past = window_size - window_future      #取目標時間點前多少為Input\n    \n    wx = 8                                         #資料padding長度的參數       \n    \n    feature_list = ['AccV', 'AccML', 'AccAP']\n    label_list = ['StartHesitation', 'Turn', 'Walking']\n\ndatacfg = DataConfig()\n    ","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:43:45.947487Z","iopub.execute_input":"2023-06-13T09:43:45.947890Z","iopub.status.idle":"2023-06-13T09:43:45.959984Z","shell.execute_reply.started":"2023-06-13T09:43:45.947837Z","shell.execute_reply":"2023-06-13T09:43:45.958989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = FOGDataset(train_fpaths, datacfg, split=\"train\")\nvalid_dataset = FOGDataset(valid_fpaths, datacfg, split=\"valid\")\nprint(f\"lengths of datasets: train - {len(train_dataset)}, valid - {len(valid_dataset)}\")\ntrain_loader = DataLoader(train_dataset, batch_size=datacfg.batch_size, num_workers=5, shuffle=True)\nvalid_loader = DataLoader(valid_dataset, batch_size=datacfg.batch_size, num_workers=5)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:43:45.961464Z","iopub.execute_input":"2023-06-13T09:43:45.961885Z","iopub.status.idle":"2023-06-13T09:44:41.113697Z","shell.execute_reply.started":"2023-06-13T09:43:45.961852Z","shell.execute_reply":"2023-06-13T09:44:41.112667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training kits","metadata":{}},{"cell_type":"code","source":"from torch.cuda.amp import GradScaler\n\ndef count_parameters(model:nn.Module):\n    return sum(p.numel() for p in model.parameters() if p.requires_grad)\n\n\ndef train_one_epoch(model, loader, optimizer, criterion, cfg):\n    loss_sum = 0.\n    scaler = GradScaler()\n    \n    model.train()\n    for x,y,t in tqdm(loader):                               #從DataLoader中取得一個Batch size的資料\n        x = x.to(cfg.device).float()\n        y = y.to(cfg.device).float()\n        t = t.to(cfg.device).float()\n        \n        y_pred = model(x)                                    #將Input x丟入模型，得到y_pred\n        loss = criterion(y_pred, y)                          #計算預測結果y_pred與GT y的loss\n        loss = torch.mean(loss*t.unsqueeze(-1), dim=1)\n        \n        t_sum = torch.sum(t)\n        if t_sum > 0:\n            loss = torch.sum(loss)/t_sum\n        else:\n            loss = torch.sum(loss)*0.\n        \n        # loss.backward()\n        scaler.scale(loss).backward()                        #根據loss做反向傳播\n        # optimizer.step()\n        scaler.step(optimizer)                               #優化器優化模型參數\n        scaler.update()                                      \n        \n        optimizer.zero_grad()                                #優化器歸零\n        \n        loss_sum += loss.item()\n    total_loss = (loss_sum/len(loader))\n    print(f\"Train Loss: {total_loss:.04f}\")\n    \n    return total_loss\n\ndef validation_one_epoch(model, loader, criterion, cfg):\n    loss_sum = 0.\n    y_true_epoch = []\n    y_pred_epoch = []\n    t_valid_epoch = []\n    \n    model.eval()\n    for x,y,t in tqdm(loader):\n        x = x.to(cfg.device).float()\n        y = y.to(cfg.device).float()\n        t = t.to(cfg.device).float()\n        \n        with torch.no_grad():                                #沒有反向傳播\n            y_pred = model(x)\n            loss = criterion(y_pred, y)\n            loss = torch.mean(loss*t.unsqueeze(-1), dim=1)\n            \n            t_sum = torch.sum(t)\n            if t_sum > 0:\n                loss = torch.sum(loss)/t_sum\n            else:\n                loss = torch.sum(loss)*0.\n        \n        loss_sum += loss.item()\n        y_true_epoch.append(y.cpu().numpy())\n        y_pred_epoch.append(y_pred.cpu().numpy())\n        t_valid_epoch.append(t.cpu().numpy())\n        \n    y_true_epoch = np.concatenate(y_true_epoch, axis=0)\n    y_pred_epoch = np.concatenate(y_pred_epoch, axis=0)\n    \n    t_valid_epoch = np.concatenate(t_valid_epoch, axis=0)\n    y_true_epoch = y_true_epoch[t_valid_epoch > 0, :]\n    y_pred_epoch = y_pred_epoch[t_valid_epoch > 0, :]\n    \n    scores = [average_precision_score(y_true_epoch[:,i], y_pred_epoch[:,i]) for i in range(3)]\n    mean_score = np.mean(scores)\n    print(f\"Validation Loss: {(loss_sum/len(loader)):.04f}, Validation Score: {mean_score:.03f}, ClassWise: {scores[0]:.03f},{scores[1]:.03f},{scores[2]:.03f}\")\n    \n    return mean_score\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:44:41.116331Z","iopub.execute_input":"2023-06-13T09:44:41.116753Z","iopub.status.idle":"2023-06-13T09:44:41.134458Z","shell.execute_reply.started":"2023-06-13T09:44:41.116701Z","shell.execute_reply":"2023-06-13T09:44:41.133437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:44:41.138164Z","iopub.execute_input":"2023-06-13T09:44:41.139277Z","iopub.status.idle":"2023-06-13T09:44:41.356497Z","shell.execute_reply.started":"2023-06-13T09:44:41.139222Z","shell.execute_reply":"2023-06-13T09:44:41.355282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Build Model & Optr","metadata":{}},{"cell_type":"code","source":"class ResNet20(nn.Module):\n    \n    def __init__(self, feature_channel, pooling_kernelsize = 3) -> None:\n        super(ResNet20, self).__init__()       \n        self.conv1 = nn.Sequential(\n            nn.Conv1d(feature_channel, 16, 3, padding=1),\n            nn.BatchNorm1d(16),\n            nn.ReLU(inplace=True)\n        )\n        self.ResBlocks = nn.Sequential(\n            self.__ResBlock(16, 16),\n            self.__ResBlock(16, 32),\n            self.__ResBlock(32, 64),\n        )\n        self.poolkernelsize = pooling_kernelsize\n    \n    class ResidualBlock(nn.Module) :\n    \n        def __init__(self, in_channel, out_channel) -> None:\n            super().__init__()\n            self.seq =nn.Sequential(\n                nn.Conv1d(in_channel, out_channel,3,padding=1),\n                nn.BatchNorm1d(out_channel),\n                nn.ReLU(inplace=True),\n                nn.Conv1d(out_channel, out_channel, 3,padding=1),\n                nn.BatchNorm1d(out_channel)\n            )\n            self.neck = nn.Sequential()\n            if in_channel != out_channel:\n                self.neck = nn.Sequential(\n                    nn.Conv1d(in_channel, out_channel,1),\n                    nn.BatchNorm1d(out_channel)\n                )\n        \n        def forward(self, x)->torch.Tensor:\n            return F.relu(self.seq(x) + self.neck(x))\n\n    def __ResBlock(self,cin, cout):\n        clist = [(cin, cout), (cout, cout), (cout, cout)]\n        return nn.Sequential(\n            *list(self.ResidualBlock(ci, co) for ci, co in clist)\n        )\n    \n    def forward(self, x)->torch.Tensor :\n        x1 = self.conv1(x.permute(0,2,1))\n        x1 = self.ResBlocks(x1)\n        x1 = x1.permute(0,2,1)\n        if self.poolkernelsize > 1:\n            x1 = F.avg_pool1d(x1, kernel_size=self.poolkernelsize)\n        return x1.reshape(x1.shape[0], -1) #flatten\n\n\nclass ResConvClassifier(nn.Module):\n    def __init__(\n            self, feature_channels, seqlen, out_features, \n            classifier_layers=[128], resnet_pooling_kernelsize = 1\n        ) -> None:\n        \n        \"\"\"\n        ResetNet20 output dimensions = floor(64/resnet_pooling_kernelsize)\n        \n        \"\"\"\n        \n        super(ResConvClassifier, self).__init__()\n        self.emd = ResNet20(feature_channels,resnet_pooling_kernelsize)\n        \n        assert resnet_pooling_kernelsize > 0\n        resnet_out = seqlen*int(math.floor(64.0/float(resnet_pooling_kernelsize)))\n        \n        layers = [resnet_out] + classifier_layers + [out_features]\n        if (layers[1] > layers[0]):\n            print(f\"Waring ! Resnet20 output are {layers[0]}\")\n            print(f\"and you give {layers[1]}, try to lift dimension ????? \")\n        if (layers[-1] > classifier_layers[-2]):\n            print(f\"Waring ! output are {classifier_layers[-1]}\")\n            print(f\"and you give {classifier_layers[-2]}, try to lift dimension ????? \")\n        \n        c = []\n        j = -1\n        for i in range(len(classifier_layers)):\n            c.append(self.__LinBlock(layers[i],layers[i+1]))\n            j = i\n        c.append(nn.Linear(layers[j+1],layers[j+2]))\n        self.classifier = nn.Sequential(*c)\n\n    def __LinBlock(self, ind, outd)->nn.Sequential:\n        return nn.Sequential(\n            nn.Linear(ind, outd),\n            nn.BatchNorm1d(outd),\n            nn.ReLU(inplace=True)\n        )\n    def forward(self, x):\n        x1 = self.emd(x)\n        x1 = self.classifier(x1)\n        return x1","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:44:41.361276Z","iopub.execute_input":"2023-06-13T09:44:41.362067Z","iopub.status.idle":"2023-06-13T09:44:41.389441Z","shell.execute_reply.started":"2023-06-13T09:44:41.362031Z","shell.execute_reply":"2023-06-13T09:44:41.388418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    \n    optimizer_name = \"Adam\"                       \n    loss_function = \"BCEWithLogitsLoss\"           \n    lr = 0.001                           \n    num_epochs = 10                           \n    device = 'cuda:0' if torch.cuda.is_available() else 'cpu'         \n    classifier_layers = [1024, 1024, 1024, 1024]\n    pooling_kernel = 1 \n    \ncfg = Config()","metadata":{"execution":{"iopub.status.busy":"2023-06-13T11:34:55.061478Z","iopub.execute_input":"2023-06-13T11:34:55.061848Z","iopub.status.idle":"2023-06-13T11:34:55.067503Z","shell.execute_reply.started":"2023-06-13T11:34:55.061818Z","shell.execute_reply":"2023-06-13T11:34:55.066495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = ResConvClassifier(\n    feature_channels=len(datacfg.feature_list), \n    seqlen=datacfg.window_size, \n    out_features=len(datacfg.label_list), \n    resnet_pooling_kernelsize = cfg.pooling_kernel,\n    classifier_layers = cfg.classifier_layers\n).to(cfg.device)\nprint(f\"Number of parameters in model - {count_parameters(model):,}\")\n\noptimizer = getattr(torch.optim, cfg.optimizer_name)(model.parameters(), lr=cfg.lr)\ncriterion = getattr(torch.nn, cfg.loss_function)(reduction='none').to(cfg.device)","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:44:41.442029Z","iopub.execute_input":"2023-06-13T09:44:41.443400Z","iopub.status.idle":"2023-06-13T09:44:44.392837Z","shell.execute_reply.started":"2023-06-13T09:44:41.443360Z","shell.execute_reply":"2023-06-13T09:44:44.391881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training and Validation","metadata":{}},{"cell_type":"code","source":"loss_per_iter = []\nvalid_per_iter = []\nmax_score = 0.0\n#sched = torch.optim.lr_scheduler.StepLR(optimizer, step_size=1, gamma=0.85)\nfor epoch in range(cfg.num_epochs):\n    print(f\"Epoch: {epoch}\")\n    loss = train_one_epoch(model, train_loader, optimizer, criterion,cfg)\n    score = validation_one_epoch(model, valid_loader, criterion, cfg)\n    #sched.step()\n    loss_per_iter.append(loss)\n    valid_per_iter.append(score)\n    \n    if score > max_score:\n        max_score = score\n        torch.save(model.state_dict(), \"best_model_state_b1.h5\")      #儲存最好的模型參數\n        print(\"Saving Model ...\")\n\n    print(\"=\"*50)\n    \ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:44:44.394138Z","iopub.execute_input":"2023-06-13T09:44:44.394818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"model =ResConvClassifier(\n    feature_channels=len(datacfg.feature_list), \n    seqlen=datacfg.window_size, \n    out_features=len(datacfg.label_list), \n    resnet_pooling_kernelsize = cfg.pooling_kernel,\n    classifier_layers = cfg.classifier_layers\n).to(cfg.device)\nmodel.load_state_dict(torch.load(\"/kaggle/working/best_model_state_b1.h5\"))             #取得最好的模型參數\nmodel.eval()\n\ntest_defog_paths = glob.glob(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/defog/*.csv\")\ntest_tdcsfog_paths = glob.glob(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog/*.csv\")\ntest_fpaths = [(f, 'de') for f in test_defog_paths] + [(f, 'tdcs') for f in test_tdcsfog_paths]\n\ntest_dataset = FOGDataset(test_fpaths, datacfg, split=\"test\")\ntest_loader = DataLoader(test_dataset, batch_size=datacfg.batch_size, num_workers=5)\n\nids = []\npreds = []\n\nfor _id, x, _ in tqdm(test_loader):\n    x = x.to(cfg.device).float()\n    with torch.no_grad():\n        y_pred = model(x)*0.1\n    \n    ids.extend(_id)\n    preds.extend(list(np.nan_to_num(y_pred.cpu().numpy())))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/sample_submission.csv\")\nsample_submission.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = np.array(preds)\nsubmission = pd.DataFrame(\n    {\n        'Id': ids, \n        'StartHesitation': np.round(preds[:,0],5),                 \n        'Turn': np.round(preds[:,1],5), \n        'Walking': np.round(preds[:,2],5)\n    }\n)\n\nsubmission = pd.merge(sample_submission[['Id']], submission, how='left', on='Id').fillna(0.0)\nsubmission.to_csv(\"submission.csv\", index=False)\nprint(submission.shape)\nsubmission.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loss \nplt.figure(dpi=800)\nplt.plot(np.arange(cfg.num_epochs), loss_per_iter)\nplt.xlabel(\"epochs\")\nplt.ylabel(\"training loss\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-13T11:35:37.883758Z","iopub.execute_input":"2023-06-13T11:35:37.884200Z","iopub.status.idle":"2023-06-13T11:35:39.262024Z","shell.execute_reply.started":"2023-06-13T11:35:37.884164Z","shell.execute_reply":"2023-06-13T11:35:39.258144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# validation score\nplt.figure(dpi=800)\nplt.plot(np.arange(cfg.num_epochs), valid_per_iter)\nplt.xlabel(\"epochs\")\nplt.ylabel(\"validation score\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-13T11:35:43.172574Z","iopub.execute_input":"2023-06-13T11:35:43.173141Z","iopub.status.idle":"2023-06-13T11:35:44.527882Z","shell.execute_reply.started":"2023-06-13T11:35:43.173106Z","shell.execute_reply":"2023-06-13T11:35:44.526990Z"},"trusted":true},"execution_count":null,"outputs":[]}]}