{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# PyTorch Solution: Rolling Window\nHere we introduce a pytorch solution which takes an window of past and future Acc readings to predict the outcomes. The different segments of the notebook can be modified and improved as per your liking to better the whole pipeline. As a way, this works as a good starter baseline!\n\nVERSION 9: \ni. Added the defog data in training phase.\nii. Using FP16 mode.\n\n**Please leave an upvote if you found this notebook helpful!**","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nimport random\nimport time\n\nimport json\nfrom tqdm import tqdm\nimport glob\nimport numpy as np\nimport pandas as pd\n\nimport torch\nimport torch.nn as nn\n\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\n\nfrom sklearn.model_selection import train_test_split, StratifiedGroupKFold\nfrom sklearn.metrics import accuracy_score, average_precision_score\n\nimport warnings\nwarnings.filterwarnings(action='ignore')\n\nfrom torch.cuda.amp import GradScaler\n\n# =========== STFT =============\nimport scipy.signal as signal\nimport matplotlib.pyplot as plt\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-25T12:39:03.707683Z","iopub.execute_input":"2023-05-25T12:39:03.708368Z","iopub.status.idle":"2023-05-25T12:39:07.554543Z","shell.execute_reply.started":"2023-05-25T12:39:03.708336Z","shell.execute_reply":"2023-05-25T12:39:07.553388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    def __init__(self):\n        # DATASET INFO\n        self.train_dir1 = \"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog\"\n        self.train_dir2 = \"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog\"\n        self.defog_sr = 100 # Sample rate for DeFog\n        self.tdcsfog_sr = 128 # Sample rate for TdcsFog\n        self.defog_units = 'm/s^2'\n        self.tdcsfog_units = 'g'\n\n        self.batch_size = 1024\n        self.window_size = 32\n        self.window_future = 8\n        self.window_past = self.window_size - self.window_future\n\n        self.wx = 8\n\n        # FOR BASELINE MODEL\n    #     self.model_dropout = 0.2\n    #     self.model_hidden = 512\n    #     self.model_nblocks = 3\n\n        self.lr = 0.00015\n    #     self.lr = 0.01\n        self.num_epochs = 5\n        self.device = 'cuda' if torch.cuda.is_available() else 'cpu'\n\n        self.feature_list = ['AccV', 'AccML', 'AccAP']\n        self.label_list = ['StartHesitation', 'Turn', 'Walking']\n        \n        self.submission = True\n    \n    \ncfg = Config()","metadata":{"execution":{"iopub.status.busy":"2023-05-24T10:29:43.989024Z","iopub.execute_input":"2023-05-24T10:29:43.989532Z","iopub.status.idle":"2023-05-24T10:29:44.000156Z","shell.execute_reply.started":"2023-05-24T10:29:43.989460Z","shell.execute_reply":"2023-05-24T10:29:43.998874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cfg.device","metadata":{"execution":{"iopub.status.busy":"2023-05-24T10:29:44.216271Z","iopub.execute_input":"2023-05-24T10:29:44.219075Z","iopub.status.idle":"2023-05-24T10:29:44.228922Z","shell.execute_reply.started":"2023-05-24T10:29:44.219019Z","shell.execute_reply":"2023-05-24T10:29:44.227675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not cfg.submission:\n    !pip3 install torchsummary\n    from torchsummary import summary","metadata":{"execution":{"iopub.status.busy":"2023-05-24T10:29:47.653944Z","iopub.execute_input":"2023-05-24T10:29:47.654377Z","iopub.status.idle":"2023-05-24T10:29:47.661536Z","shell.execute_reply.started":"2023-05-24T10:29:47.654340Z","shell.execute_reply":"2023-05-24T10:29:47.659809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Stratified Group K Fold\n\nIt's mentioned in the data that the subjects are different in the train and test set and even different between the public/private splits of the test data. So we need to use Stratified Group K Fold. But since the positive instances in the sequences are very scarce, we need to pick up the best fold which will give us the best balance of the positive/negative instances.","metadata":{}},{"cell_type":"markdown","source":"### tdcsfog","metadata":{}},{"cell_type":"code","source":"# Analysis of positive instances in each fold of our CV folds\n\nn1_sum = []\nn2_sum = []\nn3_sum = []\ncount = []\n\n# Here I am using the metadata file available during training. Since the code will run again during submission, if \n# I used the usual file from the competition folder, it would have been updated with the test files too.\nmetadata = pd.read_csv(\"/kaggle/input/copy-train-metadata/tdcsfog_metadata.csv\")\n\nfor f in tqdm(metadata['Id']):\n    fpath = f\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/{f}.csv\"\n    df = pd.read_csv(fpath)\n    \n    n1_sum.append(np.sum(df['StartHesitation']))\n    n2_sum.append(np.sum(df['Turn']))\n    n3_sum.append(np.sum(df['Walking']))\n    count.append(len(df))\n    \nprint(f\"32 files have positive values in all 3 classes\")\n\nmetadata['n1_sum'] = n1_sum\nmetadata['n2_sum'] = n2_sum\nmetadata['n3_sum'] = n3_sum\nmetadata['count'] = count\n\nsgkf = StratifiedGroupKFold(n_splits=5, random_state=42, shuffle=True)\nfor i, (train_index, valid_index) in enumerate(sgkf.split(X=metadata['Id'], y=[1]*len(metadata), groups=metadata['Subject'])):\n    print(f\"Fold = {i}\")\n    train_ids = metadata.loc[train_index, 'Id']\n    valid_ids = metadata.loc[valid_index, 'Id']\n    \n    print(f\"Length of Train = {len(train_index)}, Length of Valid = {len(valid_index)}\")\n    n1_sum = metadata.loc[train_index, 'n1_sum'].sum()\n    n2_sum = metadata.loc[train_index, 'n2_sum'].sum()\n    n3_sum = metadata.loc[train_index, 'n3_sum'].sum()\n    print(f\"Train classes: {n1_sum:,}, {n2_sum:,}, {n3_sum:,}\")\n    \n    n1_sum = metadata.loc[valid_index, 'n1_sum'].sum()\n    n2_sum = metadata.loc[valid_index, 'n2_sum'].sum()\n    n3_sum = metadata.loc[valid_index, 'n3_sum'].sum()\n    print(f\"Valid classes: {n1_sum:,}, {n2_sum:,}, {n3_sum:,}\")\n    \n# FOLD 2 is the most well balanced\n# The actual train-test split (based on Fold 2)\n\nmetadata = pd.read_csv(\"/kaggle/input/copy-train-metadata/tdcsfog_metadata.csv\")\nsgkf = StratifiedGroupKFold(n_splits=5, random_state=42, shuffle=True)\nfor i, (train_index, valid_index) in enumerate(sgkf.split(X=metadata['Id'], y=[1]*len(metadata), groups=metadata['Subject'])):\n    if i != 2:\n        continue\n    print(f\"Fold = {i}\")\n    train_ids = metadata.loc[train_index, 'Id']\n    valid_ids = metadata.loc[valid_index, 'Id']\n    print(f\"Length of Train = {len(train_ids)}, Length of Valid = {len(valid_ids)}\")\n    \n    if i == 2:\n        break\n        \ntrain_fpaths_tdcs = [f\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/{_id}.csv\" for _id in train_ids]\nvalid_fpaths_tdcs = [f\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/{_id}.csv\" for _id in valid_ids]","metadata":{"execution":{"iopub.status.busy":"2023-05-24T10:29:48.828820Z","iopub.execute_input":"2023-05-24T10:29:48.829587Z","iopub.status.idle":"2023-05-24T10:30:07.750836Z","shell.execute_reply.started":"2023-05-24T10:29:48.829545Z","shell.execute_reply":"2023-05-24T10:30:07.749435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### defog","metadata":{}},{"cell_type":"code","source":"# Analysis of positive instances in each fold of our CV folds\n\nn1_sum = []\nn2_sum = []\nn3_sum = []\ncount = []\n\n# Here I am using the metadata file available during training. Since the code will run again during submission, if \n# I used the usual file from the competition folder, it would have been updated with the test files too.\nmetadata = pd.read_csv(\"/kaggle/input/copy-train-metadata/defog_metadata.csv\")\nmetadata['n1_sum'] = 0\nmetadata['n2_sum'] = 0\nmetadata['n3_sum'] = 0\nmetadata['count'] = 0\n\nfor f in tqdm(metadata['Id']):\n    fpath = f\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog/{f}.csv\"\n    if os.path.exists(fpath) == False:\n        continue\n        \n    df = pd.read_csv(fpath)\n    metadata.loc[metadata['Id'] == f, 'n1_sum'] = np.sum(df['StartHesitation'])\n    metadata.loc[metadata['Id'] == f, 'n2_sum'] = np.sum(df['Turn'])\n    metadata.loc[metadata['Id'] == f, 'n3_sum'] = np.sum(df['Walking'])\n    metadata.loc[metadata['Id'] == f, 'count'] = len(df)\n    \nmetadata = metadata[metadata['count'] > 0].reset_index()\n\nsgkf = StratifiedGroupKFold(n_splits=5, random_state=42, shuffle=True)\nfor i, (train_index, valid_index) in enumerate(sgkf.split(X=metadata['Id'], y=[1]*len(metadata), groups=metadata['Subject'])):\n    print(f\"Fold = {i}\")\n    train_ids = metadata.loc[train_index, 'Id']\n    valid_ids = metadata.loc[valid_index, 'Id']\n    \n    print(f\"Length of Train = {len(train_index)}, Length of Valid = {len(valid_index)}\")\n    n1_sum = metadata.loc[train_index, 'n1_sum'].sum()\n    n2_sum = metadata.loc[train_index, 'n2_sum'].sum()\n    n3_sum = metadata.loc[train_index, 'n3_sum'].sum()\n    print(f\"Train classes: {n1_sum:,}, {n2_sum:,}, {n3_sum:,}\")\n    \n    n1_sum = metadata.loc[valid_index, 'n1_sum'].sum()\n    n2_sum = metadata.loc[valid_index, 'n2_sum'].sum()\n    n3_sum = metadata.loc[valid_index, 'n3_sum'].sum()\n    print(f\"Valid classes: {n1_sum:,}, {n2_sum:,}, {n3_sum:,}\")\n    \n# FOLD 2 is the most well balanced\n# The actual train-test split (based on Fold 2)\n\nsgkf = StratifiedGroupKFold(n_splits=5, random_state=42, shuffle=True)\nfor i, (train_index, valid_index) in enumerate(sgkf.split(X=metadata['Id'], y=[1]*len(metadata), groups=metadata['Subject'])):\n    if i != 1:\n        continue\n    print(f\"Fold = {i}\")\n    train_ids = metadata.loc[train_index, 'Id']\n    valid_ids = metadata.loc[valid_index, 'Id']\n    print(f\"Length of Train = {len(train_ids)}, Length of Valid = {len(valid_ids)}\")\n    \n    if i == 2:\n        break\n        \ntrain_fpaths_de = [f\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog/{_id}.csv\" for _id in train_ids]\nvalid_fpaths_de = [f\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog/{_id}.csv\" for _id in valid_ids]","metadata":{"execution":{"iopub.status.busy":"2023-05-24T10:30:07.753759Z","iopub.execute_input":"2023-05-24T10:30:07.754760Z","iopub.status.idle":"2023-05-24T10:30:31.573250Z","shell.execute_reply.started":"2023-05-24T10:30:07.754707Z","shell.execute_reply":"2023-05-24T10:30:31.571825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Define train/validation splits for one model for both datasets.","metadata":{}},{"cell_type":"code","source":"train_fpaths = [(f, 'de') for f in train_fpaths_de] + [(f, 'tdcs') for f in train_fpaths_tdcs]\nvalid_fpaths = [(f, 'de') for f in valid_fpaths_de] + [(f, 'tdcs') for f in valid_fpaths_tdcs]\n# train_fpaths = [(f, 'tdcs') for f in train_fpaths_tdcs]\n# valid_fpaths = [(f, 'tdcs') for f in valid_fpaths_tdcs]","metadata":{"execution":{"iopub.status.busy":"2023-05-24T10:30:31.575359Z","iopub.execute_input":"2023-05-24T10:30:31.576542Z","iopub.status.idle":"2023-05-24T10:30:31.585841Z","shell.execute_reply.started":"2023-05-24T10:30:31.576488Z","shell.execute_reply":"2023-05-24T10:30:31.583899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Define train/validation splits for each dataset separately - in case of using differet models for each dataset.","metadata":{}},{"cell_type":"code","source":"train_fpaths_t = [(f, 'tdcs') for f in train_fpaths_tdcs]\nvalid_fpaths_t = [(f, 'tdcs') for f in valid_fpaths_tdcs]\ntrain_fpaths_d = [(f, 'de') for f in train_fpaths_de]\nvalid_fpaths_d = [(f, 'de') for f in valid_fpaths_de]","metadata":{"execution":{"iopub.status.busy":"2023-05-24T10:30:31.589664Z","iopub.execute_input":"2023-05-24T10:30:31.591331Z","iopub.status.idle":"2023-05-24T10:30:31.600685Z","shell.execute_reply.started":"2023-05-24T10:30:31.591277Z","shell.execute_reply":"2023-05-24T10:30:31.599312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Define some feature extraction functions","metadata":{}},{"cell_type":"code","source":"def to_stft(x, debug=False):\n    # compute the STFT\n    f, t, Zxx = signal.stft(x, cfg.sample_rate, nperseg=cfg.nperseg, noverlap=cfg.noverlap)\n\n    if debug:\n        # plot the spectrogram\n        plt.pcolormesh(t, f, np.abs(Zxx), cmap='inferno')\n        plt.title('STFT Magnitude')\n        plt.ylabel('Frequency [Hz]')\n        plt.xlabel('Time [sec]')\n        plt.show()\n        \n        # print dimensions\n        print(\"Dimensions before transformation:\", x.shape, \"Dimensions after transformation:\", Zxx.shape)\n\n    return Zxx\n\n\n# TODO\ndef to_dwt(sig, widths, debug=False):\n#     t = np.linspace(-1, 1, 200, endpoint=False)\n#     sig  = np.cos(2 * np.pi * 7 * t) + signal.gausspulse(t - 0.4, fc=2)\n#     widths = np.arange(1, 31)\n    \n    # compute the CWT\n    cwtmatr = signal.cwt(sig, signal.ricker, widths)\n\n    if debug:\n        cwtmatr_yflip = np.flipud(cwtmatr)\n        plt.imshow(cwtmatr_yflip, extent=[-1, 1, 1, 31], cmap='PRGn', aspect='auto',\n                   vmax=abs(cwtmatr).max(), vmin=-abs(cwtmatr).max())\n        plt.show()\n        \n    return cwtmatr\n\n\n# ==========PLAYGROUND==========\n# ======= STFT =========\n# # generate a test signal\n# fs = 1000  # sampling frequency\n# t = np.linspace(0, 2, 2*fs, endpoint=False)\n# f0 = 50  # frequency of the signal\n# x = np.sin(2*np.pi*f0*t)\n# print(x.shape, fs)\n# stft_out = to_stft(x, fs, True)\n# print(\"STFT shape\", stft_out.shape)\n\n# # ======= DWT =========\n# t = np.linspace(-1, 1, 200, endpoint=False)\n# sig  = np.cos(2 * np.pi * 7 * t) + signal.gausspulse(t - 0.4, fc=2)\n# widths = np.arange(1, 31)\n\n# dwt_out = to_dwt(sig, widths, debug=True)\n\n# print(\"DWT \", dwt_out.shape)\n\n# =======\n\n\ndef calculate_first_derivative(signal, sampling_rate):\n    # Calculate the time step\n    dt = 1 / sampling_rate\n    \n    # Calculate the first derivative\n    first_derivative = np.gradient(signal, dt)\n    \n    return first_derivative\n\n\ndef calculate_second_derivative(signal, sampling_rate):\n    # Calculate the time step\n    dt = 1 / sampling_rate\n    \n    # Calculate the second derivative\n    second_derivative = np.gradient(np.gradient(signal, dt), dt)\n    \n    return second_derivative\n\ndef calculate_differential(time_series, h):\n    \"\"\"\n    Calculate the differential of a time series using the finite difference method.\n    \n    Arguments:\n    - time_series: The input time series.\n    - h: The time step for the finite difference approximation.\n    \n    Returns:\n    - The differential of the time series.\n    \"\"\"\n    differential = np.diff(time_series) / h\n    return differential\n\ndef moving_average(signal, window_size, axis=0):\n    window = np.ones(window_size) / window_size\n    filtered_signal = np.apply_along_axis(lambda x: np.convolve(x, window, mode='same'), axis, signal)\n    return filtered_signal\n\n\n# # Example usage\n# signal = np.array([1, 2, 4, 7, 11, 16, 22])  # Replace with your signal\n# sampling_rate = 1  # Replace with the sampling rate of your signal\n\n# first_derivative = calculate_first_derivative(signal, sampling_rate)\n# second_derivative = calculate_second_derivative(signal, sampling_rate)\n\n# print(\"First derivative:\", first_derivative)\n# print(\"Second derivative:\", second_derivative)\n\n# # Example usage\n# time_series = np.array([[1, 2, 4], [7, 11, 16], [7, 11, 16], [1, 2, 4], [5,99,6]])  # Replace with your time series\n# h = 1  # Replace with the time step\n\n# print(\"series:\", time_series)\n\n# differential = np.concatenate([np.zeros((1, 3)), calculate_differential(time_series.T, h).T], axis=0)\n\n# print(\"Differential:\", differential)\n\n# con = np.concatenate([time_series, differential], axis=1)\n\n\n\n# print(\"Concatenated:\", con)\n\n# # ========= MA ===========\n# t = [1,2,3,4,5,6,7]\n# ma = moving_average(time_series, 3, axis=0)\n# print(\"MA:\", np.concatenate([time_series, ma], axis=1))","metadata":{"execution":{"iopub.status.busy":"2023-05-24T10:30:31.603142Z","iopub.execute_input":"2023-05-24T10:30:31.603739Z","iopub.status.idle":"2023-05-24T10:30:31.626415Z","shell.execute_reply.started":"2023-05-24T10:30:31.603678Z","shell.execute_reply":"2023-05-24T10:30:31.624975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataset\n\nWe use a window comprised of past and future time Acc readings to form our dataset for a particular time instance. In case some portion of the window data is not available, we pad them with zeros.","metadata":{}},{"cell_type":"code","source":"def interpolate_time_series(series, output_length):\n    # Original length\n    original_length = len(series)\n    \n    # Calculate the factor by which to expand the original length\n    factor = output_length / (original_length - 1)\n    \n    # Create an array of indices for the interpolated values\n    interpolated_indices = np.linspace(0, original_length - 1, output_length)\n    \n    # Interpolate the time series\n    interpolated_series = np.interp(interpolated_indices, np.arange(original_length), series)\n    \n    return interpolated_series\n\ndef convert_acceleration(series, direction='m/s^2'):\n    g = 9.80665  # Acceleration due to gravity in m/s^2 (more precise value)\n    \n    if direction == 'm/s^2':\n        converted_series = [value / g for value in series]\n    elif direction == 'g':\n        converted_series = [value * g for value in series]\n    else:\n        raise ValueError(\"Invalid direction. Specify 'm/s^2' or 'g'.\")\n    \n    return converted_series\n\n\n# # Example usage\n# # Assuming you have a time series of length 10 and you want to interpolate it to have an output length of 12\n\n# # Original time series data\n# original_series = [1, 2, 3, 4, 5, 6, 7, 8, 9, 10]\n\n# # Desired output length\n# output_length = 12\n\n# # Interpolate the time series\n# interpolated_series = interpolate_time_series(original_series, output_length)\n\n# # Print the interpolated series\n# print(interpolated_series)","metadata":{"execution":{"iopub.status.busy":"2023-05-24T10:30:31.628608Z","iopub.execute_input":"2023-05-24T10:30:31.629114Z","iopub.status.idle":"2023-05-24T10:30:31.644201Z","shell.execute_reply.started":"2023-05-24T10:30:31.629063Z","shell.execute_reply":"2023-05-24T10:30:31.642796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# np.array(metadata[\"Id\"])","metadata":{"execution":{"iopub.status.busy":"2023-05-24T10:30:31.646454Z","iopub.execute_input":"2023-05-24T10:30:31.646945Z","iopub.status.idle":"2023-05-24T10:30:31.661667Z","shell.execute_reply.started":"2023-05-24T10:30:31.646901Z","shell.execute_reply":"2023-05-24T10:30:31.660352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# d = pd.read_csv(train_fpaths_tdcs[500])","metadata":{"execution":{"iopub.status.busy":"2023-05-24T10:30:31.663555Z","iopub.execute_input":"2023-05-24T10:30:31.664430Z","iopub.status.idle":"2023-05-24T10:30:31.673253Z","shell.execute_reply.started":"2023-05-24T10:30:31.664385Z","shell.execute_reply":"2023-05-24T10:30:31.671986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# d[(d[\"Valid\"]==True) & (d[\"Task\"]==True) & (d[\"Walking\"] == 1)]\n# d[(d[\"Walking\"] == 1)]","metadata":{"execution":{"iopub.status.busy":"2023-05-24T10:30:31.675222Z","iopub.execute_input":"2023-05-24T10:30:31.675720Z","iopub.status.idle":"2023-05-24T10:30:31.684566Z","shell.execute_reply.started":"2023-05-24T10:30:31.675672Z","shell.execute_reply":"2023-05-24T10:30:31.683422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Define the dataset - including data extraction","metadata":{}},{"cell_type":"code","source":"class FOGDataset(Dataset):\n    def __init__(self, fpaths, scale=9.806, split=\"train\", transformation=None, target_sr=cfg.tdcsfog_sr):\n        super(FOGDataset, self).__init__()\n        tm = time.time()\n        self.split = split\n        self.scale = scale\n        \n        self.fpaths = fpaths\n        self.dfs = [self.read(f[0], f[1]) for f in fpaths]\n        self.f_ids = [os.path.basename(f[0])[:-4] for f in self.fpaths]\n        \n        self.end_indices = []\n        self.shapes = []\n        _length = 0\n        for df in self.dfs:\n            self.shapes.append(df.shape[0])\n            _length += df.shape[0]\n            self.end_indices.append(_length)\n        \n        self.dfs = np.concatenate(self.dfs, axis=0).astype(np.float16)\n        self.length = self.dfs.shape[0]\n        \n        \n        \n        shape1 = self.dfs.shape[1]\n        \n        self.dfs = np.concatenate([np.zeros((cfg.wx*cfg.window_past, shape1)), self.dfs, np.zeros((cfg.wx*cfg.window_future, shape1))], axis=0)\n        \n        \n#         print(\"Before\", self.dfs.shape)\n        \n#         # get differential\n#         self.dfs = np.concatenate([np.zeros((1, len(cfg.feature_list))),\n#                                    calculate_differential(self.dfs.T, 1).T], axis=0)\n# #         np.concatenate([self.dfs,\n# #                                    np.concatenate([np.zeros((1, len(cfg.feature_list))),\n# #                                                    calculate_differential(self.dfs.T, 1).T], axis=0)\n# #                                   ], axis=1)\n#         print(\"After\", self.dfs.shape)\n        \n        print(f\"Dataset initialized in {time.time() - tm} secs!\")\n        gc.collect()\n        \n    def read(self, f, _type):\n        df = pd.read_csv(f)\n        \n        # TODO do interpolation of other dataset\n#         if _type == \"de\":\n#             target_length = int(df.shape[0] * (cfg.tdcsfog_sr / cfg.defog_sr))\n#              = df[cfg.feature_list[0]]\n            \n        \n        if self.split == \"test\":\n            return np.array(df)\n        \n        if _type ==\"tdcs\":\n            df['Valid'] = 1\n            df['Task'] = 1\n            df['tdcs'] = 1\n        else:\n            df['tdcs'] = 0\n        \n        return np.array(df)\n            \n    def __getitem__(self, index):\n        if self.split == \"train\":\n            row_idx = random.randint(0, self.length-1) + cfg.wx*cfg.window_past\n        elif self.split == \"test\":\n            for i,e in enumerate(self.end_indices):\n                if index >= e:\n                    continue\n                df_idx = i\n                break\n\n            row_idx_true = self.shapes[df_idx] - (self.end_indices[df_idx] - index)\n            _id = self.f_ids[df_idx] + \"_\" + str(row_idx_true)\n            row_idx = index + cfg.wx*cfg.window_past\n        else:\n            row_idx = index + cfg.wx*cfg.window_past\n            \n        #scale = 9.80665 if self.dfs[row_idx, -1] == 1 else 1.0\n        x = self.dfs[row_idx - cfg.wx*cfg.window_past : row_idx + cfg.wx*cfg.window_future, 1:4]\n        x = x[::cfg.wx, :][::-1, :]\n        \n#         differential = np.concatenate([np.zeros((1, 3)), calculate_differential(x.T, 1).T], axis=0)\n        # contactenate differential\n#         x = np.concatenate([\n#             x,\n#             np.concatenate([\n#                 np.zeros((1, 3)),\n#                 calculate_differential(x.T, 1).T\n#             ], axis=0),\n#             moving_average(x, 10, axis=0)\n#         ], axis=1)\n        \n        x = torch.tensor(x.astype('float'))#/scale\n        \n        t = self.dfs[row_idx, -3]*self.dfs[row_idx, -2]\n        \n        if self.split == \"test\":\n            return _id, x.T, t\n        \n        y = self.dfs[row_idx, 4:7].astype('float')\n        y = torch.tensor(y)\n        \n        \n        return x.T, y, t\n    \n    def __len__(self):\n        return self.length\n        if self.split == \"train\":\n            return 5_000_000\n        return self.length","metadata":{"execution":{"iopub.status.busy":"2023-05-24T10:30:31.689687Z","iopub.execute_input":"2023-05-24T10:30:31.690061Z","iopub.status.idle":"2023-05-24T10:30:31.720781Z","shell.execute_reply.started":"2023-05-24T10:30:31.690008Z","shell.execute_reply":"2023-05-24T10:30:31.719435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-24T10:30:31.723007Z","iopub.execute_input":"2023-05-24T10:30:31.724268Z","iopub.status.idle":"2023-05-24T10:30:31.909914Z","shell.execute_reply.started":"2023-05-24T10:30:31.724229Z","shell.execute_reply":"2023-05-24T10:30:31.908309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Baseline Model","metadata":{}},{"cell_type":"code","source":"# def _block(in_features, out_features, drop_rate):\n#     return nn.Sequential(\n#         nn.Linear(in_features, out_features),\n#         nn.BatchNorm1d(out_features),\n#         nn.ReLU(),\n#         nn.Dropout(drop_rate)\n#     )\n\n# class FOGModel(nn.Module):\n#     def __init__(self, p=cfg.model_dropout, dim=cfg.model_hidden, nblocks=cfg.model_nblocks):\n#         super(FOGModel, self).__init__()\n#         self.dropout = nn.Dropout(p)\n#         self.in_layer = nn.Linear(cfg.window_size*3, dim)\n#         self.blocks = nn.Sequential(*[_block(dim, dim, p) for _ in range(nblocks)])\n#         self.out_layer = nn.Linear(dim, 3)\n        \n#     def forward(self, x):\n#         x = x.view(-1, cfg.window_size*3)\n#         x = self.in_layer(x)\n#         for block in self.blocks:\n#             x = block(x)\n#         x = self.out_layer(x)\n#         return x","metadata":{"execution":{"iopub.status.busy":"2023-05-24T10:30:31.912975Z","iopub.execute_input":"2023-05-24T10:30:31.913927Z","iopub.status.idle":"2023-05-24T10:30:31.920838Z","shell.execute_reply.started":"2023-05-24T10:30:31.913874Z","shell.execute_reply":"2023-05-24T10:30:31.919601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## CNN 1D Model","metadata":{}},{"cell_type":"code","source":"class CNN1DModel(nn.Module):\n    def __init__(self, input_channels, input_size, num_classes):\n        super(CNN1DModel, self).__init__()\n        \n        self.conv1 = nn.Conv1d(input_channels, 64, kernel_size=3, stride=1, padding=1)\n        self.bn1 = nn.BatchNorm1d(64)\n        self.relu1 = nn.ReLU()\n        self.maxpool1 = nn.MaxPool1d(kernel_size=2, stride=2)\n        self.dropout1 = nn.Dropout(0.2)\n        \n        self.conv2 = nn.Conv1d(64, 128, kernel_size=3, stride=1, padding=1)\n        self.bn2 = nn.BatchNorm1d(128)\n        self.relu2 = nn.ReLU()\n        self.maxpool2 = nn.MaxPool1d(kernel_size=2, stride=2)\n        self.dropout2 = nn.Dropout(0.2)\n        \n        self.conv3 = nn.Conv1d(128, 256, kernel_size=3, stride=1, padding=1)\n        self.bn3 = nn.BatchNorm1d(256)\n        self.relu3 = nn.ReLU()\n        self.maxpool3 = nn.MaxPool1d(kernel_size=2, stride=2)\n        self.dropout3 = nn.Dropout(0.2)\n        \n        self.flatten = nn.Flatten()\n        \n        # Calculate the output size after the convolutional layers\n        conv_output_size = self._get_conv_output_size(input_size)\n        \n        self.DL_neurons = 512 # 512\n        self.fc1 = nn.Linear(conv_output_size, self.DL_neurons) # 1024\n        self.bn4 = nn.BatchNorm1d(self.DL_neurons)\n        self.relu4 = nn.ReLU()\n        self.dropout4 = nn.Dropout(0.2)\n        self.fc2 = nn.Linear(self.DL_neurons, num_classes)\n\n    \n    def _get_conv_output_size(self, input_size):\n        input = torch.zeros(1, self.conv1.in_channels, input_size)\n        output = self._forward_conv_layers(input)\n        return output.size(1)\n    \n    def _forward_conv_layers(self, x):\n        x = self.conv1(x)\n        x = self.bn1(x)\n        x = self.relu1(x)\n        x = self.maxpool1(x)\n        x = self.dropout1(x)\n        \n        x = self.conv2(x)\n        x = self.bn2(x)\n        x = self.relu2(x)\n        x = self.maxpool2(x)\n        x = self.dropout2(x)\n        \n        x = self.conv3(x)\n        x = self.bn3(x)\n        x = self.relu3(x)\n        x = self.maxpool3(x)\n        x = self.dropout3(x)\n        \n        x = self.flatten(x)\n        \n        return x\n    \n    def forward(self, x):\n        x = self._forward_conv_layers(x)\n        x = self.fc1(x)\n        x = self.bn4(x)\n        x = self.relu4(x)\n        x = self.dropout4(x)\n        x = self.fc2(x)\n        \n        return x\n","metadata":{"execution":{"iopub.status.busy":"2023-05-24T10:30:37.828367Z","iopub.execute_input":"2023-05-24T10:30:37.828800Z","iopub.status.idle":"2023-05-24T10:30:37.853052Z","shell.execute_reply.started":"2023-05-24T10:30:37.828760Z","shell.execute_reply":"2023-05-24T10:30:37.851670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"markdown","source":"define some helper functions","metadata":{}},{"cell_type":"code","source":"def count_parameters(model):\n    return sum(p.numel() for p in model.parameters() if p.requires_grad)\n\ndef get_model_architecture(model, input_dim):\n    device = next(model.parameters()).device\n    summary_str = summary(model.to(device), input_dim)\n    return summary_str\n\ndef get_input_dim(dataset: FOGDataset):\n    return dataset[0][0].size()\n\ndef create_new_directory(folder_path):\n    # Get a list of all existing directories in the specified folder\n    existing_directories = [d for d in os.listdir(folder_path) if os.path.isdir(os.path.join(folder_path, d))]\n    \n    # Extract the ID numbers from the directory names and find the highest ID\n    ids = [int(d) for d in existing_directories if d.isdigit()]\n    highest_id = max(ids) if ids else 0\n    \n    # Create a new directory with the name as the next ID number\n    new_directory_name = str(highest_id + 1)\n    new_directory_path = os.path.join(folder_path, new_directory_name)\n    os.makedirs(new_directory_path)\n    \n    return new_directory_path\n\ndef save_model_settings(model, train_dataset):\n    new_folder = create_new_directory(\"/kaggle/working\")\n    \n    if not cfg.submission:\n        architecture = get_model_architecture(model, get_input_dim(train_dataset))\n\n        # Store architecture in a txt file\n        with open(new_folder + '/' + 'architecture.txt', 'w') as file:\n            file.write(str(architecture))\n\n        print(architecture)\n\n    # Get the variables of the class\n    class_variables = vars(cfg)\n\n    # Store the variables in a JSON file\n    with open(new_folder + \"/\" + 'config.json', 'w') as file:\n        json.dump(class_variables, file, indent=4)\n\n    print(\"Config variables stored in config.json\")\n    \n    \n    print(class_variables)\n    \n    return new_folder","metadata":{"execution":{"iopub.status.busy":"2023-05-24T10:30:46.017675Z","iopub.execute_input":"2023-05-24T10:30:46.018117Z","iopub.status.idle":"2023-05-24T10:30:46.035003Z","shell.execute_reply.started":"2023-05-24T10:30:46.018079Z","shell.execute_reply":"2023-05-24T10:30:46.033475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_one_epoch(model, loader, optimizer, criterion):\n    loss_sum = 0.\n    scaler = GradScaler()\n    \n    model.train()\n    for x,y,t in tqdm(loader):\n        x = x.to(cfg.device).float()\n        y = y.to(cfg.device).float()\n        t = t.to(cfg.device).float()\n        \n        y_pred = model(x)\n        loss = criterion(y_pred, y)\n        loss = torch.mean(loss*t.unsqueeze(-1), dim=1)\n        \n#         print(\"x.shape\", x.size(), \"y:\", y, \"y pred:\", y_pred, \"loss:\", loss)\n        \n        t_sum = torch.sum(t)\n        if t_sum > 0:\n            loss = torch.sum(loss)/t_sum\n        else:\n            loss = torch.sum(loss)*0.\n        \n        # loss.backward()\n        scaler.scale(loss).backward()\n        # optimizer.step()\n        scaler.step(optimizer)\n        scaler.update()\n        \n        optimizer.zero_grad()\n        \n        loss_sum += loss.item()\n    \n    print(f\"Train Loss: {(loss_sum/len(loader)):.04f}\")\n    \n\ndef validation_one_epoch(model, loader, criterion):\n    loss_sum = 0.\n    y_true_epoch = []\n    y_pred_epoch = []\n    t_valid_epoch = []\n    \n    model.eval()\n    for x,y,t in tqdm(loader):\n        x = x.to(cfg.device).float()\n        y = y.to(cfg.device).float()\n        t = t.to(cfg.device).float()\n        \n        with torch.no_grad():\n            y_pred = model(x)\n            loss = criterion(y_pred, y)\n            loss = torch.mean(loss*t.unsqueeze(-1), dim=1)\n            \n            t_sum = torch.sum(t)\n            if t_sum > 0:\n                loss = torch.sum(loss)/t_sum\n            else:\n                loss = torch.sum(loss)*0.\n        \n        loss_sum += loss.item()\n        y_true_epoch.append(y.cpu().numpy())\n        y_pred_epoch.append(y_pred.cpu().numpy())\n        t_valid_epoch.append(t.cpu().numpy())\n        \n    y_true_epoch = np.concatenate(y_true_epoch, axis=0)\n    y_pred_epoch = np.concatenate(y_pred_epoch, axis=0)\n    \n    t_valid_epoch = np.concatenate(t_valid_epoch, axis=0)\n    y_true_epoch = y_true_epoch[t_valid_epoch > 0, :]\n    y_pred_epoch = y_pred_epoch[t_valid_epoch > 0, :]\n    \n    scores = [average_precision_score(y_true_epoch[:,i], y_pred_epoch[:,i]) for i in range(3)]\n    mean_score = np.mean(scores)\n    print(f\"Validation Loss: {(loss_sum/len(loader)):.04f}, Validation Score: {mean_score:.03f}, ClassWise: {scores[0]:.03f},{scores[1]:.03f},{scores[2]:.03f}\")\n    \n    return (loss_sum/len(loader), mean_score, scores[0], scores[1], scores[2])","metadata":{"execution":{"iopub.status.busy":"2023-05-24T10:30:49.687593Z","iopub.execute_input":"2023-05-24T10:30:49.688285Z","iopub.status.idle":"2023-05-24T10:30:49.719028Z","shell.execute_reply.started":"2023-05-24T10:30:49.688235Z","shell.execute_reply":"2023-05-24T10:30:49.717689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Get DataLoaders","metadata":{}},{"cell_type":"code","source":"train_dataset = FOGDataset(train_fpaths_t, split=\"train\")\nvalid_dataset = FOGDataset(valid_fpaths_t, split=\"valid\")\nprint(f\"lengths of datasets: train - {len(train_dataset)}, valid - {len(valid_dataset)}\")\n\ntrain_loader = DataLoader(train_dataset, batch_size=cfg.batch_size, num_workers=5, shuffle=True)\nvalid_loader = DataLoader(valid_dataset, batch_size=cfg.batch_size, num_workers=5)","metadata":{"execution":{"iopub.status.busy":"2023-05-24T10:31:03.368811Z","iopub.execute_input":"2023-05-24T10:31:03.369266Z","iopub.status.idle":"2023-05-24T10:31:14.650577Z","shell.execute_reply.started":"2023-05-24T10:31:03.369227Z","shell.execute_reply":"2023-05-24T10:31:14.648492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model = FOGModel().to(cfg.device)\nmodel = CNN1DModel(train_dataset[0][0].size(0), cfg.window_size, len(cfg.label_list)).to(cfg.device)\n# print(f\"Number of parameters in model - {count_parameters(model):,}\")\n\n# saves settings of the model\nmodel_pth = save_model_settings(model, train_dataset)\nlog_fpath = model_pth + \"/epochs_log.txt\"\n\noptimizer = torch.optim.Adam(model.parameters(), lr=cfg.lr)\ncriterion = torch.nn.BCEWithLogitsLoss(reduction='none').to(cfg.device)\n# sched = torch.optim.lr_scheduler.StepLR(optimizer, step_size=1, gamma=0.85)\n\nmax_score = 0.0\n\nprint(\"=\"*50)\nfor epoch in range(cfg.num_epochs):\n    print(f\"Epoch: {epoch}\")\n    train_loss = train_one_epoch(model, train_loader, optimizer, criterion)\n    val = validation_one_epoch(model, valid_loader, criterion)\n    score = val[1]\n\n    if score > max_score:\n        max_score = score\n        torch.save(model.state_dict(), model_pth + \"/\" + \"best_model_state.h5\")\n        print(\"Saving Model ...\")\n\n    print(\"=\"*50)\n    \n    \n    # log into CSV file\n    with open(log_fpath, 'a') as f:\n        f.write(f\"{train_loss},{val[0]},{val[1]},{val[2]},{val[3]},{val[4]}\\n\")\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-24T10:31:20.204405Z","iopub.execute_input":"2023-05-24T10:31:20.204844Z","iopub.status.idle":"2023-05-24T10:50:54.387791Z","shell.execute_reply.started":"2023-05-24T10:31:20.204806Z","shell.execute_reply":"2023-05-24T10:50:54.386495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"model.load_state_dict(torch.load(model_pth + \"/best_model_state.h5\"))\nmodel.eval()\n\ntest_tdcs_paths = glob.glob(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog/*.csv\")\ntest_fpaths = [(f, 'tdcs') for f in test_tdcs_paths]\n\ntest_dataset = FOGDataset(test_fpaths, split=\"test\")\ntest_loader = DataLoader(test_dataset, batch_size=cfg.batch_size, num_workers=5)\n\nids = []\npreds = []\n\nfor _id, x, _ in tqdm(test_loader):\n    x = x.to(cfg.device).float()\n    with torch.no_grad():\n        y_pred = model(x)*0.1\n    \n    ids.extend(_id)\n    preds.extend(list(np.nan_to_num(y_pred.cpu().numpy())))","metadata":{"execution":{"iopub.status.busy":"2023-05-24T10:50:54.390618Z","iopub.execute_input":"2023-05-24T10:50:54.391387Z","iopub.status.idle":"2023-05-24T10:50:55.654503Z","shell.execute_reply.started":"2023-05-24T10:50:54.391341Z","shell.execute_reply":"2023-05-24T10:50:55.652487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"do this for DEFOG","metadata":{}},{"cell_type":"code","source":"class CNN1DModel(nn.Module):\n    def __init__(self, input_channels, input_size, num_classes):\n        super(CNN1DModel, self).__init__()\n        \n        self.conv1 = nn.Conv1d(input_channels, 64, kernel_size=3, stride=1, padding=1)\n        self.bn1 = nn.BatchNorm1d(64)\n        self.relu1 = nn.ReLU()\n        self.maxpool1 = nn.MaxPool1d(kernel_size=2, stride=2)\n        self.dropout1 = nn.Dropout(0.2)\n        \n        self.conv2 = nn.Conv1d(64, 128, kernel_size=3, stride=1, padding=1)\n        self.bn2 = nn.BatchNorm1d(128)\n        self.relu2 = nn.ReLU()\n        self.maxpool2 = nn.MaxPool1d(kernel_size=2, stride=2)\n        self.dropout2 = nn.Dropout(0.2)\n        \n        self.conv3 = nn.Conv1d(128, 256, kernel_size=3, stride=1, padding=1)\n        self.bn3 = nn.BatchNorm1d(256)\n        self.relu3 = nn.ReLU()\n        self.maxpool3 = nn.MaxPool1d(kernel_size=2, stride=2)\n        self.dropout3 = nn.Dropout(0.2)\n        \n        self.flatten = nn.Flatten()\n        \n        # Calculate the output size after the convolutional layers\n        conv_output_size = self._get_conv_output_size(input_size)\n        \n        self.DL_neurons = 512 # 512\n        self.fc1 = nn.Linear(conv_output_size, self.DL_neurons) # 1024\n        self.bn4 = nn.BatchNorm1d(self.DL_neurons)\n        self.relu4 = nn.ReLU()\n        self.dropout4 = nn.Dropout(0.2)\n        self.fc2 = nn.Linear(self.DL_neurons, num_classes)\n\n    \n    def _get_conv_output_size(self, input_size):\n        input = torch.zeros(1, self.conv1.in_channels, input_size)\n        output = self._forward_conv_layers(input)\n        return output.size(1)\n    \n    def _forward_conv_layers(self, x):\n        x = self.conv1(x)\n        x = self.bn1(x)\n        x = self.relu1(x)\n        x = self.maxpool1(x)\n        x = self.dropout1(x)\n        \n        x = self.conv2(x)\n        x = self.bn2(x)\n        x = self.relu2(x)\n        x = self.maxpool2(x)\n        x = self.dropout2(x)\n        \n#         x = self.conv3(x)\n#         x = self.bn3(x)\n#         x = self.relu3(x)\n#         x = self.maxpool3(x)\n#         x = self.dropout3(x)\n        \n        x = self.flatten(x)\n        \n        return x\n    \n    def forward(self, x):\n        x = self._forward_conv_layers(x)\n        x = self.fc1(x)\n        x = self.bn4(x)\n        x = self.relu4(x)\n        x = self.dropout4(x)\n        x = self.fc2(x)\n        \n        return x\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = FOGDataset(train_fpaths_d, split=\"train\")\nvalid_dataset = FOGDataset(valid_fpaths_d, split=\"valid\")\nprint(f\"lengths of datasets: train - {len(train_dataset)}, valid - {len(valid_dataset)}\")\n\ntrain_loader = DataLoader(train_dataset, batch_size=cfg.batch_size, num_workers=5, shuffle=True)\nvalid_loader = DataLoader(valid_dataset, batch_size=cfg.batch_size, num_workers=5)","metadata":{"execution":{"iopub.status.busy":"2023-05-24T10:50:55.656593Z","iopub.execute_input":"2023-05-24T10:50:55.656936Z","iopub.status.idle":"2023-05-24T10:51:30.833517Z","shell.execute_reply.started":"2023-05-24T10:50:55.656902Z","shell.execute_reply":"2023-05-24T10:51:30.832193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model = FOGModel().to(cfg.device)\nmodel = CNN1DModel(train_dataset[0][0].size(0), cfg.window_size, len(cfg.label_list)).to(cfg.device)\n# print(f\"Number of parameters in model - {count_parameters(model):,}\")\n\n# saves settings of the model\nmodel_pth = save_model_settings(model, train_dataset)\nlog_fpath = model_pth + \"/epochs_log.txt\"\n\noptimizer = torch.optim.Adam(model.parameters(), lr=cfg.lr)\ncriterion = torch.nn.BCEWithLogitsLoss(reduction='none').to(cfg.device)\n# sched = torch.optim.lr_scheduler.StepLR(optimizer, step_size=1, gamma=0.85)\n\nmax_score = 0.0\n\nprint(\"=\"*50)\nfor epoch in range(cfg.num_epochs):\n    print(f\"Epoch: {epoch}\")\n    train_loss = train_one_epoch(model, train_loader, optimizer, criterion)\n    val = validation_one_epoch(model, valid_loader, criterion)\n    score = val[1]\n\n    if score > max_score:\n        max_score = score\n        torch.save(model.state_dict(), model_pth + \"/\" + \"best_model_state.h5\")\n        print(\"Saving Model ...\")\n\n    print(\"=\"*50)\n    \n    # log into CSV file\n    with open(log_fpath, 'a') as f:\n        f.write(f\"{train_loss},{val[0]},{val[1]},{val[2]},{val[3]},{val[4]}\\n\")\n    \ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-24T10:51:30.836852Z","iopub.execute_input":"2023-05-24T10:51:30.837916Z","iopub.status.idle":"2023-05-24T11:28:55.808690Z","shell.execute_reply.started":"2023-05-24T10:51:30.837869Z","shell.execute_reply":"2023-05-24T11:28:55.807354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"and submission","metadata":{}},{"cell_type":"code","source":"model.load_state_dict(torch.load(model_pth + \"/best_model_state.h5\"))\nmodel.eval()\n\ntest_defog_paths = glob.glob(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/defog/*.csv\")\ntest_fpaths = [(f, 'de') for f in test_defog_paths]\n\ntest_dataset = FOGDataset(test_fpaths, split=\"test\")\ntest_loader = DataLoader(test_dataset, batch_size=cfg.batch_size, num_workers=5)\n\n# ids = []\n# preds = []\n\nfor _id, x, _ in tqdm(test_loader):\n    x = x.to(cfg.device).float()\n    with torch.no_grad():\n        y_pred = model(x)*0.1\n    \n    ids.extend(_id)\n    preds.extend(list(np.nan_to_num(y_pred.cpu().numpy())))","metadata":{"execution":{"iopub.status.busy":"2023-05-24T11:28:55.810701Z","iopub.execute_input":"2023-05-24T11:28:55.811509Z","iopub.status.idle":"2023-05-24T11:28:57.631402Z","shell.execute_reply.started":"2023-05-24T11:28:55.811466Z","shell.execute_reply":"2023-05-24T11:28:57.628960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# OLD BASELINE Submission for concatenated model","metadata":{}},{"cell_type":"code","source":"# # model = FOGModel().to(cfg.device)\n# model.load_state_dict(torch.load(model_pth + \"/best_model_state.h5\"))\n# model.eval()\n\n# test_defog_paths = glob.glob(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/defog/*.csv\")\n# test_tdcsfog_paths = glob.glob(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog/*.csv\")\n# # test_fpaths = [(f, 'de') for f in test_defog_paths] + [(f, 'tdcs') for f in test_tdcsfog_paths]\n# test_fpaths = [(f, 'tdcs') for f in test_tdcsfog_paths]\n\n# test_dataset = FOGDataset(test_fpaths, split=\"test\")\n# test_loader = DataLoader(test_dataset, batch_size=cfg.batch_size, num_workers=5)\n\n# ids = []\n# preds = []\n\n# for _id, x, _ in tqdm(test_loader):\n#     x = x.to(cfg.device).float()\n#     with torch.no_grad():\n#         y_pred = model(x)*0.1\n    \n#     ids.extend(_id)\n#     preds.extend(list(np.nan_to_num(y_pred.cpu().numpy())))","metadata":{"execution":{"iopub.status.busy":"2023-05-21T16:11:59.331229Z","iopub.status.idle":"2023-05-21T16:11:59.331777Z","shell.execute_reply.started":"2023-05-21T16:11:59.331496Z","shell.execute_reply":"2023-05-21T16:11:59.331522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/sample_submission.csv\")\nsample_submission.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-24T11:28:57.633974Z","iopub.execute_input":"2023-05-24T11:28:57.634416Z","iopub.status.idle":"2023-05-24T11:28:57.874867Z","shell.execute_reply.started":"2023-05-24T11:28:57.634368Z","shell.execute_reply":"2023-05-24T11:28:57.873662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = np.array(preds)\nsubmission = pd.DataFrame({'Id': ids, 'StartHesitation': np.round(preds[:,0],5), \\\n                           'Turn': np.round(preds[:,1],5), 'Walking': np.round(preds[:,2],5)})\n\nsubmission = pd.merge(sample_submission[['Id']], submission, how='left', on='Id').fillna(0.0)\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-24T11:28:57.876423Z","iopub.execute_input":"2023-05-24T11:28:57.877124Z","iopub.status.idle":"2023-05-24T11:28:59.041721Z","shell.execute_reply.started":"2023-05-24T11:28:57.877083Z","shell.execute_reply":"2023-05-24T11:28:59.040527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(submission.shape)","metadata":{"execution":{"iopub.status.busy":"2023-05-24T11:28:59.043197Z","iopub.execute_input":"2023-05-24T11:28:59.043651Z","iopub.status.idle":"2023-05-24T11:28:59.048797Z","shell.execute_reply.started":"2023-05-24T11:28:59.043588Z","shell.execute_reply":"2023-05-24T11:28:59.047521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission[\"Walking\"].max()","metadata":{"execution":{"iopub.status.busy":"2023-05-24T11:28:59.050218Z","iopub.execute_input":"2023-05-24T11:28:59.051233Z","iopub.status.idle":"2023-05-24T11:28:59.063045Z","shell.execute_reply.started":"2023-05-24T11:28:59.051193Z","shell.execute_reply":"2023-05-24T11:28:59.061872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2023-05-24T11:28:59.067003Z","iopub.execute_input":"2023-05-24T11:28:59.067406Z","iopub.status.idle":"2023-05-24T11:28:59.091242Z","shell.execute_reply.started":"2023-05-24T11:28:59.067368Z","shell.execute_reply":"2023-05-24T11:28:59.089526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}