{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":41880,"databundleVersionId":5677426,"sourceType":"competition"}],"dockerImageVersionId":30747,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Project 01.09.24:\n\nAsaf Delmedigo, Romi Zarchi, Ori Armel, Tal Bocbot, Chen Zusman\n","metadata":{}},{"cell_type":"markdown","source":"**Importing libraries**","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nimport os\nimport pandas as pd\nfrom sklearn.preprocessing import StandardScaler\nimport tensorflow as tf\nfrom sklearn.model_selection import train_test_split\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, TensorDataset\nfrom sklearn.metrics import precision_score, recall_score, accuracy_score, f1_score, confusion_matrix\nfrom sklearn.metrics import classification_report, confusion_matrix, roc_curve, auc, precision_recall_curve\nimport seaborn as sns\nfrom scipy.signal import butter, filtfilt","metadata":{"execution":{"iopub.status.busy":"2024-09-01T12:13:49.586077Z","iopub.execute_input":"2024-09-01T12:13:49.586499Z","iopub.status.idle":"2024-09-01T12:14:07.751271Z","shell.execute_reply.started":"2024-09-01T12:13:49.586470Z","shell.execute_reply":"2024-09-01T12:14:07.750348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Data Loading and Standartization**","metadata":{}},{"cell_type":"code","source":"num_of_subjects = 900\n\ndef load_csv_files_from_directory(directory_path, size, exceptions='003f117e14.csv'):\n    \"\"\"\n    Loads and concatenates all CSV files from a specified directory into a single DataFrame.\n\n    Args:\n    directory_path (str): Path to the directory containing CSV files.\n\n    Returns:\n    pd.DataFrame: Concatenated DataFrame containing data from all CSV files in the directory.\n    \"\"\"\n    csv_files = [os.path.join(directory_path, file) for file in os.listdir(directory_path) \n                 if file.endswith('.csv') and file not in exceptions]\n    data_frames = []\n    for csv_file in csv_files[:size]:\n        tmp_df = pd.read_csv(csv_file)\n        tmp_df['Id'] = os.path.splitext(os.path.basename(csv_file))[0]        \n        data_frames.append(tmp_df)\n    \n    return pd.concat(data_frames, ignore_index=True)\n\ndef normalize_data(df, exclude_columns):\n    \"\"\"\n    Normalizes all numerical columns in a DataFrame except the excluded columns using StandardScaler.\n\n    Args:\n    df (pd.DataFrame): DataFrame with numerical features.\n    exclude_columns (list): List of column names to exclude from normalization.\n\n    Returns:\n    pd.DataFrame: DataFrame with normalized features.\n    \"\"\"\n    # Select columns that are numeric and not in the exclude list\n    numeric_columns = df.select_dtypes(include=['float64', 'int64']).columns\n    columns_to_normalize = [col for col in numeric_columns if col not in exclude_columns]\n    \n    # Apply normalization to the selected columns\n    scaler = StandardScaler()\n    df[columns_to_normalize] = scaler.fit_transform(df[columns_to_normalize])\n    \n    return df, scaler\n\ndef normalize_test_data(df, exclude_columns, scaler):\n    \"\"\"\n    Normalizes all numerical columns in a DataFrame except the excluded columns using StandardScaler.\n\n    Args:\n    df (pd.DataFrame): DataFrame with numerical features.\n    exclude_columns (list): List of column names to exclude from normalization.\n\n    Returns:\n    pd.DataFrame: DataFrame with normalized features.\n    \"\"\"\n    # Select columns that are numeric and not in the exclude list\n    numeric_columns = df.select_dtypes(include=['float64', 'int64']).columns\n    columns_to_normalize = [col for col in numeric_columns if col not in exclude_columns]\n    \n    # Apply normalization to the selected columns\n    df[columns_to_normalize] = scaler.transform(df[columns_to_normalize])\n    \n    return df\n\n# Define paths\ntdcsfog_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog'\ndefog_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog'\n\n# Load data\ntdcsfog_data = load_csv_files_from_directory(tdcsfog_path, 256)\ndefog_data = load_csv_files_from_directory(defog_path, 32)\n\nprint(f\"{tdcsfog_data.shape[0]:,} timestamps in tdcsfog_data\")\nprint(f\"{defog_data.shape[0]:,} timestamps in defog_data\")\n\n\nprint(f\"{len(set(tdcsfog_data['Id'].tolist())):,} unique Subjects in tdcsfog_data\")\nprint(f\"{len(set(defog_data['Id'].tolist())):,} unique Subjects in defog_data\")","metadata":{"execution":{"iopub.status.busy":"2024-09-01T13:28:39.881661Z","iopub.execute_input":"2024-09-01T13:28:39.882648Z","iopub.status.idle":"2024-09-01T13:28:47.787212Z","shell.execute_reply.started":"2024-09-01T13:28:39.882613Z","shell.execute_reply":"2024-09-01T13:28:47.786231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_data=defog_data[defog_data['Valid']==True]\ndefog_data.describe()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T13:28:48.145499Z","iopub.execute_input":"2024-09-01T13:28:48.145819Z","iopub.status.idle":"2024-09-01T13:28:48.443977Z","shell.execute_reply.started":"2024-09-01T13:28:48.145794Z","shell.execute_reply":"2024-09-01T13:28:48.443033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Combining the data and normalization**","metadata":{}},{"cell_type":"code","source":"# Add a new column to each DataFrame to indicate the source\ntdcsfog_data['source'] = 'tdcsfog'\ndefog_data['source'] = 'defog'\n\n# Combine both DataFrames into one\ncombined_data = pd.concat([tdcsfog_data, defog_data],axis=0, ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T13:28:49.373223Z","iopub.execute_input":"2024-09-01T13:28:49.373608Z","iopub.status.idle":"2024-09-01T13:28:49.760765Z","shell.execute_reply.started":"2024-09-01T13:28:49.373557Z","shell.execute_reply":"2024-09-01T13:28:49.759968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"combined_data.drop(columns=['Valid','Task'])","metadata":{"execution":{"iopub.status.busy":"2024-09-01T13:28:52.962500Z","iopub.execute_input":"2024-09-01T13:28:52.962856Z","iopub.status.idle":"2024-09-01T13:28:53.116764Z","shell.execute_reply.started":"2024-09-01T13:28:52.962822Z","shell.execute_reply":"2024-09-01T13:28:53.115825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Specify columns to exclude from normalization\nexclude_columns = ['Id', 'Time', 'source', 'StartHesitation', 'Turn', 'Walking']\n\n# Normalize the features, excluding specific columns\nnormalized_combined_data, scaler = normalize_data(combined_data, exclude_columns)\n\nnormalized_combined_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T13:28:57.694483Z","iopub.execute_input":"2024-09-01T13:28:57.695670Z","iopub.status.idle":"2024-09-01T13:28:58.075348Z","shell.execute_reply.started":"2024-09-01T13:28:57.695626Z","shell.execute_reply":"2024-09-01T13:28:58.074359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This method was taken from: https://www.kaggle.com/code/arjanso/reducing-dataframe-memory-size-by-65 @ARJANGROEN\n\ndef reduce_memory_usage(df): # Make the model run much faster by simplifying data types\n    \n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype.name\n        if ((col_type != 'datetime64[ns]') & (col_type != 'category')):\n            if (col_type != 'object'):\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                if str(col_type)[:3] == 'int':\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n\n                else:\n                    if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                        df[col] = df[col].astype(np.float16)\n                    elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        pass\n            else:\n                df[col] = df[col].astype('category')\n    mem_usg = df.memory_usage().sum() / 1024**2 \n    print(\"Memory usage became: \",mem_usg,\" MB\")\n    \n    return df\nnormalized_combined_data=reduce_memory_usage(normalized_combined_data)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T12:15:25.080469Z","iopub.execute_input":"2024-09-01T12:15:25.081072Z","iopub.status.idle":"2024-09-01T12:15:27.527093Z","shell.execute_reply.started":"2024-09-01T12:15:25.081037Z","shell.execute_reply":"2024-09-01T12:15:27.526151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T12:15:34.054654Z","iopub.execute_input":"2024-09-01T12:15:34.055027Z","iopub.status.idle":"2024-09-01T12:15:34.326293Z","shell.execute_reply.started":"2024-09-01T12:15:34.054997Z","shell.execute_reply":"2024-09-01T12:15:34.325025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"normalized_combined_data","metadata":{"execution":{"iopub.status.busy":"2024-09-01T12:15:36.293848Z","iopub.execute_input":"2024-09-01T12:15:36.294494Z","iopub.status.idle":"2024-09-01T12:15:36.314631Z","shell.execute_reply.started":"2024-09-01T12:15:36.294458Z","shell.execute_reply":"2024-09-01T12:15:36.313594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Data Visualizations + Preprocessing**","metadata":{}},{"cell_type":"markdown","source":"**Distribution of Accelerometer Signals:** plotting the distribution of the accelerometer signals (AccV, AccML, AccAP) to understand the range and behavior of the data.","metadata":{}},{"cell_type":"code","source":"def check_missing_data(df):\n    \"\"\"\n    Checks for missing data in the DataFrame and prints the summary.\n    \n    Args:\n        df (pd.DataFrame): DataFrame to check for missing values.\n    \"\"\"\n    missing_data_summary = df.isnull().sum()\n    print(\"Missing Data Summary:\")\n    print(missing_data_summary[missing_data_summary > 0])  # Only show columns with missing data\n    if missing_data_summary.sum() == 0:\n        print(\"No missing data found.\")\n    else:\n        print(\"Missing data detected in the DataFrame.\")\n\n# Check missing data in tdcsfog_data and defog_data\ncheck_missing_data(tdcsfog_data)\ncheck_missing_data(defog_data)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T12:15:40.332738Z","iopub.execute_input":"2024-09-01T12:15:40.333403Z","iopub.status.idle":"2024-09-01T12:15:41.036490Z","shell.execute_reply.started":"2024-09-01T12:15:40.333371Z","shell.execute_reply":"2024-09-01T12:15:41.035560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_signal_distribution(df, signal_columns, dataset_name):\n    plt.figure(figsize=(16, 6))\n    for col in signal_columns:\n        sns.histplot(df[col], kde=True, label=col)\n    plt.legend()\n    plt.title(f'Distribution of Accelerometer Signals - {dataset_name}')\n    plt.show()\n\nplot_signal_distribution(tdcsfog_data, ['AccV', 'AccML', 'AccAP'], dataset_name='tDCS FOG')\nplot_signal_distribution(defog_data, ['AccV', 'AccML', 'AccAP'], dataset_name='DeFOG')","metadata":{"execution":{"iopub.status.busy":"2024-09-01T12:15:46.005870Z","iopub.execute_input":"2024-09-01T12:15:46.006488Z","iopub.status.idle":"2024-09-01T12:16:59.520649Z","shell.execute_reply.started":"2024-09-01T12:15:46.006454Z","shell.execute_reply":"2024-09-01T12:16:59.519713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Filtering:**","metadata":{}},{"cell_type":"code","source":"def bandpass_filter(data, lowcut, highcut, fs, order=4):\n    nyquist = 0.5 * fs\n    low = lowcut / nyquist\n    high = highcut / nyquist\n    b, a = butter(order, [low, high], btype='band')\n    filtered_data = filtfilt(b, a, data)\n    return filtered_data\n\n# Apply bandpass filter to each accelerometer signal\nfs_tdcsfog = 128  # Sampling frequency for tDCS FOG\nfs_defog = 100    # Sampling frequency for DeFOG\nlowcut = 0.5\nhighcut = 15\n\ntdcsfog_data['AccV_filtered'] = bandpass_filter(tdcsfog_data['AccV'], lowcut, highcut, fs_tdcsfog)\ntdcsfog_data['AccML_filtered'] = bandpass_filter(tdcsfog_data['AccML'], lowcut, highcut, fs_tdcsfog)\ntdcsfog_data['AccAP_filtered'] = bandpass_filter(tdcsfog_data['AccAP'], lowcut, highcut, fs_tdcsfog)\n\ndefog_data['AccV_filtered'] = bandpass_filter(defog_data['AccV'], lowcut, highcut, fs_defog)\ndefog_data['AccML_filtered'] = bandpass_filter(defog_data['AccML'], lowcut, highcut, fs_defog)\ndefog_data['AccAP_filtered'] = bandpass_filter(defog_data['AccAP'], lowcut, highcut, fs_defog)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T12:16:59.522561Z","iopub.execute_input":"2024-09-01T12:16:59.523030Z","iopub.status.idle":"2024-09-01T12:17:00.214970Z","shell.execute_reply.started":"2024-09-01T12:16:59.522995Z","shell.execute_reply":"2024-09-01T12:17:00.213682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#smooth signal with a moving average\n# window maybe of =4, df.rolling 3","metadata":{"execution":{"iopub.status.busy":"2024-09-01T01:04:20.272321Z","iopub.execute_input":"2024-09-01T01:04:20.273201Z","iopub.status.idle":"2024-09-01T01:04:20.278155Z","shell.execute_reply.started":"2024-09-01T01:04:20.273154Z","shell.execute_reply":"2024-09-01T01:04:20.276984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_signals_before_after_filtering(df, time_column, signal_column, dataset_name, lowcut, highcut, fs):\n    # Apply bandpass filter to the signal\n    filtered_signal = bandpass_filter(df[signal_column], lowcut, highcut, fs)\n    \n    plt.figure(figsize=(16, 12))\n    \n    # Plot Raw Signal\n    plt.subplot(2, 1, 1)\n    plt.plot(df[time_column], df[signal_column], label='Raw Signal')\n    plt.xlabel('Time')\n    plt.ylabel('Signal')\n    plt.title(f'Raw {signal_column} - {dataset_name}')\n    plt.legend()\n    \n    # Plot Filtered Signal\n    plt.subplot(2, 1, 2)\n    plt.plot(df[time_column], filtered_signal, label='Filtered Signal', color='orange')\n    plt.xlabel('Time')\n    plt.ylabel('Signal')\n    plt.title(f'Filtered {signal_column} - {dataset_name}')\n    plt.legend()\n    \n    plt.tight_layout()\n    plt.show()\n\n# Define filter parameters\nlowcut = 0.5\nhighcut = 15\n\n# Plot signals before and after filtering for tDCS FOG dataset\nplot_signals_before_after_filtering(tdcsfog_data, 'Time', 'AccV', 'tDCS FOG', lowcut, highcut, fs_tdcsfog)\nplot_signals_before_after_filtering(tdcsfog_data, 'Time', 'AccML', 'tDCS FOG', lowcut, highcut, fs_tdcsfog)\nplot_signals_before_after_filtering(tdcsfog_data, 'Time', 'AccAP', 'tDCS FOG', lowcut, highcut, fs_tdcsfog)\n\n# Plot signals before and after filtering for DeFOG dataset\nplot_signals_before_after_filtering(defog_data, 'Time', 'AccV', 'DeFOG', lowcut, highcut, fs_defog)\nplot_signals_before_after_filtering(defog_data, 'Time', 'AccML', 'DeFOG', lowcut, highcut, fs_defog)\nplot_signals_before_after_filtering(defog_data, 'Time', 'AccAP', 'DeFOG', lowcut, highcut, fs_defog)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T12:17:00.216619Z","iopub.execute_input":"2024-09-01T12:17:00.217873Z","iopub.status.idle":"2024-09-01T12:17:28.639251Z","shell.execute_reply.started":"2024-09-01T12:17:00.217837Z","shell.execute_reply":"2024-09-01T12:17:28.638347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Correlation Matrix:** Visualize the correlation between the accelerometer signals and the derived features.","metadata":{}},{"cell_type":"code","source":"def plot_correlation_matrix(df, title):\n    # Select only numerical columns\n    numeric_df = df.select_dtypes(include=['float64', 'int64'])\n    \n    # Plot the correlation matrix\n    plt.figure(figsize=(12, 8))\n    sns.heatmap(numeric_df.corr(), annot=True, fmt='.2f', cmap='coolwarm')\n    plt.title(title)\n    plt.show()\n\n# Plot the correlation matrix for tDCS FOG dataset\nplot_correlation_matrix(tdcsfog_data, 'Correlation Matrix - tDCS FOG Dataset')\n\n# Plot the correlation matrix for DeFOG dataset\nplot_correlation_matrix(defog_data, 'Correlation Matrix - DeFOG Dataset')","metadata":{"execution":{"iopub.status.busy":"2024-09-01T12:17:28.641285Z","iopub.execute_input":"2024-09-01T12:17:28.641597Z","iopub.status.idle":"2024-09-01T12:17:31.207874Z","shell.execute_reply.started":"2024-09-01T12:17:28.641552Z","shell.execute_reply":"2024-09-01T12:17:31.206944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"AccAP is much lower correlation with its filtered form","metadata":{}},{"cell_type":"markdown","source":"# **Data Preparation for Modeling**","metadata":{}},{"cell_type":"code","source":"# Check column names in the DataFrame\nprint(tdcsfog_data.columns)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T12:17:31.208923Z","iopub.execute_input":"2024-09-01T12:17:31.209178Z","iopub.status.idle":"2024-09-01T12:17:31.213887Z","shell.execute_reply.started":"2024-09-01T12:17:31.209156Z","shell.execute_reply":"2024-09-01T12:17:31.212952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**TimeStamps Statistics**","metadata":{}},{"cell_type":"code","source":"#TimeStamps Statistics:\n\ntdcsfog_counts = tdcsfog_data.groupby('Id')['Id'].count()\nprint(\"tdcsfog stats:\")\nprint(f\"{np.mean(tdcsfog_counts):.1f} mean timestamps for each subject in tdcsfog\")\nprint(f\"{np.min(tdcsfog_counts):.0f} min timestamps for each subject in tdcsfog\")\nprint(f\"{np.max(tdcsfog_counts):.0f} max timestamps for each subject in tdcsfog\")\nprint(f\"{np.median(tdcsfog_counts):.0f} median timestamps for each subject in tdcsfog\")\nprint(f\"{np.std(tdcsfog_counts):.1f} std of timestamps in tdcsfog\\n\")\n\nprint(\"defog stats:\")\ndefog_counts = defog_data.groupby('Id')['Id'].count()\nprint(f\"{np.mean(defog_counts):.1f} mean timestamps for each subject in defog\")\nprint(f\"{np.min(defog_counts):.0f} min timestamps for each subject in defog\")\nprint(f\"{np.max(defog_counts):.0f} max timestamps for each subject in defog\")\nprint(f\"{np.median(defog_counts):.0f} median timestamps for each subject in defog\")\nprint(f\"{np.std(defog_counts):.1f} std of timestamps in defog\\n\\n\")\n\nmax_seq = np.max(defog_counts) ## for seq padding!\n\nStartHesitation_ratio = list(normalized_combined_data['StartHesitation'].value_counts())[1] / len(normalized_combined_data)\nTurn_ratio = list(normalized_combined_data['Turn'].value_counts())[1] / len(normalized_combined_data)\nwalking_ratio = list(normalized_combined_data['Walking'].value_counts())[1] / len(normalized_combined_data)\nprint(f\"StartHesitation_ratio: {StartHesitation_ratio:.2%}\")\nprint(f\"Turn_ratio: {Turn_ratio:.2%}\")\nprint(f\"walking_ratio: {walking_ratio:.2%}\")","metadata":{"execution":{"iopub.status.busy":"2024-09-01T12:17:31.215078Z","iopub.execute_input":"2024-09-01T12:17:31.215392Z","iopub.status.idle":"2024-09-01T12:17:31.954323Z","shell.execute_reply.started":"2024-09-01T12:17:31.215361Z","shell.execute_reply":"2024-09-01T12:17:31.953409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def analyze_events_by_subject(df, dataset_name):\n    # Group by subject and sum the events\n    events_by_subject = df.groupby('Id')[['Turn', 'Walking', 'StartHesitation']].sum()\n    \n    # Calculate the total number of timestamps for each subject\n    total_timestamps = df.groupby('Id').size()\n    \n    # Add the total timestamps to the dataframe\n    events_by_subject['Total'] = total_timestamps\n    \n    # Calculate percentages\n    for event in ['Turn', 'Walking', 'StartHesitation']:\n        events_by_subject[f'{event}_Percent'] = events_by_subject[event] / events_by_subject['Total'] * 100\n    \n    # Sort by total timestamps\n    events_by_subject = events_by_subject.sort_values('Total', ascending=False)\n    \n    print(f\"\\nEvent Distribution for {dataset_name}:\")\n    print(events_by_subject)\n    \n    # Plotting\n    plt.figure(figsize=(12, 6))\n    sns.boxplot(data=events_by_subject[[f'{event}_Percent' for event in ['Turn', 'Walking', 'StartHesitation']]])\n    plt.title(f'Distribution of Events Across Subjects - {dataset_name}')\n    plt.ylabel('Percentage of Timestamps')\n    plt.show()\n    \n    # Summary statistics\n    print(f\"\\nSummary Statistics for {dataset_name}:\")\n    print(events_by_subject[[f'{event}_Percent' for event in ['Turn', 'Walking', 'StartHesitation']]].describe())\n    \n    return events_by_subject\n\n# Analyze both datasets\ndefog_events = analyze_events_by_subject(defog_data, 'DEFOG')\ntdcsfog_events = analyze_events_by_subject(tdcsfog_data, 'TDCSFOG')\n\n# Compare the two datasets\nplt.figure(figsize=(15, 6))\nplt.subplot(1, 2, 1)\nsns.boxplot(data=defog_events[[f'{event}_Percent' for event in ['Turn', 'Walking', 'StartHesitation']]])\nplt.title('Distribution of Events - DEFOG')\nplt.ylabel('Percentage of Timestamps')\n\nplt.subplot(1, 2, 2)\nsns.boxplot(data=tdcsfog_events[[f'{event}_Percent' for event in ['Turn', 'Walking', 'StartHesitation']]])\nplt.title('Distribution of Events - TDCSFOG')\nplt.ylabel('Percentage of Timestamps')\n\nplt.tight_layout()\nplt.show()\n\n# Print subjects with highest occurrence of each event\nfor dataset_name, dataset in [('DEFOG', defog_events), ('TDCSFOG', tdcsfog_events)]:\n    print(f\"\\n{dataset_name} Subjects with Highest Event Occurrences:\")\n    for event in ['Turn', 'Walking', 'StartHesitation']:\n        top_subject = dataset.sort_values(f'{event}_Percent', ascending=False).index[0]\n        top_percent = dataset.loc[top_subject, f'{event}_Percent']\n        print(f\"{event}: Subject {top_subject} ({top_percent:.2f}%)\")","metadata":{"execution":{"iopub.status.busy":"2024-09-01T12:17:31.955502Z","iopub.execute_input":"2024-09-01T12:17:31.955820Z","iopub.status.idle":"2024-09-01T12:17:33.778091Z","shell.execute_reply.started":"2024-09-01T12:17:31.955794Z","shell.execute_reply":"2024-09-01T12:17:33.777121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T12:17:53.424531Z","iopub.execute_input":"2024-09-01T12:17:53.425196Z","iopub.status.idle":"2024-09-01T12:17:54.281869Z","shell.execute_reply.started":"2024-09-01T12:17:53.425164Z","shell.execute_reply":"2024-09-01T12:17:54.280834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train-Test Splitting","metadata":{}},{"cell_type":"markdown","source":"**Data Preparation**","metadata":{}},{"cell_type":"code","source":"from tqdm.auto import tqdm\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\ndef by_chunks(chunk, features, targets, window_size=64):\n    X, Y = [], []\n    for _, subject_data in chunk.groupby('Id'):\n        x = subject_data[features].to_numpy(dtype=np.float32)\n        y = subject_data[targets].to_numpy(dtype=np.float32)\n        \n        for idx in range(0, len(x) - window_size + 1, window_size // 2):  # 50% overlap\n            X.append(x[idx:idx+window_size])\n            Y.append(y[idx:idx+window_size])\n    \n    return np.array(X), np.array(Y)\n\n# Set window size\nwindow_size = 128\nchunk_size=1000\n\n# Copy DataFrame and extract unique subjects\ndf = normalized_combined_data.copy()\nsubjects = df['Id'].unique()\n\n# Reminder, can feed both original and/or refined features into model\nfeatures = ['AccV', 'AccML', 'AccAP']\ntargets = ['StartHesitation', 'Turn', 'Walking']\n\n# Prepare X and Y\nX_total = []\nY_total = []\ntotal_chunks = len(normalized_combined_data) // chunk_size + 1\n\n\nfor chunk in tqdm(np.array_split(normalized_combined_data, total_chunks), total=total_chunks,\n                  desc=\"Chunk by chmunk, until it's all over\\nYum yum yum, num num num\\nIt all goes in my belly\"):\n    X_chunk, Y_chunk = by_chunks(chunk, features, targets, window_size)\n    X_total.extend(X_chunk)\n    Y_total.extend(Y_chunk)\n    \nX = np.array(X_total)\nY = np.array(Y_total)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T12:18:00.278762Z","iopub.execute_input":"2024-09-01T12:18:00.279123Z","iopub.status.idle":"2024-09-01T12:33:03.211907Z","shell.execute_reply.started":"2024-09-01T12:18:00.279095Z","shell.execute_reply":"2024-09-01T12:33:03.210886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(X_total)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T12:38:30.480156Z","iopub.execute_input":"2024-09-01T12:38:30.481047Z","iopub.status.idle":"2024-09-01T12:38:30.486703Z","shell.execute_reply.started":"2024-09-01T12:38:30.481004Z","shell.execute_reply":"2024-09-01T12:38:30.485750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Free up memory\n# print(f\"X shape -> (sequences={X.shape[0]}, timestamp={X.shape[1]}, features={X.shape[2]})\")\n# print(f\"y shape -> (sequences={y.shape[0]}, timestamp={y.shape[1]}, classes={y.shape[2]})\")\n\nX_train, X_test, y_train, y_test = train_test_split(X, Y, test_size=0.2, random_state=42)\n\nprint(f\"X_train shape: {X_train.shape}\")\nprint(f\"y_train shape: {y_train.shape}\")\nprint(f\"X_test shape: {X_test.shape}\")\nprint(f\"y_test shape: {y_test.shape}\")\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T12:38:33.027916Z","iopub.execute_input":"2024-09-01T12:38:33.028920Z","iopub.status.idle":"2024-09-01T12:38:33.626447Z","shell.execute_reply.started":"2024-09-01T12:38:33.028886Z","shell.execute_reply":"2024-09-01T12:38:33.625578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train","metadata":{"execution":{"iopub.status.busy":"2024-09-01T12:38:35.526985Z","iopub.execute_input":"2024-09-01T12:38:35.527351Z","iopub.status.idle":"2024-09-01T12:38:35.535898Z","shell.execute_reply.started":"2024-09-01T12:38:35.527323Z","shell.execute_reply":"2024-09-01T12:38:35.534980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Model and Loss Definition**","metadata":{}},{"cell_type":"code","source":"# Use GPU if available\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# Convert to PyTorch tensors with float32 dtype\nX_train_tensor = torch.tensor(X_train, dtype=torch.float32).to(device)\ny_train_tensor = torch.tensor(y_train, dtype=torch.float32).to(device)\nX_test_tensor = torch.tensor(X_test, dtype=torch.float32).to(device)\ny_test_tensor = torch.tensor(y_test, dtype=torch.float32).to(device)\n\n# Create TensorDataset for training and testing\ntrain_dataset = TensorDataset(X_train_tensor, y_train_tensor)\ntest_dataset = TensorDataset(X_test_tensor, y_test_tensor)\n\n# Create DataLoaders with a batch size of 64\ntrain_loader = DataLoader(train_dataset, batch_size=64, shuffle=False) #don't want this\ntest_loader = DataLoader(test_dataset, batch_size=64, shuffle=False)\n\nclass UnidirectionalLSTM(nn.Module):\n    def __init__(self, input_size, hidden_size, output_size):\n        super(UnidirectionalLSTM, self).__init__()\n        self.lstm1 = nn.LSTM(input_size, hidden_size, batch_first=True)\n        self.dropout1 = nn.Dropout(0.5)\n        self.lstm2 = nn.LSTM(hidden_size, hidden_size // 2, batch_first=True)\n        self.dropout2 = nn.Dropout(0.5)\n        self.fc1 = nn.Linear(hidden_size // 2, 512)\n        self.dropout3 = nn.Dropout(0.25)\n        self.fc2 = nn.Linear(512, 256)\n        self.dropout4 = nn.Dropout(0.25)\n        self.fc3 = nn.Linear(256, 128)\n        self.dropout5 = nn.Dropout(0.25)\n        self.fc4 = nn.Linear(128, 64)\n        self.dropout6 = nn.Dropout(0.25)\n        self.fc5 = nn.Linear(64, output_size)\n\n    def forward(self, x):\n        x, _ = self.lstm1(x)\n        x = self.dropout1(x)\n        x, _ = self.lstm2(x)\n        x = self.dropout2(x)\n        x = torch.relu(self.fc1(x))\n        x = self.dropout3(x)\n        x = torch.relu(self.fc2(x))\n        x = self.dropout4(x)\n        x = torch.relu(self.fc3(x))\n        x = self.dropout5(x)\n        x = torch.relu(self.fc4(x))\n        x = self.dropout6(x)\n        x = self.fc5(x)\n        return x\n\n# Define Focal Loss\nclass FocalLoss(nn.Module):\n    def __init__(self, gamma=2.0, alpha=None):\n        super(FocalLoss, self).__init__()\n        self.gamma = gamma\n        self.alpha = torch.tensor(alpha).to(device) if alpha is not None else None\n\n    def forward(self, inputs, targets):\n        BCE_loss = nn.functional.binary_cross_entropy_with_logits(inputs, targets, reduction='none')\n        pt = torch.exp(-BCE_loss)\n        if self.alpha is not None:\n            F_loss = (self.alpha * (1 - pt) ** self.gamma * BCE_loss).mean()\n        else:\n            F_loss = ((1 - pt) ** self.gamma * BCE_loss).mean()\n        return F_loss\n\nnum_classes=y_train.shape[-1]\n\n# Instantiate model\nmodel = UnidirectionalLSTM(input_size=3, hidden_size=256, output_size=num_classes).to(device)\n\n\n# Define Focal Loss\nalpha = [StartHesitation_ratio, Turn_ratio, walking_ratio]  # Example values for the three classes\nfocal_loss = FocalLoss(gamma=2.0, alpha=alpha)\n\n# Define optimizer\noptimizer = optim.Adam(model.parameters(), lr=0.0001)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T12:52:22.718934Z","iopub.execute_input":"2024-09-01T12:52:22.719566Z","iopub.status.idle":"2024-09-01T12:52:22.912747Z","shell.execute_reply.started":"2024-09-01T12:52:22.719531Z","shell.execute_reply":"2024-09-01T12:52:22.911462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Training and Validation**","metadata":{}},{"cell_type":"code","source":"# Training the model\nfrom sklearn.metrics import average_precision_score\n\nnum_epochs = 20\nfor epoch in range(num_epochs):\n    # Training phase\n    model.train()\n    running_loss = 0.0\n    for X_batch, y_batch in train_loader:\n        X_batch, y_batch = X_batch.to(device), y_batch.to(device)\n        \n        optimizer.zero_grad()\n        outputs = model(X_batch)\n        loss = focal_loss(outputs, y_batch)\n        loss.backward()\n        optimizer.step()\n        \n        running_loss += loss.item()\n    \n    avg_train_loss = running_loss / len(train_loader)\n    \n    # Validation phase\n    model.eval()\n    val_loss = 0.0\n    all_preds = []\n    all_targets = []\n    \n    with torch.no_grad():\n        for X_val, y_val in test_loader:\n            X_val, y_val = X_val.to(device), y_val.to(device)\n            val_outputs = model(X_val)\n            val_loss += focal_loss(val_outputs, y_val).item()\n            \n            # Store predictions and targets for accuracy calculation\n            probs = torch.sigmoid(val_outputs)  # Convert logits to probabilities\n            all_preds.append(probs.cpu().numpy())\n            all_targets.append(y_val.cpu().numpy())\n    \n    avg_val_loss = val_loss / len(test_loader)\n    \n    # Convert lists to arrays\n    all_probs = np.concatenate(all_preds, axis=0)\n    all_targets = np.concatenate(all_targets, axis=0)\n    print(f\"Shape of all_probs before reshaping: {all_probs.shape}\")\n    print(f\"Shape of all_targets before reshaping: {all_targets.shape}\")\n\n    # Ensure shapes match\n    min_samples = min(all_probs.shape[0], all_targets.shape[0])\n    all_probs = all_probs[:min_samples]\n    all_targets = all_targets[:min_samples]\n\n    # Reshape\n    all_probs = all_probs.reshape(-1, 3)\n    all_targets = all_targets.reshape(-1, 3)\n\n    print(f\"Shape of all_probs after reshaping: {all_probs.shape}\")\n    print(f\"Shape of all_targets after reshaping: {all_targets.shape}\")\n\n    \n    # Calculate Average Precision for each class\n    average_precisions = []\n    for i in range(num_classes):\n        ap = average_precision_score(all_targets[:, i], all_probs[:, i])\n        average_precisions.append(ap)\n    \n    # Calculate Mean Average Precision\n    mean_ap = np.mean(average_precisions)\n    \n    print(f\"Epoch [{epoch+1}/{num_epochs}], Train Loss: {avg_train_loss:.4f}, Val Loss: {avg_val_loss:.4f}\")\n    print(f\"Class-wise Average Precisions: {average_precisions}\")\n    print(f\"Mean Average Precision: {mean_ap:.4f}\")\n    \n    # Optionally, print confidence scores distribution\n    for i in range(num_classes):\n        pos_scores = all_probs[:, i][all_targets[:, i] == 1]\n        neg_scores = all_probs[:, i][all_targets[:, i] == 0]\n        print(f\"Class {i}:\")\n        print(f\"  Positive samples: min={pos_scores.min():.4f}, max={pos_scores.max():.4f}, mean={pos_scores.mean():.4f}\")\n        print(f\"  Negative samples: min={neg_scores.min():.4f}, max={neg_scores.max():.4f}, mean={neg_scores.mean():.4f}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-09-01T12:39:15.612621Z","iopub.execute_input":"2024-09-01T12:39:15.613339Z","iopub.status.idle":"2024-09-01T12:44:02.395307Z","shell.execute_reply.started":"2024-09-01T12:39:15.613306Z","shell.execute_reply":"2024-09-01T12:44:02.393936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Bidirectional","metadata":{"execution":{"iopub.status.busy":"2024-09-01T01:24:52.819752Z","iopub.execute_input":"2024-09-01T01:24:52.820032Z","iopub.status.idle":"2024-09-01T01:24:52.823983Z","shell.execute_reply.started":"2024-09-01T01:24:52.820008Z","shell.execute_reply":"2024-09-01T01:24:52.823045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BidirectionalLSTM(nn.Module):\n    def __init__(self, input_size, hidden_size, output_size):\n        super(BidirectionalLSTM, self).__init__()\n        self.lstm1 = nn.LSTM(input_size, hidden_size, batch_first=True)\n        self.dropout1 = nn.Dropout(0.5)\n        self.lstm2 = nn.LSTM(hidden_size, hidden_size // 2, batch_first=True)\n        self.dropout2 = nn.Dropout(0.5)\n        self.fc1 = nn.Linear(hidden_size // 2, 512)\n        self.dropout3 = nn.Dropout(0.25)\n        self.fc2 = nn.Linear(512, 256)\n        self.dropout4 = nn.Dropout(0.25)\n        self.fc3 = nn.Linear(256, 128)\n        self.dropout5 = nn.Dropout(0.25)\n        self.fc4 = nn.Linear(128, 64)\n        self.dropout6 = nn.Dropout(0.25)\n        self.fc5 = nn.Linear(64, output_size)\n\n    def forward(self, x):\n        x, _ = self.lstm1(x)\n        x = self.dropout1(x)\n        x, _ = self.lstm2(x)\n        x = self.dropout2(x)\n        x = torch.relu(self.fc1(x))\n        x = self.dropout3(x)\n        x = torch.relu(self.fc2(x))\n        x = self.dropout4(x)\n        x = torch.relu(self.fc3(x))\n        x = self.dropout5(x)\n        x = torch.relu(self.fc4(x))\n        x = self.dropout6(x)\n        x = self.fc5(x)\n        return x\n\n# Define Focal Loss\nclass FocalLoss(nn.Module):\n    def __init__(self, gamma=2.0, alpha=None):\n        super(FocalLoss, self).__init__()\n        self.gamma = gamma\n        self.alpha = torch.tensor(alpha).to(device) if alpha is not None else None\n\n    def forward(self, inputs, targets):\n        BCE_loss = nn.functional.binary_cross_entropy_with_logits(inputs, targets, reduction='none')\n        pt = torch.exp(-BCE_loss)\n        if self.alpha is not None:\n            F_loss = (self.alpha * (1 - pt) ** self.gamma * BCE_loss).mean()\n        else:\n            F_loss = ((1 - pt) ** self.gamma * BCE_loss).mean()\n        return F_loss\n\nnum_classes=y_train.shape[-1]\n\n# Instantiate model\nmodel = BidirectionalLSTM(input_size=3, hidden_size=256, output_size=num_classes).to(device)\n\n\n# Define Focal Loss\nalpha = [StartHesitation_ratio, Turn_ratio, walking_ratio] # Example values for the three classes\nfocal_loss = FocalLoss(gamma=2.0, alpha=alpha)\n\n# Define optimizer\noptimizer = optim.Adam(model.parameters(), lr=0.001)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T12:53:02.390521Z","iopub.execute_input":"2024-09-01T12:53:02.391323Z","iopub.status.idle":"2024-09-01T12:53:02.419097Z","shell.execute_reply.started":"2024-09-01T12:53:02.391292Z","shell.execute_reply":"2024-09-01T12:53:02.418115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training the model\nfrom sklearn.metrics import average_precision_score\n\nnum_epochs = 30\nfor epoch in range(num_epochs):\n    # Training phase\n    model.train()\n    running_loss = 0.0\n    for X_batch, y_batch in train_loader:\n        X_batch, y_batch = X_batch.to(device), y_batch.to(device)\n        \n        optimizer.zero_grad()\n        outputs = model(X_batch)\n        loss = focal_loss(outputs, y_batch)\n        loss.backward()\n        optimizer.step()\n        \n        running_loss += loss.item()\n    \n    avg_train_loss = running_loss / len(train_loader)\n    \n    # Validation phase\n    model.eval()\n    val_loss = 0.0\n    all_preds = []\n    all_targets = []\n    \n    with torch.no_grad():\n        for X_val, y_val in test_loader:\n            X_val, y_val = X_val.to(device), y_val.to(device)\n            val_outputs = model(X_val)\n            val_loss += focal_loss(val_outputs, y_val).item()\n            \n            # Store predictions and targets for accuracy calculation\n            probs = torch.sigmoid(val_outputs)  # Convert logits to probabilities\n            all_preds.append(probs.cpu().numpy())\n            all_targets.append(y_val.cpu().numpy())\n    \n    avg_val_loss = val_loss / len(test_loader)\n    \n    # Convert lists to arrays\n    all_probs = np.concatenate(all_preds, axis=0)\n    all_targets = np.concatenate(all_targets, axis=0)\n\n    # Ensure shapes match\n    min_samples = min(all_probs.shape[0], all_targets.shape[0])\n    all_probs = all_probs[:min_samples]\n    all_targets = all_targets[:min_samples]\n\n    # Reshape\n    all_probs = all_probs.reshape(-1, 3)\n    all_targets = all_targets.reshape(-1, 3)    \n    # Calculate Average Precision for each class\n    average_precisions = []\n    for i in range(num_classes):\n        ap = average_precision_score(all_targets[:, i], all_probs[:, i])\n        average_precisions.append(ap)\n    \n    # Calculate Mean Average Precision\n    mean_ap = np.mean(average_precisions)\n    \n    print(f\"Epoch [{epoch+1}/{num_epochs}], Train Loss: {avg_train_loss:.4f}, Val Loss: {avg_val_loss:.4f}\")\n    print(f\"Class-wise Average Precisions: {average_precisions}\")\n    print(f\"Mean Average Precision: {mean_ap:.4f}\")\n    \n    # Optionally, print confidence scores distribution\n    for i in range(num_classes):\n        pos_scores = all_probs[:, i][all_targets[:, i] == 1]\n        neg_scores = all_probs[:, i][all_targets[:, i] == 0]\n        print(f\"Class {i}:\")\n        print(f\"  Positive samples: min={pos_scores.min():.4f}, max={pos_scores.max():.4f}, mean={pos_scores.mean():.4f}\")\n        print(f\"  Negative samples: min={neg_scores.min():.4f}, max={neg_scores.max():.4f}, mean={neg_scores.mean():.4f}\")","metadata":{"execution":{"iopub.status.busy":"2024-09-01T12:58:15.448825Z","iopub.execute_input":"2024-09-01T12:58:15.449231Z","iopub.status.idle":"2024-09-01T13:01:28.330455Z","shell.execute_reply.started":"2024-09-01T12:58:15.449205Z","shell.execute_reply":"2024-09-01T13:01:28.329193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Results Visualization**","metadata":{}},{"cell_type":"code","source":"# Visualize predictions for a specific subject\nsubject_idx = 69\nmodel.eval()\nwith torch.no_grad():\n    predictions = model(X_test_tensor[subject_idx:subject_idx+1])\n    threshold = 0.5\n    predicted_turns = [1 if p[1] > threshold else 0 for p in predictions[0]]\n\n    true_labels = y_test_tensor[subject_idx:subject_idx+1]   \n    true_turns = [torch.argmax(t).item() for t in true_labels[0]]\n\n    plt.figure(figsize=(12, 6))\n    plt.plot(predicted_turns, label='Predicted')\n    plt.plot(true_turns, label='True')\n    plt.xlabel('Time')\n    plt.ylabel('Class')\n    plt.legend()\n    plt.title('Predicted vs True Labels')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T13:02:52.882630Z","iopub.execute_input":"2024-09-01T13:02:52.883633Z","iopub.status.idle":"2024-09-01T13:02:53.176465Z","shell.execute_reply.started":"2024-09-01T13:02:52.883601Z","shell.execute_reply":"2024-09-01T13:02:53.175526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"# Load Defog and Tdcsfog Test Data\n\ndefog_test_id = \"02ab235146\"\ndefog_test_df = pd.read_csv(f\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/defog/{defog_test_id}.csv\")\ndefog_test_df['Id'] = [defog_test_id + \"_\" + str(ts) for ts in range(len(defog_test_df))]\n\ntdcsfog_test_id = \"003f117e14\"\ntdcsfog_test_df = pd.read_csv(f\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog/{tdcsfog_test_id}.csv\")\ntdcsfog_test_df['Id'] = [tdcsfog_test_id + \"_\" + str(ts) for ts in range(len(tdcsfog_test_df))]\n\ndefog_test_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T14:24:28.937363Z","iopub.execute_input":"2024-09-01T14:24:28.938201Z","iopub.status.idle":"2024-09-01T14:24:29.313352Z","shell.execute_reply.started":"2024-09-01T14:24:28.938166Z","shell.execute_reply":"2024-09-01T14:24:29.312439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# normalize data\n\ndefog_test_df = normalize_test_data(defog_test_df, exclude_columns, scaler)\ntdcsfog_test_df = normalize_test_data(tdcsfog_test_df, exclude_columns, scaler)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T14:24:31.075767Z","iopub.execute_input":"2024-09-01T14:24:31.076381Z","iopub.status.idle":"2024-09-01T14:24:31.094594Z","shell.execute_reply.started":"2024-09-01T14:24:31.076349Z","shell.execute_reply":"2024-09-01T14:24:31.093801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = defog_test_df[features].to_numpy()\n# Set window size\nwindow_size = 128\nX = []\nfor idx in range(0, len(defog_test_df) - window_size, window_size):\n    X.append(data[idx:idx+window_size].tolist())\nX = torch.tensor(X)\nX = X.to(device)\ntest_outputs = model(X)\ntest_outputs = torch.sigmoid(test_outputs)\nsub_defog_df = pd.DataFrame(np.array(test_outputs.reshape((len(X)*128,3)).detach().to(torch.device(\"cpu\"))), columns = targets)\nsub_defog_df.insert(0, column=\"Id\", value = defog_test_df['Id'])","metadata":{"execution":{"iopub.status.busy":"2024-09-01T14:24:32.970260Z","iopub.execute_input":"2024-09-01T14:24:32.970621Z","iopub.status.idle":"2024-09-01T14:24:33.585677Z","shell.execute_reply.started":"2024-09-01T14:24:32.970594Z","shell.execute_reply":"2024-09-01T14:24:33.584631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = tdcsfog_test_df[features].to_numpy()\n# Set window size\nwindow_size = 128\nX = []\nfor idx in range(0, len(tdcsfog_test_df) - window_size, window_size):\n    X.append(data[idx:idx+window_size].tolist())\nX = torch.tensor(X)\nX = X.to(device)\ntest_outputs = model(X)\ntest_outputs = torch.sigmoid(test_outputs)\nsub_tdcsfog_df = pd.DataFrame(np.array(test_outputs.reshape((len(X)*128,3)).detach().to(torch.device(\"cpu\"))), columns = targets)\nsub_tdcsfog_df.insert(0, column=\"Id\", value = tdcsfog_test_df['Id'])","metadata":{"execution":{"iopub.status.busy":"2024-09-01T14:25:43.007613Z","iopub.execute_input":"2024-09-01T14:25:43.008311Z","iopub.status.idle":"2024-09-01T14:25:43.030920Z","shell.execute_reply.started":"2024-09-01T14:25:43.008277Z","shell.execute_reply":"2024-09-01T14:25:43.029990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = pd.concat([sub_defog_df, sub_tdcsfog_df], axis=0)\nsubmission_df.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-09-01T14:28:00.480281Z","iopub.execute_input":"2024-09-01T14:28:00.481376Z","iopub.status.idle":"2024-09-01T14:28:02.093055Z","shell.execute_reply.started":"2024-09-01T14:28:00.481340Z","shell.execute_reply":"2024-09-01T14:28:02.091999Z"},"trusted":true},"execution_count":null,"outputs":[]}]}