{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30635,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-27T10:34:02.559117Z","iopub.execute_input":"2024-01-27T10:34:02.559501Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport time\n\n# plot\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2024-01-27T10:41:30.321094Z","iopub.execute_input":"2024-01-27T10:41:30.322082Z","iopub.status.idle":"2024-01-27T10:41:32.175327Z","shell.execute_reply.started":"2024-01-27T10:41:30.322039Z","shell.execute_reply":"2024-01-27T10:41:32.174215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH = '/kaggle/input/hms-harmful-brain-activity-classification/'\ndf = pd.read_csv(PATH + 'train.csv')\nTARGETS = df.columns[-6:]\nprint('Train shape:', df.shape )\nprint('Targets', list(TARGETS))\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-27T10:41:34.506445Z","iopub.execute_input":"2024-01-27T10:41:34.506939Z","iopub.status.idle":"2024-01-27T10:41:34.782601Z","shell.execute_reply.started":"2024-01-27T10:41:34.506905Z","shell.execute_reply":"2024-01-27T10:41:34.781647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_id':'first','spectrogram_label_offset_seconds':'min'})\ntrain.columns = ['spectrogram_id','min']\n\ntmp = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_label_offset_seconds':'max'})\ntrain['max'] = tmp\n\ntmp = df.groupby('eeg_id')[['patient_id']].agg('first')\ntrain['patient_id'] = tmp\n\ntmp = df.groupby('eeg_id')[TARGETS].agg('sum')\nfor t in TARGETS:\n    train[t] = tmp[t].values\n    \ny_data = train[TARGETS].values\ny_data = y_data / y_data.sum(axis=1,keepdims=True)\ntrain[TARGETS] = y_data\n\ntmp = df.groupby('eeg_id')[['expert_consensus']].agg('first')\ntrain['target'] = tmp\n\ntrain = train.reset_index()\nprint('Train non-overlapp eeg_id shape:', train.shape )\ntrain.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-01-27T10:41:36.773809Z","iopub.execute_input":"2024-01-27T10:41:36.774191Z","iopub.status.idle":"2024-01-27T10:41:36.870412Z","shell.execute_reply.started":"2024-01-27T10:41:36.774162Z","shell.execute_reply":"2024-01-27T10:41:36.869501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp","metadata":{"execution":{"iopub.status.busy":"2024-01-27T10:41:37.966871Z","iopub.execute_input":"2024-01-27T10:41:37.967559Z","iopub.status.idle":"2024-01-27T10:41:37.977383Z","shell.execute_reply.started":"2024-01-27T10:41:37.967524Z","shell.execute_reply":"2024-01-27T10:41:37.976424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.iloc[2]","metadata":{"execution":{"iopub.status.busy":"2024-01-27T10:41:38.892479Z","iopub.execute_input":"2024-01-27T10:41:38.892853Z","iopub.status.idle":"2024-01-27T10:41:38.900080Z","shell.execute_reply.started":"2024-01-27T10:41:38.892823Z","shell.execute_reply":"2024-01-27T10:41:38.899092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nEEG_PATH = PATH + 'train_eegs/'\n\nrow = train.iloc[1]\neeg_id = row['eeg_id']\neeg_file_path = os.path.join(EEG_PATH, f'{eeg_id}.parquet')\nsample_eeg_data = pd.read_parquet(eeg_file_path)\nsample_eeg_data = sample_eeg_data.iloc[:, :-1]\nsample_eeg_data = sample_eeg_data.iloc[:, [col for col in range(sample_eeg_data.shape[1]) if col not in [8, 9, 10]]]\n\n# Calculate starting time point\n\nstart_time_point = int((sample_eeg_data.shape[0] - 10_000) // 2)\n\n# Get the time point data of the middle 50 seconds\n\neeg_slice = sample_eeg_data.iloc[start_time_point : start_time_point + 10_000, :]\n\neeg_slice","metadata":{"execution":{"iopub.status.busy":"2024-01-27T10:41:39.815604Z","iopub.execute_input":"2024-01-27T10:41:39.816261Z","iopub.status.idle":"2024-01-27T10:41:40.028223Z","shell.execute_reply.started":"2024-01-27T10:41:39.816224Z","shell.execute_reply":"2024-01-27T10:41:40.027263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"row = train.iloc[2]\neeg_id = row['eeg_id']\neeg_file_path = os.path.join(EEG_PATH, f'{eeg_id}.parquet')\nsample_eeg_data = pd.read_parquet(eeg_file_path)\nsample_eeg_data = sample_eeg_data.iloc[:, :-1]\nsample_eeg_data = sample_eeg_data.iloc[:, [col for col in range(sample_eeg_data.shape[1]) if col not in [8, 9, 10]]]\n\n# Calculate starting time point\n\nstart_time_point = int((sample_eeg_data.shape[0] - 10_000) // 2)\n\n# Get the time point data of the middle 50 seconds\n\neeg_slice = sample_eeg_data.iloc[start_time_point : start_time_point + 10_000, :]\n\neeg_slice","metadata":{"execution":{"iopub.status.busy":"2024-01-27T10:41:40.774008Z","iopub.execute_input":"2024-01-27T10:41:40.774798Z","iopub.status.idle":"2024-01-27T10:41:40.828915Z","shell.execute_reply.started":"2024-01-27T10:41:40.774766Z","shell.execute_reply":"2024-01-27T10:41:40.827843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FLAG = True\nif FLAG:\n    fs = 200\n    EEG_PATH = PATH + 'train_eegs/'\n\n    # Initialize an array to save all processed EEG data\n    all_eeg_data = np.zeros((len(train), 10000, 19))\n\n    # Iterate over each eeg_id\n    for row_idx in range(len(train)):\n        row = train.iloc[row_idx]\n        eeg_id = row['eeg_id']\n\n        # Read EEG data\n        eeg_file_path = os.path.join(EEG_PATH, f'{eeg_id}.parquet')\n        eeg_data = pd.read_parquet(eeg_file_path)\n        eeg_data = eeg_data.iloc[:, :-1]\n#         eeg_data = eeg_data.iloc[:, [col for col in range(eeg_data.shape[1]) if col not in [8, 9, 10]]]\n        # Calculate starting time point\n        \n        start_time_point = int((eeg_data.shape[0] - 10_000) // 2)\n\n        # Get the time point data of the middle 50 seconds\n        eeg_slice = eeg_data.iloc[start_time_point : start_time_point + 10_000, :]\n\n        # Update all_eeg_data\n        all_eeg_data[row_idx, :, :] = eeg_slice\n\n    # The shape of the output array\n    print(\"Shape of all_eeg_data:\", all_eeg_data.shape)\nelse:\n    print(\"FLAG is set to False. The code below is not executed.\")","metadata":{"execution":{"iopub.status.busy":"2024-01-27T10:41:41.589064Z","iopub.execute_input":"2024-01-27T10:41:41.589886Z","iopub.status.idle":"2024-01-27T10:48:23.565633Z","shell.execute_reply.started":"2024-01-27T10:41:41.589856Z","shell.execute_reply":"2024-01-27T10:48:23.564632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#     Extract features from EEG data, including LL Spec, LP Spec, RP Spec, RL Spec\n    \n#     Parameters:\n#     - eeg_data: EEG data with shape (number of samples, time points, number of electrodes)\n    \n#     Returns:\n#     - Feature matrix with shape (number of samples, number of features)\n\nimport scipy.stats\n\ndef extract_eeg_features(eeg_data):\n    # Initialize the feature matrix\n    num_samples, num_time_points, num_electrodes = eeg_data.shape\n    num_features = 8 * num_electrodes  # each electrode will have 8 quantity --> Mean, standard deviation, minimum, maximum, skewness, kurtosis, energy, entropy for each electrode + 4 global features\n    features = np.zeros((num_samples, num_features))\n\n    # Extract features\n    for sample_idx in range(num_samples):\n        for electrode_idx in range(num_electrodes):\n            electrode_data = eeg_data[sample_idx, :, electrode_idx]\n            feature_idx = electrode_idx * 8\n\n            # Mean\n            features[sample_idx, feature_idx] = np.mean(electrode_data)\n            # Standard deviation\n            features[sample_idx, feature_idx + 1] = np.std(electrode_data)\n            # Minimum\n            features[sample_idx, feature_idx + 2] = np.min(electrode_data)\n            # Maximum\n            features[sample_idx, feature_idx + 3] = np.max(electrode_data)\n            # Skewness\n            features[sample_idx, feature_idx + 4] = scipy.stats.skew(electrode_data)\n            # Kurtosis\n            features[sample_idx, feature_idx + 5] = scipy.stats.kurtosis(electrode_data)\n            # Energy\n            features[sample_idx, feature_idx + 6] = np.sum(electrode_data**2) / len(electrode_data)\n            # Entropy\n            features[sample_idx, feature_idx + 7] = scipy.stats.entropy(np.abs(electrode_data))\n    return features","metadata":{"execution":{"iopub.status.busy":"2024-01-27T10:51:18.767654Z","iopub.execute_input":"2024-01-27T10:51:18.768011Z","iopub.status.idle":"2024-01-27T10:51:18.778426Z","shell.execute_reply.started":"2024-01-27T10:51:18.767982Z","shell.execute_reply":"2024-01-27T10:51:18.777507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport scipy\n\nfrom scipy.stats import skew, kurtosis\nfrom scipy.stats.mstats import moment\nfrom sklearn.preprocessing import StandardScaler\n\nif FLAG:\n\n    # Use the above functions to extract features\n    extracted_features = extract_eeg_features(all_eeg_data)\n\n    # The shape of the output feature matrix generated\n\n    print(\"Shape of extracted features:\", extracted_features.shape)\nelse:\n    print(\"FLAG is set to False. The code below is not executed.\")","metadata":{"execution":{"iopub.status.busy":"2024-01-27T10:51:23.024960Z","iopub.execute_input":"2024-01-27T10:51:23.025638Z","iopub.status.idle":"2024-01-27T10:57:03.300732Z","shell.execute_reply.started":"2024-01-27T10:51:23.025609Z","shell.execute_reply":"2024-01-27T10:57:03.299706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if FLAG:\n    # Calculate the avg of each column\n    mean_values = np.nanmean(extracted_features, axis=0)\n\n    # Find the location of the NaN value\n\n    nan_indices = np.isnan(extracted_features)\n\n    # Replace NaN values with average\n\n    extracted_features[nan_indices] = np.take(mean_values, nan_indices.nonzero()[1])\n\n    # Output the feature shape after replacing NaN\n    print(\"Shape of eeg_features after replacing NaN values:\", extracted_features.shape)\n    \n    # Save extracted features in a file\n    save_path = '/kaggle/working/extracted_eeg_features.npy'\n    np.save(save_path, extracted_features)\n    print(f\"Extracted EEG features saved at: {save_path}\")\nelse:\n    # When FLAG is False, load the previously saved extracted EEG features.\n    \n    extracted_features = np.load('/kaggle/input/8-basic-feaatures-with-eeg/extracted_eeg_features (1).npy')\n\n    # The shape of the output array\n\n    print(\"Shape of extracted features:\", extracted_features.shape)","metadata":{"execution":{"iopub.status.busy":"2024-01-27T10:57:03.302334Z","iopub.execute_input":"2024-01-27T10:57:03.302644Z","iopub.status.idle":"2024-01-27T10:57:03.352455Z","shell.execute_reply.started":"2024-01-27T10:57:03.302609Z","shell.execute_reply":"2024-01-27T10:57:03.351438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_copy = train.copy()\nycol = [c for c in train_copy.columns if c in ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']]\ncd = {'Seizure':'seizure_vote', 'GPD':'gpd_vote', 'LRDA':'lrda_vote', 'Other':'other_vote', 'GRDA':'grda_vote', 'LPD':'lpd_vote'}\ntrain_copy['target'] = train_copy['target'].map(cd)\nfor i in range(len(train_copy)):\n    c = train_copy['target'][i]\n    train_copy[c][i] = train_copy[c][i]+10 #adding weight to expert consensus\n\nysum = train_copy[ycol].sum(axis=1) \nfor c in ycol:\n    train_copy[c] = (train_copy[c] / ysum).astype(np.float64)","metadata":{"execution":{"iopub.status.busy":"2024-01-27T10:57:03.353510Z","iopub.execute_input":"2024-01-27T10:57:03.353781Z","iopub.status.idle":"2024-01-27T10:57:09.277333Z","shell.execute_reply.started":"2024-01-27T10:57:03.353757Z","shell.execute_reply":"2024-01-27T10:57:09.276430Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label = train_copy[ycol]\nlabel.shape","metadata":{"execution":{"iopub.status.busy":"2024-01-27T10:57:09.279426Z","iopub.execute_input":"2024-01-27T10:57:09.279714Z","iopub.status.idle":"2024-01-27T10:57:09.287104Z","shell.execute_reply.started":"2024-01-27T10:57:09.279688Z","shell.execute_reply":"2024-01-27T10:57:09.286092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_copy.to_csv(\"final_train.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-01-27T10:57:09.288338Z","iopub.execute_input":"2024-01-27T10:57:09.288620Z","iopub.status.idle":"2024-01-27T10:57:09.512792Z","shell.execute_reply.started":"2024-01-27T10:57:09.288596Z","shell.execute_reply":"2024-01-27T10:57:09.511879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\nfrom xgboost import XGBClassifier","metadata":{"execution":{"iopub.status.busy":"2024-01-27T10:57:09.514319Z","iopub.execute_input":"2024-01-27T10:57:09.514599Z","iopub.status.idle":"2024-01-27T10:57:09.773489Z","shell.execute_reply.started":"2024-01-27T10:57:09.514564Z","shell.execute_reply":"2024-01-27T10:57:09.772706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = extracted_features\ny_labels = np.argmax(label.values, axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-01-27T10:57:09.774603Z","iopub.execute_input":"2024-01-27T10:57:09.774885Z","iopub.status.idle":"2024-01-27T10:57:09.780720Z","shell.execute_reply.started":"2024-01-27T10:57:09.774861Z","shell.execute_reply":"2024-01-27T10:57:09.779806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y_labels, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-01-27T10:57:09.781735Z","iopub.execute_input":"2024-01-27T10:57:09.782089Z","iopub.status.idle":"2024-01-27T10:57:09.798071Z","shell.execute_reply.started":"2024-01-27T10:57:09.782053Z","shell.execute_reply":"2024-01-27T10:57:09.797287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize the XGBoost classifier and specify tree_method as \"gpu_hist\"-->\nxgb_clf = XGBClassifier(n_estimators=10000,\n                        max_depth=6,\n                        learning_rate=1e-1,\n                        objective='multi:softmax',\n                        num_class=len(np.unique(y_labels)),\n                        early_stopping_rounds=100,\n                        verbose=100,\n                        tree_method='gpu_hist')","metadata":{"execution":{"iopub.status.busy":"2024-01-27T11:09:02.387688Z","iopub.execute_input":"2024-01-27T11:09:02.388567Z","iopub.status.idle":"2024-01-27T11:09:02.394124Z","shell.execute_reply.started":"2024-01-27T11:09:02.388531Z","shell.execute_reply":"2024-01-27T11:09:02.393097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training\nxgb_clf.fit(X_train, y_train, eval_set=[(X_test, y_test)], verbose=100)","metadata":{"execution":{"iopub.status.busy":"2024-01-27T11:09:02.663505Z","iopub.execute_input":"2024-01-27T11:09:02.663858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc_xgb = accuracy_score(y_test, xgb_clf.predict(X_test))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_xgb = xgb_clf.predict(X_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Output model accuracy\naccuracy = accuracy_score(y_test, y_pred_xgb)\nprint(f\"Accuracy: {accuracy}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mutual_info_score\nkl = mutual_info_score(y_test,y_pred_xgb)","metadata":{"execution":{"iopub.status.busy":"2024-01-27T11:21:09.780700Z","iopub.execute_input":"2024-01-27T11:21:09.781494Z","iopub.status.idle":"2024-01-27T11:21:09.789621Z","shell.execute_reply.started":"2024-01-27T11:21:09.781461Z","shell.execute_reply":"2024-01-27T11:21:09.788583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kl","metadata":{"execution":{"iopub.status.busy":"2024-01-27T11:21:16.565843Z","iopub.execute_input":"2024-01-27T11:21:16.566231Z","iopub.status.idle":"2024-01-27T11:21:16.572606Z","shell.execute_reply.started":"2024-01-27T11:21:16.566199Z","shell.execute_reply":"2024-01-27T11:21:16.571538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\nprint('Test shape',test.shape)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-27T11:21:27.116152Z","iopub.execute_input":"2024-01-27T11:21:27.116528Z","iopub.status.idle":"2024-01-27T11:21:27.129004Z","shell.execute_reply.started":"2024-01-27T11:21:27.116498Z","shell.execute_reply":"2024-01-27T11:21:27.128047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FEATURE ENGINEER TEST\nPATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/'\ndata_test = np.zeros((len(test), extracted_features.shape[1]))\ntest_eeg = np.zeros((len(test), 10000, 19))\n\nfor k in range(len(test)):\n    row = test.iloc[k]\n    s = int(row.eeg_id)\n    eeg_test = pd.read_parquet(f'{PATH2}{s}.parquet')\n    eeg_test = eeg_test.iloc[:, :-1]\n    #eeg_test = eeg_test.iloc[:, [col for col in range(eeg_test.shape[1]) if col not in [8, 9, 10]]]\n    \n    start_time_point = int((eeg_test.shape[0] - 10_000) // 2)\n\n    eeg_test_slice = eeg_test.iloc[start_time_point : start_time_point + 10_000, :]\n    \n    test_eeg[k, :, :] = eeg_test_slice\n\n    \n    features_test = extract_eeg_features(test_eeg)\n    print(\"Shape of features_test:\", features_test.shape)\n\n    data_test = features_test\n\n\nprint(\"Shape of data_test:\", data_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-01-27T11:21:28.110723Z","iopub.execute_input":"2024-01-27T11:21:28.111158Z","iopub.status.idle":"2024-01-27T11:21:28.152921Z","shell.execute_reply.started":"2024-01-27T11:21:28.111127Z","shell.execute_reply":"2024-01-27T11:21:28.151954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FEATURE ENGINEER TEST\nPATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/'\ndata_test = np.zeros((len(test), extracted_features.shape[1]))\ntest_eeg = np.zeros((len(test), 10000, 19))\n\nfor k in range(len(test)):\n    row = test.iloc[k]\n    s = int(row.eeg_id)\n    eeg_test = pd.read_parquet(f'{PATH2}{s}.parquet')\n    eeg_test = eeg_test.iloc[:, :-1]\n    #eeg_test = eeg_test.iloc[:, [col for col in range(eeg_test.shape[1]) if col not in [8, 9, 10]]]\n    # Calculate starting time point\n    start_time_point = int((eeg_test.shape[0] - 10_000) // 2)\n    \n    # Get the time point data of the middle 50 seconds\n    \n    eeg_test_slice = eeg_test.iloc[start_time_point : start_time_point + 10_000, :]\n    \n    test_eeg[k, :, :] = eeg_test_slice\n\n    # Use the previously defined extract_eeg_features function for feature extraction\n    \n    features_test = extract_eeg_features(test_eeg)\n    print(\"Shape of features_test:\", features_test.shape)\n    # Put the extracted features into data_test\n    data_test = features_test\n\n# Output the shape of data_test\n\nprint(\"Shape of data_test:\", data_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-01-27T11:21:28.570673Z","iopub.execute_input":"2024-01-27T11:21:28.571378Z","iopub.status.idle":"2024-01-27T11:21:28.612317Z","shell.execute_reply.started":"2024-01-27T11:21:28.571344Z","shell.execute_reply":"2024-01-27T11:21:28.611314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use the trained model to make predictions\npredictions_proba = xgb_clf.predict_proba(data_test)\n\n# Category name\nclass_names = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n\n# Create a DataFrame to store probability values\nprobabilities_df = pd.DataFrame(predictions_proba, columns=class_names)","metadata":{"execution":{"iopub.status.busy":"2024-01-27T11:25:29.785641Z","iopub.execute_input":"2024-01-27T11:25:29.786500Z","iopub.status.idle":"2024-01-27T11:25:29.831676Z","shell.execute_reply.started":"2024-01-27T11:25:29.786459Z","shell.execute_reply":"2024-01-27T11:25:29.830668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"probabilities_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-27T11:25:34.570596Z","iopub.execute_input":"2024-01-27T11:25:34.570964Z","iopub.status.idle":"2024-01-27T11:25:34.583377Z","shell.execute_reply.started":"2024-01-27T11:25:34.570932Z","shell.execute_reply":"2024-01-27T11:25:34.582387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame({'eeg_id':test.eeg_id.values})\nsub[TARGETS] = probabilities_df\nsub.to_csv('submission.csv',index=False)\nprint('Submissionn shape',sub.shape)\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-27T11:25:53.615712Z","iopub.execute_input":"2024-01-27T11:25:53.616087Z","iopub.status.idle":"2024-01-27T11:25:53.634065Z","shell.execute_reply.started":"2024-01-27T11:25:53.616059Z","shell.execute_reply":"2024-01-27T11:25:53.633003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}