{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import keras\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom scipy.signal import spectrogram\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Dropout, Flatten, Dense\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nimport os\nfrom glob import glob\nimport cv2\nfrom tqdm.notebook import tqdm\nimport joblib","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-26T10:24:39.488674Z","iopub.execute_input":"2024-06-26T10:24:39.489124Z","iopub.status.idle":"2024-06-26T10:24:39.498828Z","shell.execute_reply.started":"2024-06-26T10:24:39.489085Z","shell.execute_reply":"2024-06-26T10:24:39.497480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set dataset paths\nBASE_PATH = \"/kaggle/input/hms-harmful-brain-activity-classification\"\nSPEC_DIR = \"/tmp/dataset/hms-hbac\"\nos.makedirs(SPEC_DIR+'/train_spectrograms', exist_ok=True)\nos.makedirs(SPEC_DIR+'/test_spectrograms', exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-06-26T10:24:39.500861Z","iopub.execute_input":"2024-06-26T10:24:39.501243Z","iopub.status.idle":"2024-06-26T10:24:39.509938Z","shell.execute_reply.started":"2024-06-26T10:24:39.501202Z","shell.execute_reply":"2024-06-26T10:24:39.508595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load meta data\ndf = pd.read_csv(f'{BASE_PATH}/train.csv')\ntest_df = pd.read_csv(f'{BASE_PATH}/test.csv')  # Assuming a similar CSV for test data\ndf['eeg_path'] = f'{BASE_PATH}/train_eegs/' + df['eeg_id'].astype(str) + '.parquet'\ndf['spec_path'] = f'{BASE_PATH}/train_spectrograms/' + df['spectrogram_id'].astype(str) + '.parquet'\ndf['spec2_path'] = f'{SPEC_DIR}/train_spectrograms/' + df['spectrogram_id'].astype(str) + '.npy'\n\n# Map labels to integers\nlabel_mapping = {'Seizure': 0, 'GPD': 1, 'LRDA': 2, 'Other': 3, 'GRDA': 4, 'LPD': 5}\ndf['class_label'] = df['expert_consensus'].map(label_mapping)","metadata":{"execution":{"iopub.status.busy":"2024-06-26T10:24:39.511023Z","iopub.execute_input":"2024-06-26T10:24:39.511367Z","iopub.status.idle":"2024-06-26T10:24:39.918191Z","shell.execute_reply.started":"2024-06-26T10:24:39.511329Z","shell.execute_reply":"2024-06-26T10:24:39.916961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to process and save spectrograms with fixed size\ndef process_spec(spec_id, split=\"train\", target_shape=(100, 100)):\n    spec_path = f\"{BASE_PATH}/{split}_spectrograms/{spec_id}.parquet\"\n    spec = pd.read_parquet(spec_path)\n    spec = spec.fillna(0).values[:, 1:].T # fill NaN values with 0, transpose for (Time, Freq) -> (Freq, Time)\n    \n    # Pad or truncate spectrogram to target shape\n    pad_width = [(0, max(0, target_shape[0] - spec.shape[0])), \n                 (0, max(0, target_shape[1] - spec.shape[1]))]\n    spec_padded = np.pad(spec, pad_width=pad_width, mode='constant')[:target_shape[0], :target_shape[1]]\n    \n    spec_padded = spec_padded.astype(\"float32\")\n    np.save(f\"{SPEC_DIR}/{split}_spectrograms/{spec_id}.npy\", spec_padded)","metadata":{"execution":{"iopub.status.busy":"2024-06-26T10:24:39.920233Z","iopub.execute_input":"2024-06-26T10:24:39.920689Z","iopub.status.idle":"2024-06-26T10:24:39.929233Z","shell.execute_reply.started":"2024-06-26T10:24:39.920650Z","shell.execute_reply":"2024-06-26T10:24:39.927576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Process train and test spectrograms\nspec_ids = df[\"spectrogram_id\"].unique()\ntest_spec_ids = test_df[\"spectrogram_id\"].unique()\n\n_ = joblib.Parallel(n_jobs=-1, backend=\"loky\")(\n    joblib.delayed(process_spec)(spec_id, \"train\")\n    for spec_id in tqdm(spec_ids, total=len(spec_ids))\n)\n\n_ = joblib.Parallel(n_jobs=-1, backend=\"loky\")(\n    joblib.delayed(process_spec)(spec_id, \"test\")\n    for spec_id in tqdm(test_spec_ids, total=len(test_spec_ids))\n)","metadata":{"execution":{"iopub.status.busy":"2024-06-26T10:24:39.932636Z","iopub.execute_input":"2024-06-26T10:24:39.933049Z","iopub.status.idle":"2024-06-26T10:27:51.477937Z","shell.execute_reply.started":"2024-06-26T10:24:39.933018Z","shell.execute_reply":"2024-06-26T10:27:51.476417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load spectrogram data\ndef load_spectrograms(spec_dir, ids, target_shape=(100, 100)):\n    spectrograms = []\n    for spec_id in tqdm(ids):\n        spec_path = f\"{spec_dir}/{spec_id}.npy\"\n        spectrogram = np.load(spec_path)\n        spectrograms.append(spectrogram)\n    return np.array(spectrograms)","metadata":{"execution":{"iopub.status.busy":"2024-06-26T10:27:51.480151Z","iopub.execute_input":"2024-06-26T10:27:51.481481Z","iopub.status.idle":"2024-06-26T10:27:51.488557Z","shell.execute_reply.started":"2024-06-26T10:27:51.481432Z","shell.execute_reply":"2024-06-26T10:27:51.487476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load training spectrograms\ntrain_spec_ids = df[\"spectrogram_id\"].unique()\ntest_spec_ids = test_df[\"spectrogram_id\"].unique()\n\nX_train = load_spectrograms(SPEC_DIR+'/train_spectrograms', train_spec_ids)\ny_train = df[df[\"spectrogram_id\"].isin(train_spec_ids)][\"class_label\"].values","metadata":{"execution":{"iopub.status.busy":"2024-06-26T10:27:51.490176Z","iopub.execute_input":"2024-06-26T10:27:51.491099Z","iopub.status.idle":"2024-06-26T10:27:56.797624Z","shell.execute_reply.started":"2024-06-26T10:27:51.491065Z","shell.execute_reply":"2024-06-26T10:27:56.796589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split into train and validation sets\nX_train, X_val, y_train, y_val = train_test_split(X_train, y_train, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-06-26T10:27:56.799055Z","iopub.execute_input":"2024-06-26T10:27:56.799495Z","iopub.status.idle":"2024-06-26T10:27:57.095792Z","shell.execute_reply.started":"2024-06-26T10:27:56.799452Z","shell.execute_reply":"2024-06-26T10:27:57.094278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load test spectrograms\nX_test = load_spectrograms(SPEC_DIR+'/test_spectrograms', test_spec_ids)","metadata":{"execution":{"iopub.status.busy":"2024-06-26T10:27:57.096734Z","iopub.status.idle":"2024-06-26T10:27:57.097138Z","shell.execute_reply.started":"2024-06-26T10:27:57.096946Z","shell.execute_reply":"2024-06-26T10:27:57.096964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reshape data for CNN input\nX_train = X_train.reshape(-1, 100, 100, 1)\nX_val = X_val.reshape(-1, 100, 100, 1)\nX_test = X_test.reshape(-1, 100, 100, 1)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-26T10:27:57.099133Z","iopub.status.idle":"2024-06-26T10:27:57.099539Z","shell.execute_reply.started":"2024-06-26T10:27:57.099357Z","shell.execute_reply":"2024-06-26T10:27:57.099373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Build and train the CNN model\nmodel = Sequential([\n    Conv2D(32, (3, 3), activation='relu', input_shape=(100, 100, 1)),\n    MaxPooling2D((2, 2)),\n    Dropout(0.25),\n    Conv2D(64, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Dropout(0.25),\n    Flatten(),\n    Dense(128, activation='relu'),\n    Dropout(0.5),\n    Dense(6, activation='softmax')\n])\n\nmodel.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-26T10:27:57.101103Z","iopub.status.idle":"2024-06-26T10:27:57.101512Z","shell.execute_reply.started":"2024-06-26T10:27:57.101334Z","shell.execute_reply":"2024-06-26T10:27:57.101351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(X_train, y_train, epochs=10, validation_data=(X_val, y_val))","metadata":{"execution":{"iopub.status.busy":"2024-06-26T10:27:57.103671Z","iopub.status.idle":"2024-06-26T10:27:57.104058Z","shell.execute_reply.started":"2024-06-26T10:27:57.103880Z","shell.execute_reply":"2024-06-26T10:27:57.103896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate on test set\ntest_loss, test_acc = model.evaluate(X_test, test_df[\"class_label\"].map(label_mapping).values, verbose=2)\nprint(f\"Test accuracy: {test_acc}\")","metadata":{"execution":{"iopub.status.busy":"2024-06-26T10:27:57.106114Z","iopub.status.idle":"2024-06-26T10:27:57.106529Z","shell.execute_reply.started":"2024-06-26T10:27:57.106343Z","shell.execute_reply":"2024-06-26T10:27:57.106361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import keras\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom scipy.signal import spectrogram\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Dropout, Flatten, Dense\nfrom sklearn.model_selection import StratifiedGroupKFold\nimport os\nfrom glob import glob\nimport cv2\nfrom tqdm.notebook import tqdm\nimport joblib\n\n# Set dataset paths\nBASE_PATH = \"/kaggle/input/hms-harmful-brain-activity-classification\"\nSPEC_DIR = \"/tmp/dataset/hms-hbac\"\nos.makedirs(SPEC_DIR+'/train_spectrograms', exist_ok=True)\nos.makedirs(SPEC_DIR+'/test_spectrograms', exist_ok=True)\n\n# Load meta data\ndf = pd.read_csv(f'{BASE_PATH}/train.csv')\ntest_df = pd.read_csv(f'{BASE_PATH}/test.csv')  # Assuming a similar CSV for test data\n\ndf['eeg_path'] = f'{BASE_PATH}/train_eegs/' + df['eeg_id'].astype(str) + '.parquet'\ndf['spec_path'] = f'{BASE_PATH}/train_spectrograms/' + df['spectrogram_id'].astype(str) + '.parquet'\ndf['spec2_path'] = f'{SPEC_DIR}/train_spectrograms/' + df['spectrogram_id'].astype(str) + '.npy'\n\ntest_df['eeg_path'] = f'{BASE_PATH}/test_eegs/' + test_df['eeg_id'].astype(str) + '.parquet'\ntest_df['spec_path'] = f'{BASE_PATH}/test_spectrograms/' + test_df['spectrogram_id'].astype(str) + '.parquet'\ntest_df['spec2_path'] = f'{SPEC_DIR}/test_spectrograms/' + test_df['spectrogram_id'].astype(str) + '.npy'\n\n# Map labels to integers\nlabel_mapping = {'Seizure': 0, 'GPD': 1, 'LRDA': 2, 'Other': 3, 'GRDA': 4, 'LPD': 5}\ndf['class_label'] = df['expert_consensus'].map(label_mapping)\ntest_df['class_label'] = test_df['expert_consensus'].map(label_mapping)\n\n# Function to process and save spectrograms with fixed size\ndef process_spec(spec_id, split=\"train\", target_shape=(100, 100)):\n    spec_path = f\"{BASE_PATH}/{split}_spectrograms/{spec_id}.parquet\"\n    spec = pd.read_parquet(spec_path)\n    spec = spec.fillna(0).values[:, 1:].T # fill NaN values with 0, transpose for (Time, Freq) -> (Freq, Time)\n    \n    # Pad or truncate spectrogram to target shape\n    pad_width = [(0, max(0, target_shape[0] - spec.shape[0])), \n                 (0, max(0, target_shape[1] - spec.shape[1]))]\n    spec_padded = np.pad(spec, pad_width=pad_width, mode='constant')[:target_shape[0], :target_shape[1]]\n    \n    spec_padded = spec_padded.astype(\"float32\")\n    np.save(f\"{SPEC_DIR}/{split}_spectrograms/{spec_id}.npy\", spec_padded)\n\n# Process train and test spectrograms\nspec_ids = df[\"spectrogram_id\"].unique()\ntest_spec_ids = test_df[\"spectrogram_id\"].unique()\n\n_ = joblib.Parallel(n_jobs=-1, backend=\"loky\")(\n    joblib.delayed(process_spec)(spec_id, \"train\")\n    for spec_id in tqdm(spec_ids, total=len(spec_ids))\n)\n\n_ = joblib.Parallel(n_jobs=-1, backend=\"loky\")(\n    joblib.delayed(process_spec)(spec_id, \"test\")\n    for spec_id in tqdm(test_spec_ids, total=len(test_spec_ids))\n)\n\n# Load spectrogram data\ndef load_spectrograms(spec_dir, ids, target_shape=(100, 100)):\n    spectrograms = []\n    for spec_id in tqdm(ids):\n        spec_path = f\"{spec_dir}/{spec_id}.npy\"\n        spectrogram = np.load(spec_path)\n        spectrograms.append(spectrogram)\n    return np.array(spectrograms)\n\n# StratifiedGroupKFold\nCFG = {\"seed\": 42, \"fold\": 0, \"batch_size\": 32}\n\nsgkf = StratifiedGroupKFold(n_splits=5, shuffle=True, random_state=CFG[\"seed\"])\n\ndf[\"fold\"] = -1\ndf.reset_index(drop=True, inplace=True)\nfor fold, (train_idx, valid_idx) in enumerate(sgkf.split(df, y=df[\"class_label\"], groups=df[\"patient_id\"])):\n    df.loc[valid_idx, \"fold\"] = fold\n\n# Sample from full data\nsample_df = df.groupby(\"spectrogram_id\").head(1).reset_index(drop=True)\ntrain_df = sample_df[sample_df.fold != CFG[\"fold\"]]\nvalid_df = sample_df[sample_df.fold == CFG[\"fold\"]]\nprint(f\"# Num Train: {len(train_df)} | Num Valid: {len(valid_df)}\")\n\n# Train\ntrain_paths = train_df.spec2_path.values\ntrain_labels = train_df.class_label.values\nX_train = load_spectrograms(SPEC_DIR+'/train_spectrograms', train_df[\"spectrogram_id\"].values)\ny_train = train_labels\n\n# Valid\nvalid_paths = valid_df.spec2_path.values\nvalid_labels = valid_df.class_label.values\nX_val = load_spectrograms(SPEC_DIR+'/train_spectrograms', valid_df[\"spectrogram_id\"].values)\ny_val = valid_labels\n\n# Reshape data for CNN input\nX_train = X_train.reshape(-1, 100, 100, 1)\nX_val = X_val.reshape(-1, 100, 100, 1)\n\n# Build and train the CNN model\nmodel = Sequential([\n    Conv2D(32, (3, 3), activation='relu', input_shape=(100, 100, 1)),\n    MaxPooling2D((2, 2)),\n    Dropout(0.25),\n    Conv2D(64, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Dropout(0.25),\n    Flatten(),\n    Dense(128, activation='relu'),\n    Dropout(0.5),\n    Dense(6, activation='softmax')\n])\n\nmodel.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\nhistory = model.fit(X_train, y_train, epochs=10, validation_data=(X_val, y_val))\n\n# Load test spectrograms\nX_test = load_spectrograms(SPEC_DIR+'/test_spectrograms', test_spec_ids)\nX_test = X_test.reshape(-1, 100, 100, 1)\n\n# Evaluate on test set\ntest_loss, test_acc = model.evaluate(X_test, test_df[\"class_label\"].map(label_mapping).values, verbose=2)\nprint(f\"Test accuracy: {test_acc}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-06-29T09:00:07.550499Z","iopub.execute_input":"2024-06-29T09:00:07.550924Z","iopub.status.idle":"2024-06-29T09:00:08.282464Z","shell.execute_reply.started":"2024-06-29T09:00:07.550888Z","shell.execute_reply":"2024-06-29T09:00:08.280865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import keras\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom scipy.signal import spectrogram\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Dropout, Flatten, Dense\nfrom sklearn.model_selection import StratifiedGroupKFold\nimport os\nfrom glob import glob\nimport cv2\nfrom tqdm.notebook import tqdm\nimport joblib\n\n# Set dataset paths\nBASE_PATH = \"/kaggle/input/hms-harmful-brain-activity-classification\"\nSPEC_DIR = \"/tmp/dataset/hms-hbac\"\nos.makedirs(SPEC_DIR+'/train_spectrograms', exist_ok=True)\nos.makedirs(SPEC_DIR+'/test_spectrograms', exist_ok=True)\n\n# Load meta data\ndf = pd.read_csv(f'{BASE_PATH}/train.csv')\n\n# Ensure test_df has the same structure\nif os.path.exists(f'{BASE_PATH}/test.csv'):\n    test_df = pd.read_csv(f'{BASE_PATH}/test.csv')\nelse:\n    # Create a mock test_df with the same columns as df for this example\n    test_df = df.sample(frac=0.1, random_state=42).copy()\n    test_df.drop(columns=['class_label'], inplace=True)\n\ndf['eeg_path'] = f'{BASE_PATH}/train_eegs/' + df['eeg_id'].astype(str) + '.parquet'\ndf['spec_path'] = f'{BASE_PATH}/train_spectrograms/' + df['spectrogram_id'].astype(str) + '.parquet'\ndf['spec2_path'] = f'{SPEC_DIR}/train_spectrograms/' + df['spectrogram_id'].astype(str) + '.npy'\n\ntest_df['eeg_path'] = f'{BASE_PATH}/test_eegs/' + test_df['eeg_id'].astype(str) + '.parquet'\ntest_df['spec_path'] = f'{BASE_PATH}/test_spectrograms/' + test_df['spectrogram_id'].astype(str) + '.parquet'\ntest_df['spec2_path'] = f'{SPEC_DIR}/test_spectrograms/' + test_df['spectrogram_id'].astype(str) + '.npy'\n\n# Map labels to integers\nlabel_mapping = {'Seizure': 0, 'GPD': 1, 'LRDA': 2, 'Other': 3, 'GRDA': 4, 'LPD': 5}\ndf['class_label'] = df['expert_consensus'].map(label_mapping)\n\n# Ensure 'expert_consensus' exists in test_df\nif 'expert_consensus' in test_df.columns:\n    test_df['class_label'] = test_df['expert_consensus'].map(label_mapping)\nelse:\n    test_df['class_label'] = np.random.choice(list(label_mapping.values()), size=len(test_df))\n\n# Function to process and save spectrograms with fixed size\ndef process_spec(spec_id, split=\"train\", target_shape=(100, 100)):\n    spec_path = f\"{BASE_PATH}/{split}_spectrograms/{spec_id}.parquet\"\n    spec = pd.read_parquet(spec_path)\n    spec = spec.fillna(0).values[:, 1:].T # fill NaN values with 0, transpose for (Time, Freq) -> (Freq, Time)\n    \n    # Pad or truncate spectrogram to target shape\n    pad_width = [(0, max(0, target_shape[0] - spec.shape[0])), \n                 (0, max(0, target_shape[1] - spec.shape[1]))]\n    spec_padded = np.pad(spec, pad_width=pad_width, mode='constant')[:target_shape[0], :target_shape[1]]\n    \n    spec_padded = spec_padded.astype(\"float32\")\n    np.save(f\"{SPEC_DIR}/{split}_spectrograms/{spec_id}.npy\", spec_padded)\n\n# Process train and test spectrograms\nspec_ids = df[\"spectrogram_id\"].unique()\ntest_spec_ids = test_df[\"spectrogram_id\"].unique()\n\n_ = joblib.Parallel(n_jobs=-1, backend=\"loky\")(\n    joblib.delayed(process_spec)(spec_id, \"train\")\n    for spec_id in tqdm(spec_ids, total=len(spec_ids))\n)\n\n_ = joblib.Parallel(n_jobs=-1, backend=\"loky\")(\n    joblib.delayed(process_spec)(spec_id, \"test\")\n    for spec_id in tqdm(test_spec_ids, total=len(test_spec_ids))\n)\n\n# Load spectrogram data\ndef load_spectrograms(spec_dir, ids, target_shape=(100, 100)):\n    spectrograms = []\n    for spec_id in tqdm(ids):\n        spec_path = f\"{spec_dir}/{spec_id}.npy\"\n        spectrogram = np.load(spec_path)\n        spectrograms.append(spectrogram)\n    return np.array(spectrograms)\n\n# StratifiedGroupKFold\nCFG = {\"seed\": 42, \"fold\": 0, \"batch_size\": 32}\n\nsgkf = StratifiedGroupKFold(n_splits=5, shuffle=True, random_state=CFG[\"seed\"])\n\ndf[\"fold\"] = -1\ndf.reset_index(drop=True, inplace=True)\nfor fold, (train_idx, valid_idx) in enumerate(sgkf.split(df, y=df[\"class_label\"], groups=df[\"patient_id\"])):\n    df.loc[valid_idx, \"fold\"] = fold\n\n# Sample from full data\nsample_df = df.groupby(\"spectrogram_id\").head(1).reset_index(drop=True)\ntrain_df = sample_df[sample_df.fold != CFG[\"fold\"]]\nvalid_df = sample_df[sample_df.fold == CFG[\"fold\"]]\nprint(f\"# Num Train: {len(train_df)} | Num Valid: {len(valid_df)}\")\n\n# Train\ntrain_paths = train_df.spec2_path.values\ntrain_labels = train_df.class_label.values\nX_train = load_spectrograms(SPEC_DIR+'/train_spectrograms', train_df[\"spectrogram_id\"].values)\ny_train = train_labels\n\n# Valid\nvalid_paths = valid_df.spec2_path.values\nvalid_labels = valid_df.class_label.values\nX_val = load_spectrograms(SPEC_DIR+'/train_spectrograms', valid_df[\"spectrogram_id\"].values)\ny_val = valid_labels\n\n# Reshape data for CNN input\nX_train = X_train.reshape(-1, 100, 100, 1)\nX_val = X_val.reshape(-1, 100, 100, 1)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-30T13:26:04.544930Z","iopub.execute_input":"2024-06-30T13:26:04.545364Z","iopub.status.idle":"2024-06-30T13:45:31.630191Z","shell.execute_reply.started":"2024-06-30T13:26:04.545328Z","shell.execute_reply":"2024-06-30T13:45:31.627912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Dropout, Flatten, Dense, BatchNormalization, GlobalAveragePooling2D","metadata":{"execution":{"iopub.status.busy":"2024-06-30T13:59:23.772686Z","iopub.execute_input":"2024-06-30T13:59:23.773532Z","iopub.status.idle":"2024-06-30T13:59:23.779234Z","shell.execute_reply.started":"2024-06-30T13:59:23.773497Z","shell.execute_reply":"2024-06-30T13:59:23.777300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Build and train the CNN model\nmodel = Sequential([\n    Conv2D(32, (3, 3), activation='relu', input_shape=(100, 100, 1)),\n    BatchNormalization(),\n    MaxPooling2D((2, 2)),\n    Dropout(0.25),\n    \n    Conv2D(64, (3, 3), activation='relu'),\n    BatchNormalization(),\n    MaxPooling2D((2, 2)),\n    Dropout(0.25),\n    \n    Conv2D(128, (3, 3), activation='relu'),\n    BatchNormalization(),\n    MaxPooling2D((2, 2)),\n    Dropout(0.25),\n    \n    Conv2D(256, (3, 3), activation='relu'),\n    BatchNormalization(),\n    MaxPooling2D((2, 2)),\n    Dropout(0.25),\n    \n    GlobalAveragePooling2D(),\n    Dense(128, activation='relu'),\n    BatchNormalization(),\n    Dropout(0.5),\n    \n    Dense(6, activation='softmax')\n])\nmodel.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\nhistory = model.fit(X_train, y_train, epochs=10, validation_data=(X_val, y_val))\n\n# Load test spectrograms\nX_test = load_spectrograms(SPEC_DIR+'/test_spectrograms', test_spec_ids)\nX_test = X_test.reshape(-1, 100, 100, 1)\n\n# Evaluate on test set\ntest_loss, test_acc = model.evaluate(X_test, test_df[\"class_label\"].values, verbose=2)\nprint(f\"Test accuracy: {test_acc}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-06-30T13:59:25.754137Z","iopub.execute_input":"2024-06-30T13:59:25.754572Z","iopub.status.idle":"2024-06-30T14:23:42.073801Z","shell.execute_reply.started":"2024-06-30T13:59:25.754542Z","shell.execute_reply":"2024-06-30T14:23:42.072395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}