{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30839,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ---------------------------\n# 1. Imports and Setup\n# ---------------------------\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout\nimport pandas as pd\nimport numpy as np\nimport os\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, confusion_matrix\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm import tqdm\nfrom joblib import Parallel, delayed  # Import Joblib\n\nprint(\"Libraries imported successfully\")\nprint(\"TensorFlow version:\", tf.__version__)\nprint(\"Pandas version:\", pd.__version__)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T14:10:04.004164Z","iopub.execute_input":"2025-01-23T14:10:04.004493Z","iopub.status.idle":"2025-01-23T14:10:04.010901Z","shell.execute_reply.started":"2025-01-23T14:10:04.004466Z","shell.execute_reply":"2025-01-23T14:10:04.010068Z"}},"outputs":[{"name":"stdout","text":"Libraries imported successfully\nTensorFlow version: 2.17.1\nPandas version: 2.2.2\n","output_type":"stream"}],"execution_count":4},{"cell_type":"code","source":"# ---------------------------\n# 2. Configuration Parameters\n# ---------------------------\n# Spectrogram dimensions\nSPECTROGRAM_TIME_STEPS = 10  # 10 seconds (1 per second)\nSPECTROGRAM_FREQ_BINS = 60   # 0.5Hz steps from 0.5Hz to 30Hz\nSPECTROGRAM_REGIONS = 4      # Brain regions: LL, RL, LP, RP\nNUM_CLASSES = 6              # Output classes\n\n# Training parameters\nBATCH_SIZE = 32\nEPOCHS = 10\n\n# Data paths (example Kaggle paths)\nTRAIN_CSV_PATH = \"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\"\nTEST_CSV_PATH = \"/kaggle/input/hms-harmful-brain-activity-classification/test.csv\"\nTRAIN_SPEC_DIR = \"/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms\"\nTEST_SPEC_DIR = \"/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms\"\n\nprint(\"\\nConfiguration set:\")\nprint(f\"Time steps: {SPECTROGRAM_TIME_STEPS}, Freq bins: {SPECTROGRAM_FREQ_BINS}\")\nprint(f\"Regions: {SPECTROGRAM_REGIONS}, Classes: {NUM_CLASSES}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T14:10:32.229746Z","iopub.execute_input":"2025-01-23T14:10:32.230048Z","iopub.status.idle":"2025-01-23T14:10:32.236563Z","shell.execute_reply.started":"2025-01-23T14:10:32.230026Z","shell.execute_reply":"2025-01-23T14:10:32.23562Z"}},"outputs":[{"name":"stdout","text":"\nConfiguration set:\nTime steps: 10, Freq bins: 60\nRegions: 4, Classes: 6\n","output_type":"stream"}],"execution_count":5},{"cell_type":"code","source":"# ---------------------------\n# 3. Data Loading\n# ---------------------------\n# Load metadata\ntrain_df = pd.read_csv(TRAIN_CSV_PATH)\ntest_df = pd.read_csv(TEST_CSV_PATH)\n\nprint(\"\\nData loading complete:\")\nprint(f\"Training samples: {len(train_df)}\")\nprint(f\"Test samples: {len(test_df)}\")\nprint(\"Sample training data:\")\nprint(train_df.head(2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T14:10:42.95405Z","iopub.execute_input":"2025-01-23T14:10:42.95438Z","iopub.status.idle":"2025-01-23T14:10:43.121212Z","shell.execute_reply.started":"2025-01-23T14:10:42.954355Z","shell.execute_reply":"2025-01-23T14:10:43.120374Z"}},"outputs":[{"name":"stdout","text":"\nData loading complete:\nTraining samples: 106800\nTest samples: 1\nSample training data:\n       eeg_id  eeg_sub_id  eeg_label_offset_seconds  spectrogram_id  \\\n0  1628180742           0                       0.0          353733   \n1  1628180742           1                       6.0          353733   \n\n   spectrogram_sub_id  spectrogram_label_offset_seconds    label_id  \\\n0                   0                               0.0   127492639   \n1                   1                               6.0  3887563113   \n\n   patient_id expert_consensus  seizure_vote  lpd_vote  gpd_vote  lrda_vote  \\\n0       42516          Seizure             3         0         0          0   \n1       42516          Seizure             3         0         0          0   \n\n   grda_vote  other_vote  \n0          0           0  \n1          0           0  \n","output_type":"stream"}],"execution_count":6},{"cell_type":"code","source":"# ---------------------------\n# 4. Spectrogram Preprocessing (Fixed)\n# ---------------------------\ndef preprocess_spectrogram(spec_id, data_dir):\n    \"\"\"Process a spectrogram file into a normalized 3D array with fixed 60 frequency bins\"\"\"\n    # Load raw data\n    spec_path = os.path.join(data_dir, f\"{spec_id}.parquet\")\n    spec = pd.read_parquet(spec_path)\n    \n    # Extract central 10 seconds (295-305s in 600s spectrogram)\n    central_start = 295\n    spec_central = spec.iloc[central_start:central_start+10]\n    \n    # Define expected frequencies (0.5Hz to 30Hz in 0.5Hz steps)\n    expected_freqs = np.round(np.arange(0.5, 30.5, 0.5), 1)\n    expected_regions = ['LL', 'RL', 'LP', 'RP']\n    \n    # Create empty 3D array (Time, Frequency, Region)\n    spec_3d = np.zeros((SPECTROGRAM_TIME_STEPS, len(expected_freqs), len(expected_regions)))\n    \n    # Fill the 3D array\n    for t in range(SPECTROGRAM_TIME_STEPS):\n        for r_idx, region in enumerate(expected_regions):\n            for f_idx, freq in enumerate(expected_freqs):\n                col_name = f\"{region}_{freq}Hz\"\n                if col_name in spec_central.columns:\n                    spec_3d[t, f_idx, r_idx] = spec_central[col_name].iloc[t]\n    \n    # Normalize each frequency-region channel independently\n    mean = np.mean(spec_3d, axis=(0, 1))\n    std = np.std(spec_3d, axis=(0, 1))\n    spec_3d = (spec_3d - mean) / (std + 1e-8)\n    \n    return spec_3d.astype(np.float32)\n\n# Test preprocessing on a sample\nsample_spec = preprocess_spectrogram(train_df['spectrogram_id'][0], TRAIN_SPEC_DIR)\nprint(\"\\nPreprocessing check:\")\nprint(f\"Input shape: {sample_spec.shape} (Time, Frequency, Regions)\")\nprint(f\"Data range: {np.min(sample_spec):.2f} to {np.max(sample_spec):.2f}\")\nprint(f\"NaN values: {np.isnan(sample_spec).sum()}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T14:11:00.554547Z","iopub.execute_input":"2025-01-23T14:11:00.55487Z","iopub.status.idle":"2025-01-23T14:11:00.595296Z","shell.execute_reply.started":"2025-01-23T14:11:00.554847Z","shell.execute_reply":"2025-01-23T14:11:00.594437Z"}},"outputs":[{"name":"stdout","text":"\nPreprocessing check:\nInput shape: (10, 60, 4) (Time, Frequency, Regions)\nData range: 0.00 to 0.00\nNaN values: 0\n","output_type":"stream"}],"execution_count":7},{"cell_type":"code","source":"# ---------------------------\n# 5. Prepare Training Data\n# ---------------------------\n# Convert votes to probabilities (normalize)\ntrain_labels = train_df[['seizure_vote', 'lpd_vote', 'gpd_vote', \n                        'lrda_vote', 'grda_vote', 'other_vote']].values\ntrain_labels = train_labels / train_labels.sum(axis=1, keepdims=True)\n\n# Process all training spectrograms in parallel\nprint(\"\\nProcessing training spectrograms in parallel...\")\nX_train = Parallel(n_jobs=-1)(  # Use all available CPU cores\n    delayed(preprocess_spectrogram)(spec_id, TRAIN_SPEC_DIR)\n    for spec_id in tqdm(train_df['spectrogram_id'], total=len(train_df))\n)\nX_train = np.array(X_train)\n\n# Split into training and validation\nX_train, X_val, y_train, y_val = train_test_split(X_train, train_labels, \n                                                 test_size=0.2, random_state=42)\n\nprint(\"\\nData shapes:\")\nprint(f\"Training data: {X_train.shape}\")\nprint(f\"Validation data: {X_val.shape}\")\nprint(f\"Training labels: {y_train.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T14:11:10.205073Z","iopub.execute_input":"2025-01-23T14:11:10.20539Z"}},"outputs":[{"name":"stdout","text":"\nProcessing training spectrograms in parallel...\n","output_type":"stream"},{"name":"stderr","text":" 55%|█████▌    | 59164/106800 [12:44<09:39, 82.14it/s]","output_type":"stream"}],"execution_count":null},{"cell_type":"code","source":"# ---------------------------\n# 6. Model Architecture\n# ---------------------------\nmodel = Sequential([\n    # First convolution block\n    Conv2D(32, (3, 3), activation='relu', \n          input_shape=(SPECTROGRAM_TIME_STEPS, SPECTROGRAM_FREQ_BINS, SPECTROGRAM_REGIONS)),\n    MaxPooling2D((2, 2)),\n    \n    # Second convolution block\n    Conv2D(64, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    \n    # Classifier\n    Flatten(),\n    Dense(128, activation='relu'),\n    Dropout(0.5),  # Reduces overfitting\n    Dense(NUM_CLASSES, activation='softmax')\n])\n\nmodel.compile(\n    optimizer='adam',\n    loss=tf.keras.losses.KLDivergence(),  # Matches competition evaluation metric\n    metrics=['accuracy']\n)\n\nprint(\"\\nModel summary:\")\nmodel.summary()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ---------------------------\n# 7. Training Process\n# ---------------------------\nprint(\"\\nStarting training...\")\nhistory = model.fit(\n    X_train, y_train,\n    validation_data=(X_val, y_val),\n    epochs=EPOCHS,\n    batch_size=BATCH_SIZE,\n    verbose=1\n)\n\n# Plot training history\nplt.figure(figsize=(12, 5))\nplt.subplot(1, 2, 1)\nplt.plot(history.history['loss'], label='Training Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.title('Loss Over Epochs')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\n\nplt.subplot(1, 2, 2)\nplt.plot(history.history['accuracy'], label='Training Accuracy')\nplt.plot(history.history['val_accuracy'], label='Validation Accuracy')\nplt.title('Accuracy Over Epochs')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ---------------------------\n# 8. Evaluation\n# ---------------------------\n# Calculate validation metrics\nval_pred = model.predict(X_val)\n\n# KL Divergence (competition metric)\nval_kl = tf.keras.losses.KLDivergence()(y_val, val_pred).numpy()\n\n# Accuracy (agreement with majority vote)\ntrue_classes = np.argmax(y_val, axis=1)\npred_classes = np.argmax(val_pred, axis=1)\nval_accuracy = accuracy_score(true_classes, pred_classes)\n\nprint(\"\\nValidation Results:\")\nprint(f\"KL Divergence: {val_kl:.4f} (Lower is better)\")\nprint(f\"Accuracy: {val_accuracy:.4f} (Higher is better)\")\n\n# Confusion Matrix\nconf_matrix = confusion_matrix(true_classes, pred_classes)\nplt.figure(figsize=(8, 6))\nsns.heatmap(conf_matrix, annot=True, fmt='d', cmap='Blues', \n            xticklabels=['SZ', 'LPD', 'GPD', 'LRDA', 'GRDA', 'Other'],\n            yticklabels=['SZ', 'LPD', 'GPD', 'LRDA', 'GRDA', 'Other'])\nplt.title('Confusion Matrix')\nplt.xlabel('Predicted')\nplt.ylabel('True')\nplt.show()\n\n# Class Distribution\nplt.figure(figsize=(8, 5))\nsns.countplot(x=true_classes, palette='viridis')\nplt.title('Class Distribution in Validation Set')\nplt.xlabel('Class')\nplt.ylabel('Count')\nplt.xticks(ticks=range(6), labels=['SZ', 'LPD', 'GPD', 'LRDA', 'GRDA', 'Other'])\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ---------------------------\n# 9. Test Predictions\n# ---------------------------\nprint(\"\\nGenerating test predictions in parallel...\")\ndef predict_spec(spec_id):\n    spec_data = preprocess_spectrogram(spec_id, TEST_SPEC_DIR)\n    spec_data = np.expand_dims(spec_data, axis=0)  # Add batch dimension\n    pred = model.predict(spec_data, verbose=0)[0]\n    return pred\n\ntest_preds = Parallel(n_jobs=-1)(  # Use all available CPU cores\n    delayed(predict_spec)(spec_id)\n    for spec_id in tqdm(test_df['spectrogram_id'], total=len(test_df))\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ---------------------------\n# 10. Create Submission File\n# ---------------------------\nsubmission = pd.DataFrame({\n    'eeg_id': test_df['eeg_id'],\n    'seizure_vote': [p[0] for p in test_preds],\n    'lpd_vote': [p[1] for p in test_preds],\n    'gpd_vote': [p[2] for p in test_preds],\n    'lrda_vote': [p[3] for p in test_preds],\n    'grda_vote': [p[4] for p in test_preds],\n    'other_vote': [p[5] for p in test_preds]\n})\n\nsubmission.to_csv('submission.csv', index=False)\nprint(\"\\nSubmission file saved successfully!\")\nprint(\"First 2 rows of submission:\")\nprint(submission.head(2))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}