{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":31011,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n    #for filename in filenames:\n        #print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-29T16:26:13.692164Z","iopub.execute_input":"2025-04-29T16:26:13.692488Z","iopub.status.idle":"2025-04-29T16:26:14.018816Z","shell.execute_reply.started":"2025-04-29T16:26:13.692456Z","shell.execute_reply":"2025-04-29T16:26:14.018171Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport os\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder\nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix\nfrom tensorflow.keras.utils import to_categorical\nimport pyarrow.parquet as pq\nfrom PIL import Image\n\n# Paths\ndata_csv = \"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\"\nparquet_dir = \"/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms\"\n\n# Load metadata\ndf = pd.read_csv(data_csv)\nexisting_files = set(os.listdir(parquet_dir))\ndf = df[df[\"spectrogram_id\"].apply(lambda x: f\"{x}.parquet\" in existing_files)]\ndf = df.head(5000)\n\nif df.empty:\n    raise ValueError(\"No matching .parquet files found.\")\n\n# Encode labels\ny = df[\"expert_consensus\"].dropna()\ndf = df.loc[y.index]\nle = LabelEncoder()\ny_encoded = le.fit_transform(y)\ny_categorical = to_categorical(y_encoded)\n\n# Metadata features\nX_meta = df[['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']].values\nscaler = StandardScaler()\nX_meta = scaler.fit_transform(X_meta)\n\n# Helper to load and preprocess spectrogram image\ndef load_spectrogram_as_image(spectrogram_id, size=(224, 224)):\n    file_path = os.path.join(parquet_dir, f\"{spectrogram_id}.parquet\")\n    table = pq.read_table(file_path)\n    data = table.to_pandas().values.T.astype(np.float32)\n    data = np.nan_to_num(data, nan=0.0, posinf=0.0, neginf=0.0)\n    data = np.log1p(np.clip(data, a_min=1e-6, a_max=None))\n    data = np.nan_to_num(data, nan=0.0, posinf=0.0, neginf=0.0)\n    data = (data - np.min(data)) / (np.max(data) - np.min(data) + 1e-8)\n    img = Image.fromarray(np.uint8(data * 255)).convert(\"RGB\").resize(size)\n    return np.array(img)\n\n# Load all images\nX_images = np.stack([load_spectrogram_as_image(sid) for sid in df[\"spectrogram_id\"]])\nX_images = X_images / 255.0  # Normalize\n\n# Train-val-test split\nX_img_train, X_img_temp, X_meta_train, X_meta_temp, y_train, y_temp = train_test_split(\n    X_images, X_meta, y_categorical, test_size=0.3, random_state=42, stratify=y_categorical)\nX_img_val, X_img_test, X_meta_val, X_meta_test, y_val, y_test = train_test_split(\n    X_img_temp, X_meta_temp, y_temp, test_size=0.5, random_state=42, stratify=y_temp)\n\n# VGG16 image branch\nimage_input = keras.Input(shape=(224, 224, 3), name=\"image_input\")\nbase_vgg16 = keras.applications.VGG16(weights=\"imagenet\", include_top=False, input_shape=(224, 224, 3))\nbase_vgg16.trainable = False  # Freeze base\nx = base_vgg16(image_input)\nx = layers.GlobalAveragePooling2D()(x)\nx = layers.Dense(256, activation='relu')(x)\nx = layers.Dropout(0.5)(x)\nimage_features = layers.Dense(128, activation='relu')(x)\n\n# Metadata branch\nmeta_input = keras.Input(shape=(X_meta.shape[1],), name=\"meta_input\")\ny_meta = layers.Dense(64, activation='relu')(meta_input)\ny_meta = layers.Dropout(0.3)(y_meta)\nmeta_features = layers.Dense(32, activation='relu')(y_meta)\n\n# Combine both\ncombined = layers.concatenate([image_features, meta_features])\nz = layers.Dense(128, activation='relu')(combined)\nz = layers.Dropout(0.5)(z)\noutput = layers.Dense(y_categorical.shape[1], activation='softmax')(z)\n\n# Final model\nmodel = keras.Model(inputs=[image_input, meta_input], outputs=output)\nmodel.compile(optimizer=keras.optimizers.Adam(1e-4), loss='categorical_crossentropy', metrics=['accuracy'])\n\n# Callbacks\ncallbacks = [\n    keras.callbacks.EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True),\n    keras.callbacks.ModelCheckpoint(\"best_vgg16_model.keras\", monitor='val_loss', save_best_only=True)\n]\n\n# Train\nhistory = model.fit(\n    x=[X_img_train, X_meta_train],\n    y=y_train,\n    validation_data=([X_img_val, X_meta_val], y_val),\n    epochs=10,\n    batch_size=32,\n    callbacks=callbacks,\n    verbose=1\n)\n\n# Evaluate\ny_pred = model.predict([X_img_test, X_meta_test])\ny_pred_classes = np.argmax(y_pred, axis=1)\ny_true_classes = np.argmax(y_test, axis=1)\n\nprint(\"Test Accuracy:\", accuracy_score(y_true_classes, y_pred_classes))\nprint(\"Classification Report:\\n\", classification_report(y_true_classes, y_pred_classes))\n\n# Confusion Matrix\nplt.figure(figsize=(6, 4))\nplt.imshow(confusion_matrix(y_true_classes, y_pred_classes), cmap='Blues', interpolation='nearest')\nplt.colorbar()\nplt.title(\"Confusion Matrix - Test Set\")\nplt.xlabel(\"Predicted\")\nplt.ylabel(\"True\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T16:09:51.689688Z","iopub.execute_input":"2025-04-29T16:09:51.689948Z","iopub.status.idle":"2025-04-29T16:15:44.197110Z","shell.execute_reply.started":"2025-04-29T16:09:51.689929Z","shell.execute_reply":"2025-04-29T16:15:44.196493Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}