{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport glob\nfrom tqdm import tqdm\ntqdm.pandas()\npd.options.display.max_colwidth = 10000\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-29T07:59:27.270704Z","iopub.execute_input":"2024-05-29T07:59:27.271147Z","iopub.status.idle":"2024-05-29T07:59:28.39899Z","shell.execute_reply.started":"2024-05-29T07:59:27.271115Z","shell.execute_reply":"2024-05-29T07:59:28.397841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-05-29T07:59:28.400711Z","iopub.execute_input":"2024-05-29T07:59:28.401132Z","iopub.status.idle":"2024-05-29T07:59:28.645326Z","shell.execute_reply.started":"2024-05-29T07:59:28.401104Z","shell.execute_reply":"2024-05-29T07:59:28.644297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(1)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-29T07:59:28.64666Z","iopub.execute_input":"2024-05-29T07:59:28.647081Z","iopub.status.idle":"2024-05-29T07:59:28.670003Z","shell.execute_reply.started":"2024-05-29T07:59:28.647046Z","shell.execute_reply":"2024-05-29T07:59:28.668966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def parquet_to_numpy(parquet_path):\n    # Read the Parquet file into a DataFrame\n    spec_df = pd.read_parquet(parquet_path)\n    \n    # Process the DataFrame to convert it into a numpy array\n    spec_array = spec_df.fillna(0).values[:, 1:].T  # fill NaN values with 0, transpose for (Time, Freq) -> (Freq, Time)\n    spec_array = spec_array.astype(\"float32\")\n    \n    return spec_array","metadata":{"execution":{"iopub.status.busy":"2024-05-29T07:59:28.672428Z","iopub.execute_input":"2024-05-29T07:59:28.672777Z","iopub.status.idle":"2024-05-29T07:59:28.678111Z","shell.execute_reply.started":"2024-05-29T07:59:28.672749Z","shell.execute_reply":"2024-05-29T07:59:28.677048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\ndef preprocess_spectrogram(image):\n     # Normalization: Ensures that the pixel values are within a certain range\n    # This helps in stabilizing the training process and ensures faster convergence\n    image_array = image.astype('float32')\n    image_array -= np.min(image_array)\n    image_array /= np.max(image_array) + 1e-4\n    \n    # Log Transformation: Enhances contrast and reduces the effect of outliers\n    # It helps in better visualization of the spectrogram features\n    image_array = np.log(image_array + 1e-4)\n    \n    # Mean Subtraction: Centers the data around zero\n    # This helps in reducing bias and improving the stability of the model\n    mean = np.mean(image_array)\n    image_array -= mean\n    \n    # Standardization: Scales the data to have zero mean and unit variance\n    # It ensures that all features are on a similar scale, which can improve model performance\n    std = np.std(image_array)\n    image_array /= std + 1e-6\n    return image_array\n\n# Generate a random image array with shape (400, 300)\nimage1 = np.random.rand(400, 300)\n\n# Check the shape of the image_array\nif image1.shape != (400, 300):\n    raise ValueError(f\"Invalid shape {image1.shape} for image_array\")\n\npreprocessed_images = []\nfor row in image1:\n    preprocessed_image = preprocess_spectrogram(row.reshape(1, -1))\n    preprocessed_images.append(preprocessed_image)\n    \npreprocessed_images = np.vstack(preprocessed_images)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-29T07:59:28.679827Z","iopub.execute_input":"2024-05-29T07:59:28.680534Z","iopub.status.idle":"2024-05-29T07:59:28.731422Z","shell.execute_reply.started":"2024-05-29T07:59:28.680475Z","shell.execute_reply":"2024-05-29T07:59:28.730439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess_image(image):\n    # Normalization: Ensures that the pixel values are within a certain range\n    # This helps in stabilizing the training process and ensures faster convergence\n    image_array = image.astype('float32')\n    image_array -= np.min(image_array)\n    image_array /= np.max(image_array) + 1e-4\n\n    # Mean Subtraction: Centers the data around zero\n    # This helps in reducing bias and improving the stability of the model\n    mean = np.mean(image_array)\n    image_array -= mean\n\n    # Standardization: Scales the data to have zero mean and unit variance\n    # It ensures that all features are on a similar scale, which can improve model performance\n    std = np.std(image_array)\n    image_array /= std + 1e-6\n    return image_array\n\n# Generate a random image array with shape (400, 300)\nimage1 = np.random.rand(400, 300)\n\n# Check the shape of the image_array\nif image1.shape != (400, 300):\n    raise ValueError(f\"Invalid shape {image1.shape} for image_array\")\n\npreprocessed_images = []\nfor row in image1:\n    preprocessed_image = preprocess_image(row.reshape(1, -1))\n    preprocessed_images.append(preprocessed_image)\n\npreprocessed_images = np.vstack(preprocessed_images)\n\n# Split preprocessed_images into train and test sets\nnum_train_samples = int(0.8 * preprocessed_images.shape[0])  # Assuming 80% train and 20% test split\n\ntrain_images = preprocessed_images[:num_train_samples]\ntest_images = preprocessed_images[num_train_samples:]\n\n\n# Reshape the input tensor to have a shape of (batch_size, timesteps, input_size)\ntrain_images = train_images.reshape((train_images.shape[0], 1, train_images.shape[1]))\ntest_images = test_images.reshape((test_images.shape[0], 1, test_images.shape[1]))\n\n# Normalize the pixel values to be between 0 and 1\ntrain_images = train_images / 255.0\ntest_images = test_images / 255.0\n","metadata":{"execution":{"iopub.status.busy":"2024-05-29T07:59:28.732662Z","iopub.execute_input":"2024-05-29T07:59:28.732975Z","iopub.status.idle":"2024-05-29T07:59:28.779119Z","shell.execute_reply.started":"2024-05-29T07:59:28.732949Z","shell.execute_reply":"2024-05-29T07:59:28.777922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import GRU, Dense, InputLayer\n\n# Example input data: assuming train_images has shape (num_samples, 784)\ntrain_images = np.random.rand(60000, 784)  # Example data\ntrain_labels = np.random.rand(60000, 10)   # Example labels (one-hot encoded)\n\n# Reshape input data to (num_samples, timesteps, features)\ntrain_images = train_images.reshape(train_images.shape[0], 1, 784)\n\n# Define the model architecture\nmodel = Sequential()\nmodel.add(InputLayer(input_shape=(1, 784)))  # Adjust input shape to (timesteps, features)\nmodel.add(GRU(units=128))  # GRU layer\nmodel.add(Dense(units=64, activation='relu'))  # Dense layer\nmodel.add(Dense(units=10, activation='softmax'))  # Output layer\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n\n# Train the model\nmodel.fit(train_images, train_labels, epochs=20, batch_size=32)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-29T08:08:18.959352Z","iopub.execute_input":"2024-05-29T08:08:18.960216Z","iopub.status.idle":"2024-05-29T08:11:23.70211Z","shell.execute_reply.started":"2024-05-29T08:08:18.960173Z","shell.execute_reply":"2024-05-29T08:11:23.700897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def scheduler(epoch, lr):\n      if epoch < 8:\n        return lr\n      else:\n        return lr * tf.math.exp(-0.1)","metadata":{"execution":{"iopub.status.busy":"2024-05-29T07:59:43.38881Z","iopub.status.idle":"2024-05-29T07:59:43.389359Z","shell.execute_reply.started":"2024-05-29T07:59:43.389106Z","shell.execute_reply":"2024-05-29T07:59:43.389127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vote_cols = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n\ndef confidence(r):\n    \n    vote_vals = r[vote_cols].values\n    \n    confidence = max(vote_vals)/sum(vote_vals) * 100\n    \n    return confidence\n\ntrain_df['expert_confidence'] = train_df.apply(confidence, axis=1)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-29T07:59:43.390681Z","iopub.status.idle":"2024-05-29T07:59:43.39116Z","shell.execute_reply.started":"2024-05-29T07:59:43.390919Z","shell.execute_reply":"2024-05-29T07:59:43.390937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import LSTM, Dense, Dropout\nfrom tensorflow.keras.callbacks import TensorBoard","metadata":{"execution":{"iopub.status.busy":"2024-05-29T07:59:43.392542Z","iopub.status.idle":"2024-05-29T07:59:43.392995Z","shell.execute_reply.started":"2024-05-29T07:59:43.392765Z","shell.execute_reply":"2024-05-29T07:59:43.392782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split data into features (X) and labels (y)\nX = train_df.drop(columns=['eeg_id', 'eeg_sub_id', 'spectrogram_id', 'spectrogram_sub_id', 'label_id', 'patient_id', 'expert_consensus'])\ny = train_df[['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']]\n","metadata":{"execution":{"iopub.status.busy":"2024-05-29T07:59:43.394331Z","iopub.status.idle":"2024-05-29T07:59:43.394835Z","shell.execute_reply.started":"2024-05-29T07:59:43.394585Z","shell.execute_reply":"2024-05-29T07:59:43.394604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-05-29T07:59:43.397923Z","iopub.status.idle":"2024-05-29T07:59:43.398881Z","shell.execute_reply.started":"2024-05-29T07:59:43.398612Z","shell.execute_reply":"2024-05-29T07:59:43.398635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_val_scaled = scaler.transform(X_val)","metadata":{"execution":{"iopub.status.busy":"2024-05-29T07:59:43.400376Z","iopub.status.idle":"2024-05-29T07:59:43.401422Z","shell.execute_reply.started":"2024-05-29T07:59:43.40116Z","shell.execute_reply":"2024-05-29T07:59:43.401183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_reshaped = X_train_scaled.reshape((X_train_scaled.shape[0], X_train_scaled.shape[1], 1))\nX_val_reshaped = X_val_scaled.reshape((X_val_scaled.shape[0], X_val_scaled.shape[1], 1))","metadata":{"execution":{"iopub.status.busy":"2024-05-29T07:59:43.402723Z","iopub.status.idle":"2024-05-29T07:59:43.403645Z","shell.execute_reply.started":"2024-05-29T07:59:43.403358Z","shell.execute_reply":"2024-05-29T07:59:43.40338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\nfrom sklearn.tree import DecisionTreeClassifier\n\nfrom sklearn.ensemble import RandomForestClassifier\n\nfrom sklearn.naive_bayes import GaussianNB\n\nfrom sklearn.neighbors import KNeighborsClassifier","metadata":{"execution":{"iopub.status.busy":"2024-05-29T07:59:43.405127Z","iopub.status.idle":"2024-05-29T07:59:43.40584Z","shell.execute_reply.started":"2024-05-29T07:59:43.40554Z","shell.execute_reply":"2024-05-29T07:59:43.405563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, confusion_matrix, f1_score\n\nfrom sklearn.metrics import classification_report\n\nfrom sklearn.model_selection import KFold\n\nfrom sklearn.model_selection import cross_validate, cross_val_score\n\nfrom sklearn.svm import SVC\n\nfrom sklearn import metrics","metadata":{"execution":{"iopub.status.busy":"2024-05-29T07:59:43.407277Z","iopub.status.idle":"2024-05-29T07:59:43.407774Z","shell.execute_reply.started":"2024-05-29T07:59:43.407527Z","shell.execute_reply":"2024-05-29T07:59:43.407548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\n\n\nclass RNNModel(nn.Module):\n    def __init__(\n        self,\n        input_dim=4,\n        lstm_dim=256,\n        dense_dim=256,\n        logit_dim=256,\n        num_classes=1,\n    ):\n        super().__init__()\n\n        self.mlp = nn.Sequential(\n            nn.Linear(input_dim, dense_dim // 2),\n            nn.ReLU(),\n            nn.Linear(dense_dim // 2, dense_dim),\n            nn.ReLU(),\n        )\n\n        self.lstm = nn.LSTM(dense_dim, lstm_dim, batch_first=True, bidirectional=True)\n\n        self.logits = nn.Sequential(\n            nn.Linear(lstm_dim * 2, logit_dim),\n            nn.ReLU(),\n            nn.Linear(logit_dim, num_classes),\n        )\n\n    def forward(self, x):\n        features = self.mlp(x)\n        features, _ = self.lstm(features)\n        pred = self.logits(features)\n        return pred","metadata":{"execution":{"iopub.status.busy":"2024-05-29T07:59:43.409426Z","iopub.status.idle":"2024-05-29T07:59:43.409928Z","shell.execute_reply.started":"2024-05-29T07:59:43.40968Z","shell.execute_reply":"2024-05-29T07:59:43.4097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport numpy as np\n\ndef compute_metric(df, preds):\n    \n    y = np.array(df['patient_id'].values.tolist())\n    w = np.array(df['expert_consensus'].values.tolist())\n    \n    assert y.shape == preds.shape and w.shape == y.shape, (y.shape, preds.shape, w.shape)\n    \n    mae = w * np.abs(y - preds)\n    mae = mae.sum() / w.sum()\n    \n    return mae\n\n\nclass VentilatorLoss(nn.Module):\n   \n    def __call__(self, preds, y, w):\n        mae = w * (y - preds).abs()\n        mae = mae.sum(-1) / w.sum(-1)\n\n        return mae","metadata":{"execution":{"iopub.status.busy":"2024-05-29T08:59:50.968996Z","iopub.execute_input":"2024-05-29T08:59:50.969386Z","iopub.status.idle":"2024-05-29T08:59:54.803277Z","shell.execute_reply.started":"2024-05-29T08:59:50.96936Z","shell.execute_reply":"2024-05-29T08:59:54.802131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}