{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30635,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\nimport seaborn as sns\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.preprocessing import LabelEncoder\nfrom keras.models import Sequential\nfrom keras.layers import BatchNormalization\nfrom keras.regularizers import l2\nfrom keras.regularizers import l1\nfrom keras.layers import Dropout\nfrom keras.layers import Dense\nfrom keras.optimizers import Adam\nfrom sklearn.metrics import classification_report, confusion_matrix\nfrom sklearn.metrics import confusion_matrix, accuracy_score\nfrom keras.models import load_model","metadata":{"execution":{"iopub.status.busy":"2024-01-17T23:28:14.192744Z","iopub.execute_input":"2024-01-17T23:28:14.193218Z","iopub.status.idle":"2024-01-17T23:28:14.202873Z","shell.execute_reply.started":"2024-01-17T23:28:14.193180Z","shell.execute_reply":"2024-01-17T23:28:14.201313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the data\ndata = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')","metadata":{"execution":{"iopub.status.busy":"2024-01-17T22:38:24.952311Z","iopub.execute_input":"2024-01-17T22:38:24.953239Z","iopub.status.idle":"2024-01-17T22:38:25.278330Z","shell.execute_reply.started":"2024-01-17T22:38:24.953194Z","shell.execute_reply":"2024-01-17T22:38:25.277063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(data.head())","metadata":{"execution":{"iopub.status.busy":"2024-01-17T22:39:36.969768Z","iopub.execute_input":"2024-01-17T22:39:36.970360Z","iopub.status.idle":"2024-01-17T22:39:36.983631Z","shell.execute_reply.started":"2024-01-17T22:39:36.970315Z","shell.execute_reply":"2024-01-17T22:39:36.982601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(data.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2024-01-17T22:42:24.346651Z","iopub.execute_input":"2024-01-17T22:42:24.347401Z","iopub.status.idle":"2024-01-17T22:42:24.373734Z","shell.execute_reply.started":"2024-01-17T22:42:24.347337Z","shell.execute_reply":"2024-01-17T22:42:24.371967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(data.dtypes)","metadata":{"execution":{"iopub.status.busy":"2024-01-17T22:42:54.970064Z","iopub.execute_input":"2024-01-17T22:42:54.970569Z","iopub.status.idle":"2024-01-17T22:42:54.979590Z","shell.execute_reply.started":"2024-01-17T22:42:54.970504Z","shell.execute_reply":"2024-01-17T22:42:54.977926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert 'expert_consensus' to a categorical type\ndata['expert_consensus'] = data['expert_consensus'].astype('category')\n\n# Assign encoded variable back to data['expert_consensus']\ndata['expert_consensus'] = data['expert_consensus'].cat.codes","metadata":{"execution":{"iopub.status.busy":"2024-01-17T22:44:25.865879Z","iopub.execute_input":"2024-01-17T22:44:25.866367Z","iopub.status.idle":"2024-01-17T22:44:25.885685Z","shell.execute_reply.started":"2024-01-17T22:44:25.866326Z","shell.execute_reply":"2024-01-17T22:44:25.884232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(data.dtypes)","metadata":{"execution":{"iopub.status.busy":"2024-01-17T22:45:17.403425Z","iopub.execute_input":"2024-01-17T22:45:17.404655Z","iopub.status.idle":"2024-01-17T22:45:17.414158Z","shell.execute_reply.started":"2024-01-17T22:45:17.404601Z","shell.execute_reply":"2024-01-17T22:45:17.412669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert 'expert_consensus' to int64\ndata['expert_consensus'] = data['expert_consensus'].astype('int64')","metadata":{"execution":{"iopub.status.busy":"2024-01-17T22:46:47.645433Z","iopub.execute_input":"2024-01-17T22:46:47.646137Z","iopub.status.idle":"2024-01-17T22:46:47.655768Z","shell.execute_reply.started":"2024-01-17T22:46:47.646089Z","shell.execute_reply":"2024-01-17T22:46:47.654489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(data.dtypes)","metadata":{"execution":{"iopub.status.busy":"2024-01-17T22:46:55.937893Z","iopub.execute_input":"2024-01-17T22:46:55.938448Z","iopub.status.idle":"2024-01-17T22:46:55.947596Z","shell.execute_reply.started":"2024-01-17T22:46:55.938405Z","shell.execute_reply":"2024-01-17T22:46:55.946031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loop through all columns in the DataFrame\nfor column in data.columns:\n    # Check if the column is of a numerical data type\n    if data[column].dtype in ['int64', 'float64']:\n        # Create a histogram for the column\n        plt.hist(data[column], bins=10, edgecolor='black')\n        \n        # Add a title and labels\n        plt.title(f'Histogram of {column}')\n        plt.xlabel(column)\n        plt.ylabel('Frequency')\n        \n        # Show the plot\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-17T22:50:59.967691Z","iopub.execute_input":"2024-01-17T22:50:59.968222Z","iopub.status.idle":"2024-01-17T22:51:04.801939Z","shell.execute_reply.started":"2024-01-17T22:50:59.968182Z","shell.execute_reply":"2024-01-17T22:51:04.800658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prepare the data\nX = data.drop('expert_consensus', axis=1)\ny = data['expert_consensus']","metadata":{"execution":{"iopub.status.busy":"2024-01-17T23:36:35.162796Z","iopub.execute_input":"2024-01-17T23:36:35.164739Z","iopub.status.idle":"2024-01-17T23:36:35.177258Z","shell.execute_reply.started":"2024-01-17T23:36:35.164665Z","shell.execute_reply":"2024-01-17T23:36:35.175989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encode the labels\nencoder = LabelEncoder()\ny = encoder.fit_transform(y)","metadata":{"execution":{"iopub.status.busy":"2024-01-17T23:36:37.927805Z","iopub.execute_input":"2024-01-17T23:36:37.929098Z","iopub.status.idle":"2024-01-17T23:36:37.939640Z","shell.execute_reply.started":"2024-01-17T23:36:37.929046Z","shell.execute_reply":"2024-01-17T23:36:37.938121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split the data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-01-17T23:36:40.521035Z","iopub.execute_input":"2024-01-17T23:36:40.521491Z","iopub.status.idle":"2024-01-17T23:36:40.549832Z","shell.execute_reply.started":"2024-01-17T23:36:40.521456Z","shell.execute_reply":"2024-01-17T23:36:40.548278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print the shapes of the training and testing sets\nprint(\"X_train shape:\", X_train.shape)\nprint(\"y_train shape:\", y_train.shape)\nprint(\"X_test shape:\", X_test.shape)\nprint(\"y_test shape:\", y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-01-17T23:36:41.897677Z","iopub.execute_input":"2024-01-17T23:36:41.898208Z","iopub.status.idle":"2024-01-17T23:36:41.905910Z","shell.execute_reply.started":"2024-01-17T23:36:41.898166Z","shell.execute_reply":"2024-01-17T23:36:41.904764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential()\n\n# Input layer\nmodel.add(Dense(512, input_dim=X_train.shape[1], activation='relu', kernel_regularizer=l2(0.01)))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.5))\n\n# Hidden layer 1\nmodel.add(Dense(512, activation='relu', kernel_regularizer=l2(0.01)))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.5))\n\n# Hidden layer 2\nmodel.add(Dense(256, activation='relu', kernel_regularizer=l2(0.01)))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.5))\n\n# Hidden layer 3\nmodel.add(Dense(128, activation='relu', kernel_regularizer=l2(0.01)))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.5))\n\n# Output layer\nmodel.add(Dense(len(np.unique(y)), activation='softmax'))\n\n# Compile the model with Adam optimizer\nmodel.compile(loss='sparse_categorical_crossentropy', \n              optimizer=Adam(learning_rate=0.001), \n              metrics=['accuracy'])\n\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-17T23:36:45.841643Z","iopub.execute_input":"2024-01-17T23:36:45.842131Z","iopub.status.idle":"2024-01-17T23:36:46.170187Z","shell.execute_reply.started":"2024-01-17T23:36:45.842075Z","shell.execute_reply":"2024-01-17T23:36:46.168765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(X_train, y_train, epochs=30, batch_size=32, validation_data=(X_test, y_test))","metadata":{"execution":{"iopub.status.busy":"2024-01-17T23:37:03.103410Z","iopub.execute_input":"2024-01-17T23:37:03.103933Z","iopub.status.idle":"2024-01-17T23:54:21.720664Z","shell.execute_reply.started":"2024-01-17T23:37:03.103895Z","shell.execute_reply":"2024-01-17T23:54:21.718890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the model\nmodel.save('my_model1.h5')","metadata":{"execution":{"iopub.status.busy":"2024-01-17T23:54:55.976752Z","iopub.execute_input":"2024-01-17T23:54:55.977306Z","iopub.status.idle":"2024-01-17T23:54:56.064818Z","shell.execute_reply.started":"2024-01-17T23:54:55.977260Z","shell.execute_reply":"2024-01-17T23:54:56.063269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot training & validation accuracy values\nplt.figure(figsize=(12, 6))\nplt.plot(history.history['accuracy'])\nplt.plot(history.history['val_accuracy'])\nplt.title('Model accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.legend(['Train', 'Test'], loc='upper left')\nplt.show()\n\n# Plot training & validation loss values\nplt.figure(figsize=(12, 6))\nplt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.title('Model loss')\nplt.ylabel('Loss')\nplt.xlabel('Epoch')\nplt.legend(['Train', 'Test'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-17T23:55:17.176007Z","iopub.execute_input":"2024-01-17T23:55:17.176483Z","iopub.status.idle":"2024-01-17T23:55:17.959502Z","shell.execute_reply.started":"2024-01-17T23:55:17.176439Z","shell.execute_reply":"2024-01-17T23:55:17.957856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the model from disk\nmodel = load_model('/kaggle/working/my_model1.h5')","metadata":{"execution":{"iopub.status.busy":"2024-01-17T23:55:31.393951Z","iopub.execute_input":"2024-01-17T23:55:31.394439Z","iopub.status.idle":"2024-01-17T23:55:31.758106Z","shell.execute_reply.started":"2024-01-17T23:55:31.394398Z","shell.execute_reply":"2024-01-17T23:55:31.756581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predicting the Test set results\ny_pred = model.predict(X_test)\ny_pred = np.argmax(y_pred, axis=1)  # Convert probabilities to class labels","metadata":{"execution":{"iopub.status.busy":"2024-01-17T23:55:34.751589Z","iopub.execute_input":"2024-01-17T23:55:34.753026Z","iopub.status.idle":"2024-01-17T23:55:38.187016Z","shell.execute_reply.started":"2024-01-17T23:55:34.752964Z","shell.execute_reply":"2024-01-17T23:55:38.185376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Making the Confusion Matrix\ncm = confusion_matrix(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2024-01-17T23:55:51.801825Z","iopub.execute_input":"2024-01-17T23:55:51.803756Z","iopub.status.idle":"2024-01-17T23:55:51.816218Z","shell.execute_reply.started":"2024-01-17T23:55:51.803682Z","shell.execute_reply":"2024-01-17T23:55:51.814772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualizing the Confusion Matrix\nplt.figure(figsize=(10,7))\nsns.heatmap(cm, annot=True, fmt='d')\nplt.xlabel('Predicted')\nplt.ylabel('Truth')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-17T23:55:55.226894Z","iopub.execute_input":"2024-01-17T23:55:55.227362Z","iopub.status.idle":"2024-01-17T23:55:55.733956Z","shell.execute_reply.started":"2024-01-17T23:55:55.227322Z","shell.execute_reply":"2024-01-17T23:55:55.732588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test Accuracy\ntest_accuracy = accuracy_score(y_test, y_pred)\nprint(\"Test Accuracy: \", test_accuracy)","metadata":{"execution":{"iopub.status.busy":"2024-01-17T23:56:02.747433Z","iopub.execute_input":"2024-01-17T23:56:02.748910Z","iopub.status.idle":"2024-01-17T23:56:02.761012Z","shell.execute_reply.started":"2024-01-17T23:56:02.748849Z","shell.execute_reply":"2024-01-17T23:56:02.759338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predict the classes\ny_pred_probs = model.predict(X_test)\ny_pred_classes = np.argmax(y_pred_probs, axis=1)\n\n# Select 10 random indices\nrandom_indices = np.random.choice(range(len(y_test)), 10)\n\n# Map the class indices to the actual class labels\nclass_labels = {i: label for i, label in enumerate(encoder.classes_)}\n\n# Check actual class and predicted class for the randomly selected instances\nfor i in random_indices:\n    actual_class = class_labels[y_test[i]]\n    predicted_class = class_labels[y_pred_classes[i]]\n    print(f\"Instance {i+1}: Actual Class - {actual_class}, Predicted Class - {predicted_class}\")","metadata":{"execution":{"iopub.status.busy":"2024-01-17T23:56:05.808262Z","iopub.execute_input":"2024-01-17T23:56:05.808737Z","iopub.status.idle":"2024-01-17T23:56:08.799608Z","shell.execute_reply.started":"2024-01-17T23:56:05.808702Z","shell.execute_reply":"2024-01-17T23:56:08.798258Z"},"trusted":true},"execution_count":null,"outputs":[]}]}