{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30684,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport glob\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\ntqdm.pandas()\npd.options.display.max_colwidth = 10000\n\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-14T04:19:14.793722Z","iopub.execute_input":"2024-04-14T04:19:14.794042Z","iopub.status.idle":"2024-04-14T04:19:15.690570Z","shell.execute_reply.started":"2024-04-14T04:19:14.794016Z","shell.execute_reply":"2024-04-14T04:19:15.689772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Loading Train Data ","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:19:15.871454Z","iopub.execute_input":"2024-04-14T04:19:15.871749Z","iopub.status.idle":"2024-04-14T04:19:16.035470Z","shell.execute_reply.started":"2024-04-14T04:19:15.871724Z","shell.execute_reply":"2024-04-14T04:19:16.034582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Mapping train_eegs and train_spectogram  to  its Full path","metadata":{}},{"cell_type":"code","source":"def create_id_mapping(paths_list):\n    id_map = {}\n    for path in paths_list:\n        file_id = os.path.basename(path).split('.')[0]  # Corrected to extract the file ID correctly\n        id_map[file_id] = path\n    return id_map\n\ndef mapping_id(ids, id_map):\n    return id_map.get(ids)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:19:16.120138Z","iopub.execute_input":"2024-04-14T04:19:16.120597Z","iopub.status.idle":"2024-04-14T04:19:16.126229Z","shell.execute_reply.started":"2024-04-14T04:19:16.120560Z","shell.execute_reply":"2024-04-14T04:19:16.125397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create ID mappings for train_eeg_path_list and train_spectrograms_path_list\ntrain_eeg_path_list = glob.glob(\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/*\")\ntrain_spectrograms_path_list = glob.glob(\"/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/*\")\neeg_id_map = create_id_mapping(train_eeg_path_list)\nspectrograms_id_map = create_id_mapping(train_spectrograms_path_list)\n\n# Example usage:\nprint(\"Mapping train_eeg_path_list\",\"-\"*60)\ntrain_df['eeg_path'] = train_df['eeg_id'].astype(str).progress_apply(lambda x: mapping_id(x, eeg_id_map))\n\nprint(\"Mapping train_spectrograms_path_list\",\"-\"*60)\ntrain_df['spectrograms_path'] = train_df['spectrogram_id'].astype(str).progress_apply(lambda x: mapping_id(x, spectrograms_id_map))","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:19:17.638107Z","iopub.execute_input":"2024-04-14T04:19:17.638795Z","iopub.status.idle":"2024-04-14T04:19:18.300235Z","shell.execute_reply.started":"2024-04-14T04:19:17.638761Z","shell.execute_reply":"2024-04-14T04:19:18.299315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(1)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:19:18.301617Z","iopub.execute_input":"2024-04-14T04:19:18.301895Z","iopub.status.idle":"2024-04-14T04:19:18.320924Z","shell.execute_reply.started":"2024-04-14T04:19:18.301871Z","shell.execute_reply":"2024-04-14T04:19:18.319973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Mapping test_eegs and test_spectogram  to  its Full path","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/test.csv\")\n\n# Create ID mappings for train_eeg_path_list and train_spectrograms_path_list\ntest_eeg_path_list = glob.glob(\"/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/*\")\ntest_spectrograms_path_list = glob.glob(\"/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/*\")\neeg_id_map = create_id_mapping(test_eeg_path_list)\nspectrograms_id_map = create_id_mapping(test_spectrograms_path_list)\n\n# Example usage:\nprint(\"Mapping train_eeg_path_list\",\"-\"*60)\ntest_df['eeg_path'] = test_df['eeg_id'].astype(str).progress_apply(lambda x: mapping_id(x, eeg_id_map))\n\nprint(\"Mapping train_spectrograms_path_list\",\"-\"*60)\ntest_df['spectrograms_path'] = test_df['spectrogram_id'].astype(str).progress_apply(lambda x: mapping_id(x, spectrograms_id_map))","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:19:21.312937Z","iopub.execute_input":"2024-04-14T04:19:21.313760Z","iopub.status.idle":"2024-04-14T04:19:21.334020Z","shell.execute_reply.started":"2024-04-14T04:19:21.313729Z","shell.execute_reply":"2024-04-14T04:19:21.333118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head(1)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:19:22.839891Z","iopub.execute_input":"2024-04-14T04:19:22.840753Z","iopub.status.idle":"2024-04-14T04:19:22.850805Z","shell.execute_reply.started":"2024-04-14T04:19:22.840723Z","shell.execute_reply":"2024-04-14T04:19:22.849724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Now We Will Create A Function Which Will Convert The parquet file to .npy format i.e numpy format","metadata":{}},{"cell_type":"code","source":"def parquet_to_numpy(parquet_path):\n    # Read the Parquet file into a DataFrame\n    spec_df = pd.read_parquet(parquet_path)\n    \n    # Process the DataFrame to convert it into a numpy array\n    spec_array = spec_df.fillna(0).values[:, 1:].T  # fill NaN values with 0, transpose for (Time, Freq) -> (Freq, Time)\n    spec_array = spec_array.astype(\"float32\")\n    \n    return spec_array","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:19:23.519605Z","iopub.execute_input":"2024-04-14T04:19:23.520198Z","iopub.status.idle":"2024-04-14T04:19:23.527215Z","shell.execute_reply.started":"2024-04-14T04:19:23.520170Z","shell.execute_reply":"2024-04-14T04:19:23.526380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Preprocesing The Image To Get The Better Visualization of the Image\n[Taking Help To Preprocess The Image From This Notebook](https://www.kaggle.com/code/awsaf49/hms-hbac-kerascv-starter-notebook//)","metadata":{}},{"cell_type":"code","source":"def preprocess_spectrogram(image_array):\n\n    # Normalization: Ensures that the pixel values are within a certain range\n    # This helps in stabilizing the training process and ensures faster convergence\n    image_array = image_array.astype('float32')\n    image_array -= np.min(image_array)\n    image_array /= np.max(image_array) + 1e-4\n    \n    # Log Transformation: Enhances contrast and reduces the effect of outliers\n    # It helps in better visualization of the spectrogram features\n    image_array = np.log(image_array + 1e-4)\n    \n    # Mean Subtraction: Centers the data around zero\n    # This helps in reducing bias and improving the stability of the model\n    mean = np.mean(image_array)\n    image_array -= mean\n    \n    # Standardization: Scales the data to have zero mean and unit variance\n    # It ensures that all features are on a similar scale, which can improve model performance\n    std = np.std(image_array)\n    image_array /= std + 1e-6\n    \n    return image_array","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:19:24.723720Z","iopub.execute_input":"2024-04-14T04:19:24.724093Z","iopub.status.idle":"2024-04-14T04:19:24.730403Z","shell.execute_reply.started":"2024-04-14T04:19:24.724064Z","shell.execute_reply":"2024-04-14T04:19:24.729470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating Train And Test Data Directory","metadata":{}},{"cell_type":"code","source":"current_dir = os.getcwd()\ntrain_dir = os.path.join(current_dir, \"train\")\ntest_dir  = os.path.join(current_dir, \"test\")\nprint(\"train_dir--->\", train_dir)\nprint(\"test_dir---->\", test_dir)\nos.makedirs(train_dir, exist_ok = True)\nos.makedirs(test_dir, exist_ok = True)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:19:25.454440Z","iopub.execute_input":"2024-04-14T04:19:25.455253Z","iopub.status.idle":"2024-04-14T04:19:25.461475Z","shell.execute_reply.started":"2024-04-14T04:19:25.455218Z","shell.execute_reply":"2024-04-14T04:19:25.460455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Dumping Train Spectograms Data To Train Dir","metadata":{}},{"cell_type":"code","source":"from PIL import Image\n\nfor i in tqdm(train_df['spectrograms_path'].unique()):\n    img_array = preprocess_spectrogram(parquet_to_numpy(i))\n\n    img_name = os.path.basename(i).split('.')[0] + \".jpeg\"\n    spectrograms_path = os.path.join(train_dir, \"spectrograms\")\n    os.makedirs(spectrograms_path, exist_ok=True)\n    \n    plt.imsave(os.path.join(spectrograms_path, img_name), img_array)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:19:26.201120Z","iopub.execute_input":"2024-04-14T04:19:26.201838Z","iopub.status.idle":"2024-04-14T04:25:51.602653Z","shell.execute_reply.started":"2024-04-14T04:19:26.201807Z","shell.execute_reply":"2024-04-14T04:25:51.601641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Dumping Test Spectograms to the Test Dir","metadata":{}},{"cell_type":"code","source":"from PIL import Image\n\nfor i in tqdm(test_df['spectrograms_path'].unique()):\n    img_array = preprocess_spectrogram(parquet_to_numpy(i))\n\n    img_name = os.path.basename(i).split('.')[0] + \".jpeg\"\n    spectrograms_path = os.path.join(test_dir, \"spectrograms\")\n    os.makedirs(spectrograms_path, exist_ok=True)\n    \n    plt.imsave(os.path.join(spectrograms_path, img_name), img_array)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:25:51.604702Z","iopub.execute_input":"2024-04-14T04:25:51.605073Z","iopub.status.idle":"2024-04-14T04:25:51.646866Z","shell.execute_reply.started":"2024-04-14T04:25:51.605039Z","shell.execute_reply":"2024-04-14T04:25:51.646048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Renaming Train Image In Train Directory Adding Class To The Train Image","metadata":{}},{"cell_type":"code","source":"xx = train_df['spectrograms_path expert_consensus'.split()].drop_duplicates().reset_index(drop = True)\ndict_xx = dict(zip(xx['spectrograms_path'], xx['expert_consensus']))","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:25:51.648051Z","iopub.execute_input":"2024-04-14T04:25:51.648385Z","iopub.status.idle":"2024-04-14T04:25:51.694654Z","shell.execute_reply.started":"2024-04-14T04:25:51.648353Z","shell.execute_reply":"2024-04-14T04:25:51.693891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for original_path in tqdm(glob.glob(\"/kaggle/working/train/spectrograms/*\")):\n    # Extract the image ID from the file name\n    img_id = os.path.basename(original_path).split('.')[-2]\n    # Find the corresponding value in dict_xx based on the image ID\n    xx = [v for k, v in dict_xx.items() if img_id in k][0]\n    \n    # Construct the new file name with the same directory path\n    directory_path = os.path.dirname(original_path)\n    new_filename = f\"{xx}.{img_id}.jpeg\"\n    \n    # Construct the new file path\n    new_path = os.path.join(directory_path, new_filename)\n    \n    # Rename the file\n    os.rename(original_path, new_path)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:25:51.696472Z","iopub.execute_input":"2024-04-14T04:25:51.696744Z","iopub.status.idle":"2024-04-14T04:26:22.762453Z","shell.execute_reply.started":"2024-04-14T04:25:51.696721Z","shell.execute_reply":"2024-04-14T04:26:22.761530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del xx, dict_xx","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:26:22.764477Z","iopub.execute_input":"2024-04-14T04:26:22.764834Z","iopub.status.idle":"2024-04-14T04:26:22.770050Z","shell.execute_reply.started":"2024-04-14T04:26:22.764800Z","shell.execute_reply":"2024-04-14T04:26:22.769153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Creating A Function Which Will Plot The Random Image From The Directory","metadata":{}},{"cell_type":"code","source":"import os\nimport random\nimport matplotlib.pyplot as plt\nfrom PIL import Image\n\ndef plot_random_images(dir_path, num_images, folder_name):\n    if not os.path.exists(dir_path):\n        print(\"Directory does not exist.\")\n        return\n    \n    image_files = [f for f in os.listdir(dir_path) if f.endswith('.jpeg') or f.endswith('.png') or f.endswith('.jpg')]\n    if len(image_files) == 0:\n        print(\"No image files found in the directory.\")\n        return\n    \n    # Shuffle the list of image files\n    random.shuffle(image_files)\n    \n    # Limit the number of images to plot\n    num_images = min(num_images, len(image_files))\n    \n    # Calculate the number of rows and columns based on the aspect ratio of the images\n    num_rows = int(num_images ** 0.5)\n    num_cols = (num_images + num_rows - 1) // num_rows\n\n    fig, axes = plt.subplots(num_rows, num_cols, figsize=(15, 10))\n\n    for i in range(num_images):\n        img_path = os.path.join(dir_path, image_files[i])\n        img = Image.open(img_path)\n\n        row = i // num_cols\n        col = i % num_cols\n\n        ax = axes[row, col]\n        ax.imshow(img, cmap=\"gray\")\n        ax.set_xticks([])\n        ax.set_yticks([])\n        ax.set_xlabel(image_files[i], fontsize=8, wrap=True)\n\n    # Hide any remaining empty subplots\n    for i in range(num_images, num_rows * num_cols):\n        row = i // num_cols\n        col = i % num_cols\n        axes[row, col].axis('off')\n\n    plt.suptitle(folder_name, fontsize=16)\n    plt.subplots_adjust(wspace=0.2, hspace=0.2)  # Adjust spacing between subplots\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:26:22.771135Z","iopub.execute_input":"2024-04-14T04:26:22.771392Z","iopub.status.idle":"2024-04-14T04:26:22.783044Z","shell.execute_reply.started":"2024-04-14T04:26:22.771369Z","shell.execute_reply":"2024-04-14T04:26:22.782185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_dir_spectrograms = '/kaggle/working/train/spectrograms'\nplot_random_images(training_dir_spectrograms, 5, \"Train_Images\")","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:26:22.784276Z","iopub.execute_input":"2024-04-14T04:26:22.784998Z","iopub.status.idle":"2024-04-14T04:26:23.481685Z","shell.execute_reply.started":"2024-04-14T04:26:22.784966Z","shell.execute_reply":"2024-04-14T04:26:23.480744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" ### Dumping of the data is completed","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"#### Now We Will Create A Train Data Frame Which Will Contain  Training Image Path And Their Class Along With Their EEG_ID","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### This will Be Our Final Train Data Frame","metadata":{}},{"cell_type":"code","source":"def create_id_mapping(paths_list):\n    id_map = {}\n    for path in paths_list:\n        file_id = os.path.basename(path).split('.')[1]  # Corrected to extract the file ID correctly\n        id_map[file_id] = path\n    return id_map\n\ndef mapping_id(ids, id_map):\n    return id_map.get(ids)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:26:23.483173Z","iopub.execute_input":"2024-04-14T04:26:23.483538Z","iopub.status.idle":"2024-04-14T04:26:23.489969Z","shell.execute_reply.started":"2024-04-14T04:26:23.483503Z","shell.execute_reply":"2024-04-14T04:26:23.488961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create ID mappings for train_eeg_path_list and train_spectrograms_path_list\ntrain_spectrograms_path_list = glob.glob(\"/kaggle/working/train/spectrograms/*\")\nspectrograms_id_map = create_id_mapping(train_spectrograms_path_list)\n\ntrain_df = train_df['eeg_id spectrogram_id expert_consensus'.split()].drop_duplicates().reset_index(drop = True)\n\ntrain_df['spectrograms_path'] = train_df['spectrogram_id'].astype(str).progress_apply(lambda x: mapping_id(x, spectrograms_id_map))","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:26:23.491100Z","iopub.execute_input":"2024-04-14T04:26:23.491382Z","iopub.status.idle":"2024-04-14T04:26:23.625007Z","shell.execute_reply.started":"2024-04-14T04:26:23.491354Z","shell.execute_reply":"2024-04-14T04:26:23.624121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Creating Train And Validation Split","metadata":{}},{"cell_type":"code","source":"train_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:26:23.628234Z","iopub.execute_input":"2024-04-14T04:26:23.628510Z","iopub.status.idle":"2024-04-14T04:26:23.637712Z","shell.execute_reply.started":"2024-04-14T04:26:23.628486Z","shell.execute_reply":"2024-04-14T04:26:23.636767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrain_dataset, valid_dataset = train_test_split(train_df ,test_size = 0.3 , random_state = 42, shuffle = True,\n                                               stratify = train_df['expert_consensus'])","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:26:23.638786Z","iopub.execute_input":"2024-04-14T04:26:23.639127Z","iopub.status.idle":"2024-04-14T04:26:23.731693Z","shell.execute_reply.started":"2024-04-14T04:26:23.639103Z","shell.execute_reply":"2024-04-14T04:26:23.730798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training Set Validation Set","metadata":{}},{"cell_type":"code","source":"train_dataset.head(1)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:26:23.732862Z","iopub.execute_input":"2024-04-14T04:26:23.733371Z","iopub.status.idle":"2024-04-14T04:26:23.743095Z","shell.execute_reply.started":"2024-04-14T04:26:23.733344Z","shell.execute_reply":"2024-04-14T04:26:23.742100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Image Data Generator","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n# Define data generators\ntrain_datagen = ImageDataGenerator()\ntest_datagen = ImageDataGenerator()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:26:23.744177Z","iopub.execute_input":"2024-04-14T04:26:23.744484Z","iopub.status.idle":"2024-04-14T04:26:26.918693Z","shell.execute_reply.started":"2024-04-14T04:26:23.744459Z","shell.execute_reply":"2024-04-14T04:26:26.917826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training Set Validation Set","metadata":{}},{"cell_type":"code","source":"training_set = train_datagen.flow_from_dataframe(\n                        dataframe = train_dataset,\n                        x_col = 'spectrograms_path',\n                        y_col = 'expert_consensus',\n                        target_size=(299,299),\n                        color_mode='rgb',\n                        class_mode = 'categorical',\n                        batch_size= 64)\n\nvalidation_set = test_datagen.flow_from_dataframe(\n                dataframe = valid_dataset,\n                x_col = 'spectrograms_path',\n                y_col = 'expert_consensus',\n                color_mode='rgb',\n                target_size = (299, 299),\n                class_mode = 'categorical',\n                batch_size = 64)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:26:26.920012Z","iopub.execute_input":"2024-04-14T04:26:26.920770Z","iopub.status.idle":"2024-04-14T04:26:27.138797Z","shell.execute_reply.started":"2024-04-14T04:26:26.920736Z","shell.execute_reply":"2024-04-14T04:26:27.137898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Importing Multiple Transffer Learning Technique ","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.layers import Input, Concatenate, Flatten, Dropout, Dense\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.applications import ResNet50V2, InceptionResNetV2\n\ndef create_combined_model(input_shape=(299, 299, 3)):\n    # Input layer\n    input_tensor = Input(shape=input_shape)\n\n    # Initialize lists to store outputs of each model\n    outputs = []\n\n    # Loop through each base model and perform transfer learning\n    for base_model in [ResNet50V2, InceptionResNetV2]:\n        # Load pre-trained model without top layers\n        base_model = base_model(weights='imagenet', include_top=False, input_tensor=input_tensor)\n        # Freeze layers\n        for layer in base_model.layers[:90]:\n            layer.trainable = False\n        # Get output of the base model\n        output = base_model.output\n        # Flatten the output\n        output = Flatten()(output)\n        # Append to outputs list\n        outputs.append(output)\n\n    # Concatenate outputs vertically\n    concatenated_output = Concatenate(axis=1)(outputs)\n\n    # Apply dropout\n    output = Dropout(0.8)(concatenated_output)\n\n    # Output layer\n    output = Dense(6, activation='softmax')(output)\n\n    # Combine models\n    combine_model = Model(inputs=[input_tensor], outputs=output)\n\n    return combine_model","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:26:27.139940Z","iopub.execute_input":"2024-04-14T04:26:27.140204Z","iopub.status.idle":"2024-04-14T04:26:27.151779Z","shell.execute_reply.started":"2024-04-14T04:26:27.140180Z","shell.execute_reply":"2024-04-14T04:26:27.150867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create and compile the model\ncombine_model = create_combined_model()\ncombine_model.compile(optimizer='adam', loss='categorical_crossentropy', \n                      metrics=['accuracy'])\nprint(\"Total number of layers in the model:---> \", len(combine_model.layers))","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:26:27.153063Z","iopub.execute_input":"2024-04-14T04:26:27.153771Z","iopub.status.idle":"2024-04-14T04:26:33.707471Z","shell.execute_reply.started":"2024-04-14T04:26:27.153740Z","shell.execute_reply":"2024-04-14T04:26:33.706568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Defining CallBack Function","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.callbacks import ReduceLROnPlateau, EarlyStopping\n\n#learning_rate reduce module\nlr_reduce = ReduceLROnPlateau('val_loss', patience=3, \n                                              factor=0.5, min_lr=1e-6)\n\n# Stop early if model doesn't improve after n epochs\nearly_stopper = EarlyStopping(monitor='val_loss', patience=4,\n                              verbose=0, restore_best_weights=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:26:33.708711Z","iopub.execute_input":"2024-04-14T04:26:33.709088Z","iopub.status.idle":"2024-04-14T04:26:33.714932Z","shell.execute_reply.started":"2024-04-14T04:26:33.709054Z","shell.execute_reply":"2024-04-14T04:26:33.714137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Fitting The Model","metadata":{}},{"cell_type":"code","source":"history = combine_model.fit(training_set,\n                         batch_size = 512,\n                         epochs=30,\n                         validation_data=validation_set,\n                         callbacks = [lr_reduce, early_stopper],\n                         verbose=1, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:26:33.716151Z","iopub.execute_input":"2024-04-14T04:26:33.716425Z","iopub.status.idle":"2024-04-14T04:41:08.887084Z","shell.execute_reply.started":"2024-04-14T04:26:33.716402Z","shell.execute_reply":"2024-04-14T04:41:08.886183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Plot Accuracy Loss","metadata":{}},{"cell_type":"code","source":"# Create subplots with 1 row and 2 columns\nfig, axs = plt.subplots(1, 2, figsize=(15, 5))\n\n# Plot training and validation accuracy values\naxs[0].plot(history.history['accuracy'], label='Training Accuracy')\naxs[0].plot(history.history['val_accuracy'], label='Validation Accuracy')\naxs[0].set_title('Training and Validation Accuracy')\naxs[0].set_xlabel('Epoch')\naxs[0].set_ylabel('Accuracy')\naxs[0].legend()\n\n# Plot training and validation loss values\naxs[1].plot(history.history['loss'], label='Training Loss')\naxs[1].plot(history.history['val_loss'], label='Validation Loss')\naxs[1].set_title('Training and Validation Loss')\naxs[1].set_xlabel('Epoch')\naxs[1].set_ylabel('Loss')\naxs[1].legend()\n\n# Show the plots\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:41:08.888378Z","iopub.execute_input":"2024-04-14T04:41:08.888640Z","iopub.status.idle":"2024-04-14T04:41:09.509832Z","shell.execute_reply.started":"2024-04-14T04:41:08.888616Z","shell.execute_reply":"2024-04-14T04:41:09.508932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Create A Model Prediction Function ","metadata":{}},{"cell_type":"code","source":"from PIL import Image\nfrom IPython.display import display\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport cv2","metadata":{"execution":{"iopub.status.busy":"2024-04-14T05:03:26.794039Z","iopub.execute_input":"2024-04-14T05:03:26.794402Z","iopub.status.idle":"2024-04-14T05:03:26.961192Z","shell.execute_reply.started":"2024-04-14T05:03:26.794371Z","shell.execute_reply":"2024-04-14T05:03:26.960370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Class Indeces Are :-->\", validation_set.class_indices)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:41:09.517511Z","iopub.execute_input":"2024-04-14T04:41:09.517853Z","iopub.status.idle":"2024-04-14T04:41:09.530374Z","shell.execute_reply.started":"2024-04-14T04:41:09.517828Z","shell.execute_reply":"2024-04-14T04:41:09.529488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_dataset.head(1)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:41:09.531568Z","iopub.execute_input":"2024-04-14T04:41:09.531900Z","iopub.status.idle":"2024-04-14T04:41:09.545359Z","shell.execute_reply.started":"2024-04-14T04:41:09.531870Z","shell.execute_reply":"2024-04-14T04:41:09.544550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Model Prediction Function\n\ndef model_prediction(img_path, model, target_size=(299, 299)):\n    # Load and preprocess the image\n    img = Image.open(img_path)\n    img_resized = img.resize(target_size)  # Resize the image to the input size of the model\n    \n    # Expand the dimensions of the image array to match the input shape expected by the model\n    img_expanded = np.expand_dims(img_resized, axis=0)\n    \n    # Make prediction using the model\n    predictions = model.predict(img_expanded, verbose=False)\n    predicted_class_index = np.argmax(predictions)\n    \n    # Map the predicted class index to its corresponding label\n    predicted_label = next((class_label for class_label, index in validation_set.class_indices.items() if index == predicted_class_index), None)\n    \n    return predicted_label","metadata":{"execution":{"iopub.status.busy":"2024-04-14T05:02:32.779335Z","iopub.execute_input":"2024-04-14T05:02:32.780000Z","iopub.status.idle":"2024-04-14T05:02:32.786874Z","shell.execute_reply.started":"2024-04-14T05:02:32.779966Z","shell.execute_reply":"2024-04-14T05:02:32.785831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_path = '/kaggle/working/train/spectrograms/GPD.1023320297.jpeg'\nmodel_prediction(image_path , combine_model)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T05:02:34.353761Z","iopub.execute_input":"2024-04-14T05:02:34.354581Z","iopub.status.idle":"2024-04-14T05:02:49.745409Z","shell.execute_reply.started":"2024-04-14T05:02:34.354549Z","shell.execute_reply":"2024-04-14T05:02:49.744384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Top 10 Prediction On The Validation Data","metadata":{}},{"cell_type":"code","source":"valid_test =  valid_dataset['spectrograms_path expert_consensus'.split()].drop_duplicates().reset_index(drop = True).sample(10)\nvalid_test['ModelPrediction'] = valid_test['spectrograms_path'].progress_apply(lambda x: model_prediction(x, combine_model))","metadata":{"execution":{"iopub.status.busy":"2024-04-14T05:02:56.213077Z","iopub.execute_input":"2024-04-14T05:02:56.213929Z","iopub.status.idle":"2024-04-14T05:02:57.323692Z","shell.execute_reply.started":"2024-04-14T05:02:56.213876Z","shell.execute_reply":"2024-04-14T05:02:57.322696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Plotting Actual/Model Prediction ","metadata":{}},{"cell_type":"code","source":"def plot_predict(img_path, model):\n    # Load the image\n    img_array = cv2.imread(img_path)\n\n    # Display the image\n    plt.imshow(img_array)\n    \n    # Make prediction using the model\n    prediction = model_prediction(img_path, model)\n    \n    # Set the title and x label\n    plt.title(\"Prediction: \" + prediction, color='red')\n    plt.xlabel(\"Actual Image: \" + img_path.split('/')[-1].split('.')[0])\n\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T05:03:36.241027Z","iopub.execute_input":"2024-04-14T05:03:36.241397Z","iopub.status.idle":"2024-04-14T05:03:36.247957Z","shell.execute_reply.started":"2024-04-14T05:03:36.241369Z","shell.execute_reply":"2024-04-14T05:03:36.246976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_dataset.head(1)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T05:03:38.211228Z","iopub.execute_input":"2024-04-14T05:03:38.211598Z","iopub.status.idle":"2024-04-14T05:03:38.221300Z","shell.execute_reply.started":"2024-04-14T05:03:38.211567Z","shell.execute_reply":"2024-04-14T05:03:38.220337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_path  = '/kaggle/working/train/spectrograms/GPD.1023320297.jpeg'\nplot_predict(img_path, model= combine_model)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T05:03:39.116878Z","iopub.execute_input":"2024-04-14T05:03:39.117255Z","iopub.status.idle":"2024-04-14T05:03:39.585742Z","shell.execute_reply.started":"2024-04-14T05:03:39.117225Z","shell.execute_reply":"2024-04-14T05:03:39.584763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Now We Will Do The Prediction On The Test Data ","metadata":{}},{"cell_type":"code","source":"test_dir = '/kaggle/working/test'\ntest_prediction = pd.DataFrame(glob.glob(test_dir+'/spectrograms/*'),columns = ['img_path'])\ntest_prediction['ModelPrediction'] = test_prediction['img_path'].progress_apply(lambda x: model_prediction(x, model= combine_model))\n","metadata":{"execution":{"iopub.status.busy":"2024-04-14T05:03:49.280018Z","iopub.execute_input":"2024-04-14T05:03:49.280388Z","iopub.status.idle":"2024-04-14T05:03:49.398020Z","shell.execute_reply.started":"2024-04-14T05:03:49.280361Z","shell.execute_reply":"2024-04-14T05:03:49.397053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_prediction","metadata":{"execution":{"iopub.status.busy":"2024-04-14T05:03:51.520072Z","iopub.execute_input":"2024-04-14T05:03:51.520458Z","iopub.status.idle":"2024-04-14T05:03:51.529853Z","shell.execute_reply.started":"2024-04-14T05:03:51.520426Z","shell.execute_reply":"2024-04-14T05:03:51.528957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 📩 | Submission¶\n","metadata":{}},{"cell_type":"code","source":"def model_probability(img_path, model, target_size=(299, 299)):\n    # Load and preprocess the image\n    img = Image.open(img_path)\n    img = img.resize(target_size)  # Resize the image to the input size of the model\n    \n    # Expand the dimensions of the image array to match the input shape expected by the model\n    expand_dim = np.expand_dims(img, axis=0)\n    \n    # Make prediction using the model\n    predictions = model.predict(expand_dim, verbose=False)[0]\n    \n    # Get class labels and indices\n    class_indices = {v: k for k, v in validation_set.class_indices.items()}\n    \n    # Create dictionary with class labels as keys and model probabilities as values\n    probabilities = {class_indices[i]: prob for i, prob in enumerate(predictions)}\n    \n    return probabilities","metadata":{"execution":{"iopub.status.busy":"2024-04-14T05:04:16.186261Z","iopub.execute_input":"2024-04-14T05:04:16.187103Z","iopub.status.idle":"2024-04-14T05:04:16.194081Z","shell.execute_reply.started":"2024-04-14T05:04:16.187065Z","shell.execute_reply":"2024-04-14T05:04:16.192983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_path = '/kaggle/working/test/spectrograms/853520.jpeg'\nmodel_prob = model_probability(img_path, model= combine_model, target_size=(299, 299))\nmodel_prob","metadata":{"execution":{"iopub.status.busy":"2024-04-14T05:04:28.338481Z","iopub.execute_input":"2024-04-14T05:04:28.339101Z","iopub.status.idle":"2024-04-14T05:04:28.449743Z","shell.execute_reply.started":"2024-04-14T05:04:28.339068Z","shell.execute_reply":"2024-04-14T05:04:28.448838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a DataFrame from the model probabilities\nsubmission_df = pd.DataFrame([model_prob], columns=model_prob.keys())\n\n# Add 'eeg_id' column from test_df as the first column\nsubmission_df.insert(0, 'eeg_id', test_df['eeg_id'].copy())\n\n# Rename the columns\nsubmission_df.rename(columns={\n    'Seizure': 'seizure_vote',\n    'LPD': 'lpd_vote',\n    'GPD': 'gpd_vote',\n    'LRDA': 'lrda_vote',\n    'GRDA': 'grda_vote',\n    'Other': 'other_vote'\n}, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T05:04:35.771526Z","iopub.execute_input":"2024-04-14T05:04:35.771888Z","iopub.status.idle":"2024-04-14T05:04:35.779900Z","shell.execute_reply.started":"2024-04-14T05:04:35.771858Z","shell.execute_reply":"2024-04-14T05:04:35.779054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df","metadata":{"execution":{"iopub.status.busy":"2024-04-14T05:04:38.518136Z","iopub.execute_input":"2024-04-14T05:04:38.518479Z","iopub.status.idle":"2024-04-14T05:04:38.529846Z","shell.execute_reply.started":"2024-04-14T05:04:38.518454Z","shell.execute_reply":"2024-04-14T05:04:38.528855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Submitting Result","metadata":{}},{"cell_type":"code","source":"submission_df.to_csv(\"submission.csv\", index=False)\nsubmission_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-14T05:04:42.567072Z","iopub.execute_input":"2024-04-14T05:04:42.567460Z","iopub.status.idle":"2024-04-14T05:04:42.585674Z","shell.execute_reply.started":"2024-04-14T05:04:42.567429Z","shell.execute_reply":"2024-04-14T05:04:42.584668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}