{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30683,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport glob\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\ntqdm.pandas()\npd.options.display.max_colwidth = 10000\n\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-21T10:47:03.117966Z","iopub.execute_input":"2024-04-21T10:47:03.118706Z","iopub.status.idle":"2024-04-21T10:47:05.131100Z","shell.execute_reply.started":"2024-04-21T10:47:03.118672Z","shell.execute_reply":"2024-04-21T10:47:05.130282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Loading Train Data ","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:47:05.132787Z","iopub.execute_input":"2024-04-21T10:47:05.133167Z","iopub.status.idle":"2024-04-21T10:47:05.375431Z","shell.execute_reply.started":"2024-04-21T10:47:05.133140Z","shell.execute_reply":"2024-04-21T10:47:05.374597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Mapping train_eegs and train_spectogram  to  its Full path","metadata":{}},{"cell_type":"code","source":"def create_id_mapping(paths_list):\n    id_map = {}\n    for path in paths_list:\n        file_id = os.path.basename(path).split('.')[0]  # Corrected to extract the file ID correctly\n        id_map[file_id] = path\n    return id_map\n\ndef mapping_id(ids, id_map):\n    return id_map.get(ids)","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:47:05.376795Z","iopub.execute_input":"2024-04-21T10:47:05.377516Z","iopub.status.idle":"2024-04-21T10:47:05.383325Z","shell.execute_reply.started":"2024-04-21T10:47:05.377474Z","shell.execute_reply":"2024-04-21T10:47:05.382485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create ID mappings for train_eeg_path_list and train_spectrograms_path_list\ntrain_eeg_path_list = glob.glob(\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/*\")\ntrain_spectrograms_path_list = glob.glob(\"/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/*\")\neeg_id_map = create_id_mapping(train_eeg_path_list)\nspectrograms_id_map = create_id_mapping(train_spectrograms_path_list)\n\n# Example usage:\nprint(\"Mapping train_eeg_path_list\",\"-\"*60)\ntrain_df['eeg_path'] = train_df['eeg_id'].astype(str).progress_apply(lambda x: mapping_id(x, eeg_id_map))\n\nprint(\"Mapping train_spectrograms_path_list\",\"-\"*60)\ntrain_df['spectrograms_path'] = train_df['spectrogram_id'].astype(str).progress_apply(lambda x: mapping_id(x, spectrograms_id_map))","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:47:05.385892Z","iopub.execute_input":"2024-04-21T10:47:05.386161Z","iopub.status.idle":"2024-04-21T10:47:06.545870Z","shell.execute_reply.started":"2024-04-21T10:47:05.386137Z","shell.execute_reply":"2024-04-21T10:47:06.544960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(1)","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:47:06.547033Z","iopub.execute_input":"2024-04-21T10:47:06.547350Z","iopub.status.idle":"2024-04-21T10:47:06.567519Z","shell.execute_reply.started":"2024-04-21T10:47:06.547323Z","shell.execute_reply":"2024-04-21T10:47:06.566572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Mapping test_eegs and test_spectogram  to  its Full path","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/test.csv\")\n\n# Create ID mappings for train_eeg_path_list and train_spectrograms_path_list\ntest_eeg_path_list = glob.glob(\"/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/*\")\ntest_spectrograms_path_list = glob.glob(\"/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/*\")\neeg_id_map = create_id_mapping(test_eeg_path_list)\nspectrograms_id_map = create_id_mapping(test_spectrograms_path_list)\n\n# Example usage:\nprint(\"Mapping train_eeg_path_list\",\"-\"*60)\ntest_df['eeg_path'] = test_df['eeg_id'].astype(str).progress_apply(lambda x: mapping_id(x, eeg_id_map))\n\nprint(\"Mapping train_spectrograms_path_list\",\"-\"*60)\ntest_df['spectrograms_path'] = test_df['spectrogram_id'].astype(str).progress_apply(lambda x: mapping_id(x, spectrograms_id_map))","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:47:06.568853Z","iopub.execute_input":"2024-04-21T10:47:06.569522Z","iopub.status.idle":"2024-04-21T10:47:06.595100Z","shell.execute_reply.started":"2024-04-21T10:47:06.569488Z","shell.execute_reply":"2024-04-21T10:47:06.594108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head(1)","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:47:06.596150Z","iopub.execute_input":"2024-04-21T10:47:06.596404Z","iopub.status.idle":"2024-04-21T10:47:06.606860Z","shell.execute_reply.started":"2024-04-21T10:47:06.596381Z","shell.execute_reply":"2024-04-21T10:47:06.605991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Now We Will Create A Function Which Will Convert The parquet file to .npy format i.e numpy format","metadata":{}},{"cell_type":"code","source":"def parquet_to_numpy(parquet_path):\n    # Read the Parquet file into a DataFrame\n    spec_df = pd.read_parquet(parquet_path)\n    \n    # Process the DataFrame to convert it into a numpy array\n    spec_array = spec_df.fillna(0).values[:, 1:].T  # fill NaN values with 0, transpose for (Time, Freq) -> (Freq, Time)\n    spec_array = spec_array.astype(\"float32\")\n    \n    return spec_array","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:47:06.608246Z","iopub.execute_input":"2024-04-21T10:47:06.608858Z","iopub.status.idle":"2024-04-21T10:47:06.614678Z","shell.execute_reply.started":"2024-04-21T10:47:06.608823Z","shell.execute_reply":"2024-04-21T10:47:06.613722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Preprocesing The Image To Get The Better Visualization of the Image\n[Taking Help To Preprocess The Image From This Notebook](https://www.kaggle.com/code/awsaf49/hms-hbac-kerascv-starter-notebook//)","metadata":{}},{"cell_type":"code","source":"def preprocess_spectrogram(image_array):\n\n    # Normalization: Ensures that the pixel values are within a certain range\n    # This helps in stabilizing the training process and ensures faster convergence\n    image_array = image_array.astype('float32')\n    image_array -= np.min(image_array)\n    image_array /= np.max(image_array) + 1e-4\n    \n    # Log Transformation: Enhances contrast and reduces the effect of outliers\n    # It helps in better visualization of the spectrogram features\n    image_array = np.log(image_array + 1e-4)\n    \n    # Mean Subtraction: Centers the data around zero\n    # This helps in reducing bias and improving the stability of the model\n    mean = np.mean(image_array)\n    image_array -= mean\n    \n    # Standardization: Scales the data to have zero mean and unit variance\n    # It ensures that all features are on a similar scale, which can improve model performance\n    std = np.std(image_array)\n    image_array /= std + 1e-6\n    \n    return image_array","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:47:06.616007Z","iopub.execute_input":"2024-04-21T10:47:06.616743Z","iopub.status.idle":"2024-04-21T10:47:06.627724Z","shell.execute_reply.started":"2024-04-21T10:47:06.616719Z","shell.execute_reply":"2024-04-21T10:47:06.626791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating Train And Test Data Directory","metadata":{}},{"cell_type":"code","source":"current_dir = os.getcwd()\ntrain_dir = os.path.join(current_dir, \"train\")\ntest_dir  = os.path.join(current_dir, \"test\")\nprint(\"train_dir--->\", train_dir)\nprint(\"test_dir---->\", test_dir)\nos.makedirs(train_dir, exist_ok = True)\nos.makedirs(test_dir, exist_ok = True)","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:47:06.631355Z","iopub.execute_input":"2024-04-21T10:47:06.631656Z","iopub.status.idle":"2024-04-21T10:47:06.638012Z","shell.execute_reply.started":"2024-04-21T10:47:06.631613Z","shell.execute_reply":"2024-04-21T10:47:06.637081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Dumping Train Spectograms Data To Train Dir","metadata":{}},{"cell_type":"code","source":"from PIL import Image\n\nfor i in tqdm(train_df['spectrograms_path'].unique()):\n    img_array = preprocess_spectrogram(parquet_to_numpy(i))\n\n    img_name = os.path.basename(i).split('.')[0] + \".jpeg\"\n    spectrograms_path = os.path.join(train_dir, \"spectrograms\")\n    os.makedirs(spectrograms_path, exist_ok=True)\n    \n    plt.imsave(os.path.join(spectrograms_path, img_name), img_array)","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:47:06.639134Z","iopub.execute_input":"2024-04-21T10:47:06.639395Z","iopub.status.idle":"2024-04-21T10:55:59.445046Z","shell.execute_reply.started":"2024-04-21T10:47:06.639373Z","shell.execute_reply":"2024-04-21T10:55:59.444013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Dumping Test Spectograms to the Test Dir","metadata":{}},{"cell_type":"code","source":"from PIL import Image\n\nfor i in tqdm(test_df['spectrograms_path'].unique()):\n    img_array = preprocess_spectrogram(parquet_to_numpy(i))\n\n    img_name = os.path.basename(i).split('.')[0] + \".jpeg\"\n    spectrograms_path = os.path.join(test_dir, \"spectrograms\")\n    os.makedirs(spectrograms_path, exist_ok=True)\n    \n    plt.imsave(os.path.join(spectrograms_path, img_name), img_array)","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:55:59.446350Z","iopub.execute_input":"2024-04-21T10:55:59.446675Z","iopub.status.idle":"2024-04-21T10:55:59.498870Z","shell.execute_reply.started":"2024-04-21T10:55:59.446618Z","shell.execute_reply":"2024-04-21T10:55:59.497934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Renaming Train Image In Train Directory Adding Class To The Train Image","metadata":{}},{"cell_type":"code","source":"xx = train_df['spectrograms_path expert_consensus'.split()].drop_duplicates().reset_index(drop = True)\ndict_xx = dict(zip(xx['spectrograms_path'], xx['expert_consensus']))","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:55:59.500042Z","iopub.execute_input":"2024-04-21T10:55:59.500393Z","iopub.status.idle":"2024-04-21T10:55:59.548459Z","shell.execute_reply.started":"2024-04-21T10:55:59.500360Z","shell.execute_reply":"2024-04-21T10:55:59.547761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for original_path in tqdm(glob.glob(\"/kaggle/working/train/spectrograms/*\")):\n    # Extract the image ID from the file name\n    img_id = os.path.basename(original_path).split('.')[-2]\n    # Find the corresponding value in dict_xx based on the image ID\n    xx = [v for k, v in dict_xx.items() if img_id in k][0]\n    \n    # Construct the new file name with the same directory path\n    directory_path = os.path.dirname(original_path)\n    new_filename = f\"{xx}.{img_id}.jpeg\"\n    \n    # Construct the new file path\n    new_path = os.path.join(directory_path, new_filename)\n    \n    # Rename the file\n    os.rename(original_path, new_path)","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:55:59.549470Z","iopub.execute_input":"2024-04-21T10:55:59.549730Z","iopub.status.idle":"2024-04-21T10:56:15.283726Z","shell.execute_reply.started":"2024-04-21T10:55:59.549706Z","shell.execute_reply":"2024-04-21T10:56:15.282790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del xx, dict_xx","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:56:15.285013Z","iopub.execute_input":"2024-04-21T10:56:15.285377Z","iopub.status.idle":"2024-04-21T10:56:15.290062Z","shell.execute_reply.started":"2024-04-21T10:56:15.285343Z","shell.execute_reply":"2024-04-21T10:56:15.289175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Creating A Function Which Will Plot The Random Image From The Directory","metadata":{}},{"cell_type":"code","source":"import os\nimport random\nimport matplotlib.pyplot as plt\nfrom PIL import Image\n\ndef plot_random_images(dir_path, num_images, folder_name):\n    if not os.path.exists(dir_path):\n        print(\"Directory does not exist.\")\n        return\n    \n    image_files = [f for f in os.listdir(dir_path) if f.endswith('.jpeg') or f.endswith('.png') or f.endswith('.jpg')]\n    if len(image_files) == 0:\n        print(\"No image files found in the directory.\")\n        return\n    \n    # Shuffle the list of image files\n    random.shuffle(image_files)\n    \n    # Limit the number of images to plot\n    num_images = min(num_images, len(image_files))\n    \n    # Calculate the number of rows and columns based on the aspect ratio of the images\n    num_rows = int(num_images ** 0.5)\n    num_cols = (num_images + num_rows - 1) // num_rows\n\n    fig, axes = plt.subplots(num_rows, num_cols, figsize=(15, 10))\n\n    for i in range(num_images):\n        img_path = os.path.join(dir_path, image_files[i])\n        img = Image.open(img_path)\n\n        row = i // num_cols\n        col = i % num_cols\n\n        ax = axes[row, col]\n        ax.imshow(img, cmap=\"gray\")\n        ax.set_xticks([])\n        ax.set_yticks([])\n        ax.set_xlabel(image_files[i], fontsize=8, wrap=True)\n\n    # Hide any remaining empty subplots\n    for i in range(num_images, num_rows * num_cols):\n        row = i // num_cols\n        col = i % num_cols\n        axes[row, col].axis('off')\n\n    plt.suptitle(folder_name, fontsize=16)\n    plt.subplots_adjust(wspace=0.2, hspace=0.2)  # Adjust spacing between subplots\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:47:03.095868Z","iopub.execute_input":"2024-04-21T10:47:03.096699Z","iopub.status.idle":"2024-04-21T10:47:03.116030Z","shell.execute_reply.started":"2024-04-21T10:47:03.096665Z","shell.execute_reply":"2024-04-21T10:47:03.114976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_dir_spectrograms = '/kaggle/working/train/spectrograms'\nplot_random_images(training_dir_spectrograms, 5, \"Train_Images\")","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:56:15.291313Z","iopub.execute_input":"2024-04-21T10:56:15.291674Z","iopub.status.idle":"2024-04-21T10:56:16.118898Z","shell.execute_reply.started":"2024-04-21T10:56:15.291623Z","shell.execute_reply":"2024-04-21T10:56:16.117794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" ### Dumping of the data is completed","metadata":{}},{"cell_type":"markdown","source":"#### Now We Will Create A Train Data Frame Which Will Contain  Training Image Path And Their Class Along With Their EEG_ID","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### This will Be Our Final Train Data Frame","metadata":{}},{"cell_type":"code","source":"def create_id_mapping(paths_list):\n    id_map = {}\n    for path in paths_list:\n        file_id = os.path.basename(path).split('.')[1]  # Corrected to extract the file ID correctly\n        id_map[file_id] = path\n    return id_map\n\ndef mapping_id(ids, id_map):\n    return id_map.get(ids)","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:56:16.120069Z","iopub.execute_input":"2024-04-21T10:56:16.120350Z","iopub.status.idle":"2024-04-21T10:56:16.126020Z","shell.execute_reply.started":"2024-04-21T10:56:16.120326Z","shell.execute_reply":"2024-04-21T10:56:16.125039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create ID mappings for train_eeg_path_list and train_spectrograms_path_list\ntrain_spectrograms_path_list = glob.glob(\"/kaggle/working/train/spectrograms/*\")\nspectrograms_id_map = create_id_mapping(train_spectrograms_path_list)\n\ntrain_df = train_df['eeg_id spectrogram_id expert_consensus'.split()].drop_duplicates().reset_index(drop = True)\n\ntrain_df['spectrograms_path'] = train_df['spectrogram_id'].astype(str).progress_apply(lambda x: mapping_id(x, spectrograms_id_map))","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:56:16.127260Z","iopub.execute_input":"2024-04-21T10:56:16.127564Z","iopub.status.idle":"2024-04-21T10:56:16.265933Z","shell.execute_reply.started":"2024-04-21T10:56:16.127537Z","shell.execute_reply":"2024-04-21T10:56:16.264925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Creating Train And Validation Split","metadata":{}},{"cell_type":"code","source":"train_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:56:16.267143Z","iopub.execute_input":"2024-04-21T10:56:16.267429Z","iopub.status.idle":"2024-04-21T10:56:16.277000Z","shell.execute_reply.started":"2024-04-21T10:56:16.267403Z","shell.execute_reply":"2024-04-21T10:56:16.275820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrain_dataset, valid_dataset = train_test_split(train_df ,test_size = 0.3 , random_state = 42, shuffle = True,\n                                               stratify = train_df['expert_consensus'])","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:56:16.278194Z","iopub.execute_input":"2024-04-21T10:56:16.278517Z","iopub.status.idle":"2024-04-21T10:56:16.506854Z","shell.execute_reply.started":"2024-04-21T10:56:16.278482Z","shell.execute_reply":"2024-04-21T10:56:16.505660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training Set Validation Set","metadata":{}},{"cell_type":"code","source":"train_dataset.head(1)","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:56:16.508105Z","iopub.execute_input":"2024-04-21T10:56:16.508684Z","iopub.status.idle":"2024-04-21T10:56:16.518590Z","shell.execute_reply.started":"2024-04-21T10:56:16.508646Z","shell.execute_reply":"2024-04-21T10:56:16.517639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Image Data Generator","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n# Define data generators\ntrain_datagen = ImageDataGenerator()\ntest_datagen = ImageDataGenerator()","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:56:16.519841Z","iopub.execute_input":"2024-04-21T10:56:16.520205Z","iopub.status.idle":"2024-04-21T10:56:27.521298Z","shell.execute_reply.started":"2024-04-21T10:56:16.520180Z","shell.execute_reply":"2024-04-21T10:56:27.520296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training Set Validation Set","metadata":{}},{"cell_type":"code","source":"training_set = train_datagen.flow_from_dataframe(\n                        dataframe = train_dataset,\n                        x_col = 'spectrograms_path',\n                        y_col = 'expert_consensus',\n                        target_size=(331,331),\n                        color_mode='rgb',\n                        class_mode = 'categorical',\n                        batch_size= 64)\n\nvalidation_set = test_datagen.flow_from_dataframe(\n                dataframe = valid_dataset,\n                x_col = 'spectrograms_path',\n                y_col = 'expert_consensus',\n                color_mode='rgb',\n                target_size = (331, 331),\n                class_mode = 'categorical',\n                batch_size = 64)","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:56:27.522731Z","iopub.execute_input":"2024-04-21T10:56:27.523765Z","iopub.status.idle":"2024-04-21T10:56:27.739322Z","shell.execute_reply.started":"2024-04-21T10:56:27.523727Z","shell.execute_reply":"2024-04-21T10:56:27.738409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Importing Multiple Transffer Learning Technique ","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.layers import (Input, Lambda , Dense, Flatten, \n                                     ReLU, LeakyReLU, PReLU, BatchNormalization,\n                                    Conv2D, MaxPool2D, Dropout, \n                                     GlobalAveragePooling2D)\n\nfrom tensorflow.keras.models import (Model, Sequential)","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:56:27.740419Z","iopub.execute_input":"2024-04-21T10:56:27.740740Z","iopub.status.idle":"2024-04-21T10:56:27.748119Z","shell.execute_reply.started":"2024-04-21T10:56:27.740713Z","shell.execute_reply":"2024-04-21T10:56:27.747167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Creating A Function Which Will Extract Feature From Transffer Learning Technique","metadata":{}},{"cell_type":"code","source":"## function to extract features from the dataset by a given pretrained model\nimg_size = (331,331,3)\n\ndef get_features(model_name, model_preprocessor, input_size, data):\n    input_layer = Input(input_size)\n    preprocessor = Lambda(model_preprocessor)(input_layer)\n    base_model   = model_name(weights='imagenet', include_top=False,\n                            input_shape = input_size)(preprocessor)\n    avg = GlobalAveragePooling2D()(base_model)\n    feature_extractor = Model(inputs = input_layer, outputs = avg)\n    \n    #Extract feature.\n    \n    feature_maps = feature_extractor.predict(training_set, verbose=1)\n    print(\"{0} Feature Map Shape Are {1}\".format(model_name, feature_maps.shape))\n    return feature_maps","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:56:27.749295Z","iopub.execute_input":"2024-04-21T10:56:27.749622Z","iopub.status.idle":"2024-04-21T10:56:27.756526Z","shell.execute_reply.started":"2024-04-21T10:56:27.749590Z","shell.execute_reply":"2024-04-21T10:56:27.755519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Extracting  From InceptionV3","metadata":{}},{"cell_type":"code","source":"# Extract features using InceptionV3 \n\nfrom keras.applications.inception_v3 import InceptionV3, preprocess_input\ninception_preprocessor = preprocess_input\ninception_features = get_features(InceptionV3,\n                                  inception_preprocessor,\n                                  img_size, data = training_set)","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:56:27.757720Z","iopub.execute_input":"2024-04-21T10:56:27.757993Z","iopub.status.idle":"2024-04-21T10:57:59.623418Z","shell.execute_reply.started":"2024-04-21T10:56:27.757969Z","shell.execute_reply":"2024-04-21T10:57:59.622422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Extracting  From Xception","metadata":{}},{"cell_type":"code","source":"# Extract features using Xception \nfrom keras.applications.xception import Xception, preprocess_input\nxception_preprocessor = preprocess_input\nxception_features = get_features(Xception,\n                                 xception_preprocessor,\n                                 img_size, data = training_set)","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:57:59.624806Z","iopub.execute_input":"2024-04-21T10:57:59.625083Z","iopub.status.idle":"2024-04-21T10:59:26.328265Z","shell.execute_reply.started":"2024-04-21T10:57:59.625058Z","shell.execute_reply":"2024-04-21T10:59:26.327180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Extracting  From InceptionResNetV2","metadata":{}},{"cell_type":"code","source":"# Extract features using InceptionResNetV2 \nfrom keras.applications.inception_resnet_v2 import InceptionResNetV2, preprocess_input\ninc_resnet_preprocessor = preprocess_input\ninc_resnet_features = get_features(InceptionResNetV2,\n                                   inc_resnet_preprocessor,\n                                   img_size, data = training_set)","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:59:26.333289Z","iopub.execute_input":"2024-04-21T10:59:26.333588Z","iopub.status.idle":"2024-04-21T11:01:17.977299Z","shell.execute_reply.started":"2024-04-21T10:59:26.333565Z","shell.execute_reply":"2024-04-21T11:01:17.976188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Extract features using NASNetLarge \n","metadata":{}},{"cell_type":"code","source":"# Extract features using NASNetLarge \nfrom keras.applications.nasnet import NASNetLarge, preprocess_input\nnasnet_preprocessor = preprocess_input\nnasnet_features = get_features(NASNetLarge,\n                               nasnet_preprocessor,\n                               img_size, data = training_set)","metadata":{"execution":{"iopub.status.busy":"2024-04-21T11:01:17.978990Z","iopub.execute_input":"2024-04-21T11:01:17.979274Z","iopub.status.idle":"2024-04-21T11:05:13.245513Z","shell.execute_reply.started":"2024-04-21T11:01:17.979251Z","shell.execute_reply":"2024-04-21T11:05:13.244475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Extract features using VGG19 \n","metadata":{}},{"cell_type":"code","source":"from keras.applications.vgg19 import VGG19, preprocess_input\nvgg19_preprocessor = preprocess_input\n\n# Assuming you have a function `get_features` defined similarly as in your example\nvgg19_features = get_features(VGG19,\n                              vgg19_preprocessor,\n                              img_size, data=training_set)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-21T11:28:08.118476Z","iopub.execute_input":"2024-04-21T11:28:08.118920Z","iopub.status.idle":"2024-04-21T11:30:24.165788Z","shell.execute_reply.started":"2024-04-21T11:28:08.118888Z","shell.execute_reply":"2024-04-21T11:30:24.164691Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Concatenating All The Feature","metadata":{}},{"cell_type":"code","source":"#Creating final featuremap by combining all extracted features\n\nfinal_features = np.concatenate([inception_features,\n                                 xception_features,\n                                 nasnet_features,\n                                 inc_resnet_features,\n                                vgg19_features], axis=-1) #axis=-1 to concatinate horizontally\n\nprint('Final feature maps shape', final_features.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-21T11:30:24.167887Z","iopub.execute_input":"2024-04-21T11:30:24.168222Z","iopub.status.idle":"2024-04-21T11:30:24.360033Z","shell.execute_reply.started":"2024-04-21T11:30:24.168193Z","shell.execute_reply":"2024-04-21T11:30:24.359082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Now We will Define The CallBack Function Using Tensorflow¶\n","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.callbacks import ReduceLROnPlateau,EarlyStopping\n\n## Learning Rate Annealer \nlrr = ReduceLROnPlateau(monitor = \"val_acc\", factor = .01, patience = 3, \n                       min_lr= 1e-5, verbose = 1)\n## Prepare Callbacks\nEarlyStop = EarlyStopping(monitor='val_loss', patience = 10, \n                          restore_best_weights = True)","metadata":{"execution":{"iopub.status.busy":"2024-04-21T11:31:33.054799Z","iopub.execute_input":"2024-04-21T11:31:33.055175Z","iopub.status.idle":"2024-04-21T11:31:33.060865Z","shell.execute_reply.started":"2024-04-21T11:31:33.055149Z","shell.execute_reply":"2024-04-21T11:31:33.059822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model Building","metadata":{}},{"cell_type":"code","source":"n_classes = len(list(validation_set.class_indices.keys()))","metadata":{"execution":{"iopub.status.busy":"2024-04-21T11:31:36.716764Z","iopub.execute_input":"2024-04-21T11:31:36.717142Z","iopub.status.idle":"2024-04-21T11:31:36.721946Z","shell.execute_reply.started":"2024-04-21T11:31:36.717107Z","shell.execute_reply":"2024-04-21T11:31:36.720908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validation_set.class_indices","metadata":{"execution":{"iopub.status.busy":"2024-04-21T11:31:40.362580Z","iopub.execute_input":"2024-04-21T11:31:40.362979Z","iopub.status.idle":"2024-04-21T11:31:40.369419Z","shell.execute_reply.started":"2024-04-21T11:31:40.362950Z","shell.execute_reply":"2024-04-21T11:31:40.368505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.utils import to_categorical\n\n# Assuming train_dataset is the DataFrame containing expert consensus labels\ny = to_categorical(train_dataset['expert_consensus'].replace(validation_set.class_indices), num_classes=6)\n\nprint(\"final_data shape:\", final_features.shape)\nprint(\"shape of y_one_hot_encoder:\", y.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-21T11:33:29.321220Z","iopub.execute_input":"2024-04-21T11:33:29.321615Z","iopub.status.idle":"2024-04-21T11:33:29.340537Z","shell.execute_reply.started":"2024-04-21T11:33:29.321567Z","shell.execute_reply":"2024-04-21T11:33:29.339691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras import optimizers\n\nmodel = Sequential()\nmodel.add(Dropout(0.7, input_shape=(final_features.shape[1],)))\nmodel.add(Dense(n_classes, activation='softmax'))\n\n# Create optimizer instance\nadam = optimizers.Adam()\n\nmodel.compile(optimizer=adam,\n              loss='categorical_crossentropy',\n              metrics=['accuracy'])\n\n# Training the model.\nhistory = model.fit(final_features, y,\n                    batch_size=64,\n                    epochs=50,\n                    validation_split=0.2,\n                    callbacks=[lrr, EarlyStop])","metadata":{"execution":{"iopub.status.busy":"2024-04-21T11:33:49.945821Z","iopub.execute_input":"2024-04-21T11:33:49.946192Z","iopub.status.idle":"2024-04-21T11:34:00.151573Z","shell.execute_reply.started":"2024-04-21T11:33:49.946164Z","shell.execute_reply":"2024-04-21T11:34:00.150489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create subplots with 1 row and 2 columns\nfig, axs = plt.subplots(1, 2, figsize=(15, 5))\n\n# Plot training and validation accuracy values\naxs[0].plot(history.history['accuracy'], label='Training Accuracy')\naxs[0].plot(history.history['val_accuracy'], label='Validation Accuracy')\naxs[0].set_title('Training and Validation Accuracy')\naxs[0].set_xlabel('Epoch')\naxs[0].set_ylabel('Accuracy')\naxs[0].legend()\n\n# Plot training and validation loss values\naxs[1].plot(history.history['loss'], label='Training Loss')\naxs[1].plot(history.history['val_loss'], label='Validation Loss')\naxs[1].set_title('Training and Validation Loss')\naxs[1].set_xlabel('Epoch')\naxs[1].set_ylabel('Loss')\naxs[1].legend()\n\n# Show the plots\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-21T11:34:50.845564Z","iopub.execute_input":"2024-04-21T11:34:50.846372Z","iopub.status.idle":"2024-04-21T11:34:51.412771Z","shell.execute_reply.started":"2024-04-21T11:34:50.846337Z","shell.execute_reply":"2024-04-21T11:34:51.411536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}