{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport glob\nfrom tqdm import tqdm\ntqdm.pandas()\npd.options.display.max_colwidth = 10000\n\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-10T08:37:54.910226Z","iopub.execute_input":"2024-04-10T08:37:54.913029Z","iopub.status.idle":"2024-04-10T08:37:56.045130Z","shell.execute_reply.started":"2024-04-10T08:37:54.912986Z","shell.execute_reply":"2024-04-10T08:37:56.043936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Loading Train Data ","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-04-10T09:38:13.823577Z","iopub.execute_input":"2024-04-10T09:38:13.824527Z","iopub.status.idle":"2024-04-10T09:38:14.042097Z","shell.execute_reply.started":"2024-04-10T09:38:13.824484Z","shell.execute_reply":"2024-04-10T09:38:14.041224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Mapping train_eegs and train_spectogram  to  its Full path","metadata":{}},{"cell_type":"code","source":"print(\"Mapping train_eeg_path_list\",\"-\"*60)\n\ntrain_eeg_path_list = glob.glob(\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/*\")\ntrain_df['eeg_path'] = train_df['eeg_id'].astype(str).progress_apply(lambda x: [i for i in train_eeg_path_list if x in i][0])\n\nprint(\"Mapping train_spectrograms_path_list\",\"-\"*60)\n\ntrain_spectrograms_path_list = glob.glob(\"/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/*\")\ntrain_df['spectrograms_path'] = train_df['spectrogram_id'].astype(str).progress_apply(lambda x: [i for i in train_spectrograms_path_list if x in i][0])\n","metadata":{"execution":{"iopub.status.busy":"2024-04-10T09:16:39.522460Z","iopub.execute_input":"2024-04-10T09:16:39.522872Z","iopub.status.idle":"2024-04-10T09:20:56.835558Z","shell.execute_reply.started":"2024-04-10T09:16:39.522839Z","shell.execute_reply":"2024-04-10T09:20:56.834535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(1)","metadata":{"execution":{"iopub.status.busy":"2024-04-10T09:37:23.855756Z","iopub.execute_input":"2024-04-10T09:37:23.856175Z","iopub.status.idle":"2024-04-10T09:37:23.872537Z","shell.execute_reply.started":"2024-04-10T09:37:23.856142Z","shell.execute_reply":"2024-04-10T09:37:23.871540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Mapping test_eegs and test_spectogram  to  its Full path","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/test.csv\")\n\nprint(\"Mapping test_eeg_path_list\", \"-\"*60)\n\n# Mapping test EEG paths\ntest_eeg_path_list = glob.glob(\"/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/*\")\ntest_df['eeg_path'] = test_df['eeg_id'].astype(str).progress_apply(lambda x: [i for i in test_eeg_path_list if x in i][0])\n\nprint(\"Mapping test_spectrograms_path_list\", \"-\"*60)\n\n# Mapping test spectrograms paths\ntest_spectrograms_path_list = glob.glob(\"/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/*\")\ntest_df['spectrograms_path'] = test_df['spectrogram_id'].astype(str).progress_apply(lambda x: [i for i in test_spectrograms_path_list if x in i][0])","metadata":{"execution":{"iopub.status.busy":"2024-04-10T09:56:58.918171Z","iopub.execute_input":"2024-04-10T09:56:58.918542Z","iopub.status.idle":"2024-04-10T09:56:58.940112Z","shell.execute_reply.started":"2024-04-10T09:56:58.918511Z","shell.execute_reply":"2024-04-10T09:56:58.939027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2024-04-10T09:57:02.212792Z","iopub.execute_input":"2024-04-10T09:57:02.213242Z","iopub.status.idle":"2024-04-10T09:57:02.226636Z","shell.execute_reply.started":"2024-04-10T09:57:02.213208Z","shell.execute_reply":"2024-04-10T09:57:02.225424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Now We Will Create A Function Which Will Convert The parquet file to .npy format i.e numpy format","metadata":{}},{"cell_type":"code","source":"def parquet_to_numpy(parquet_path):\n    # Read the Parquet file into a DataFrame\n    spec_df = pd.read_parquet(parquet_path)\n    \n    # Process the DataFrame to convert it into a numpy array\n    spec_array = spec_df.fillna(0).values[:, 1:].T  # fill NaN values with 0, transpose for (Time, Freq) -> (Freq, Time)\n    spec_array = spec_array.astype(\"float32\")\n    \n    return spec_array","metadata":{"execution":{"iopub.status.busy":"2024-04-10T10:19:34.475953Z","iopub.execute_input":"2024-04-10T10:19:34.476356Z","iopub.status.idle":"2024-04-10T10:19:34.482742Z","shell.execute_reply.started":"2024-04-10T10:19:34.476325Z","shell.execute_reply":"2024-04-10T10:19:34.481532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spec_path = \"/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/853520.parquet\"\nspec_array = parquet_to_numpy(spec_path)\nspec_array","metadata":{"execution":{"iopub.status.busy":"2024-04-10T10:21:25.133289Z","iopub.execute_input":"2024-04-10T10:21:25.133749Z","iopub.status.idle":"2024-04-10T10:21:25.178863Z","shell.execute_reply.started":"2024-04-10T10:21:25.133715Z","shell.execute_reply":"2024-04-10T10:21:25.177702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess_spectrogram(image_array):\n\n    # Normalization: Ensures that the pixel values are within a certain range\n    # This helps in stabilizing the training process and ensures faster convergence\n    image_array = image_array.astype('float32')\n    image_array -= np.min(image_array)\n    image_array /= np.max(image_array) + 1e-4\n    \n    # Log Transformation: Enhances contrast and reduces the effect of outliers\n    # It helps in better visualization of the spectrogram features\n    image_array = np.log(image_array + 1e-4)\n    \n    # Mean Subtraction: Centers the data around zero\n    # This helps in reducing bias and improving the stability of the model\n    mean = np.mean(image_array)\n    image_array -= mean\n    \n    # Standardization: Scales the data to have zero mean and unit variance\n    # It ensures that all features are on a similar scale, which can improve model performance\n    std = np.std(image_array)\n    image_array /= std + 1e-6\n    \n    return image_array\n","metadata":{"execution":{"iopub.status.busy":"2024-04-10T10:54:34.968776Z","iopub.execute_input":"2024-04-10T10:54:34.969251Z","iopub.status.idle":"2024-04-10T10:54:34.977548Z","shell.execute_reply.started":"2024-04-10T10:54:34.969215Z","shell.execute_reply":"2024-04-10T10:54:34.976201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimage_array = np.random.rand(400, 300)\n\npreprocessed_image = preprocess_spectrogram(spec_array)","metadata":{"execution":{"iopub.status.busy":"2024-04-10T11:05:37.053163Z","iopub.execute_input":"2024-04-10T11:05:37.053592Z","iopub.status.idle":"2024-04-10T11:05:37.061079Z","shell.execute_reply.started":"2024-04-10T11:05:37.053559Z","shell.execute_reply":"2024-04-10T11:05:37.059947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### original spectogram image","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.imshow(spec_array)","metadata":{"execution":{"iopub.status.busy":"2024-04-10T11:17:25.997787Z","iopub.execute_input":"2024-04-10T11:17:25.999391Z","iopub.status.idle":"2024-04-10T11:17:26.029568Z","shell.execute_reply.started":"2024-04-10T11:17:25.999337Z","shell.execute_reply":"2024-04-10T11:17:26.027803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(preprocessed_image)","metadata":{"execution":{"iopub.status.busy":"2024-04-10T11:05:46.377928Z","iopub.execute_input":"2024-04-10T11:05:46.378307Z","iopub.status.idle":"2024-04-10T11:05:46.751412Z","shell.execute_reply.started":"2024-04-10T11:05:46.378278Z","shell.execute_reply":"2024-04-10T11:05:46.750461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}