{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import librosa, glob, os, itertools, time, multiprocessing\nfrom PIL import Image\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom scipy.signal import wiener, stft, butter, lfilter\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Conv2D, MaxPooling2D, Flatten, Dropout, BatchNormalization\nfrom keras.optimizers import Adam\nfrom keras.utils import to_categorical","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":10.444964,"end_time":"2023-03-21T13:20:48.037572","exception":false,"start_time":"2023-03-21T13:20:37.592608","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-03-27T19:09:43.695228Z","iopub.execute_input":"2023-03-27T19:09:43.695688Z","iopub.status.idle":"2023-03-27T19:09:43.704777Z","shell.execute_reply.started":"2023-03-27T19:09:43.695650Z","shell.execute_reply":"2023-03-27T19:09:43.702801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dictionary of hyperparameters\nhp = {\n    # STFT parameters\n    'nperseg': 1024,\n    'noverlap': 0,\n    'window': 'rect',\n    # Length of each STFT block\n    'nperblock': 1024 * 300,\n    # Preprocessing\n    'lowpass': 8000,\n    'highpass': 200,\n    'denoising': False,\n     }","metadata":{"papermill":{"duration":0.013707,"end_time":"2023-03-21T13:20:48.054589","exception":false,"start_time":"2023-03-21T13:20:48.040882","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-03-27T18:53:58.031093Z","iopub.execute_input":"2023-03-27T18:53:58.031570Z","iopub.status.idle":"2023-03-27T18:53:58.038278Z","shell.execute_reply.started":"2023-03-27T18:53:58.031526Z","shell.execute_reply":"2023-03-27T18:53:58.036689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def highpass(signal, f0, fs):\n    b, a = butter(3, f0/(fs/2),  btype='highpass')\n    filtered_signal = lfilter(b, a, signal)\n    return filtered_signal\n\ndef lowpass(signal, f0, fs):\n    b, a = butter(3, f0/(fs/2),  btype='lowpass')\n    filtered_signal = lfilter(b, a, signal)\n    return filtered_signal\n\ndef create_power_spectrograms(filepath, nperblock):\n    '''Function that slices the audiofile into equal size parts and transforms them into spectrograms.'''\n    # Filepath -> audiofile\n    audio, sr = librosa.load(filepath, sr=None)\n    if sr != 32000: print('Warning: Samplerate is not 32 kHz. ', filepath)\n    # Normalize audio\n    audio = audio / max(audio)\n    # Bandpass filtering\n    if hp['highpass']: audio = highpass(audio, hp['highpass'], sr)\n    if hp['lowpass']: audio = lowpass(audio, hp['lowpass'], sr)\n    # Denoising\n    if hp['denoising']: audio = wiener(wiener(audio))\n    \n    # Wrap first audio samples around to the end to make the whole file an integer multiple of 'nperblock'\n    audio = np.append(audio, audio[:nperblock-audio.size%nperblock])\n    if len(audio) < nperblock:\n        audio = np.append(audio, np.zeros(nperblock-audio.size%nperblock))\n    # Split audio file into segments\n    segments = np.array_split(audio, len(audio)//nperblock)\n    # Make a list of spectrograms \n    spectrograms = []\n    for segment in segments:\n        # Compute spectrogram\n        _,_,spec = stft(segment, nperseg=hp['nperseg'], noverlap=hp['noverlap'], window=hp['window'])\n        # Convert to power spectrogram\n        spec_power = librosa.power_to_db(np.abs(spec) ** 2, ref=np.max)\n        spectrograms.append(spec_power)\n\n    # Return the list of spectrogram matrices\n    return spectrograms\n\n# Plotting spectrogram function\ndef plot_spectrogram(specs):\n    for spec in specs:\n        plt.pcolormesh(spec)\n        plt.ylabel('Frequency segments')\n        plt.xlabel('Time segments')\n        plt.show()","metadata":{"papermill":{"duration":0.022519,"end_time":"2023-03-21T13:20:48.080210","exception":false,"start_time":"2023-03-21T13:20:48.057691","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-03-27T18:53:58.557464Z","iopub.execute_input":"2023-03-27T18:53:58.557929Z","iopub.status.idle":"2023-03-27T18:53:58.574732Z","shell.execute_reply.started":"2023-03-27T18:53:58.557884Z","shell.execute_reply":"2023-03-27T18:53:58.573412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Example plot of spectrogams\nspecs = create_power_spectrograms('/kaggle/input/birdclef-2023/train_audio/abethr1/XC128013.ogg', hp['nperblock'])\nplot_spectrogram(specs[:1])\nprint(type(specs[:1][0]))\nnp.savez_compressed('compressed', specs[:1])\nnp.save('not_compressed', specs[:1])\n# Create the Image object from the array\n\nspec = specs[:1][0] + specs[:1][0].min()\nspec/= spec.max()\n\nspec = (spec * 255).astype(np.uint8)\nimg = Image.fromarray(spec)\nprint(np.amax(spec))\nprint(np.amin(spec))\n# Save the Image object as a JPEG file\nimg.save('compressed_image.jpg', format='JPEG', quality=80)\n\n\nprint(type(specs[:1][0][0][0]))","metadata":{"papermill":{"duration":14.212833,"end_time":"2023-03-21T13:21:02.296138","exception":false,"start_time":"2023-03-21T13:20:48.083305","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-03-27T19:20:19.300206Z","iopub.execute_input":"2023-03-27T19:20:19.300730Z","iopub.status.idle":"2023-03-27T19:20:20.003600Z","shell.execute_reply.started":"2023-03-27T19:20:19.300689Z","shell.execute_reply":"2023-03-27T19:20:20.001909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"start = time.time()\n\n# Create DataFrame to store the spectrograms and the corresponding bird names\ndf = pd.DataFrame(columns=['Bird_Name', 'Spectrogram'])\n\n# Loop to fill the DataFrame with spectrograms of all audiofiles. \n# TODO: Currently really slow and exceeds the memory limit.\nfor dirname, subdirs,_ in os.walk('/kaggle/input/birdclef-2023/train_audio/'):\n    for subdir in subdirs[:4]:\n        spec_list = []\n        files = glob.glob('/kaggle/input/birdclef-2023/train_audio/' + subdir + '/*.ogg')\n        for file in files:\n            spec_list.extend(create_power_spectrograms(file, hp['nperblock']))\n        for spec in spec_list:\n            df = df.append({'Bird_Name': subdir, 'Spectrogram': spec}, ignore_index=True)\n            \nend = time.time()\nprint(end-start)\n                        \nprint(df.describe())","metadata":{"papermill":{"duration":1369.678646,"end_time":"2023-03-21T13:43:51.980170","exception":false,"start_time":"2023-03-21T13:21:02.301524","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-03-27T18:54:58.232084Z","iopub.execute_input":"2023-03-27T18:54:58.233410Z","iopub.status.idle":"2023-03-27T18:56:52.569332Z","shell.execute_reply.started":"2023-03-27T18:54:58.233335Z","shell.execute_reply":"2023-03-27T18:56:52.567511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport glob\nimport pandas as pd\nfrom concurrent.futures import ProcessPoolExecutor\n\nstart = time.time()\n\ndef process_subdir(subdir):\n    spec_list = []\n    files = glob.glob('/kaggle/input/birdclef-2023/train_audio/' + subdir + '/*.ogg')\n    for file in files:\n        spec_list.extend(create_power_spectrograms(file, hp['nperblock']))\n    results = [(subdir, spec) for spec in spec_list]\n    print(len(results))\n    return results\n\n\n# Create DataFrame to store the spectrograms and the corresponding bird names\n\n# Get the subdirectories\nsubdirs = []\nfor dirname, subdir_list, _ in os.walk('/kaggle/input/birdclef-2023/train_audio/'):\n    subdirs.extend(subdir_list)\n    break\n\n# Process the subdirectories in parallel using a ProcessPoolExecutor\nnum_cores = multiprocessing.cpu_count()\nwith ProcessPoolExecutor(max_workers=num_cores) as executor:\n    results = list(executor.map(process_subdir, subdirs[:4]))\nflattened_results = list(itertools.chain(*results))\n\n# Fill the DataFrame with spectrograms of all audiofiles\ndf = pd.DataFrame(flattened_results, columns=['Bird_Name', 'Spectrogram'])\n\nend = time.time()\nprint(end-start)","metadata":{"execution":{"iopub.status.busy":"2023-03-27T18:56:52.572663Z","iopub.execute_input":"2023-03-27T18:56:52.573370Z","iopub.status.idle":"2023-03-27T18:57:11.354246Z","shell.execute_reply.started":"2023-03-27T18:56:52.573303Z","shell.execute_reply":"2023-03-27T18:57:11.352643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head(50)","metadata":{"execution":{"iopub.status.busy":"2023-03-27T17:52:29.130933Z","iopub.execute_input":"2023-03-27T17:52:29.131363Z","iopub.status.idle":"2023-03-27T17:52:29.382131Z","shell.execute_reply.started":"2023-03-27T17:52:29.131319Z","shell.execute_reply":"2023-03-27T17:52:29.380724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get spectrograms as training data\nX = df['Spectrogram']\n# Get bird names as labels\ny = df['Bird_Name']\n\n# Prepare data for the model\nle = LabelEncoder()\ny_encoded = le.fit_transform(y)\n\nX_stack = np.stack(X.values)\nX_stack = X_stack.reshape(X_stack.shape[0], X_stack.shape[1], X_stack.shape[2], 1)  # Reshape for CNN input\ny_categorical = to_categorical(y_encoded)\n# Split the data into training and test data\nX_train, X_valid, y_train, y_valid = train_test_split(X_stack, y_categorical, test_size=0.2, random_state=0)","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-03-27T14:10:34.712773Z","iopub.execute_input":"2023-03-27T14:10:34.713242Z","iopub.status.idle":"2023-03-27T14:10:35.810376Z","shell.execute_reply.started":"2023-03-27T14:10:34.713201Z","shell.execute_reply":"2023-03-27T14:10:35.808531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Implementation of a model goes here ...\n\n# Compile the model\nmodel.compile(loss='categorical_crossentropy', optimizer=Adam(learning_rate=0.001), metrics=['accuracy'])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T09:11:05.450978Z","iopub.status.idle":"2023-03-21T09:11:05.451716Z","shell.execute_reply":"2023-03-21T09:11:05.451475Z","shell.execute_reply.started":"2023-03-21T09:11:05.451451Z"},"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the model\nmodel.fit(X_train, y_train, batch_size=32, epochs=50, verbose=1, validation_data=(X_valid, y_valid))\n\n# Evaluate the model\nscore = model.evaluate(X_valid, y_valid, verbose=0)\nprint('Test loss:', score[0])\nprint('Test accuracy:', score[1])","metadata":{"execution":{"iopub.status.busy":"2023-03-21T09:11:05.453009Z","iopub.status.idle":"2023-03-21T09:11:05.453711Z","shell.execute_reply":"2023-03-21T09:11:05.453500Z","shell.execute_reply.started":"2023-03-21T09:11:05.453474Z"},"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[]},"execution_count":null,"outputs":[]}]}