{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%pip install audiomentations\n%pip install pillow\n%pip install tqdm","metadata":{"execution":{"iopub.status.busy":"2024-06-09T06:16:11.471622Z","iopub.execute_input":"2024-06-09T06:16:11.472129Z","iopub.status.idle":"2024-06-09T06:17:00.164601Z","shell.execute_reply.started":"2024-06-09T06:16:11.472079Z","shell.execute_reply":"2024-06-09T06:17:00.162391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\nId=[]\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-08T03:47:57.928417Z","iopub.execute_input":"2024-06-08T03:47:57.928793Z","iopub.status.idle":"2024-06-08T03:48:02.699133Z","shell.execute_reply.started":"2024-06-08T03:47:57.928761Z","shell.execute_reply":"2024-06-08T03:48:02.698072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np \nfrom sklearn.decomposition import PCA\nimport glob\nfrom skimage.transform import resize\nimport torch\nimport torch.nn.functional as F\nimport gc\nfrom PIL import Image\nfrom io import BytesIO\nimport shutil\nimport zipfile\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nimport plotly.express as px\nimport librosa\nfrom IPython.display import Audio\nimport IPython.display as ipd\nimport librosa.display\nfrom pydub import AudioSegment\nfrom librosa import feature\nfrom audiomentations import Compose,AddGaussianNoise,PitchShift,HighPassFilter\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Input, Dense, Concatenate, Layer\nfrom tensorflow.keras.models import Sequential, Model","metadata":{"execution":{"iopub.status.busy":"2024-06-09T06:17:00.168722Z","iopub.execute_input":"2024-06-09T06:17:00.170266Z","iopub.status.idle":"2024-06-09T06:17:18.253449Z","shell.execute_reply.started":"2024-06-09T06:17:00.170198Z","shell.execute_reply":"2024-06-09T06:17:18.25186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"folder='birdclef-2024'\nif not os.path.exists(folder):\n    os.mkdir(folder)\nos.chdir('/kaggle/working/birdclef-2024')\nsub_folder='train_audio'\nprint(os.getcwd())\nif not os.path.exists(sub_folder):\n    os.mkdir(sub_folder)\nos.chdir('/kaggle/working/birdclef-2024/train_audio')\n","metadata":{"execution":{"iopub.status.busy":"2024-06-09T06:05:23.823398Z","iopub.execute_input":"2024-06-09T06:05:23.824662Z","iopub.status.idle":"2024-06-09T06:05:23.834041Z","shell.execute_reply.started":"2024-06-09T06:05:23.824617Z","shell.execute_reply":"2024-06-09T06:05:23.832271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for root, dirs, files in os.walk('/kaggle/input/birdclef-2024/train_audio/'): \n    for file in files:\n        path_file = os.path.join(root,file)\n        if not os.path.exists('/kaggle/working/birdclef-2024/train_audio/'+file):\n            shutil.copy2(path_file,'./')\n#meta_data.info()","metadata":{"execution":{"iopub.status.busy":"2024-06-09T04:13:55.863142Z","iopub.status.idle":"2024-06-09T04:13:55.863727Z","shell.execute_reply.started":"2024-06-09T04:13:55.863457Z","shell.execute_reply":"2024-06-09T04:13:55.86348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_data = pd.read_csv('/kaggle/input/birdclef-2024/train_metadata.csv')\nebird_taxonomy = pd.read_csv('/kaggle/input/birdclef-2024/eBird_Taxonomy_v2021.csv')\nmeta_data\nsample_submission = pd.read_csv('/kaggle/input/birdclef-2024/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-06-09T06:17:18.255165Z","iopub.execute_input":"2024-06-09T06:17:18.255985Z","iopub.status.idle":"2024-06-09T06:17:18.567451Z","shell.execute_reply.started":"2024-06-09T06:17:18.255939Z","shell.execute_reply":"2024-06-09T06:17:18.566129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def audio_waveframe(file_path):\n#     # Load the audio file\n#     audio_data, sampling_rate = librosa.load(file_path)\n#     # Calculate the duration of the audio file\n#     duration = len(audio_data) / sampling_rate\n#     # Create a time array for plotting\n#     time = np.arange(0, duration, 1/sampling_rate)\n#     # Plot the waveform\n#     plt.figure(figsize=(30, 4))\n#     plt.plot(time, audio_data, color='blue')\n#     plt.title('Audio Waveform')\n#     plt.xlabel('Time (s)')\n#     plt.ylabel('Amplitude')\n#     plot = plt.show()\n#     return plot\n\n# def spectrogram(file_path):\n#     # Compute the short-time Fourier transform (STFT)\n#     n_fft = 500  # Number of FFT points 2048\n#     hop_length = 50  # Hop length for STFT 512\n#     audio_data, sampling_rate = librosa.load(file_path)\n#     stft = librosa.stft(audio_data, n_fft=n_fft, hop_length=hop_length)\n#     # Convert the magnitude spectrogram to decibels (log scale)\n#     spectrogram = librosa.amplitude_to_db(np.abs(stft))\n#     # Plot the spectrogram\n#     plt.figure(figsize=(30, 6))\n#     librosa.display.specshow(spectrogram, sr=sampling_rate, hop_length=hop_length, x_axis='time', y_axis='linear')\n#     plt.colorbar(format='%+2.0f dB')\n#     plt.title('Spectrogram')\n#     plt.xlabel('Time (s)')\n#     plt.ylabel('Frequency (Hz)')\n#     plt.tight_layout()\n#     plot = plt.show()\n#     return plot\n# def audio_analysis(file_path):\n#     aw = audio_waveframe(file_path)\n#     spg = spectrogram(file_path)\n#     return aw, spg","metadata":{"execution":{"iopub.status.busy":"2024-06-08T03:51:38.764538Z","iopub.execute_input":"2024-06-08T03:51:38.764889Z","iopub.status.idle":"2024-06-08T03:51:38.769781Z","shell.execute_reply.started":"2024-06-08T03:51:38.764862Z","shell.execute_reply":"2024-06-08T03:51:38.768736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# audio_analysis('/kaggle/input/birdclef-2024/train_audio/litegr/XC518525.ogg')\n# Audio('/kaggle/input/birdclef-2024/train_audio/litegr/XC518525.ogg')","metadata":{"execution":{"iopub.status.busy":"2024-06-08T03:51:38.771497Z","iopub.execute_input":"2024-06-08T03:51:38.772272Z","iopub.status.idle":"2024-06-08T03:51:38.786554Z","shell.execute_reply.started":"2024-06-08T03:51:38.772233Z","shell.execute_reply":"2024-06-08T03:51:38.785401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"augment=Compose([\n    AddGaussianNoise(min_amplitude=0.01,max_amplitude=0.04,p=1),\n    PitchShift(min_semitones=-4,max_semitones=4,p=1),\n    HighPassFilter(min_cutoff_freq=20.0, max_cutoff_freq=2400.0,p=1)\n]\n)","metadata":{"execution":{"iopub.status.busy":"2024-06-08T03:51:38.788087Z","iopub.execute_input":"2024-06-08T03:51:38.7885Z","iopub.status.idle":"2024-06-08T03:51:38.796914Z","shell.execute_reply.started":"2024-06-08T03:51:38.788463Z","shell.execute_reply":"2024-06-08T03:51:38.795912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# aud,sr=librosa.load('/kaggle/input/birdclef-2024/train_audio/litegr/XC518525.ogg')\n# aud=augment(aud,sr)\n# n_fft = 500  # Number of FFT points 2048\n# hop_length = 50  # Hop length for STFT 512 \n# stft = librosa.stft(aud, n_fft=n_fft, hop_length=hop_length)\n# # Convert the magnitude spectrogram to decibels (log scale)\n# spectrogram = librosa.amplitude_to_db(np.abs(stft))\n# # Plot the spectrogram\n# plt.figure(figsize=(30, 6))\n# librosa.display.specshow(spectrogram, sr=sr, hop_length=hop_length, x_axis='time', y_axis='linear')\n# plt.colorbar(format='%+2.0f dB')\n# plt.title('Spectrogram')\n# plt.xlabel('Time (s)')\n# plt.ylabel('Frequency (Hz)')\n# plt.tight_layout()\n# plot = plt.show()\n# plot\n\n# # Calculate the duration of the audio file\n# duration = len(aud) / sr\n# # Create a time array for plotting\n# time = np.arange(0, duration, 1/sr)\n# # Plot the waveform\n# plt.figure(figsize=(30, 4))\n# plt.plot(time, aud, color='blue')\n# plt.title('Audio Waveform')\n# plt.xlabel('Time (s)')\n# plt.ylabel('Amplitude')\n# plo = plt.show()\n# plo\n# Audio(data=aud,rate=sr)","metadata":{"execution":{"iopub.status.busy":"2024-06-08T03:51:38.80031Z","iopub.execute_input":"2024-06-08T03:51:38.800652Z","iopub.status.idle":"2024-06-08T03:51:38.810401Z","shell.execute_reply.started":"2024-06-08T03:51:38.80062Z","shell.execute_reply":"2024-06-08T03:51:38.80923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef audioCut(filename):\n    song=AudioSegment.from_ogg(filename);\n    duration = song.duration_seconds\n    if(duration<5):\n        newd=1000-duration*1000\n        \n        end_sec_segment = AudioSegment.silent(duration=newd)\n        finalaudio=song+end_sec_segment;\n        finalaudio.export(filename,format=\"ogg\")\n    else:\n        five_seconds=5*1000\n        first_five_seconds=song[:five_seconds]\n        first_five_seconds.export(filename,format=\"ogg\")\n    \n# audioCut('./XC468456.ogg')\n# features('./XC468456.ogg')","metadata":{"execution":{"iopub.status.busy":"2024-06-09T02:47:25.361223Z","iopub.execute_input":"2024-06-09T02:47:25.361602Z","iopub.status.idle":"2024-06-09T02:47:25.37212Z","shell.execute_reply.started":"2024-06-09T02:47:25.361573Z","shell.execute_reply":"2024-06-09T02:47:25.369898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***feature engineering***","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mfcc(audio,sr):\n    return librosa.feature.mfcc(y=audio, sr=sr, n_mfcc=13)\ndef chroma(audio,sr):\n    return feature.chroma_stft(y=audio,sr=sr,n_chroma=12)\ndef torintz(audio,sr):\n    librosa.feature.tonnetz(y=audio,sr=sr,)\ndef logMelSpec(filename,png_filename):\n    y,sr=librosa.load(filename)\n    mel_bins = 224 # Number of mel bands\n    fmin = 0\n    fmax= None\n    Mel_spectrogram = librosa.feature.melspectrogram(y=y, sr=sr, n_fft=1024, hop_length=int(y.size/224), win_length=1024, window='hann', n_mels = mel_bins, power=2.0)\n    mel_spectrogram_db = librosa.power_to_db(Mel_spectrogram, ref=np.max)\n#     librosa.display.specshow(mel_spectrogram_db, sr=sr, x_axis='time', y_axis='mel',hop_length=int(y.size/224))\n#     plt.colorbar(format='%+2.0f dB')\n#     plt.title('Log Mel spectrogram')\n#     plt.tight_layout()\n#     plt.show()\n    fig, ax = plt.subplots(1, figsize=(2.24,2.24))\n    fig.subplots_adjust(top=1.0, bottom=0, right=1.0, left=0, hspace=0, wspace=0)\n#     librosa.display.specshow(mel_spectrogram_db,ax=ax, sr=sr, x_axis='time', y_axis='mel',hop_length=int(y.size/224))\n    ax.axes.get_xaxis().set_visible(False)\n    ax.axes.get_yaxis().set_visible(False)\n    ax.set_frame_on(False)\n    ax.set_xlabel(None)\n    ax.set_ylabel(None)\n    ax.set_axis_off()\n    librosa.display.specshow(mel_spectrogram_db,ax=ax, sr=sr, x_axis='time', y_axis='mel',hop_length=int(y.size/224))\n    return fig\n","metadata":{"execution":{"iopub.status.busy":"2024-06-08T03:51:38.827981Z","iopub.execute_input":"2024-06-08T03:51:38.828312Z","iopub.status.idle":"2024-06-08T03:51:38.839833Z","shell.execute_reply.started":"2024-06-08T03:51:38.828268Z","shell.execute_reply":"2024-06-08T03:51:38.838693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Id=[]\nfor dirname, _, filenames in os.walk('/kaggle/working/birdclef-2024/train_audio'):\n    for filename in filenames:\n        Id.append(os.path.join(dirname, filename))\ntrain=pd.DataFrame()\ntrain=train.assign(filename=Id)\n\n\ntrain['label']=meta_data['filename']\n# train['label']=train['label'].str.replace('/kaggle/working/birdclef-2024/train_audio/','')\n\n\ntrain['file'] = train['label'].str.split('/').str[1]\ntrain['label'] = train['label'].str.split('/').str[0]\n# train['filename']='/kaggle/working/birdclef-2024/train_audio/'+train['label']+'/'+train['file']\ntrain.tail()","metadata":{"execution":{"iopub.status.busy":"2024-06-09T06:17:18.570522Z","iopub.execute_input":"2024-06-09T06:17:18.571602Z","iopub.status.idle":"2024-06-09T06:17:18.888315Z","shell.execute_reply.started":"2024-06-09T06:17:18.571552Z","shell.execute_reply":"2024-06-09T06:17:18.886941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.tail()","metadata":{"execution":{"iopub.status.busy":"2024-06-09T06:06:32.856211Z","iopub.execute_input":"2024-06-09T06:06:32.856701Z","iopub.status.idle":"2024-06-09T06:06:32.870041Z","shell.execute_reply.started":"2024-06-09T06:06:32.856659Z","shell.execute_reply":"2024-06-09T06:06:32.868688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['filename'][0]","metadata":{"execution":{"iopub.status.busy":"2024-06-08T03:51:39.005969Z","iopub.execute_input":"2024-06-08T03:51:39.006292Z","iopub.status.idle":"2024-06-08T03:51:39.014411Z","shell.execute_reply.started":"2024-06-08T03:51:39.006265Z","shell.execute_reply":"2024-06-08T03:51:39.013214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"","metadata":{"execution":{"iopub.status.busy":"2024-05-27T04:36:26.868177Z","iopub.execute_input":"2024-05-27T04:36:26.868443Z","iopub.status.idle":"2024-05-27T04:36:30.355466Z","shell.execute_reply.started":"2024-05-27T04:36:26.86842Z","shell.execute_reply":"2024-05-27T04:36:30.354475Z"}}},{"cell_type":"code","source":"# def save_mel_spectrogram():\n#     \"\"\"\n#     Extracts Mel spectrograms from a list of audio files and saves them as PNG images with specified dimensions.\n\n#     Parameters:\n#     audio_files (list of str): List of paths to the audio files.\n#     output_dir (str): Directory where the PNG images will be saved.\n#     img_height (int): Desired height of the output image.\n#     img_width (int): Desired width of the output image.\n\n#     Returns:\n#     None\n#     \"\"\"\n#     gc.collect()\n#     spectrogram_dict = {}\n#     from tensorflow.keras.preprocessing.image import img_to_array\n#     os.chdir('/kaggle/working/birdclef-2024')\n#     if not os.path.exists('numpy'):\n#         os.mkdir('numpy')\n#     output='/kaggle/working/birdclef-2024'\n#     for i, file_path in enumerate(train['filename']):\n#         # Load the audio file\n#         y, sr = librosa.load(file_path, sr=None)\n        \n#         # Compute the Mel spectrogram\n#         S = librosa.feature.melspectrogram(y=y, sr=sr, n_mels=128)\n#         S_dB = librosa.power_to_db(S, ref=np.max)\n\n#         # Normalize the spectrogram to the range [0, 1]\n#         S_dB_normalized = (S_dB - S_dB.min()) / (S_dB.max() - S_dB.min())\n\n#         # Resize the spectrogram to the desired dimensions\n#         S_dB_resized = resize(S_dB_normalized, (124, 124), anti_aliasing=True)\n\n#         # Convert the single-channel spectrogram to three channels (RGB)\n#         S_dB_rgb = np.stack([S_dB_resized] * 3, axis=-1)\n\n#         # Add to dictionary\n#         spectrogram_dict[file_path] = S_dB_rgb\n\n#     # Save the dictionary as a pickle file\n#     with open(output, 'wb') as f:\n#         pickle.dump(spectrogram_dict, f)\n\n            \n\n# save_mel_spectrogram()","metadata":{"execution":{"iopub.status.busy":"2024-06-08T03:51:39.016112Z","iopub.execute_input":"2024-06-08T03:51:39.01644Z","iopub.status.idle":"2024-06-08T03:51:39.025534Z","shell.execute_reply.started":"2024-06-08T03:51:39.016414Z","shell.execute_reply":"2024-06-08T03:51:39.024536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from concurrent.futures import ThreadPoolExecutor","metadata":{"execution":{"iopub.status.busy":"2024-06-09T06:17:18.889851Z","iopub.execute_input":"2024-06-09T06:17:18.890209Z","iopub.status.idle":"2024-06-09T06:17:18.896111Z","shell.execute_reply.started":"2024-06-09T06:17:18.89018Z","shell.execute_reply":"2024-06-09T06:17:18.894741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\n\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\ndef process_single_file(file_path, img_height, img_width):\n    # Load the audio file\n    y, sr = librosa.load(file_path, sr=None)\n    \n    # Compute the Mel spectrogram\n    S = librosa.feature.melspectrogram(y=y, sr=sr, n_mels=128)\n    S_dB = librosa.power_to_db(S, ref=np.max)\n\n    # Normalize the spectrogram to the range [0, 1]\n    S_dB_normalized = (S_dB - S_dB.min()) / (S_dB.max() - S_dB.min())\n\n    # Convert to PyTorch tensor and move to GPU\n    S_dB_tensor = torch.tensor(S_dB_normalized, dtype=torch.float32).unsqueeze(0).to(device)\n\n    # Resize the spectrogram using bilinear interpolation on GPU\n    S_dB_resized = F.interpolate(S_dB_tensor.unsqueeze(0), size=(img_height, img_width), mode='bilinear', align_corners=False).squeeze()\n\n    # Move back to CPU and convert to NumPy array\n    S_dB_resized = S_dB_resized.cpu().numpy()\n\n    # Convert the single-channel spectrogram to three channels (RGB)\n    S_dB_rgb = np.stack([S_dB_resized] * 3, axis=-1)\n\n    return file_path, S_dB_rgb\n\ndef process_audio_files(img_height=224, img_width=224, batch_size=16):\n    \"\"\"\n    Process audio files and return a dictionary of spectrograms.\n    \"\"\"\n    spectrogram_dict = {}\n    os.chdir('/kaggle/working/birdclef-2024')\n\n    filenames = train['filename']\n    total_files = len(filenames)\n\n    for start in range(0, total_files, batch_size):\n        end = min(start + batch_size, total_files)\n        batch_filenames = filenames[start:end]\n\n        # Use ThreadPoolExecutor for parallel processing\n        with ThreadPoolExecutor() as executor:\n            futures = []\n            for file_path in batch_filenames:\n                futures.append(executor.submit(process_single_file, file_path, img_height, img_width))\n            \n            for future in futures:\n                file_path, S_dB_rgb = future.result()\n                spectrogram_dict[file_path] = S_dB_rgb\n\n        # Clear memory after each batch\n        del futures\n        gc.collect()\n\n        print(f\"Processed {end} / {total_files}\")\n\n    return spectrogram_dict\n\n# Usage\nspec = process_audio_files()","metadata":{"execution":{"iopub.status.busy":"2024-06-09T06:17:38.905371Z","iopub.execute_input":"2024-06-09T06:17:38.906733Z","iopub.status.idle":"2024-06-09T07:14:49.185995Z","shell.execute_reply.started":"2024-06-09T06:17:38.906687Z","shell.execute_reply":"2024-06-09T07:14:49.184498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%pip install h5py","metadata":{"execution":{"iopub.status.busy":"2024-06-09T07:38:52.942444Z","iopub.execute_input":"2024-06-09T07:38:52.942988Z","iopub.status.idle":"2024-06-09T07:39:10.779824Z","shell.execute_reply.started":"2024-06-09T07:38:52.942949Z","shell.execute_reply":"2024-06-09T07:39:10.778068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame.from_dict(spec, orient='index', columns=['spectrogram'])\n\n# Reset index to have filenames as a column\ndf.reset_index(inplace=True)\ndf.rename(columns={'index': 'filename'}, inplace=True)\n\n# Display DataFrame\nprint(df.head())","metadata":{"execution":{"iopub.status.busy":"2024-06-09T07:43:07.492642Z","iopub.execute_input":"2024-06-09T07:43:07.493194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from tensorflow.keras.preprocessing.image import img_to_array\n# os.chdir('/kaggle/working/birdclef-2024')\n# if not os.path.exists('mel_spectogram'):\n#     os.mkdir('mel_spectogram')\n# os.chdir('/kaggle/working/birdclef-2024/mel_spectogram')\n# spectogram={}\n# for i in train['filename'] :\n#     audioCut(i)\n#     png_filename=i.split('/')[-1]\n#     a=logMelSpec(i,png_filename)\n#     a=img_to_arr(a)\n#     png_filename=png_filename.split(\".\")[0]+\".png\"\n#     spectogram[png_filename]=a\n#     gc.collect()\n# audioCut('/kaggle/working/birdclef-2024/train_audio/XC803673.ogg')\n# logMelSpec('/kaggle/working/birdclef-2024/train_audio/XC803673.ogg','XC803673')","metadata":{"execution":{"iopub.status.busy":"2024-06-08T05:05:33.4442Z","iopub.execute_input":"2024-06-08T05:05:33.444706Z","iopub.status.idle":"2024-06-08T05:05:33.457083Z","shell.execute_reply.started":"2024-06-08T05:05:33.444664Z","shell.execute_reply":"2024-06-08T05:05:33.455366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from matplotlib import image as mpimg\n# img=mpimg.imread('/kaggle/working/birdclef-2024/mel_spectogram/XC803673.png')\n# print(img.shape)\n# plt.imshow(img)\n\n    ","metadata":{"execution":{"iopub.status.busy":"2024-06-08T05:05:33.458676Z","iopub.execute_input":"2024-06-08T05:05:33.459143Z","iopub.status.idle":"2024-06-08T05:05:33.475992Z","shell.execute_reply.started":"2024-06-08T05:05:33.459099Z","shell.execute_reply":"2024-06-08T05:05:33.474798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# resnet and efficientnet feature extraction","metadata":{}},{"cell_type":"code","source":"from keras.applications.resnet50 import ResNet50,preprocess_input\nfrom tensorflow import keras\nfrom keras.layers import Dense,BatchNormalization\nfrom keras import Sequential\n\nfrom tensorflow.keras.preprocessing.image import load_img,img_to_array\nfrom tensorflow.keras.models import Model\nmodel=ResNet50()\nmodel=Model(inputs=model.inputs,outputs=model.layers[-3].output)\nprint(model.summary())","metadata":{"execution":{"iopub.status.busy":"2024-06-09T04:15:10.594792Z","iopub.execute_input":"2024-06-09T04:15:10.595317Z","iopub.status.idle":"2024-06-09T04:15:18.890931Z","shell.execute_reply.started":"2024-06-09T04:15:10.595278Z","shell.execute_reply":"2024-06-09T04:15:18.889804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.applications.efficientnet import EfficientNetB0\nfrom keras.applications.efficientnet import preprocess_input as preprocess_input2\nfrom tensorflow.keras.preprocessing.image import load_img,img_to_array\nfrom tensorflow.keras.models import Model\nmodel2=EfficientNetB0(weights='imagenet')\nmodel2=Model(inputs=model2.inputs,outputs=model2.layers[-4].output)\nmodel2.summary()","metadata":{"execution":{"iopub.status.busy":"2024-06-09T04:15:44.532275Z","iopub.execute_input":"2024-06-09T04:15:44.532765Z","iopub.status.idle":"2024-06-09T04:15:49.205811Z","shell.execute_reply.started":"2024-06-09T04:15:44.53273Z","shell.execute_reply":"2024-06-09T04:15:49.204669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tqdm","metadata":{"execution":{"iopub.status.busy":"2024-06-09T04:20:01.519553Z","iopub.execute_input":"2024-06-09T04:20:01.519981Z","iopub.status.idle":"2024-06-09T04:20:01.525897Z","shell.execute_reply.started":"2024-06-09T04:20:01.519948Z","shell.execute_reply":"2024-06-09T04:20:01.524329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pretrain_feature(filename):\n    img=spec[filename]\n    f1 = model.predict(np.expand_dims(img, axis=0))\n    f2 = model2.predict(np.expand_dims(img, axis=0))\n    a = np.concatenate((f1, f2), axis=-1)\n    a=keras.layers.GlobalAveragePooling2D()(a)\n    return a\n\ndef feature1():\n    d={}\n    i=0\n    for f in train['filename'] :\n        if(i%1000):\n            print(i)\n        a=pretrain_feature(f)\n        d[f]=a;\n        i=i+1\n    return d;\n    ","metadata":{"execution":{"iopub.status.busy":"2024-06-09T04:21:51.492715Z","iopub.execute_input":"2024-06-09T04:21:51.493175Z","iopub.status.idle":"2024-06-09T04:21:51.502955Z","shell.execute_reply.started":"2024-06-09T04:21:51.49314Z","shell.execute_reply":"2024-06-09T04:21:51.500942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-06-09T05:48:37.541539Z","iopub.execute_input":"2024-06-09T05:48:37.54205Z","iopub.status.idle":"2024-06-09T05:48:37.552315Z","shell.execute_reply.started":"2024-06-09T05:48:37.542013Z","shell.execute_reply":"2024-06-09T05:48:37.55073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pretrain_feature_batch(filenames):\n    images = np.array([spec[filename] for filename in filenames])\n    f1 = model.predict(images)\n    f2 = model2.predict(images)\n    a = np.concatenate((f1, f2), axis=-1)\n    a = GlobalAveragePooling2D()(a)\n    return {filename: feature for filename, feature in zip(filenames, a)}\n\ndef feature1(batch_size=32):\n    d = {}\n    filenames = train['filename']\n    total_files = len(filenames)\n    \n    with ThreadPoolExecutor() as executor:\n        futures = []\n        for i in range(0, total_files, batch_size):\n            batch_filenames = filenames[i:i+batch_size]\n            futures.append(executor.submit(pretrain_feature_batch, batch_filenames))\n        \n        for future in futures:\n            d.update(future.result())\n\n    return d\n\n# Usage\nfeatures_dict = feature1()","metadata":{"execution":{"iopub.status.busy":"2024-06-09T05:48:42.699306Z","iopub.execute_input":"2024-06-09T05:48:42.699721Z","iopub.status.idle":"2024-06-09T05:58:27.457788Z","shell.execute_reply.started":"2024-06-09T05:48:42.699689Z","shell.execute_reply":"2024-06-09T05:58:27.453145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature2():\n    d={}\n    for file in train['filename']:\n        audio,sr=librosa.load(filename)\n        f1=mfcc(audio,sr)\n        f2=chroma(audio,sr)\n        f3=torintz(audio,sr)\n        d[file]=np.concatenate((f1,f2,f3));\n    return d;\nf1=feature1()\ncf1_train=pd.DataFrame(list(f1.items()), columns=['label', 'f1'])","metadata":{"execution":{"iopub.status.busy":"2024-06-09T04:21:54.145451Z","iopub.execute_input":"2024-06-09T04:21:54.14588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f2=feature2()\nf2_train=pd.DataFrame(list(f2.items()),columns=['filename','f2'])\ndf=pd.concat([f1, f2], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-06-08T05:05:46.855091Z","iopub.execute_input":"2024-06-08T05:05:46.855428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # fmodel=Sequential()\n# # fmodel.add(Dense(512,input_shape=(3340,),activation='relu'))\n# # fmodel.add(BatchNormalization())\n# # a=fmodel.predict(arr)\n# count=0;\n# for dirname, _, filenames in os.walk('/kaggle/working/birdclef-2024/mel_spectogram'):\n#     for filename in filenames:\n#         count=count+1;\n# print(count)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# pca for features dimensionality reduction","metadata":{}},{"cell_type":"code","source":"pca=PCA(n_components=512)\npca.fit(f2)\nf2=pca.transform(f2)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# transformer encoder","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# label encoding and splitting","metadata":{}},{"cell_type":"code","source":"original_labels = df['label'].unique()\n\n# Fitting the label encoder to the list of labels from the metadata\nlabel_encoder = LabelEncoder()\nlabel_encoder.fit(original_labels)\n\n# encoding the primary_labels\nencoded_labels = label_encoder.transform(df['label'])\ndf['encoded_label'] = encoded_labels\n\n\n\n#train test split\nX, Y = df.drop(columns = ['label', 'encoded_label','filename']), df['encoded_label']\nX.head()\nsampler = RandomOverSampler(random_state = 0)\n\n# fit oversampler and resample\nX_resampled, Y_resampled = sampler.fit_resample(X, Y)\n\ndf_train_resampled = pd.DataFrame(X_resampled)\ndf_train_resampled['encoded_label'] = Y_resampled\n\nX_train, X_test, y_train, y_test = train_test_split(X_resampled, Y_resampled, test_size = 0.2, random_state = 12)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# model for classification","metadata":{}},{"cell_type":"code","source":"\n# Define a custom layer for concatenation\nclass ConcatenateLayer(Layer):\n    def call(self, inputs):\n        return Concatenate()([inputs[0], inputs[1]])\n\n# Define input dimension and number of classes\ninput_dim = 3328 # example input dimension\nnum_classes = 184 # example number of classes\n\n# Define Model 1 using Sequential API\nmodel_1 = Sequential([\n    Dense(1024, activation='relu', input_shape=(input_dim,)),\n    Dense(512, activation='relu'),\n])\nconcatenated = Concatenate()([output_1, feature_vector])\n\n# Define Model 2 using Sequential API\nmodel_2 = Sequential([\n    Dense(603, activation='relu', input_shape=(1024,)),  # input_dim + output of Model 1\n    BatchNormalization()\n    Dense(364, activation='relu')\n    BatchNormalization()\n    Dense(182, activation='sigmoid')\n])\n\n# Define the combined model using the Functional API to handle concatenation\ninput_1 = Input(shape=(input_dim,))\noutput_1 = model_1(input_1)\nconcatenated = Concatenate()([input_1, output_1])\noutput_2 = model_2(concatenated)\n\ncombined_model = Model(inputs=[input_1, feature_vector], outputs=output_2)\n\n# Compile the combined model\ncombined_model.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n\n# Train the combined model\ncombined_model.fit([x_train,feature_vector], y_train, epochs=10, batch_size=32, validation_data=(x_val, y_val))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_dict_from_pickle(filename):\n    with open(filename, 'rb') as pickle_file:\n        dictionary = pickle.load(pickle_file)\n    return dictionary\n\n# Example usage\nloaded_spec = load_dict_from_pickle('/kaggle/working/birdclef-2024/spec.pkl')\n","metadata":{"execution":{"iopub.status.busy":"2024-06-09T07:46:23.781682Z","iopub.execute_input":"2024-06-09T07:46:23.782196Z","iopub.status.idle":"2024-06-09T07:46:24.435934Z","shell.execute_reply.started":"2024-06-09T07:46:23.782155Z","shell.execute_reply":"2024-06-09T07:46:24.434037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}