{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30716,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Import Required Libraries**","metadata":{}},{"cell_type":"code","source":"!pip install soundfile\n!pip install librosa soundfile","metadata":{"execution":{"iopub.status.busy":"2024-06-05T17:49:38.750045Z","iopub.execute_input":"2024-06-05T17:49:38.750488Z","iopub.status.idle":"2024-06-05T17:50:06.136177Z","shell.execute_reply.started":"2024-06-05T17:49:38.750457Z","shell.execute_reply":"2024-06-05T17:50:06.135039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport librosa\nimport os\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout, Activation\nfrom tensorflow.keras.utils import to_categorical\nimport tensorflow as tf\nimport resampy\nimport soundfile as sf\nfrom tensorflow.keras.layers import Input, Conv1D, MaxPooling1D, Flatten, Dense, Dropout\nfrom tensorflow.keras.optimizers import Adam","metadata":{"execution":{"iopub.status.busy":"2024-06-05T17:50:14.509515Z","iopub.execute_input":"2024-06-05T17:50:14.509905Z","iopub.status.idle":"2024-06-05T17:50:14.516463Z","shell.execute_reply.started":"2024-06-05T17:50:14.509879Z","shell.execute_reply":"2024-06-05T17:50:14.515558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --upgrade --force-reinstall resampy\n","metadata":{"execution":{"iopub.status.busy":"2024-06-05T17:13:28.496019Z","iopub.execute_input":"2024-06-05T17:13:28.496370Z","iopub.status.idle":"2024-06-05T17:13:55.075402Z","shell.execute_reply.started":"2024-06-05T17:13:28.496346Z","shell.execute_reply":"2024-06-05T17:13:55.074447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Num GPUs Available: \", len(tf.config.experimental.list_physical_devices('GPU')))\nif tf.config.experimental.list_physical_devices('GPU'):\n    print('Using GPU:', tf.test.gpu_device_name())\nelse:\n    print(\"No GPU found!\")","metadata":{"execution":{"iopub.status.busy":"2024-06-05T17:48:20.844921Z","iopub.execute_input":"2024-06-05T17:48:20.845869Z","iopub.status.idle":"2024-06-05T17:48:20.853032Z","shell.execute_reply.started":"2024-06-05T17:48:20.845836Z","shell.execute_reply":"2024-06-05T17:48:20.851901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Load and Explore the Dataset**","metadata":{}},{"cell_type":"code","source":"metadata = pd.read_csv('/kaggle/input/birdclef-2024/train_metadata.csv')\nprint(metadata.head())\n\nprint(metadata['primary_label'].value_counts())\n","metadata":{"execution":{"iopub.status.busy":"2024-06-05T17:50:19.171520Z","iopub.execute_input":"2024-06-05T17:50:19.172285Z","iopub.status.idle":"2024-06-05T17:50:19.295401Z","shell.execute_reply.started":"2024-06-05T17:50:19.172255Z","shell.execute_reply":"2024-06-05T17:50:19.294461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Audio Augmentation Function**","metadata":{}},{"cell_type":"code","source":"# def add_noise(data, noise_factor=0.005):\n#     noise = np.random.randn(len(data))\n#     augmented_data = data + noise_factor * noise\n#     # Cast back to same data type\n#     augmented_data = augmented_data.astype(type(data[0]))\n#     return augmented_data\n\n# def time_shift(data, sampling_rate, shift_max=2):\n#     shift = np.random.randint(sampling_rate * shift_max)\n#     direction = np.random.randint(0, 2)\n#     if direction == 1:\n#         shift = -shift\n#     augmented_data = np.roll(data, shift)\n#     # Set to zero the samples that have been rolled\n#     if shift > 0:\n#         augmented_data[:shift] = 0\n#     else:\n#         augmented_data[shift:] = 0\n#     return augmented_data\n\n# # def time_stretch(data, rate=0.8):\n# #     # rate less than 1.0 makes the data faster, greater than 1.0 makes it slower\n# #     return librosa.effects.time_stretch(data, rate)\n\n# def time_stretch(audio, rate):\n#     n_samples = int(len(audio) / rate)\n#     audio_stretched = np.zeros(n_samples)\n#     for i in range(n_samples):\n#         audio_stretched[i] = audio[int(i * rate) % len(audio)]\n#     return audio_stretched\n\n# # def pitch_shift(data, sampling_rate, shift_max=5):\n# #     shift = np.random.randint(-shift_max, shift_max)\n# #     return librosa.effects.pitch_shift(data, sampling_rate, shift)\n\n# def pitch_shift(audio, sr, min_semitones=-5, max_semitones=5):\n#         # Choose a random number of semitones to shift within the range\n#     n_steps = np.random.randint(min_semitones, max_semitones + 1)\n\n#     # Calculate the pitch shifting factor\n#     factor = 2 ** (n_steps / 12.0)\n\n#     # Calculate the new length of the sample\n#     new_length = int(len(audio) / factor)\n\n#     # Create a new sample array with the new length\n#     new_sample = np.zeros(new_length)\n\n#     # Resample the original audio to the new length\n#     for i in range(new_length):\n#         index = int(i * factor)\n#         if index < len(audio):\n#             new_sample[i] = audio[index]\n\n#     # To maintain the original length, we need to interpolate back to the original length\n#     original_length = len(audio)\n#     resampled_audio = np.interp(np.linspace(0, new_length, original_length), np.arange(new_length), new_sample)\n\n#     return resampled_audio\n\n# # def extract_features(file_name):\n# #     try:\n# #         audio, sample_rate = sf.read(file_name)\n# #         print(\"Loaded using PySoundFile.\")\n# #                 # Ensuring audio is mono\n# #         if audio.ndim > 1:\n# #             audio = audio.mean(axis=1)\n# #         # Apply augmentations\n# #         audio = add_noise(audio)\n# #         audio = time_shift(audio, sample_rate)\n# #         audio = time_stretch(audio, rate=np.random.uniform(0.8, 1.2))\n# #         audio = pitch_shift(audio, sample_rate)\n# #         mfccs = librosa.feature.mfcc(y=audio, sr=sample_rate, n_mfcc=40)\n# #         mfccs_scaled = np.mean(mfccs.T, axis=0)\n# #     except Exception as e:\n# #         print(\"Error encountered while parsing file: \", file_name)\n# #         return None \n# #     return mfccs_scaled\n\n# def extract_features(file_name):\n#     try:\n#         audio, sample_rate = sf.read(file_name)\n        \n#         # Ensure audio is mono\n#         if audio.ndim > 1:\n#             audio = audio.mean(axis=1)\n#             print(\"Converted stereo to mono.\")\n        \n#         # Apply augmentations\n#         audio = add_noise(audio)\n#         audio = time_shift(audio, sample_rate)\n        \n#         rate=np.random.uniform(0.8, 1.2)\n#         audio = time_stretch(audio, rate)\n        \n#         audio = pitch_shift(audio, sample_rate)\n        \n#         # Extract MFCCs\n#         mfccs = librosa.feature.mfcc(y=audio, sr=sample_rate, n_mfcc=15)\n#         mfccs_scaled = np.mean(mfccs.T, axis=0)\n        \n#     except Exception as e:\n#         print(\"Error encountered while processing file: \", file_name)\n#         print(\"Exception:\", e)\n#         return None \n    \n#     return mfccs_scaled","metadata":{"execution":{"iopub.status.busy":"2024-06-05T20:18:34.981194Z","iopub.execute_input":"2024-06-05T20:18:34.982113Z","iopub.status.idle":"2024-06-05T20:18:35.000202Z","shell.execute_reply.started":"2024-06-05T20:18:34.982081Z","shell.execute_reply":"2024-06-05T20:18:34.999111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Feature Extraction Function**","metadata":{}},{"cell_type":"code","source":"# def extract_features(file_name):\n#     try:\n#         # Using soundfile to read .ogg files\n#         audio, sample_rate = sf.read(file_name)\n#         # Ensuring audio is mono\n#         if audio.ndim > 1:\n#             audio = audio.mean(axis=1)\n#         # Calculate MFCCs\n#         mfccs = librosa.feature.mfcc(y=audio, sr=sample_rate, n_mfcc=40)\n#         mfccs_scaled = np.mean(mfccs.T, axis=0)\n#     except Exception as e:\n#         print(f\"Error encountered while parsing file: {file_name}\", e)\n#         return None \n#     return mfccs_scaled\n","metadata":{"execution":{"iopub.status.busy":"2024-06-05T17:51:57.110537Z","iopub.execute_input":"2024-06-05T17:51:57.110927Z","iopub.status.idle":"2024-06-05T17:51:57.117480Z","shell.execute_reply.started":"2024-06-05T17:51:57.110896Z","shell.execute_reply":"2024-06-05T17:51:57.116471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install audiomentations\n","metadata":{"execution":{"iopub.status.busy":"2024-06-05T20:29:40.704867Z","iopub.execute_input":"2024-06-05T20:29:40.705259Z","iopub.status.idle":"2024-06-05T20:29:53.876268Z","shell.execute_reply.started":"2024-06-05T20:29:40.705230Z","shell.execute_reply":"2024-06-05T20:29:53.874327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from audiomentations import Compose, AddGaussianNoise, TimeStretch, PitchShift, Shift\n\naugment = Compose([\n    AddGaussianNoise(min_amplitude=0.001, max_amplitude=0.015, p=0.5),\n    TimeStretch(min_rate=0.8, max_rate=1.25, p=0.5),\n    PitchShift(min_semitones=-4, max_semitones=4, p=0.5),\n    Shift(min_shift=-0.5, max_shift=0.5, p=0.5, rollover=False)  # shift without rolling over\n])\n\ndef extract_features(file_name):\n    try:\n        audio, sample_rate = sf.read(file_name)\n        \n        if audio.ndim > 1:\n            audio = np.mean(audio, axis=1)\n        \n        audio = augment(samples=audio, sample_rate=sample_rate)\n    \n        mfccs = librosa.feature.mfcc(y=audio, sr=sample_rate, n_mfcc=15)\n        mfccs_scaled = np.mean(mfccs.T, axis=0)\n    \n    except Exception as e:\n        print(\"Error encountered while processing file: \", file_name)\n        print(\"Exception:\", e)\n        return None \n    \n    return mfccs_scaled\n","metadata":{"execution":{"iopub.status.busy":"2024-06-05T20:32:27.173373Z","iopub.execute_input":"2024-06-05T20:32:27.174335Z","iopub.status.idle":"2024-06-05T20:32:27.183612Z","shell.execute_reply.started":"2024-06-05T20:32:27.174302Z","shell.execute_reply":"2024-06-05T20:32:27.182560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfile_path = '/kaggle/input/birdclef-2024/train_audio/asbfly/XC134896.ogg'\ndata = extract_features(file_path)\nif data is not None:\n    print(\"File loaded successfully!\")\nelse:\n    print(\"Failed to load the file.\")","metadata":{"execution":{"iopub.status.busy":"2024-06-05T20:32:34.621932Z","iopub.execute_input":"2024-06-05T20:32:34.622521Z","iopub.status.idle":"2024-06-05T20:32:35.600335Z","shell.execute_reply.started":"2024-06-05T20:32:34.622478Z","shell.execute_reply":"2024-06-05T20:32:35.599041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = []\nlabels = []\ncounter = 0\n\nfor index, row in metadata.iterrows():\n    file_path = os.path.join('/kaggle/input/birdclef-2024/train_audio/', row['filename'])\n    class_label = row['primary_label']\n    data = extract_features(file_path)\n    \n    if data is not None:\n        features.append(data)\n        labels.append(class_label)\n        \n    counter += 1\n    if counter % 1000 == 0:\n        print(\"Done processing {} samples.\".format(counter))\n\n\nprint(\"Total processed samples: {}\".format(counter))\n\nfeatures = np.array(features)\nlabels = np.array(labels)\n\nle = LabelEncoder()\nlabels_encoded = to_categorical(le.fit_transform(labels))","metadata":{"execution":{"iopub.status.busy":"2024-06-05T20:33:03.456837Z","iopub.execute_input":"2024-06-05T20:33:03.457642Z","iopub.status.idle":"2024-06-05T20:42:56.252008Z","shell.execute_reply.started":"2024-06-05T20:33:03.457610Z","shell.execute_reply":"2024-06-05T20:42:56.250521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(features)","metadata":{"execution":{"iopub.status.busy":"2024-06-03T21:23:18.985261Z","iopub.execute_input":"2024-06-03T21:23:18.986101Z","iopub.status.idle":"2024-06-03T21:23:18.993980Z","shell.execute_reply.started":"2024-06-03T21:23:18.986061Z","shell.execute_reply":"2024-06-03T21:23:18.992884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(labels)","metadata":{"execution":{"iopub.status.busy":"2024-06-04T05:08:55.750253Z","iopub.execute_input":"2024-06-04T05:08:55.750922Z","iopub.status.idle":"2024-06-04T05:08:55.788135Z","shell.execute_reply.started":"2024-06-04T05:08:55.750888Z","shell.execute_reply":"2024-06-04T05:08:55.787035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, x_test, y_train, y_test = train_test_split(features, labels_encoded, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-06-03T21:27:27.022536Z","iopub.execute_input":"2024-06-03T21:27:27.022920Z","iopub.status.idle":"2024-06-03T21:27:27.050443Z","shell.execute_reply.started":"2024-06-03T21:27:27.022888Z","shell.execute_reply":"2024-06-03T21:27:27.049241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# model = Sequential([\n#     Input(shape=(40, 1)),  # Define the input shape directly in the Input layer\n#     Conv1D(64, kernel_size=3, activation='relu'),\n#     MaxPooling1D(2),\n#     Conv1D(128, kernel_size=3, activation='relu'),\n#     MaxPooling1D(2),\n#     Flatten(),\n#     Dense(256, activation='relu'),\n#     Dropout(0.5),\n#     Dense(len(np.unique(labels)), activation='softmax')  # Ensure this matches the number of classes\n# ])\n\nmodel = Sequential([\n    Input(shape=(40, 1)),\n    Conv1D(128, kernel_size=3, activation='relu'),  # Increased filters\n    MaxPooling1D(2),\n    Conv1D(256, kernel_size=5, activation='relu'),  # Larger kernel size\n    MaxPooling1D(2),\n    Flatten(),\n    Dense(512, activation='relu'),\n    Dropout(0.5),\n    Dense(len(np.unique(labels)), activation='softmax')\n])\n\n# Display the model summary to check the overall structure\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-04T05:06:11.818474Z","iopub.execute_input":"2024-06-04T05:06:11.819122Z","iopub.status.idle":"2024-06-04T05:06:12.542640Z","shell.execute_reply.started":"2024-06-04T05:06:11.819093Z","shell.execute_reply":"2024-06-04T05:06:12.541407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train = x_train.reshape(-1, 40, 1)  # Reshape from (num_samples, 40) to (num_grouped_samples, 40, 1)\nx_test = x_test.reshape(-1, 40, 1)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-03T21:54:34.422495Z","iopub.execute_input":"2024-06-03T21:54:34.423061Z","iopub.status.idle":"2024-06-03T21:54:34.428268Z","shell.execute_reply.started":"2024-06-03T21:54:34.423025Z","shell.execute_reply":"2024-06-03T21:54:34.427173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nmodel.compile(optimizer=Adam(), loss='categorical_crossentropy', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2024-06-03T22:19:48.146896Z","iopub.execute_input":"2024-06-03T22:19:48.147686Z","iopub.status.idle":"2024-06-03T22:19:48.159579Z","shell.execute_reply.started":"2024-06-03T22:19:48.147652Z","shell.execute_reply":"2024-06-03T22:19:48.158729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with tf.device('/GPU:0'):\n    model.fit(x_train, y_train, batch_size=32, epochs=50, validation_data=(x_test, y_test))","metadata":{"execution":{"iopub.status.busy":"2024-06-03T22:19:51.111837Z","iopub.execute_input":"2024-06-03T22:19:51.112665Z","iopub.status.idle":"2024-06-03T22:21:24.754085Z","shell.execute_reply.started":"2024-06-03T22:19:51.112631Z","shell.execute_reply":"2024-06-03T22:21:24.753168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss, accuracy = model.evaluate(x_test, y_test, verbose=1)\nprint(f'Test accuracy: {accuracy*100:.2f}%')","metadata":{"execution":{"iopub.status.busy":"2024-06-03T22:21:26.160624Z","iopub.execute_input":"2024-06-03T22:21:26.161302Z","iopub.status.idle":"2024-06-03T22:21:26.491480Z","shell.execute_reply.started":"2024-06-03T22:21:26.161270Z","shell.execute_reply":"2024-06-03T22:21:26.490545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}