{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":25954,"databundleVersionId":2091745,"sourceType":"competition"},{"sourceId":44224,"databundleVersionId":5188730,"sourceType":"competition"},{"sourceId":33246,"databundleVersionId":3221581,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Pick a file\naudio_path = '../input/birdclef-2021/train_short_audio/banana/XC112602.ogg'\n\n# Listen to it\nimport IPython.display as ipd\nipd.Audio(audio_path)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T06:27:32.366543Z","iopub.execute_input":"2023-11-24T06:27:32.367464Z","iopub.status.idle":"2023-11-24T06:27:32.388448Z","shell.execute_reply.started":"2023-11-24T06:27:32.367426Z","shell.execute_reply":"2023-11-24T06:27:32.387475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport warnings\nwarnings.filterwarnings(action='ignore')\n\n# Librosa is the most versatile audio library for Python \n# and uses FFMPEG to load and open audio files\n# For more information visit: https://librosa.org/doc/latest/index.html\nimport librosa\n\n# Load the first 15 seconds this file using librosa\nsig, rate = librosa.load(audio_path, sr=32000, offset=None, duration=15)\n\n# The result is a 1D numpy array that conatains audio samples. \n# Take a look at the shape (seconds * sample rate == 15 * 32000 == 480000)\nprint('SIGNAL SHAPE:', sig.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T05:56:11.072517Z","iopub.execute_input":"2023-11-24T05:56:11.072858Z","iopub.status.idle":"2023-11-24T05:56:11.101814Z","shell.execute_reply.started":"2023-11-24T05:56:11.072817Z","shell.execute_reply":"2023-11-24T05:56:11.100772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport librosa.display\n\nplt.figure(figsize=(15, 5))\nlibrosa.display.waveshow(sig, sr=32000)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T05:56:11.102823Z","iopub.execute_input":"2023-11-24T05:56:11.103091Z","iopub.status.idle":"2023-11-24T05:56:11.662557Z","shell.execute_reply.started":"2023-11-24T05:56:11.103066Z","shell.execute_reply":"2023-11-24T05:56:11.661351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# First, compute the spectrogram using the \"short-time Fourier transform\" (stft)\nspec = librosa.stft(sig)\n\n# Scale the amplitudes according to the decibel scale\nspec_db = librosa.amplitude_to_db(spec, ref=np.max)\n\n# Plot the spectrogram\nplt.figure(figsize=(15, 5))\nlibrosa.display.specshow(spec_db, \n                         sr=32000, \n                         x_axis='time', \n                         y_axis='hz', \n                         cmap=plt.get_cmap('viridis'))","metadata":{"execution":{"iopub.status.busy":"2023-11-24T05:56:11.666328Z","iopub.execute_input":"2023-11-24T05:56:11.666649Z","iopub.status.idle":"2023-11-24T05:56:13.105129Z","shell.execute_reply.started":"2023-11-24T05:56:11.666621Z","shell.execute_reply":"2023-11-24T05:56:13.104037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('SPEC SHAPE:', spec_db.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T05:56:13.106541Z","iopub.execute_input":"2023-11-24T05:56:13.106914Z","iopub.status.idle":"2023-11-24T05:56:13.112189Z","shell.execute_reply.started":"2023-11-24T05:56:13.106880Z","shell.execute_reply":"2023-11-24T05:56:13.111194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Try a few window lengths (should be a power of 2)\nfor win_length in [128, 256, 512, 1024]:\n    \n    # We want 50% overlap between samples\n    hop_length = win_length // 2\n    \n    # Compute spec (win_length implicity also sets n_fft and vice versa)\n    spec = librosa.stft(sig, \n                        n_fft=win_length, \n                        hop_length=hop_length)\n    \n    # Scale to decibel scale\n    spec_db = librosa.amplitude_to_db(spec, ref=np.max)\n    \n    # Show plot\n    plt.figure(figsize=(15, 5))\n    plt.title('Window length: ' + str(win_length) + ', Shape: ' + str(spec_db.shape))\n    librosa.display.specshow(spec_db, \n                             sr=32000, \n                             hop_length=hop_length, \n                             x_axis='time', \n                             y_axis='hz', \n                             cmap=plt.get_cmap('viridis'))","metadata":{"execution":{"iopub.status.busy":"2023-11-24T05:56:13.113471Z","iopub.execute_input":"2023-11-24T05:56:13.113832Z","iopub.status.idle":"2023-11-24T05:56:16.869277Z","shell.execute_reply.started":"2023-11-24T05:56:13.113798Z","shell.execute_reply":"2023-11-24T05:56:16.868176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Desired shape of the input spectrogram\nSPEC_HEIGHT = 64\nSPEC_WIDTH = 256\n\n# Derive num_mels and hop_length from desired spec shape\n# num_mels is easy, that's just spec_height\n# hop_length is a bit more complicated\nNUM_MELS = SPEC_HEIGHT\nHOP_LENGTH = int(32000 * 5 / (SPEC_WIDTH - 1)) # sample rate * duration / spec width - 1 == 627\n\n# High- and low-pass frequencies\n# For many birds, these are a good choice\nFMIN = 500\nFMAX = 12500\n\n# Let's get all three spectrograms\nfor second in [5, 10, 15]:  \n    \n    # Get start and stop sample\n    s_start = (second - 5) * 32000\n    s_end = second * 32000\n\n    # Compute the spectrogram and apply the mel scale\n    mel_spec = librosa.feature.melspectrogram(y=sig[s_start:s_end], \n                                              sr=32000, \n                                              n_fft=1024, \n                                              hop_length=HOP_LENGTH, \n                                              n_mels=NUM_MELS, \n                                              fmin=FMIN, \n                                              fmax=FMAX)\n    \n    mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max)\n\n    # Show the spec\n    plt.figure(figsize=(15, 5))\n    plt.title('Second: ' + str(second) + ', Shape: ' + str(mel_spec_db.shape))\n    librosa.display.specshow(mel_spec_db, \n                             sr=32000, \n                             hop_length=HOP_LENGTH, \n                             x_axis='time', \n                             y_axis='mel',\n                             fmin=FMIN, \n                             fmax=FMAX, \n                             cmap=plt.get_cmap('viridis'))","metadata":{"execution":{"iopub.status.busy":"2023-11-24T05:56:16.870642Z","iopub.execute_input":"2023-11-24T05:56:16.870995Z","iopub.status.idle":"2023-11-24T05:56:18.036393Z","shell.execute_reply.started":"2023-11-24T05:56:16.870963Z","shell.execute_reply":"2023-11-24T05:56:18.035480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Traning","metadata":{}},{"cell_type":"markdown","source":"\n在這個筆記本中，我們將訓練我們的第一個模型並將這個模型應用到一個音境中。為了使執行時間盡可能短，我們將保持訓練樣本、物種和音境的數量最小。請記住，這只是一個示範實現，請隨意探索您自己的工作流程。\n\n這是我們將要涵蓋的步驟：\n\n1. 選擇我們想用於訓練的音訊文件。\n2. 從這些文件中提取頻譜圖並將它們保存在一個工作目錄中。\n3. 加載所選樣本到一個大型的內存數據集中。\n4. 構建一個簡單的初學者 CNN（卷積神經網絡）。\n5. 訓練模型。\n6. 將模型應用到選定的音境並查看結果","metadata":{}},{"cell_type":"markdown","source":"## 1.設置和導入\n讓我們開始導入所需的庫和進行一些基本的設置。","metadata":{}},{"cell_type":"code","source":"import os\n\nimport warnings\nwarnings.filterwarnings(action='ignore')\n\nimport pandas as pd\nimport librosa\nimport numpy as np\n\nfrom sklearn.utils import shuffle\nfrom PIL import Image\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\n\nimport tensorflow as tf\n\n# Global vars\nRANDOM_SEED = 1337\nSAMPLE_RATE = 32000\nSIGNAL_LENGTH = 5 # seconds\nSPEC_SHAPE = (48, 128) # height x width\nFMIN = 500\nFMAX = 12500\nMAX_AUDIO_FILES = 1500","metadata":{"execution":{"iopub.status.busy":"2023-11-24T05:56:18.037916Z","iopub.execute_input":"2023-11-24T05:56:18.039124Z","iopub.status.idle":"2023-11-24T05:56:18.046645Z","shell.execute_reply.started":"2023-11-24T05:56:18.039071Z","shell.execute_reply":"2023-11-24T05:56:18.045621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.數據準備：\n\n這次比賽的訓練數據包含了數萬個涉及 397 種物種的音訊文件。這對於這個教程來說太多了，因此我們將限制我們的物種選擇為擁有至少 200 條評分為 4 或更高的錄音的物種。","metadata":{}},{"cell_type":"code","source":"# Code adapted from: \n# https://www.kaggle.com/frlemarchand/bird-song-classification-using-an-efficientnet\n# Make sure to check out the entire notebook.\n\n# Load metadata file\ntrain = pd.read_csv('../input/birdclef-2021/train_metadata.csv',)\n\n# Limit the number of training samples and classes\n# First, only use high quality samples\ntrain = train.query('rating>=4')\n\n# Second, assume that birds with the most training samples are also the most common\n# A species needs at least 200 recordings with a rating above 4 to be considered common\nbirds_count = {}\nfor bird_species, count in zip(train.primary_label.unique(), \n                               train.groupby('primary_label')['primary_label'].count().values):\n    birds_count[bird_species] = count\nmost_represented_birds = [key for key,value in birds_count.items() if value >= 200] \n\nTRAIN = train.query('primary_label in @most_represented_birds')\nLABELS = sorted(TRAIN.primary_label.unique())\n\n# Let's see how many species and samples we have left\nprint('NUMBER OF SPECIES IN TRAIN DATA:', len(LABELS))\nprint('NUMBER OF SAMPLES IN TRAIN DATA:', len(TRAIN))\nprint('LABELS:', most_represented_birds)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T05:56:18.048068Z","iopub.execute_input":"2023-11-24T05:56:18.049346Z","iopub.status.idle":"2023-11-24T05:56:18.399507Z","shell.execute_reply.started":"2023-11-24T05:56:18.049306Z","shell.execute_reply":"2023-11-24T05:56:18.398340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3.提取訓練樣本\n我們需要定義一個函數，用於提取給定音訊文件的頻譜圖。這個函數需要使用 Librosa 載入文件（在這個教程中只使用前 15 秒），提取梅爾頻譜圖，並將每個頻譜圖保存為 PNG 圖片在一個工作目錄中，以便稍後訪問。","metadata":{}},{"cell_type":"code","source":"# Shuffle the training data and limit the number of audio files to MAX_AUDIO_FILES\nTRAIN = shuffle(TRAIN, random_state=RANDOM_SEED)[:MAX_AUDIO_FILES]\n\n# Define a function that splits an audio file, \n# extracts spectrograms and saves them in a working directory\ndef get_spectrograms(filepath, primary_label, output_dir):\n    \n    # Open the file with librosa (limited to the first 15 seconds)\n    sig, rate = librosa.load(filepath, sr=SAMPLE_RATE, offset=None, duration=15)\n    \n    # Split signal into five second chunks\n    sig_splits = []\n    for i in range(0, len(sig), int(SIGNAL_LENGTH * SAMPLE_RATE)):\n        split = sig[i:i + int(SIGNAL_LENGTH * SAMPLE_RATE)]\n\n        # End of signal?\n        if len(split) < int(SIGNAL_LENGTH * SAMPLE_RATE):\n            break\n        \n        sig_splits.append(split)\n        \n    # Extract mel spectrograms for each audio chunk\n    s_cnt = 0\n    saved_samples = []\n    for chunk in sig_splits:\n        \n        hop_length = int(SIGNAL_LENGTH * SAMPLE_RATE / (SPEC_SHAPE[1] - 1))\n        mel_spec = librosa.feature.melspectrogram(y=chunk, \n                                                  sr=SAMPLE_RATE, \n                                                  n_fft=1024, \n                                                  hop_length=hop_length, \n                                                  n_mels=SPEC_SHAPE[0], \n                                                  fmin=FMIN, \n                                                  fmax=FMAX)\n    \n        mel_spec = librosa.power_to_db(mel_spec, ref=np.max) \n        \n        # Normalize\n        mel_spec -= mel_spec.min()\n        mel_spec /= mel_spec.max()\n        \n        # Save as image file\n        save_dir = os.path.join(output_dir, primary_label)\n        if not os.path.exists(save_dir):\n            os.makedirs(save_dir)\n        save_path = os.path.join(save_dir, filepath.rsplit(os.sep, 1)[-1].rsplit('.', 1)[0] + \n                                 '_' + str(s_cnt) + '.png')\n        im = Image.fromarray(mel_spec * 255.0).convert(\"L\")\n        im.save(save_path)\n        \n        saved_samples.append(save_path)\n        s_cnt += 1\n        \n        \n    return saved_samples\n\nprint('FINAL NUMBER OF AUDIO FILES IN TRAINING DATA:', len(TRAIN))","metadata":{"execution":{"iopub.status.busy":"2023-11-24T05:56:18.402925Z","iopub.execute_input":"2023-11-24T05:56:18.403258Z","iopub.status.idle":"2023-11-24T05:56:18.419227Z","shell.execute_reply.started":"2023-11-24T05:56:18.403228Z","shell.execute_reply":"2023-11-24T05:56:18.418298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Parse audio files and extract training samples\ninput_dir = '../input/birdclef-2021/train_short_audio/'\noutput_dir = '../working/melspectrogram_dataset/'\nsamples = []\nwith tqdm(total=len(TRAIN)) as pbar:\n    for idx, row in TRAIN.iterrows():\n        pbar.update(1)\n        \n        if row.primary_label in most_represented_birds:\n            audio_file_path = os.path.join(input_dir, row.primary_label, row.filename)\n            samples += get_spectrograms(audio_file_path, row.primary_label, output_dir)\n            \nTRAIN_SPECS = shuffle(samples, random_state=RANDOM_SEED)\nprint('SUCCESSFULLY EXTRACTED {} SPECTROGRAMS'.format(len(TRAIN_SPECS)))","metadata":{"execution":{"iopub.status.busy":"2023-11-24T05:56:18.420673Z","iopub.execute_input":"2023-11-24T05:56:18.421064Z","iopub.status.idle":"2023-11-24T05:58:00.702696Z","shell.execute_reply.started":"2023-11-24T05:56:18.421028Z","shell.execute_reply":"2023-11-24T05:58:00.701464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the first 12 spectrograms of TRAIN_SPECS\nplt.figure(figsize=(15, 7))\nfor i in range(12):\n    spec = Image.open(TRAIN_SPECS[i])\n    plt.subplot(3, 4, i + 1)\n    plt.title(TRAIN_SPECS[i].split(os.sep)[-1])\n    plt.imshow(spec, origin='lower')","metadata":{"execution":{"iopub.status.busy":"2023-11-24T05:58:00.704884Z","iopub.execute_input":"2023-11-24T05:58:00.705646Z","iopub.status.idle":"2023-11-24T05:58:03.141964Z","shell.execute_reply.started":"2023-11-24T05:58:00.705589Z","shell.execute_reply":"2023-11-24T05:58:03.141022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4.載入訓練樣本\n目前，我們的頻譜圖位於一個工作目錄中。如果我們想要訓練一個模型，我們必須將它們載入內存。然而，由於可能有數十萬個提取的頻譜圖，將其放入內存數據集不是一個好主意。但就目前而言，從磁盤加載樣本並將它們組合成一個大的 NumPy 數組是可以的。這是使用 Keras 進行訓練的最簡單方法。","metadata":{}},{"cell_type":"code","source":"# Parse all samples and add spectrograms into train data, primary_labels into label data\ntrain_specs, train_labels = [], []\nwith tqdm(total=len(TRAIN_SPECS)) as pbar:\n    for path in TRAIN_SPECS:\n        pbar.update(1)\n\n        # Open image\n        spec = Image.open(path)\n\n        # Convert to numpy array\n        spec = np.array(spec, dtype='float32')\n        \n        # Normalize between 0.0 and 1.0\n        # and exclude samples with nan \n        spec -= spec.min()\n        spec /= spec.max()\n        if not spec.max() == 1.0 or not spec.min() == 0.0:\n            continue\n\n        # Add channel axis to 2D array\n        spec = np.expand_dims(spec, -1)\n\n        # Add new dimension for batch size\n        spec = np.expand_dims(spec, 0)\n\n        # Add to train data\n        if len(train_specs) == 0:\n            train_specs = spec\n        else:\n            train_specs = np.vstack((train_specs, spec))\n\n        # Add to label data\n        target = np.zeros((len(LABELS)), dtype='float32')\n        bird = path.split(os.sep)[-2]\n        target[LABELS.index(bird)] = 1.0\n        if len(train_labels) == 0:\n            train_labels = target\n        else:\n            train_labels = np.vstack((train_labels, target))","metadata":{"execution":{"iopub.status.busy":"2023-11-24T05:58:03.143381Z","iopub.execute_input":"2023-11-24T05:58:03.143720Z","iopub.status.idle":"2023-11-24T05:59:14.818159Z","shell.execute_reply.started":"2023-11-24T05:58:03.143682Z","shell.execute_reply":"2023-11-24T05:59:14.817136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5.構建一個簡單的模型\n好的，我們的數據集已經準備好了，現在我們需要定義一個模型架構。在這個教程中，我們將使用一個非常簡單的類似 AlexNet 的設計，其中包含四個卷積層和三個全連接層。選擇一個在音訊數據上預先訓練的 TF 模型可能是有道理的，但我們需要調整輸入（即我們的頻譜圖的解析度）以適應外部模型。因此，我們保持它簡單，建立自己的模型。","metadata":{}},{"cell_type":"code","source":"# Make sure your experiments are reproducible\ntf.random.set_seed(RANDOM_SEED)\n\n# Build a simple model as a sequence of  convolutional blocks.\n# Each block has the sequence CONV --> RELU --> BNORM --> MAXPOOL.\n# Finally, perform global average pooling and add 2 dense layers.\n# The last layer is our classification layer and is softmax activated.\n# (Well it's a multi-label task so sigmoid might actually be a better choice)\nmodel = tf.keras.Sequential([\n    \n    # First conv block\n    tf.keras.layers.Conv2D(16, (3, 3), activation='relu', \n                           input_shape=(SPEC_SHAPE[0], SPEC_SHAPE[1], 1)),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)),\n    \n    # Second conv block\n    tf.keras.layers.Conv2D(32, (3, 3), activation='relu'),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)), \n    \n    # Third conv block\n    tf.keras.layers.Conv2D(64, (3, 3), activation='relu'),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)), \n    \n    # Fourth conv block\n    tf.keras.layers.Conv2D(128, (3, 3), activation='relu'),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)),\n    \n    # Global pooling instead of flatten()\n    tf.keras.layers.GlobalAveragePooling2D(), \n    \n    # Dense block\n    tf.keras.layers.Dense(256, activation='relu'),   \n    tf.keras.layers.Dropout(0.5),  \n    tf.keras.layers.Dense(256, activation='relu'),   \n    tf.keras.layers.Dropout(0.5),\n    \n    # Classification layer\n    tf.keras.layers.Dense(len(LABELS), activation='softmax')\n])\nprint('MODEL HAS {} PARAMETERS.'.format(model.count_params()))","metadata":{"execution":{"iopub.status.busy":"2023-11-24T05:59:14.819563Z","iopub.execute_input":"2023-11-24T05:59:14.819850Z","iopub.status.idle":"2023-11-24T05:59:15.065486Z","shell.execute_reply.started":"2023-11-24T05:59:14.819823Z","shell.execute_reply":"2023-11-24T05:59:15.064483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compile the model and specify optimizer, loss and metric\nmodel.compile(optimizer=tf.keras.optimizers.Adam(lr=0.001),\n              loss=tf.keras.losses.CategoricalCrossentropy(label_smoothing=0.01),\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-11-24T05:59:15.067154Z","iopub.execute_input":"2023-11-24T05:59:15.068016Z","iopub.status.idle":"2023-11-24T05:59:15.082199Z","shell.execute_reply.started":"2023-11-24T05:59:15.067975Z","shell.execute_reply":"2023-11-24T05:59:15.081408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add callbacks to reduce the learning rate if needed, early stopping, and checkpoint saving\ncallbacks = [tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', \n                                                  patience=2, \n                                                  verbose=1, \n                                                  factor=0.5),\n             tf.keras.callbacks.EarlyStopping(monitor='val_loss', \n                                              verbose=1,\n                                              patience=5),\n             tf.keras.callbacks.ModelCheckpoint(filepath='best_model.h5', \n                                                monitor='val_loss',\n                                                verbose=0,\n                                                save_best_only=True)]","metadata":{"execution":{"iopub.status.busy":"2023-11-24T05:59:15.083448Z","iopub.execute_input":"2023-11-24T05:59:15.083784Z","iopub.status.idle":"2023-11-24T05:59:15.090437Z","shell.execute_reply.started":"2023-11-24T05:59:15.083750Z","shell.execute_reply":"2023-11-24T05:59:15.089352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's train the model for a few epochs\nmodel.fit(train_specs, \n          train_labels,\n          batch_size=32,\n          validation_split=0.2,\n          callbacks=callbacks,\n          epochs=25)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T05:59:15.091855Z","iopub.execute_input":"2023-11-24T05:59:15.092317Z","iopub.status.idle":"2023-11-24T05:59:38.009915Z","shell.execute_reply.started":"2023-11-24T05:59:15.092282Z","shell.execute_reply":"2023-11-24T05:59:38.009145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 6.音境分析\n在這個教程中，我們將簡單地從訓練數據中挑選一個音境，但整個過程可以輕松地自動化，然後應用於所有音境文件。同樣地，我們需要使用 Librosa 加載文件，為每個 5 秒的片段提取頻譜圖，將每個片段通過模型，最終為這個 5 秒的音頻片段分配一個標籤。\n\n讓我們選擇一個實際包含我們模型訓練的一些物種的音境。文件 \"28933_SSW_20170408.ogg\" 似乎包含了很多歌麻雀（sonspa）的聲音，讓我們嘗試這個。","metadata":{}},{"cell_type":"code","source":"# Load the best checkpoint\nmodel = tf.keras.models.load_model('best_model.h5')\n\n# Pick a soundscape\nsoundscape_path = '../input/birdclef-2021/train_soundscapes/28933_SSW_20170408.ogg'\n\n# Open it with librosa\nsig, rate = librosa.load(soundscape_path, sr=SAMPLE_RATE)\n\n# Store results so that we can analyze them later\ndata = {'row_id': [], 'prediction': [], 'score': []}\n\n# Split signal into 5-second chunks\n# Just like we did before (well, this could actually be a seperate function)\nsig_splits = []\nfor i in range(0, len(sig), int(SIGNAL_LENGTH * SAMPLE_RATE)):\n    split = sig[i:i + int(SIGNAL_LENGTH * SAMPLE_RATE)]\n\n    # End of signal?\n    if len(split) < int(SIGNAL_LENGTH * SAMPLE_RATE):\n        break\n\n    sig_splits.append(split)\n    \n# Get the spectrograms and run inference on each of them\n# This should be the exact same process as we used to\n# generate training samples!\nseconds, scnt = 0, 0\nfor chunk in sig_splits:\n    \n    # Keep track of the end time of each chunk\n    seconds += 5\n        \n    # Get the spectrogram\n    hop_length = int(SIGNAL_LENGTH * SAMPLE_RATE / (SPEC_SHAPE[1] - 1))\n    mel_spec = librosa.feature.melspectrogram(y=chunk, \n                                              sr=SAMPLE_RATE, \n                                              n_fft=1024, \n                                              hop_length=hop_length, \n                                              n_mels=SPEC_SHAPE[0], \n                                              fmin=FMIN, \n                                              fmax=FMAX)\n\n    mel_spec = librosa.power_to_db(mel_spec, ref=np.max) \n\n    # Normalize to match the value range we used during training.\n    # That's something you should always double check!\n    mel_spec -= mel_spec.min()\n    mel_spec /= mel_spec.max()\n    \n    # Add channel axis to 2D array\n    mel_spec = np.expand_dims(mel_spec, -1)\n\n    # Add new dimension for batch size\n    mel_spec = np.expand_dims(mel_spec, 0)\n    \n    # Predict\n    p = model.predict(mel_spec)[0]\n    \n    # Get highest scoring species\n    idx = p.argmax()\n    species = LABELS[idx]\n    score = p[idx]\n    \n    # Prepare submission entry\n    data['row_id'].append(soundscape_path.split(os.sep)[-1].rsplit('_', 1)[0] + \n                          '_' + str(seconds))    \n    \n    # Decide if it's a \"nocall\" or a species by applying a threshold\n    if score > 0.25:\n        data['prediction'].append(species)\n        scnt += 1\n    else:\n        data['prediction'].append('nocall')\n        \n    # Add the confidence score as well\n    data['score'].append(score)\n        \nprint('SOUNSCAPE ANALYSIS DONE. FOUND {} BIRDS.'.format(scnt))","metadata":{"execution":{"iopub.status.busy":"2023-11-24T05:59:38.011430Z","iopub.execute_input":"2023-11-24T05:59:38.011870Z","iopub.status.idle":"2023-11-24T05:59:54.149130Z","shell.execute_reply.started":"2023-11-24T05:59:38.011829Z","shell.execute_reply":"2023-11-24T05:59:54.145985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make a new data frame\nresults = pd.DataFrame(data, columns = ['row_id', 'prediction', 'score'])\n\n# Merge with ground truth so we can inspect\ngt = pd.read_csv('../input/birdclef-2021/train_soundscape_labels.csv',)\nresults = pd.merge(gt, results, on='row_id')\n\n# Let's look at the first 50 entries\nresults.head(50)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T05:59:54.151229Z","iopub.execute_input":"2023-11-24T05:59:54.151747Z","iopub.status.idle":"2023-11-24T05:59:54.190183Z","shell.execute_reply.started":"2023-11-24T05:59:54.151690Z","shell.execute_reply":"2023-11-24T05:59:54.189252Z"},"trusted":true},"execution_count":null,"outputs":[]}]}