{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":8098641,"sourceType":"datasetVersion","datasetId":4782229},{"sourceId":3836,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":2739,"modelId":319}],"dockerImageVersionId":30408,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This work is inspired by the post from the last year's first place winner, which used the google bird vocalization model to clean the data by using the pretrained model by Google. THis notebook implement the data cleaning part which removes the audio that has low quality in the train_audio.","metadata":{}},{"cell_type":"code","source":"debug_mode = False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T04:39:37.644694Z","iopub.execute_input":"2025-05-26T04:39:37.644922Z","iopub.status.idle":"2025-05-26T04:39:37.668651Z","shell.execute_reply.started":"2025-05-26T04:39:37.644898Z","shell.execute_reply":"2025-05-26T04:39:37.667843Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 1: Imports","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_hub as hub\nimport tensorflow_io as tfio\n\nimport pandas as pd\nimport numpy as np\nimport librosa\nimport glob\n\nimport csv\nimport io\nfrom tqdm import tqdm\nimport os\n\nimport plotly.express as px\nfrom matplotlib import pyplot as plt \n\nfrom IPython.display import Audio","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2025-05-26T04:40:28.314418Z","iopub.execute_input":"2025-05-26T04:40:28.315137Z","iopub.status.idle":"2025-05-26T04:40:38.065695Z","shell.execute_reply.started":"2025-05-26T04:40:28.315103Z","shell.execute_reply":"2025-05-26T04:40:38.064506Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 2: Match the model's output with the bird species in the competition\n\nThe competition includes 264 classes of birds, 261 of which exist in this model. We'll set up a way to map the model's output logits to our competition.","metadata":{}},{"cell_type":"code","source":"model = hub.load('https://kaggle.com/models/google/bird-vocalization-classifier/frameworks/tensorFlow2/variations/bird-vocalization-classifier/versions/1')\nlabels_path = hub.resolve('https://kaggle.com/models/google/bird-vocalization-classifier/frameworks/tensorFlow2/variations/bird-vocalization-classifier/versions/1') + \"/assets/label.csv\"","metadata":{"execution":{"iopub.status.busy":"2025-05-26T04:40:38.067344Z","iopub.execute_input":"2025-05-26T04:40:38.067680Z","iopub.status.idle":"2025-05-26T04:40:48.084253Z","shell.execute_reply.started":"2025-05-26T04:40:38.067649Z","shell.execute_reply":"2025-05-26T04:40:48.083193Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Find the name of the class with the top score when mean-aggregated across frames.\ndef class_names_from_csv(class_map_csv_text):\n    \"\"\"Returns list of class names corresponding to score vector.\"\"\"\n    with open(labels_path) as csv_file:\n        csv_reader = csv.reader(csv_file, delimiter=',')\n        class_names = [mid for mid, desc in csv_reader]\n        return class_names[1:]\n\n## note that the bird classifier classifies a much larger set of birds than the\n## competition, so we need to load the model's set of class names or else our \n## indices will be off.\nclasses = class_names_from_csv(labels_path)","metadata":{"execution":{"iopub.status.busy":"2025-05-26T04:40:48.085623Z","iopub.execute_input":"2025-05-26T04:40:48.085976Z","iopub.status.idle":"2025-05-26T04:40:48.106041Z","shell.execute_reply.started":"2025-05-26T04:40:48.085941Z","shell.execute_reply":"2025-05-26T04:40:48.105106Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_metadata = pd.read_csv(\"/kaggle/input/birdclef-2025/train.csv\")\ntrain_metadata.head()\ncompetition_classes = sorted(train_metadata.primary_label.unique())\n\nforced_defaults = 0\ncompetition_class_map = []\nmissing_class_map = []\nfor c in competition_classes:\n    try:\n        i = classes.index(c)\n        competition_class_map.append(i)\n    except:\n        competition_class_map.append(0)\n        missing_class_map.append(c)\n        forced_defaults += 1\n        \n## this is the count of classes not supported by our pretrained model\n## you could choose to simply not predict these, set a default as above,\n## or create your own model using the pretrained model as a base.\nprint(f\"There are {forced_defaults} classes that are not supported by the model\")","metadata":{"execution":{"iopub.status.busy":"2025-05-26T04:40:48.108495Z","iopub.execute_input":"2025-05-26T04:40:48.108941Z","iopub.status.idle":"2025-05-26T04:40:48.309254Z","shell.execute_reply.started":"2025-05-26T04:40:48.108905Z","shell.execute_reply":"2025-05-26T04:40:48.308261Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 3: Preprocess the data\n\nThe following functions are one way to load the audio provided and break it up into the five-second samples with a sample rate of 32,000 required by the competition.","metadata":{}},{"cell_type":"code","source":"def frame_audio(\n      audio_array: np.ndarray,\n      window_size_s: float = 5.0,\n      hop_size_s: float = 5.0,\n      sample_rate = 32000,\n      ) -> np.ndarray:\n    \n    \"\"\"Helper function for framing audio for inference.\"\"\"\n    \"\"\" using tf.signal \"\"\"\n    if window_size_s is None or window_size_s < 0:\n        return audio_array[np.newaxis, :]\n    frame_length = int(window_size_s * sample_rate)\n    hop_length = int(hop_size_s * sample_rate)\n    framed_audio = tf.signal.frame(audio_array, frame_length, hop_length, pad_end=True)\n    return framed_audio\n\ndef ensure_sample_rate(waveform, original_sample_rate,\n                       desired_sample_rate=32000):\n    \"\"\"Resample waveform if required.\"\"\"\n    if original_sample_rate != desired_sample_rate:\n        waveform = tfio.audio.resample(waveform, original_sample_rate, desired_sample_rate)\n    return desired_sample_rate, waveform","metadata":{"execution":{"iopub.status.busy":"2025-05-26T04:40:48.310512Z","iopub.execute_input":"2025-05-26T04:40:48.310856Z","iopub.status.idle":"2025-05-26T04:40:48.317824Z","shell.execute_reply.started":"2025-05-26T04:40:48.310828Z","shell.execute_reply":"2025-05-26T04:40:48.316705Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 4: Make predictions for train_data\n\nEach test sample is cut into 5-second chunks. We use the pretrained model to return probabilities for all 10k birds included in the model, then pull out the classes used in this competition to create a final submission row. Note that we are NOT doing anything special to handle the 63 missing classes; those will need fine-tuning / transfer learning, which will be handled in a separate notebook.","metadata":{}},{"cell_type":"code","source":"def predict_for_sample(filename, sample_submission, frame_limit_secs=None):\n    file_id = filename.split(\".ogg\")[0].split(\"/\")[-1]\n    \n    audio, sample_rate = librosa.load(filename)\n    sample_rate, wav_data = ensure_sample_rate(audio, sample_rate)\n    \n    fixed_tm = frame_audio(wav_data)\n    \n    frame = 5\n    all_logits, all_embeddings = model.infer_tf(fixed_tm[:1])\n    for window in fixed_tm[1:]:\n        if frame_limit_secs and frame > frame_limit_secs:\n            continue\n        \n        logits, embeddings = model.infer_tf(window[np.newaxis, :])\n        all_logits = np.concatenate([all_logits, logits], axis=0)\n        frame += 5\n    \n    frame = 5\n    all_probabilities = []\n    for frame_logits in all_logits:\n        probabilities = tf.nn.softmax(frame_logits).numpy()\n        \n        ## set the appropriate row in the sample submission\n        sample_submission.loc[sample_submission.row_id == file_id + \"_\" + str(frame), competition_classes] = probabilities[competition_class_map]\n        frame += 5","metadata":{"execution":{"iopub.status.busy":"2025-05-26T04:40:48.318910Z","iopub.execute_input":"2025-05-26T04:40:48.319228Z","iopub.status.idle":"2025-05-26T04:40:48.328424Z","shell.execute_reply.started":"2025-05-26T04:40:48.319186Z","shell.execute_reply":"2025-05-26T04:40:48.327391Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 6: Evaluation","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, confusion_matrix\nimport os\n\nif debug_mode:\n    df = pd.read_csv('/kaggle/input/birdclef-2025/train.csv')\n    X = list(df['filename'][:1000])\n    base_path = \"/kaggle/input/birdclef-2025/train_audio/\"\nelse:\n    df = pd.read_csv('/kaggle/input/birdclef-2025/train.csv')\n    X = list(df['filename'])\n    base_path = \"/kaggle/input/birdclef-2025/train_audio/\"\n","metadata":{"execution":{"iopub.status.busy":"2025-05-26T04:40:48.329664Z","iopub.execute_input":"2025-05-26T04:40:48.329928Z","iopub.status.idle":"2025-05-26T04:40:48.521398Z","shell.execute_reply.started":"2025-05-26T04:40:48.329894Z","shell.execute_reply":"2025-05-26T04:40:48.520334Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict(x):\n    Y_true = []\n    y_audio = []\n    y_audio_type = []\n    y_predict = []\n    y_prob = []\n    for audio in tqdm(x):\n        Y_true.append(df.loc[df['filename'] == audio, 'primary_label'].iloc[0])\n        y_audio.append(audio)\n        y_audio_type.append(df.loc[df['filename'] == audio, 'type'].iloc[0])\n        path = os.path.join(base_path, audio)\n        audio, sample_rate = librosa.load(path)\n        sample_rate, wav_data = ensure_sample_rate(audio, sample_rate)\n        fixed_tm = frame_audio(wav_data)\n        i = 0\n        max_prob = 0\n        max_pred = 0\n               \n        while(i+1<=fixed_tm.shape[0]):\n            logits, embeddings = model.infer_tf(fixed_tm[i:i+1])\n            probabilities = tf.nn.softmax(logits)\n            argmax = np.argmax(probabilities)\n            i += 1\n            if (probabilities[0][argmax] > max_prob):\n                max_prob = probabilities[0][argmax]\n                max_pred = classes[argmax]\n        y_predict.append(max_pred)\n        y_prob.append(float(max_prob))\n    \n    return Y_true, y_audio, y_audio_type, y_predict, y_prob","metadata":{"execution":{"iopub.status.busy":"2025-05-26T04:40:48.522447Z","iopub.execute_input":"2025-05-26T04:40:48.522725Z","iopub.status.idle":"2025-05-26T04:40:48.532831Z","shell.execute_reply.started":"2025-05-26T04:40:48.522698Z","shell.execute_reply":"2025-05-26T04:40:48.531742Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def evaluate_model(y_true, y_pred):\n    # Calculate accuracy\n    accuracy = accuracy_score(y_true, y_pred)\n    # Calculate precision\n    precision = precision_score(y_true, y_pred, average='weighted')\n    # Calculate recall\n    recall = recall_score(y_true, y_pred, average='weighted')\n    # Calculate F1 score\n    f1 = f1_score(y_true, y_pred, average='weighted')\n    \n    return accuracy, precision, recall, f1","metadata":{"execution":{"iopub.status.busy":"2025-05-26T04:40:48.534286Z","iopub.execute_input":"2025-05-26T04:40:48.534610Z","iopub.status.idle":"2025-05-26T04:40:48.546405Z","shell.execute_reply.started":"2025-05-26T04:40:48.534566Z","shell.execute_reply":"2025-05-26T04:40:48.545412Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Y_true, y_audio, y_audio_type, y_predict, y_prob = predict(X)\naccuracy, precision, recall, f1 = evaluate_model(Y_true, y_predict)\nprint('Accuracy: ',accuracy)\nprint('Precision: ', precision)\nprint('Recall: ', recall)\nprint('F1: ', f1)","metadata":{"execution":{"iopub.status.busy":"2025-05-26T04:40:48.549252Z","iopub.execute_input":"2025-05-26T04:40:48.549530Z","iopub.status.idle":"2025-05-26T04:46:41.495141Z","shell.execute_reply.started":"2025-05-26T04:40:48.549506Z","shell.execute_reply":"2025-05-26T04:46:41.494061Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"d = {'filename': y_audio, 'Prediction': y_predict, 'Probability':y_prob}\ndf_predict = pd.DataFrame(data=d)\ndf_predict.to_csv('/kaggle/working/prediction.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2025-05-26T04:46:41.496746Z","iopub.execute_input":"2025-05-26T04:46:41.498169Z","iopub.status.idle":"2025-05-26T04:46:41.509640Z","shell.execute_reply.started":"2025-05-26T04:46:41.498122Z","shell.execute_reply":"2025-05-26T04:46:41.508645Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"replace_count = 0  # Global counter\n\ndef replace_label(x):\n    global replace_count  # Tell Python to use the global variable\n\n    if x['primary_label'] in missing_class_map:\n        return x\n    else:\n        if x['primary_label'] == classes.index(x['Prediction']):\n            return x\n        else:\n            if str(classes.index(x['Prediction'])) in x['secondary_labels']:\n                x['primary_label'] = classes.index(x['Prediction'])\n                replace_count += 1  # Increment correctly\n            else:\n                x['primary_label'] = 'Bad'\n    return x\n\n# Load data\npred = pd.read_csv('/kaggle/working/prediction.csv')\ndf = pd.read_csv('/kaggle/input/birdclef-2025/train.csv')\nrow_before = df.shape[0]\nprint(row_before)\n\n# Merge prediction with original data\ndf = pd.merge(df, pred, on='filename')\n\n# Apply label correction\ndf = df.apply(replace_label, axis=1)\n\n# Remove rows marked as 'Bad'\ndf = df[df['primary_label'] != 'Bad']\nrow_after = df.shape[0]\n\n# Difference in rows\nrow_diff = row_before - row_after\n\n# Save result\ndf.to_csv('/kaggle/working/cleaned_train.csv', index=False)\ndf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T04:48:14.082667Z","iopub.execute_input":"2025-05-26T04:48:14.083515Z","iopub.status.idle":"2025-05-26T04:48:14.271562Z","shell.execute_reply.started":"2025-05-26T04:48:14.083471Z","shell.execute_reply":"2025-05-26T04:48:14.270594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f'{row_diff} number of data have been deleted because of the quality')\nprint(f'{replace_count} number of data have been changed')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T04:47:39.677896Z","iopub.execute_input":"2025-05-26T04:47:39.678475Z","iopub.status.idle":"2025-05-26T04:47:39.685695Z","shell.execute_reply.started":"2025-05-26T04:47:39.678424Z","shell.execute_reply":"2025-05-26T04:47:39.683990Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}