{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div style=\"display: flex; justify-content: space-between; align-items: flex-start;\">\n    <div style=\"text-align: left;\">\n        <p style=\"color:#FFD700; font-size: 15px; font-weight: bold; margin-bottom: 1px; text-align: left;\">Published on  March 17, 2025</p>\n        <h4 style=\"color:#4B0082; font-weight: bold; text-align: left; margin-top: 6px;\">Author: Jocelyn C. Dumlao</h4>\n        <p style=\"font-size: 17px; line-height: 1.7; color: #333; text-align: center; margin-top: 20px;\"></p>\n        <a href=\"https://www.linkedin.com/in/jocelyn-dumlao-168921a8/\" target=\"_blank\" style=\"display: inline-block; background-color: #003f88; color: #fff; text-decoration: none; padding: 5px 10px; border-radius: 10px; margin: 15px;\">LinkedIn</a>\n        <a href=\"https://github.com/jcdumlao14\" target=\"_blank\" style=\"display: inline-block; background-color: transparent; color: #059c99; text-decoration: none; padding: 5px 10px; border-radius: 10px; margin: 15px; border: 2px solid #007bff;\">GitHub</a>\n        <a href=\"https://www.youtube.com/@CogniCraftedMinds\" target=\"_blank\" style=\"display: inline-block; background-color: #ff0054; color: #fff; text-decoration: none; padding: 5px 10px; border-radius: 10px; margin: 15px;\">YouTube</a>\n        <a href=\"https://www.kaggle.com/jocelyndumlao\" target=\"_blank\" style=\"display: inline-block; background-color: #3a86ff; color: #fff; text-decoration: none; padding: 5px 10px; border-radius: 10px; margin: 15px;\">Kaggle</a>\n    </div>\n</div>","metadata":{}},{"cell_type":"markdown","source":"# <p style=\"padding:10px;background-color:#ffdf5b;font-family:newtimeroman;font-size:100%;text-align:center;border-radius:12px;font-weight:300;border: 6px outset #f2102e;\">BirdCLEF 2025: Inference with SimpleCNN Spectrogram Model</p>","metadata":{}},{"cell_type":"markdown","source":"# <p style=\"padding:10px;background-color:#ffdf5b;font-family:newtimeroman;font-size:100%;text-align:center;border-radius:12px;font-weight:300;border: 6px outset #f2102e;\">Setup and Imports</p>\n- This section prepares the environment by loading necessary tools and defining key data locations.\n- It starts by importing the required libraries for deep learning, audio processing, and data manipulation. Then, it sets up the paths to the data files used in the project. Crucially, it loads the sample submission to get the list of bird species being predicted and maps them to numerical indices. Finally, a debug mode is configured to allow for easier testing with a smaller amount of data.\n","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\nimport os\nimport librosa\nimport numpy as np\nimport pandas as pd\n\nimport gc\nimport dataclasses\nfrom concurrent.futures import ThreadPoolExecutor\nfrom typing import Optional, Callable, Tuple, List\nimport traceback  # Import traceback module","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T05:46:15.163135Z","iopub.execute_input":"2025-03-18T05:46:15.163477Z","iopub.status.idle":"2025-03-18T05:46:17.437616Z","shell.execute_reply.started":"2025-03-18T05:46:15.163452Z","shell.execute_reply":"2025-03-18T05:46:17.436781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define file paths \ntest_data = \"/kaggle/input/birdclef-2025/test_soundscapes\"\nsubmission = \"/kaggle/input/birdclef-2025/sample_submission.csv\"\ntrain_csv = \"/kaggle/input/birdclef-2025/train.csv\"\ntaxonomy_csv = \"/kaggle/input/birdclef-2025/taxonomy.csv\"\n\ntransform: Optional[Callable] = None  # Type hint for transform\naudio_transform: Optional[Callable] = None # Type hint for audio_transform\n\n@dataclasses.dataclass\nclass AudioParam:\n    SR: int = 32_000  # Sample rate\n    NFFT: int = 2048  # Number of FFT points\n    NMEL: int = 128   # Number of Mel bands\n    FMAX: int = 16_000 # Maximum frequency\n    FMIN: int = 20   # Minimum frequency\n    HOP_LENGTH: int = NFFT // 4  # Hop length\n\naudio_param = AudioParam()\n\n# Load submission CSV to get class names\ntry:\n    sub_csv = pd.read_csv(submission)\n    idx2cls = sub_csv.columns.drop(\"row_id\").tolist()  # List of bird species (class names)\n    cls2idx = {c: i for i, c in enumerate(idx2cls)} # Class name to index mapping\nexcept FileNotFoundError as e:\n    print(f\"Error: sample_submission.csv not found! {e}\")\n    idx2cls = [] # Provide a default for testing, but the code will likely fail\n    cls2idx = {}\n\n\nDEBUG = True # Enable Debugging\nfile_names = [os.path.join(test_data, fp) for fp in os.listdir(test_data) if fp.endswith(\".ogg\")]\n\n# Use a single file for debugging.  This makes the matrix dimension calculations easier.\nif len(file_names) == 0:\n    file_names = [\n        \"/kaggle/input/birdclef-2025/train_soundscapes/H02_20230420_074000.ogg\",\n    ]\n    DEBUG = True\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T05:46:25.081303Z","iopub.execute_input":"2025-03-18T05:46:25.081835Z","iopub.status.idle":"2025-03-18T05:46:25.112564Z","shell.execute_reply.started":"2025-03-18T05:46:25.081804Z","shell.execute_reply":"2025-03-18T05:46:25.111734Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <p style=\"padding:10px;background-color:#ffdf5b;font-family:newtimeroman;font-size:100%;text-align:center;border-radius:12px;font-weight:300;border: 6px outset #f2102e;\">SimpleCNN Model Definition</p>\n- The SimpleCNN class defines a basic convolutional neural network for processing audio spectrograms. It consists of two convolutional layers with ReLU activation and max pooling, followed by a flattening layer and a dynamically sized linear (fully connected) layer. The model learns to extract features from the spectrogram and predict the presence of different bird species. The size of the final linear layer is dynamically adjusted based on the input size, making it flexible for different spectrogram dimensions.","metadata":{}},{"cell_type":"code","source":"#  a simpler, randomly initialized CNN model\nclass SimpleCNN(nn.Module):\n    def __init__(self, num_classes: int = 1):\n        super().__init__()\n        self.conv1 = nn.Conv2d(1, 16, kernel_size=3, padding=1)\n        self.relu1 = nn.ReLU()\n        self.pool1 = nn.MaxPool2d(kernel_size=2, stride=2)\n        self.conv2 = nn.Conv2d(16, 32, kernel_size=3, padding=1)\n        self.relu2 = nn.ReLU()\n        self.pool2 = nn.MaxPool2d(kernel_size=2, stride=2)\n        self.flatten = nn.Flatten()\n\n        # Calculate the input size to the linear layer dynamically\n        self._to_linear = None  # Placeholder, will be calculated during the first forward pass\n        self.fc1 = nn.Linear(1, num_classes)  # Placeholder Linear layer\n\n    def forward(self, x: torch.Tensor) -> torch.Tensor:\n        try:\n            x = self.pool1(self.relu1(self.conv1(x)))\n            x = self.pool2(self.relu2(self.conv2(x)))\n            x = self.flatten(x)\n\n            # Dynamically determine the input size of the linear layer\n            if self._to_linear is None:\n                self._to_linear = x.shape[1]\n                if self._to_linear == 0:\n                   print(\"Error: self._to_linear is zero!\")\n                   #Handle this better - e.g., skip or set a min size\n                   return torch.zeros((1, len(idx2cls)))  # or a zero tensor of the right size\n                self.fc1 = nn.Linear(self._to_linear, len(idx2cls))  # Update the linear layer\n            x = self.fc1(x)\n            return x\n        except Exception as e:\n            print(f\"Error in SimpleCNN.forward: {e}\")\n            return torch.zeros((1, len(idx2cls)))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T05:46:30.443729Z","iopub.execute_input":"2025-03-18T05:46:30.444091Z","iopub.status.idle":"2025-03-18T05:46:30.452396Z","shell.execute_reply.started":"2025-03-18T05:46:30.444028Z","shell.execute_reply":"2025-03-18T05:46:30.451253Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <p style=\"padding:10px;background-color:#ffdf5b;font-family:newtimeroman;font-size:100%;text-align:center;border-radius:12px;font-weight:300;border: 6px outset #f2102e;\">Audio Processing Pipeline</p>\n- This section defines how raw audio is transformed into a format suitable for the CNN.\n- It takes a raw audio waveform as input and converts it into a Mel spectrogram. This spectrogram representation highlights the frequencies present in the audio in a way that is perceptually relevant. The spectrogram is then normalized to a specific range, and a \"channel\" dimension is added to meet the input requirements of the neural network.\n","metadata":{}},{"cell_type":"code","source":"# Instantiate the SimpleCNN model.\nmodel = SimpleCNN(num_classes=len(idx2cls))\nmodel.eval() # Set the model to evaluation mode.\n\ndef pipeline(x: np.ndarray) -> np.ndarray:\n    \"\"\"\n    Converts audio data to a mel spectrogram and then to a dB scale.\n    \"\"\"\n    try:\n        mels = librosa.feature.melspectrogram(\n            y=x,\n            sr=audio_param.SR,\n            n_fft=audio_param.NFFT,\n            n_mels=audio_param.NMEL,\n            fmax=audio_param.FMAX,\n            fmin=audio_param.FMIN,\n            hop_length=audio_param.HOP_LENGTH,\n        )\n        db_map = librosa.power_to_db(mels, ref=np.max)\n        db_map = (db_map + 80) / (80 + 1e-6)  # Normalize to [0, 1] - Added small constant\n        if np.isnan(db_map).any():\n            print(\"Warning: NaN values detected in db_map!\")\n            db_map = np.nan_to_num(db_map) #Replace with 0\n\n        return db_map[None, :, :]  # Add a channel dimension (1, height, width)\n    except Exception as e:\n        print(f\"Error in pipeline: {e}\")\n        return np.zeros((1, audio_param.NMEL, 1)) # return a zero array\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T05:46:35.394864Z","iopub.execute_input":"2025-03-18T05:46:35.395282Z","iopub.status.idle":"2025-03-18T05:46:35.425865Z","shell.execute_reply.started":"2025-03-18T05:46:35.395252Z","shell.execute_reply":"2025-03-18T05:46:35.424882Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <p style=\"padding:10px;background-color:#ffdf5b;font-family:newtimeroman;font-size:100%;text-align:center;border-radius:12px;font-weight:300;border: 6px outset #f2102e;\">Prediction Function</p>\n- This section defines the core prediction process for a single audio file.\n- It loads an audio file, divides it into 5-second segments, and then processes each segment individually. It applies optional audio and image augmentations to improve robustness. The processed segment (the Mel spectrogram) is then passed to the neural network to generate a prediction. The function also creates unique identifiers for each segment and stores the prediction results along with these identifiers.\n\n","metadata":{}},{"cell_type":"code","source":"@torch.no_grad()\ndef predict(fp: str) -> Tuple[np.ndarray, List[str]]:\n    \"\"\"\n    Predicts bird calls in a given audio file.\n\n    Args:\n        fp (str): File path of the audio file.\n\n    Returns:\n        Tuple[np.ndarray, List[str]]: Tuple containing the model output and the list of row IDs.\n    \"\"\"\n    try:\n        x, _ = librosa.load(fp, sr=audio_param.SR)  # Load the audio file.\n\n        if x.size == 0:\n            print(f\"Warning: Audio file {fp} is empty!\")\n            return np.array([]), [] #return empty arrays\n    except Exception as e:\n        print(f\"Error loading file {fp}: {e}\")\n        return np.array([]), []\n\n    # Number of 5-second segments\n    num_segments = int(np.floor(len(x) / audio_param.SR / 5))\n    all_outs = []\n    all_row_ids = []\n    for i in range(num_segments):\n        start = i * audio_param.SR * 5\n        end = (i + 1) * audio_param.SR * 5\n        segment = x[start:end]\n\n\n        if audio_transform is not None:\n            try:\n                segment = audio_transform(sample=segment, sample_rate=audio_param.SR) #Apply audio transform\n            except Exception as e:\n                print(f\"Audio Transform Failed {e}\")\n\n        try:\n            segment = pipeline(segment)  #Convert waveform to mel spectrogram.\n        except Exception as e:\n            print(f\"Pipeline failed {e}\")\n            continue\n\n        if transform is not None:\n            try:\n                segment = transform(image=segment)[\"image\"] #Apply image transform.\n            except Exception as e:\n                print(f\"Transform failed {e}\")\n                continue\n\n        try:\n            segment = torch.from_numpy(segment).float().unsqueeze(0)  # Convert to tensor and add batch dimension.\n            out = model(segment).sigmoid().detach().cpu().numpy() # Get the model output.\n            all_outs.append(out[0])\n\n            fp_name = os.path.basename(fp).split(\".\")[0] #Extract the base filename.\n            row_id = f\"{fp_name}_{(i + 1) * 5}\" #Create row IDs.  Correct the slice name\n            all_row_ids.append(row_id)\n        except Exception as e:\n            print(f\"Error during processing of segment {i} in {fp}: {e}  {traceback.format_exc()}\") #Print trace\n\n    return np.array(all_outs), all_row_ids # return all values\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T05:46:40.441113Z","iopub.execute_input":"2025-03-18T05:46:40.441489Z","iopub.status.idle":"2025-03-18T05:46:40.451542Z","shell.execute_reply.started":"2025-03-18T05:46:40.441461Z","shell.execute_reply":"2025-03-18T05:46:40.450245Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <p style=\"padding:10px;background-color:#ffdf5b;font-family:newtimeroman;font-size:100%;text-align:center;border-radius:12px;font-weight:300;border: 6px outset #f2102e;\">Prediction Loop and Submission File</p>\n- This section orchestrates the prediction process for multiple audio files and creates the final submission file.\n- It initializes data structures to store the prediction results. It then uses parallel processing to speed up the prediction process across multiple audio files. The prediction results and identifiers are collected, reshaped, and combined into a Pandas DataFrame. Finally, this DataFrame is saved to a CSV file in the format required for submission. A small sample of the DataFrame is displayed for verification.","metadata":{}},{"cell_type":"code","source":"row_id = []\nmatrix = []\n\n#Using a ThreadPoolExecutor to parallelize the predictions\nwith ThreadPoolExecutor(max_workers=4) as executor:\n    for fp_idx, (fp) in enumerate(file_names): #Enumerate so you know the file index\n        try:\n            out, rid = predict(fp)\n            if len(rid) > 0:  # Only extend if there are valid results\n                row_id.extend(rid)  # Changed append to extend to unpack the list of strings\n                matrix.extend(out)  # Append the output, which should have shape (num_classes,)\n            else:\n                print(f\"Warning: No predictions generated for file: {fp}\")\n        except Exception as e:\n            print(f\"Failed to run predict for file {fp} {e}\") #Major problem.\n        gc.collect() #Collect after each file\n        print(f\"Finished {fp_idx+1}/{len(file_names)}\") #Track progress\n\ntry:\n    matrix = np.array(matrix).reshape(-1, len(idx2cls))  # Reshape to (num_segments, num_classes)\n    row_id = np.array(row_id).reshape(-1, 1)  # Ensure row_id is a 2D array\n    matrix = np.hstack([row_id, matrix])  # Now both arrays have the same dimensions\n\n    # Create a Pandas DataFrame from the results.\n    sub = pd.DataFrame(matrix, columns=[\"row_id\", *idx2cls])\n    sub.to_csv('submission.csv', index=False)\n\n    print(sub.head())\nexcept Exception as e:\n    print(f\"Error creating submission file {e}\") #Most likely problem.\n\nprint(\"Finished!\") #If you see this, then great!\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T05:46:45.828189Z","iopub.execute_input":"2025-03-18T05:46:45.828541Z","iopub.status.idle":"2025-03-18T05:47:03.959491Z","shell.execute_reply.started":"2025-03-18T05:46:45.828516Z","shell.execute_reply":"2025-03-18T05:47:03.958513Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <p style=\"padding:10px;background-color:#ffdf5b;font-family:newtimeroman;font-size:100%;text-align:center;border-radius:12px;font-weight:300;border: 6px outset #f2102e;\">Visualization</p>","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport librosa\nimport librosa.display\nimport IPython.display as ipd\nimport soundfile as sf\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.preprocessing import MultiLabelBinarizer\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Define paths\nDATA_PATH = '/kaggle/input/birdclef-2025'\nTRAIN_AUDIO_PATH = os.path.join(DATA_PATH, 'train_audio')\nTEST_SOUNDSCAPES_PATH = os.path.join(DATA_PATH, 'test_soundscapes')\nTRAIN_SOUNDSCAPES_PATH = os.path.join(DATA_PATH, 'train_soundscapes')\n\n# Load datasets\ntrain_df = pd.read_csv(os.path.join(DATA_PATH, 'train.csv'))\ntaxonomy_df = pd.read_csv(os.path.join(DATA_PATH, 'taxonomy.csv'))\n\n# ----------------------------------------------------------------------\n# Visualization\n# ----------------------------------------------------------------------\n\n\n# --- 1.4. Recording Location Data ---\ntry:\n    with open(os.path.join(DATA_PATH, 'recording_location.txt'), 'r') as f:\n        print(\"\\nRecording Location:\")\n        print(f.read())\nexcept FileNotFoundError:\n    print(\"\\nrecording_location.txt not found.\")\n\n# --- 1.5. Train Audio Examples ---\n\ndef plot_audio_example(filename, title):\n    \"\"\"Plots waveform and spectrogram of an audio file.\"\"\"\n    file_path = os.path.join(TRAIN_AUDIO_PATH, filename)\n    try:\n        y, sr = librosa.load(file_path)\n        plt.figure(figsize=(14, 5))\n        plt.subplot(1, 2, 1)\n        librosa.display.waveshow(y, sr=sr)\n        plt.title(f'Waveform: {title}')\n\n        plt.subplot(1, 2, 2)\n        D = librosa.stft(y)\n        S_db = librosa.amplitude_to_db(np.abs(D), ref=np.max)\n        librosa.display.specshow(S_db, sr=sr, x_axis='time', y_axis='log')\n        plt.colorbar(format='%+2.0f dB')\n        plt.title(f'Spectrogram: {title}')\n        plt.tight_layout()\n        plt.show()\n\n        print(f\"Playing audio: {title}\")\n        ipd.display(ipd.Audio(file_path))\n\n    except Exception as e:\n        print(f\"Error processing {filename}: {e}\")\n\nprint(\"\\nTrain Audio Examples:\")\nfor i in range(4):\n    example_filename = train_df['filename'].iloc[i]\n    example_common_name = train_df['common_name'].iloc[i]\n    plot_audio_example(example_filename, example_common_name)\n\n\n# --- 1.6. Soundscape Examples ---\ndef plot_soundscape_example(filename, title):\n    \"\"\"Plots waveform and spectrogram of a soundscape audio file.\"\"\"\n    file_path = os.path.join(TRAIN_SOUNDSCAPES_PATH, filename)\n    try:\n        y, sr = librosa.load(file_path)\n        plt.figure(figsize=(14, 5))\n        plt.subplot(1, 2, 1)\n        librosa.display.waveshow(y, sr=sr)\n        plt.title(f'Waveform: {title}')\n\n        plt.subplot(1, 2, 2)\n        D = librosa.stft(y)\n        S_db = librosa.amplitude_to_db(np.abs(D), ref=np.max)\n        librosa.display.specshow(S_db, sr=sr, x_axis='time', y_axis='log')\n        plt.colorbar(format='%+2.0f dB')\n        plt.title(f'Spectrogram: {title}')\n        plt.tight_layout()\n        plt.show()\n\n        print(f\"Playing audio: {title}\")\n        ipd.display(ipd.Audio(file_path))\n\n    except Exception as e:\n        print(f\"Error processing {filename}: {e}\")\n\n\nprint(\"\\nSoundscape Examples:\")\nsoundscape_files = [f for f in os.listdir(TRAIN_SOUNDSCAPES_PATH) if f.endswith('.ogg')]\nfor i in range(min(4, len(soundscape_files))):  # Display up to 4 examples\n    plot_soundscape_example(soundscape_files[i], f\"Soundscape {i+1}\")\n\n\n\n# --- 1.7. Species Distribution ---\nplt.figure(figsize=(12, 6))\nspecies_counts = train_df['common_name'].value_counts().nlargest(20)\nsns.barplot(x=species_counts.index, y=species_counts.values, palette=\"viridis\")\nplt.xticks(rotation=90)\nplt.title('Top 20 Most Frequent Species')\nplt.xlabel('Species')\nplt.ylabel('Frequency')\nplt.tight_layout()\nplt.show()\n\n\n# --- 1.8. Geographical Distribution (Map) ---\n\ntry:\n    import folium\n\n    # Create a map centered around the average latitude and longitude\n    m = folium.Map(location=[train_df['latitude'].mean(), train_df['longitude'].mean()], zoom_start=6)\n\n    # Add markers for each recording location\n    for index, row in train_df.iterrows():\n        folium.Marker([row['latitude'], row['longitude']],\n                      popup=f\"{row['common_name']} ({row['primary_label']})\").add_to(m)\n\n    # Display the map (you may need to save it to an HTML file and display that in Kaggle)\n    m  # Display the map in the output.  If it doesn't render, save to HTML and display that.\n    m.save(\"recording_locations.html\") # Save the map to an HTML file.\n    print(\"Map saved to recording_locations.html.  Display this file to see the map.\")\n\n\nexcept ImportError:\n    print(\"Folium is not installed. Install it to visualize the map: pip install folium\")\nexcept Exception as e:\n    print(f\"Error creating map: {e}\")\n\n\n# --- 1.9. Taxonomy Visualization ---\n\nplt.figure(figsize=(12,6))\nclass_counts = taxonomy_df['class_name'].value_counts()\nsns.barplot(x=class_counts.index, y=class_counts.values, palette=\"magma\")\nplt.xticks(rotation=45)\nplt.title(\"Distribution of Classes in Taxonomy\")\nplt.xlabel(\"Class\")\nplt.ylabel(\"Frequency\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T05:47:15.822168Z","iopub.execute_input":"2025-03-18T05:47:15.822533Z","iopub.status.idle":"2025-03-18T05:47:50.208403Z","shell.execute_reply.started":"2025-03-18T05:47:15.822502Z","shell.execute_reply":"2025-03-18T05:47:50.207354Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}