{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## BirdCLEF+ 2025 Sample Submission\n\nThis is a quick run through the submission process. Test data is hidden, so we can't access it before submission. In order to make a valid submission, here's what we'll do:\n\n1. Make sure we predict for all 206 classes in the train data\n2. Load a list of test soundscapes\n3. Process each soundscape\n     - load audio\n     - split into 5-second chunks\n     - run model inference for each chunk\n     - save predictions\n4. Make submission csv file\n5. Submit\n\nOk, so here we go.","metadata":{}},{"cell_type":"code","source":"import os\nimport librosa\nimport numpy as np\nimport pandas as pd\n\n# Set seed\nnp.random.seed(42)\n\n# Class labels from train audio\nclass_labels = sorted(os.listdir('/kaggle/input/birdclef-2025/train_audio/'))\n\n# List of test soundscapes (only visible during submission)\ntest_soundscape_path = '/kaggle/input/birdclef-2025/test_soundscapes/'\ntest_soundscapes = [os.path.join(test_soundscape_path, afile) for afile in sorted(os.listdir(test_soundscape_path)) if afile.endswith('.ogg')]\n\n# Open each soundscape and make predictions for 5-second segments\n# Use pandas df with 'row_id' plus class labels as columns\npredictions = pd.DataFrame(columns=['row_id'] + class_labels)\n\nfor soundscape in test_soundscapes:\n\n    # Load audio\n    sig, rate = librosa.load(path=soundscape, sr=None)\n\n    # Split into 5-second chunks\n    chunks = []\n    for i in range(0, len(sig), rate*5):\n        chunk = sig[i:i+rate*5]\n        chunks.append(chunk)\n        \n    # Make predictions for each chunk\n    for i, chunk in enumerate(chunks):\n        \n        # Get row id  (soundscape id + end time of 5s chunk)      \n        row_id = os.path.basename(soundscape).split('.')[0] + f'_{i * 5 + 5}'\n        \n        # Make prediction (let's use random scores for now)\n        # scores = model.predict...\n        scores = np.random.rand(len(class_labels))\n\n     # Append to predictions as new row\n        new_row = pd.DataFrame([[row_id] + list(scores)], columns=['row_id'] + class_labels)\n        predictions = pd.concat([predictions, new_row], axis=0, ignore_index=True)\n        \n# Save prediction as csv\npredictions.to_csv('submission.csv', index=False)\npredictions.head()\n        ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-26T16:17:18.884644Z","iopub.execute_input":"2025-05-26T16:17:18.885055Z","iopub.status.idle":"2025-05-26T16:17:20.514489Z","shell.execute_reply.started":"2025-05-26T16:17:18.885014Z","shell.execute_reply":"2025-05-26T16:17:20.513073Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"In order to make a submission, we need to:\n- disable internet for this notebook (Settings --> Turn off internet)\n- make sure the notebook runs without errors and a submission file gets created\n- submit to competition (panel on the right)\n- wait for the notebook to finish (this may take a while, remember there's a 90-min time limit)\n\nIf all goes well, we should see our submission scores on the leaderboard.","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nfrom glob import glob\nimport librosa  # Make sure librosa is imported\nimport librosa.display \nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T16:17:20.515408Z","iopub.execute_input":"2025-05-26T16:17:20.515807Z","iopub.status.idle":"2025-05-26T16:17:38.812241Z","shell.execute_reply.started":"2025-05-26T16:17:20.515778Z","shell.execute_reply":"2025-05-26T16:17:38.810888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n\n# Get list of audio files\naudio_files = glob('/kaggle/input/birdclef-2025/train_audio/*/*.ogg')\n\n# Output directory for saved spectrograms\noutput_dir = '/kaggle/working/mel_spectrograms/'\nos.makedirs(output_dir, exist_ok=True)\ndata_dir = '/kaggle/input/birdclef-2025/train_audio/'  # Input audio directory\noutput_dir = '/kaggle/working/mel_spectrograms/'  # Output directory for .npy files\nos.makedirs(output_dir, exist_ok=True)\nsample_percentage = 0.1  # Process 10% of the data\nsr = 32000  # Sample rate\nn_mels = 128  # Number of Mel bands\ntest_size = 0.2 # Size of the validation set\nrandom_state = 42 # Random state for reproducibility\nepochs = 20  # Adjust as needed\nbatch_size = 32  # Adjust as needed\nmodel_filename = 'birdclef_cnn_model.h5' # Filename to save the trained model\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T16:17:38.813379Z","iopub.execute_input":"2025-05-26T16:17:38.814028Z","iopub.status.idle":"2025-05-26T16:17:40.894565Z","shell.execute_reply.started":"2025-05-26T16:17:38.813998Z","shell.execute_reply":"2025-05-26T16:17:40.893572Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_mel_spectrogram(audio_path, sr=32000, n_mels=128):\n    y, sr = librosa.load(audio_path, sr=sr)\n    y, _ = librosa.effects.trim(y, top_db=20)  # remove silence\n    mel = librosa.feature.melspectrogram(y=y, sr=sr, n_mels=n_mels)\n    mel_db = librosa.power_to_db(mel, ref=np.max)\n    return mel_db","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T16:17:40.895555Z","iopub.execute_input":"2025-05-26T16:17:40.895918Z","iopub.status.idle":"2025-05-26T16:17:40.901673Z","shell.execute_reply.started":"2025-05-26T16:17:40.89589Z","shell.execute_reply":"2025-05-26T16:17:40.900503Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_files = len(audio_files)\nnum_to_process = int(sample_percentage * num_files)\nsampled_audio_files = np.random.choice(audio_files, num_to_process, replace=False) # Use numpy.random.choice\n\nmel_spectrograms = []\nlabels = []\n\nfor audio_file in tqdm(sampled_audio_files, desc=\"Processing Audio Files\"): # Use tqdm\n    try:\n        mel = extract_mel_spectrogram(audio_file, sr=sr, n_mels=n_mels)  # Use the defined function\n\n        # Save as .npy (efficient for CNN input)\n        species_folder = os.path.basename(os.path.dirname(audio_file))\n        file_id = os.path.splitext(os.path.basename(audio_file))[0]\n        species_dir = os.path.join(output_dir, species_folder)\n        os.makedirs(species_dir, exist_ok=True)\n        np.save(os.path.join(species_dir, f\"{file_id}.npy\"), mel)  # Save the mel spectrogram\n\n        mel_spectrograms.append(mel)\n        labels.append(species_folder)\n\n    except Exception as e:\n        print(f\"Failed to process {audio_file}: {e}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T16:17:40.902837Z","iopub.execute_input":"2025-05-26T16:17:40.903185Z","iopub.status.idle":"2025-05-26T16:27:54.684246Z","shell.execute_reply.started":"2025-05-26T16:17:40.90315Z","shell.execute_reply":"2025-05-26T16:27:54.682677Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T16:27:54.687333Z","iopub.execute_input":"2025-05-26T16:27:54.688429Z","iopub.status.idle":"2025-05-26T16:27:54.695626Z","shell.execute_reply.started":"2025-05-26T16:27:54.688387Z","shell.execute_reply":"2025-05-26T16:27:54.693779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert to NumPy arrays\nmel_spectrograms = np.array(output_dir)\nlabels = np.array(labels)\n\n\n# Add a channel dimension (for CNN input) - assuming your Mel spectrograms are 2D\nif mel_spectrograms.ndim == 3:\n    mel_spectrograms = np.expand_dims(mel_spectrograms, axis=-1)\n\n# Encode labels\nlabel_encoder = LabelEncoder()\nencoded_labels = label_encoder.fit_transform(labels)\nnum_classes = len(label_encoder.classes_)\nencoded_labels = tf.keras.utils.to_categorical(encoded_labels, num_classes=num_classes)\n\n# Split data into training and validation sets\nX_train, X_val, y_train, y_val = train_test_split(\n    mel_spectrograms, encoded_labels, test_size=test_size, stratify=encoded_labels, random_state=random_state\n)\n\n# --- 3. CNN Model Definition ---\ndef create_cnn(input_shape, num_classes):\n    model = models.Sequential([\n        layers.Conv2D(32, (3, 3), activation='relu', input_shape=input_shape),\n        layers.MaxPooling2D((2, 2)),\n        layers.Conv2D(64, (3, 3), activation='relu'),\n        layers.MaxPooling2D((2, 2)),\n        layers.Flatten(),\n        layers.Dense(128, activation='relu'),\n        layers.Dropout(0.5),\n        layers.Dense(num_classes, activation='softmax')  # Output layer for classification\n    ])\n    return model\n\n# Get the input shape of your Mel spectrograms\ninput_shape = X_train.shape[1:]\nmodel = create_cnn(input_shape, num_classes)\n\n# --- 4. Model Compilation ---\nmodel.compile(optimizer='adam',\n              loss='categorical_crossentropy',\n              metrics=['accuracy'])\n\n# --- 5. Model Training ---\nhistory = model.fit(X_train, y_train,\n                    epochs=epochs,\n                    batch_size=batch_size,\n                    validation_data=(X_val, y_val))\n\n# --- 6. Evaluate the Model ---\nloss, accuracy = model.evaluate(X_val, y_val, verbose=0)\nprint(f\"Validation Loss: {loss:.4f}\")\nprint(f\"Validation Accuracy: {accuracy:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T16:27:54.696931Z","iopub.execute_input":"2025-05-26T16:27:54.69725Z","iopub.status.idle":"2025-05-26T16:27:54.789674Z","shell.execute_reply.started":"2025-05-26T16:27:54.697218Z","shell.execute_reply":"2025-05-26T16:27:54.787955Z"}},"outputs":[],"execution_count":null}]}