{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\nimport cv2\nimport math\nimport time\nimport librosa\nfrom tqdm.notebook import tqdm\nimport torch\nimport matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T02:51:00.073259Z","iopub.execute_input":"2025-05-01T02:51:00.073987Z","iopub.status.idle":"2025-05-01T02:51:00.079182Z","shell.execute_reply.started":"2025-05-01T02:51:00.073957Z","shell.execute_reply":"2025-05-01T02:51:00.078441Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SAMPLE_RATE = 32000\nFFT_SIZE = 1024\nHOP_SIZE = 512\nNUM_MELS = 128\nFREQ_MIN = 50\nFREQ_MAX = 14000\nCLIP_DURATION = 5.0\nSPEC_SHAPE = (256, 256)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T02:21:07.716344Z","iopub.execute_input":"2025-05-01T02:21:07.716596Z","iopub.status.idle":"2025-05-01T02:21:07.720294Z","shell.execute_reply.started":"2025-05-01T02:21:07.716581Z","shell.execute_reply":"2025-05-01T02:21:07.719548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"taxonomy = pd.read_csv('/kaggle/input/birdclef-2025/taxonomy.csv')\ntrain_meta = pd.read_csv('/kaggle/input/birdclef-2025/train.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T02:21:07.901687Z","iopub.execute_input":"2025-05-01T02:21:07.901913Z","iopub.status.idle":"2025-05-01T02:21:08.041748Z","shell.execute_reply.started":"2025-05-01T02:21:07.901898Z","shell.execute_reply":"2025-05-01T02:21:08.04109Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"species_to_class = dict(zip(taxonomy['primary_label'], taxonomy['class_name']))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T02:21:08.133398Z","iopub.execute_input":"2025-05-01T02:21:08.134155Z","iopub.status.idle":"2025-05-01T02:21:08.13907Z","shell.execute_reply.started":"2025-05-01T02:21:08.134125Z","shell.execute_reply":"2025-05-01T02:21:08.138372Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"species_labels = sorted(train_meta['primary_label'].unique())\nlabel_ids = list(range(len(species_labels)))\nlabel_to_id = dict(zip(species_labels, label_ids))\nid_to_label = dict(zip(label_ids, species_labels))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T02:21:08.343765Z","iopub.execute_input":"2025-05-01T02:21:08.344049Z","iopub.status.idle":"2025-05-01T02:21:08.350121Z","shell.execute_reply.started":"2025-05-01T02:21:08.34401Z","shell.execute_reply":"2025-05-01T02:21:08.349396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Number of distinct species found: {len(species_labels)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T02:21:09.89059Z","iopub.execute_input":"2025-05-01T02:21:09.891199Z","iopub.status.idle":"2025-05-01T02:21:09.895047Z","shell.execute_reply.started":"2025-05-01T02:21:09.891181Z","shell.execute_reply":"2025-05-01T02:21:09.894249Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = train_meta[['primary_label', 'rating', 'filename']].copy()\ndata['label_id'] = data['primary_label'].map(label_to_id)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T02:21:10.05468Z","iopub.execute_input":"2025-05-01T02:21:10.054911Z","iopub.status.idle":"2025-05-01T02:21:10.065046Z","shell.execute_reply.started":"2025-05-01T02:21:10.054896Z","shell.execute_reply":"2025-05-01T02:21:10.064454Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data['file_path'] = '/kaggle/input/birdclef-2025/train_audio/' + data['filename']\ndata['sample_id'] = data['filename'].apply(lambda f: f.split('/')[0] + '-' + f.split('/')[-1].split('.')[0])\ndata['class_name'] = data['primary_label'].map(lambda k: species_to_class.get(k, 'Unknown'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T02:21:10.459524Z","iopub.execute_input":"2025-05-01T02:21:10.460174Z","iopub.status.idle":"2025-05-01T02:21:10.497954Z","shell.execute_reply.started":"2025-05-01T02:21:10.460153Z","shell.execute_reply":"2025-05-01T02:21:10.497328Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"max_samples = len(data)\nprint(f\"Preparing {max_samples} samples from a total of {len(data)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T02:21:11.537617Z","iopub.execute_input":"2025-05-01T02:21:11.537856Z","iopub.status.idle":"2025-05-01T02:21:11.541883Z","shell.execute_reply.started":"2025-05-01T02:21:11.537841Z","shell.execute_reply":"2025-05-01T02:21:11.541205Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Class distribution:\")\nprint(data['class_name'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T02:21:11.903258Z","iopub.execute_input":"2025-05-01T02:21:11.903891Z","iopub.status.idle":"2025-05-01T02:21:11.912634Z","shell.execute_reply.started":"2025-05-01T02:21:11.903868Z","shell.execute_reply":"2025-05-01T02:21:11.911863Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def waveform_to_melspectrogram(waveform):\n    if np.isnan(waveform).any():\n        avg_value = np.nanmean(waveform)\n        waveform = np.nan_to_num(waveform, nan=avg_value)\n\n    mel = librosa.feature.melspectrogram(\n        y=waveform,\n        sr=SAMPLE_RATE,\n        n_fft=FFT_SIZE,\n        hop_length=HOP_SIZE,\n        n_mels=NUM_MELS,\n        fmin=FREQ_MIN,\n        fmax=FREQ_MAX,\n        power=2.0\n    )\n\n    mel_db = librosa.power_to_db(mel, ref=np.max)\n\n    mel_normalized = (mel_db - mel_db.min()) / (mel_db.max() - mel_db.min() + 1e-8)\n\n    return mel_normalized","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T02:21:13.267358Z","iopub.execute_input":"2025-05-01T02:21:13.267609Z","iopub.status.idle":"2025-05-01T02:21:13.272592Z","shell.execute_reply.started":"2025-05-01T02:21:13.267594Z","shell.execute_reply":"2025-05-01T02:21:13.271878Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Beginning spectrogram extraction...\")\nstart_time = time.time()\n\nprocessed_specs = {}\nfailed_files = []\n\nfor idx, row in tqdm(data.iterrows(), total=len(data)):\n    try:\n        # Load and resample audio to fixed sample rate\n        waveform, _ = librosa.load(row.file_path, sr=SAMPLE_RATE)\n\n        required_len = int(CLIP_DURATION * SAMPLE_RATE)\n\n        # Repeat audio if it's too short\n        if len(waveform) < required_len:\n            repeat_factor = math.ceil(required_len / len(waveform))\n            if repeat_factor > 1:\n                waveform = np.tile(waveform, repeat_factor)\n\n        # Extract center segment\n        mid = len(waveform) // 2\n        half_window = required_len // 2\n        start = max(0, mid - half_window)\n        end = min(len(waveform), start + required_len)\n        clip = waveform[start:end]\n\n        # Pad if still not enough samples\n        if len(clip) < required_len:\n            pad_width = required_len - len(clip)\n            clip = np.pad(clip, (0, pad_width), mode='constant')\n\n        # Generate mel-spectrogram\n        mel_output = waveform_to_melspectrogram(clip)\n\n        # Resize to target shape if needed\n        if mel_output.shape != SPEC_SHAPE:\n            mel_output = cv2.resize(mel_output, SPEC_SHAPE, interpolation=cv2.INTER_LINEAR)\n\n        # Store the result\n        processed_specs[row.sample_id] = mel_output.astype(np.float32)\n\n    except Exception as err:\n        print(f\"Failed to process {row.file_path}: {err}\")\n        failed_files.append((row.file_path, str(err)))\n\nelapsed = time.time() - start_time\nprint(f\"Finished processing in {elapsed:.2f} seconds.\")\n# print(f\"Successfully processed {len(processed_specs)} / {total_samples} files.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T02:21:13.455711Z","iopub.execute_input":"2025-05-01T02:21:13.455962Z","iopub.status.idle":"2025-05-01T02:41:19.065192Z","shell.execute_reply.started":"2025-05-01T02:21:13.455935Z","shell.execute_reply":"2025-05-01T02:41:19.064192Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected_samples = []\nseen_classes = set()\n\n# Define how many spectrograms to display\nnum_to_display = min(4, len(processed_specs))\nlimit = len(data)\n\n# Collect unique class examples\nfor idx, row in data.iterrows():\n    if idx >= limit:\n        break\n\n    sample_id = row['sample_id']\n    class_label = row['class_name']\n    species_id = row['primary_label']\n\n    if sample_id not in processed_specs:\n        continue\n\n    if class_label not in seen_classes:\n        selected_samples.append((sample_id, class_label, species_id))\n        seen_classes.add(class_label)\n\n    if len(selected_samples) >= num_to_display:\n        break\n# Plot if any samples were collected\nif selected_samples:\n    plt.figure(figsize=(16, 12))\n\n    for i, (sample_id, class_label, species_id) in enumerate(selected_samples):\n        plt.subplot(2, 2, i + 1)\n        plt.imshow(processed_specs[sample_id], aspect='auto', origin='lower', cmap='viridis')\n        plt.title(f\"{class_label}: {species_id}\")\n        plt.colorbar(format='%+2.0f dB')\n\n    plt.tight_layout()\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T02:51:03.769505Z","iopub.execute_input":"2025-05-01T02:51:03.770193Z","iopub.status.idle":"2025-05-01T02:51:05.535444Z","shell.execute_reply.started":"2025-05-01T02:51:03.770173Z","shell.execute_reply":"2025-05-01T02:51:05.534422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"np.save('falcon_birdclef_cnn_preprocessed_dataset.npy', processed_specs, allow_pickle=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T02:56:26.309608Z","iopub.execute_input":"2025-05-01T02:56:26.309848Z","iopub.status.idle":"2025-05-01T02:57:05.564045Z","shell.execute_reply.started":"2025-05-01T02:56:26.309834Z","shell.execute_reply":"2025-05-01T02:57:05.563263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}