{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Hello fellow Kagglers,\n\nThis notebook demonstrates the generation of a dataset with melspectrograms in PNG format that are saved in a single dictionary.\n\nThe audio files are converted to melspectrograms and saved in raw PNG bytes for efficient saving.\n\nIn a training notebook the dictionary can be loaded in memory and decoding is quick as the raw PNG bytes are already loaded in working memory.\n\nThe next step is to select the parts of the melspectrograms which are likely to contain a bird call, which will be shown in the soon to be released training notebook.\n\nHappy Kaggling!\n\n**UPDATES**\n\n**V2**\n* hardcoded sampling rate of 32K\n* updated spectrogram settings to get 5s windows of exactly 320 pixels\n* normalizing audio","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport imageio.v3 as imageio\n\nfrom tqdm.notebook import tqdm\n\nimport librosa\nimport cv2\nimport pickle\nimport lzma","metadata":{"execution":{"iopub.status.busy":"2024-04-08T13:50:41.587030Z","iopub.execute_input":"2024-04-08T13:50:41.587549Z","iopub.status.idle":"2024-04-08T13:50:43.269974Z","shell.execute_reply.started":"2024-04-08T13:50:41.587513Z","shell.execute_reply":"2024-04-08T13:50:43.268921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configuration","metadata":{}},{"cell_type":"code","source":"class Config():\n    # Horizontal melspectrogram resolution\n    MELSPEC_H = 128\n    # Competition Root Folder\n    ROOT_FOLDER = '/kaggle/input/birdclef-2024'\n    # Maximum decibel to clip audio to\n    TOP_DB = 100\n    # Minimum rating\n    MIN_RATING = 3.0\n    # Sample rate as provided in competition description\n    SR = 32000\n    N_FFT = 2000\n    HOP_LENGTH = 500\n    \nCONFIG = Config()","metadata":{"execution":{"iopub.status.busy":"2024-04-08T13:50:43.271752Z","iopub.execute_input":"2024-04-08T13:50:43.272184Z","iopub.status.idle":"2024-04-08T13:50:43.278406Z","shell.execute_reply.started":"2024-04-08T13:50:43.272156Z","shell.execute_reply":"2024-04-08T13:50:43.277278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Sample Submission","metadata":{}},{"cell_type":"code","source":"sample_submission = pd.read_csv('/kaggle/input/birdclef-2024/sample_submission.csv')\n\n# Set labels\nCONFIG.LABELS = sample_submission.columns[1:]\nCONFIG.N_LABELS = len(CONFIG.LABELS)\nprint(f'# labels: {CONFIG.N_LABELS}')\n\ndisplay(sample_submission.head())","metadata":{"execution":{"iopub.status.busy":"2024-04-08T13:50:43.279542Z","iopub.execute_input":"2024-04-08T13:50:43.280607Z","iopub.status.idle":"2024-04-08T13:50:43.348756Z","shell.execute_reply.started":"2024-04-08T13:50:43.280564Z","shell.execute_reply":"2024-04-08T13:50:43.347594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train MetaData","metadata":{}},{"cell_type":"code","source":"train_metadata_df = pd.read_csv(\n        '/kaggle/input/birdclef-2024/train_metadata.csv',\n        dtype={\n            'secondary_labels': 'string',\n            'primary_label': 'category',\n        },\n    )\n\n# Convert secondary_labels to iterable tuple\ndef parse_secondary_labels(s):\n    s = s.strip(\"[']\")\n    s = s.split(\"', '\")\n    return tuple([e for e in s if len(e) > 0])\n\ntrain_metadata_df['secondary_labels'] = train_metadata_df['secondary_labels'].apply(parse_secondary_labels)\n\n# Number of samples\nCONFIG.N_SAMPLES = len(train_metadata_df)\nprint(f'# Samples: {CONFIG.N_SAMPLES:,}')\n\ndisplay(train_metadata_df.head())\ndisplay(train_metadata_df.info())","metadata":{"execution":{"iopub.status.busy":"2024-04-08T13:50:43.351291Z","iopub.execute_input":"2024-04-08T13:50:43.351643Z","iopub.status.idle":"2024-04-08T13:50:43.664439Z","shell.execute_reply.started":"2024-04-08T13:50:43.351613Z","shell.execute_reply":"2024-04-08T13:50:43.663428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploratory Data Analysis","metadata":{}},{"cell_type":"code","source":"# Rating value counts\nvc = train_metadata_df['rating'].value_counts().sort_index(ascending=False)\n\nplt.figure(figsize=(8, 6))\nplt.title('Rating Distribution')\nplt.pie(vc, labels=vc.index, autopct='%1.1f%%')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-08T13:50:43.665696Z","iopub.execute_input":"2024-04-08T13:50:43.666106Z","iopub.status.idle":"2024-04-08T13:50:43.971451Z","shell.execute_reply.started":"2024-04-08T13:50:43.666072Z","shell.execute_reply":"2024-04-08T13:50:43.969947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# OGG → Melspectrogram Conversion","metadata":{}},{"cell_type":"code","source":"# Convert OOG audio files to melspectrogram encoded as PNG bytes\ndef ogg2melspectrogram(file_path):\n    # Load the audio file\n    y, _ = librosa.load(file_path, sr=CONFIG.SR)\n    # Normalize audio\n    y = librosa.util.normalize(y)\n    # Convert to mel spectrogram\n    spec = librosa.feature.melspectrogram(\n        y=y,\n        sr=CONFIG.SR, # sample rate\n        n_fft=CONFIG.N_FFT, # number of samples in window \n        hop_length=CONFIG.HOP_LENGTH, # step size of window\n        n_mels=CONFIG.MELSPEC_H, # horizontal resolution from fmin→fmax in log scale\n        fmin=40, # minimum frequency\n        fmax=15000, # maximum frequency\n        power=2.0, # intensity^power for log scale\n    )\n    # Convert to Db\n    spec = librosa.power_to_db(spec, ref=CONFIG.TOP_DB)\n    # Normalize 0-min\n    spec = spec - spec.min()\n    # Normalize 0-255\n    spec = (spec / spec.max() * 255).astype(np.uint8)\n    # Convert to PNG bytes\n    _, spec_png_uint8 = cv2.imencode('.png', spec)\n    spec_png_bytes = bytes(spec_png_uint8)\n    \n    return spec_png_bytes","metadata":{"execution":{"iopub.status.busy":"2024-04-08T13:50:43.973834Z","iopub.execute_input":"2024-04-08T13:50:43.974377Z","iopub.status.idle":"2024-04-08T13:50:43.982829Z","shell.execute_reply.started":"2024-04-08T13:50:43.974347Z","shell.execute_reply":"2024-04-08T13:50:43.981784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Maps a class to corresponding integer label\nCLASS2LABEL = dict(zip(CONFIG.LABELS, np.arange(CONFIG.N_LABELS)))\n# Label to class mapping\nLABEL2CLASS = dict([(v,k) for k, v in CLASS2LABEL.items()])\n# Create dataset\nX = {}\ny = {}\nfor idx, row in tqdm(train_metadata_df.iterrows(), total=CONFIG.N_SAMPLES):\n    # Guard for absence of secundary label and minimum rating\n    if row['rating'] >= CONFIG.MIN_RATING and len(row['secondary_labels']) == 0:\n        # Save PNGs in X\n        X[idx] = ogg2melspectrogram(f'{CONFIG.ROOT_FOLDER}/train_audio/{row.filename}')\n        # Save labels in y\n        y[idx] = CLASS2LABEL.get(row['primary_label'])\n        \nprint(f'# Training Samples: {len(X):,}')","metadata":{"execution":{"iopub.status.busy":"2024-04-08T13:50:43.984209Z","iopub.execute_input":"2024-04-08T13:50:43.984933Z","iopub.status.idle":"2024-04-08T13:51:21.887388Z","shell.execute_reply.started":"2024-04-08T13:50:43.984891Z","shell.execute_reply":"2024-04-08T13:51:21.885543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Example Plots\ndef example_plots(X, y, N):\n    np.random.seed(42)\n    random_keys = np.random.choice(list(X.keys()), N)\n    \n    for k in random_keys:\n        spec = imageio.imread(X[k])\n        plt.figure(figsize=(12,5))\n        plt.title(\n                f'Label: {y[k]}, Class: {LABEL2CLASS[y[k]]}, shape: {spec.shape}, ' +\n                f'min: {spec.min():.0f}, max: {spec.max():.0f}, ' +\n                f'µ: {spec.mean():.1f}, σ: {spec.std():.1f}'\n            )\n        plt.imshow(spec)\n        plt.show()\n        \nexample_plots(X, y, 8)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T13:51:21.888807Z","iopub.status.idle":"2024-04-08T13:51:21.890137Z","shell.execute_reply.started":"2024-04-08T13:51:21.889819Z","shell.execute_reply":"2024-04-08T13:51:21.889853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Write X\nwith open('X.pkl', 'wb') as f:\n    pickle.dump(X, f)\n    \n# Write y\nwith open('y.pkl', 'wb') as f:\n    pickle.dump(y, f)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T13:51:21.891987Z","iopub.status.idle":"2024-04-08T13:51:21.892874Z","shell.execute_reply.started":"2024-04-08T13:51:21.892637Z","shell.execute_reply":"2024-04-08T13:51:21.892660Z"},"trusted":true},"execution_count":null,"outputs":[]}]}