{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":70203,"databundleVersionId":8068726}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nfrom tensorflow.keras import layers, optimizers , datasets , models\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nimport librosa\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-18T18:35:20.161047Z","iopub.execute_input":"2026-05-18T18:35:20.161782Z","iopub.status.idle":"2026-05-18T18:35:20.165888Z","shell.execute_reply.started":"2026-05-18T18:35:20.161753Z","shell.execute_reply":"2026-05-18T18:35:20.165181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"meta = pd.read_csv(\"/kaggle/input/competitions/birdclef-2024/train_metadata.csv\")\nmeta = meta.drop([\"author\",\"license\",\"url\"],axis = 1 )\nmeta.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-18T18:35:20.178380Z","iopub.execute_input":"2026-05-18T18:35:20.178892Z","iopub.status.idle":"2026-05-18T18:35:20.285743Z","shell.execute_reply.started":"2026-05-18T18:35:20.178864Z","shell.execute_reply":"2026-05-18T18:35:20.285091Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"meta['file_path'] = '/kaggle/input/competitions/birdclef-2024/train_audio/' + meta['filename']\nmeta = meta[meta['rating'] >= 3.0].reset_index(drop=True)\nmeta.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-18T18:35:20.287125Z","iopub.execute_input":"2026-05-18T18:35:20.287440Z","iopub.status.idle":"2026-05-18T18:35:20.307723Z","shell.execute_reply.started":"2026-05-18T18:35:20.287416Z","shell.execute_reply":"2026-05-18T18:35:20.307114Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(meta.shape)\nprint(len(meta[\"scientific_name\"].unique()))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-18T18:35:20.308559Z","iopub.execute_input":"2026-05-18T18:35:20.308864Z","iopub.status.idle":"2026-05-18T18:35:20.314140Z","shell.execute_reply.started":"2026-05-18T18:35:20.308828Z","shell.execute_reply":"2026-05-18T18:35:20.313399Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dur = 5\nN_MELS = 128\nsamples = 1000\n\ndef audio2img(filepath):\n    audio, sr = librosa.load(filepath, duration=dur, mono=True)\n    target = sr * dur\n    # Sample rate in HZ\n    if len(audio) < target:\n        audio = np.pad(audio, (0, target - len(audio)))\n        \n    img = librosa.feature.melspectrogram(\n        y=audio,\n        sr=sr,\n        n_mels=N_MELS,\n        hop_length=512\n    )\n    # freq. axis is scaled according to mel\n    # Time , frequency , amplitude\n    img = librosa.power_to_db(img, ref=np.max)\n    img = (img - img.min()) / (img.max() - img.min() + 1e-6)\n    # Min-Max Scaling\n    img = img[:, :, np.newaxis]\n    return img","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-18T18:35:20.316060Z","iopub.execute_input":"2026-05-18T18:35:20.316348Z","iopub.status.idle":"2026-05-18T18:35:20.325516Z","shell.execute_reply.started":"2026-05-18T18:35:20.316326Z","shell.execute_reply":"2026-05-18T18:35:20.324905Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = []   \ny = []  \n\n\n\nfor i, (_, row) in enumerate(meta.iterrows()):\n    filepath = row[\"file_path\"]\n    \n    img = audio2img(filepath)\n    if img is not None:\n        X.append(img)\n        y.append(row['primary_label'])\n    if (i + 1) % 100 == 0 or (i + 1) == len(meta):\n        print(f'   Progress: {i+1}/{len(meta)} files processed | Collected: {len(X)}')\n        \nX = np.array(X, dtype=np.float32)\ny = np.array(y)\n\nprint(f'   X shape : {X.shape}')\nprint(f'   y shape : {y.shape}')\nprint(f'   Species : {np.unique(y)[:5]}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-18T18:35:20.326338Z","iopub.execute_input":"2026-05-18T18:35:20.326712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"le = LabelEncoder()\ny = le.fit_transform(y)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_train, x_val, y_train, y_val = train_test_split(\n    X,\n    y, \n    test_size=0.2, \n    random_state=42,\n    stratify=y\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inputs = tf.keras.Input(shape=(N_MELS, 216, 1))\n\nx = layers.Conv2D(32, (3, 3), activation='relu', padding='same')(inputs)\nx = layers.Conv2D(32, (3, 3), activation='relu', padding='same')(x)\nx = layers.MaxPooling2D((2, 2))(x)\nx = layers.Dropout(0.2)(x)\n\n\nx = layers.Conv2D(64, (3, 3), activation='relu', padding='same')(x)\nx = layers.Conv2D(64, (3, 3), activation='relu', padding='same')(x)\nx = layers.MaxPooling2D((2, 2))(x)\nx = layers.Dropout(0.25)(x)\n\n\nx = layers.Conv2D(128, (3, 3), activation='relu', padding='same')(x)\nx = layers.Conv2D(128, (3, 3), activation='relu', padding='same')(x)\nx = layers.MaxPooling2D((2, 2))(x)\nx = layers.Dropout(0.3)(x)\n\nx = layers.GlobalAveragePooling2D()(x)\nx = layers.Dense(256, activation='relu')(x)\nx = layers.Dropout(0.5)(x)\noutputs = layers.Dense(182, activation='softmax')(x)\n\nmodel = models.Model(inputs=inputs, outputs=outputs, name=\"audio_Model\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.summary()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.compile(\n    optimizer='Adam', \n    loss='sparse_categorical_crossentropy', \n    metrics=['accuracy']\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"epochs = 50\nBATCH_SIZE = 128\n\nhistory = model.fit(x_train, y_train, epochs=epochs, validation_split=0.2 , batch_size=BATCH_SIZE)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 5))\n\nplt.subplot(1, 2, 1)\nplt.plot(history.history['accuracy'], label='Train Accuracy')\nplt.plot(history.history['val_accuracy'], label='Validation Accuracy')\nplt.title('Model Accuracy')\nplt.legend()\n\nplt.subplot(1, 2, 2)\nplt.plot(history.history['loss'], label='Train Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.title('Model Loss')\nplt.legend()\n\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"score = model.evaluate(x_val,y_val)\nprint(\"Test loss:\", score[0])\nprint(\"Test accuracy:\", score[1])","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}