{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nimport os\nfrom glob import glob\nimport sys\nimport ast \nimport random \nfrom tqdm import tqdm\nimport time \nimport json\nimport soundfile as sf\nimport librosa\nimport librosa.display\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom IPython.display import Audio\n\nfrom sklearn.preprocessing import OneHotEncoder, StandardScaler\nfrom sklearn.model_selection import train_test_split\n\nfrom sklearn.metrics import accuracy_score, f1_score, precision_score, recall_score\n\n\nimport timm \nimport torch\nfrom torch import nn\nfrom torch.utils.data import Dataset, DataLoader\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T10:06:50.286660Z","iopub.execute_input":"2025-05-13T10:06:50.287350Z","iopub.status.idle":"2025-05-13T10:07:06.113179Z","shell.execute_reply.started":"2025-05-13T10:06:50.287312Z","shell.execute_reply":"2025-05-13T10:07:06.112619Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CONFIG:\n    TRAIN = '/kaggle/input/birdclef-2025/train_audio'\n    OUTPUT = '/kaggle/working/'\n    TRAIN_CSV = '/kaggle/input/birdclef-2025/train.csv'\n    TAXONOMY = '/kaggle/input/birdclef-2025/taxonomy.csv'\n    SR = 32000\n    DURATION = 10\n    HOP_LENGTH = 512\n    N_MELS = 128\n    FMIN = 20\n    FMAX = SR//2\n    # Model Parameters\n    MODEL_NAME = \"tf_efficientnet_b0\"  \n    PRETRAINED = True\n    NUM_CLASSES = 206                  \n\n    # Training Parameters\n    EPOCHS = 1\n    BATCH_SIZE = 3\n    LR = 1e-4\n    SEED = 42\n    NUM_WORKERS = 0\n\n    # Inference\n    THRESHOLD = 0.5\n    DEVICE = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n\n    # Debug Mode\n    DEBUG = False\n\n\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T10:07:06.114214Z","iopub.execute_input":"2025-05-13T10:07:06.114620Z","iopub.status.idle":"2025-05-13T10:07:06.180356Z","shell.execute_reply.started":"2025-05-13T10:07:06.114589Z","shell.execute_reply":"2025-05-13T10:07:06.179490Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv(CONFIG.TRAIN_CSV)\ntaxonomy = pd.read_csv(CONFIG.TAXONOMY)\n\ntaxonomy.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T10:07:06.181068Z","iopub.execute_input":"2025-05-13T10:07:06.181253Z","iopub.status.idle":"2025-05-13T10:07:06.371147Z","shell.execute_reply.started":"2025-05-13T10:07:06.181237Z","shell.execute_reply":"2025-05-13T10:07:06.370491Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T10:07:06.371814Z","iopub.execute_input":"2025-05-13T10:07:06.372070Z","iopub.status.idle":"2025-05-13T10:07:06.384304Z","shell.execute_reply.started":"2025-05-13T10:07:06.372043Z","shell.execute_reply":"2025-05-13T10:07:06.383725Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_path = pd.merge(train,taxonomy[['scientific_name','class_name']],how='left',on='scientific_name')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T10:07:06.385948Z","iopub.execute_input":"2025-05-13T10:07:06.386211Z","iopub.status.idle":"2025-05-13T10:07:06.414713Z","shell.execute_reply.started":"2025-05-13T10:07:06.386187Z","shell.execute_reply":"2025-05-13T10:07:06.414154Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert stringified lists to actual Python lists\nfor col in ['secondary_labels', 'type']:\n    data_path[col] = data_path[col].apply(lambda x: ast.literal_eval(x))\n\n# Add full path to audio files\ndata_path['filepath'] = data_path['filename'].apply(lambda x: os.path.join(CONFIG.TRAIN, x))\n\n# Preview\ndata_path.sample(5)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T10:07:06.415398Z","iopub.execute_input":"2025-05-13T10:07:06.415561Z","iopub.status.idle":"2025-05-13T10:07:07.004066Z","shell.execute_reply.started":"2025-05-13T10:07:06.415549Z","shell.execute_reply":"2025-05-13T10:07:07.003475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Shape of training data:\", data_path.shape)\nprint(\"Columns:\", data_path.columns.tolist())\ndata_path.info()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T10:07:07.004706Z","iopub.execute_input":"2025-05-13T10:07:07.004893Z","iopub.status.idle":"2025-05-13T10:07:07.038591Z","shell.execute_reply.started":"2025-05-13T10:07:07.004880Z","shell.execute_reply":"2025-05-13T10:07:07.038040Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 8))\nsns.countplot(y='primary_label', data=data_path,\n              order=data_path['primary_label'].value_counts().iloc[:30].index)\nplt.title(\"Top 30 Most Frequent Bird Species\")\nplt.xlabel(\"Frequency\")\nplt.ylabel(\"Bird Species\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T10:07:07.039234Z","iopub.execute_input":"2025-05-13T10:07:07.039469Z","iopub.status.idle":"2025-05-13T10:07:07.444699Z","shell.execute_reply.started":"2025-05-13T10:07:07.039451Z","shell.execute_reply":"2025-05-13T10:07:07.443975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(8, 5))\nsns.barplot(x=data_path['rating'].value_counts().sort_index().index, y=data_path['rating'].value_counts().sort_index().values, palette=\"viridis\")\n\nplt.title(\"Distribution of Ratings in Training Data\")\nplt.xlabel(\"Rating\")\nplt.ylabel(\"Count\")\nplt.xticks(rotation=0)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T10:07:07.445614Z","iopub.execute_input":"2025-05-13T10:07:07.446348Z","iopub.status.idle":"2025-05-13T10:07:07.641731Z","shell.execute_reply.started":"2025-05-13T10:07:07.446328Z","shell.execute_reply":"2025-05-13T10:07:07.641112Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.title('Count of Bird Classes', size=16)\nsns.countplot(data = data_path, y ='class_name')\nplt.ylabel('Count', size=12)\nplt.xlabel('Classes', size=12)\nsns.despine(top=True, right=True, left=False, bottom=False)\nplt.show()\n## so we have 6 emotions, almost equally distributed\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T10:08:22.405727Z","iopub.execute_input":"2025-05-13T10:08:22.406105Z","iopub.status.idle":"2025-05-13T10:08:22.531928Z","shell.execute_reply.started":"2025-05-13T10:08:22.406071Z","shell.execute_reply":"2025-05-13T10:08:22.531327Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Compute duration of all audio files ","metadata":{}},{"cell_type":"code","source":"def get_duration(path):\n    f = sf.SoundFile(path)\n    return len(f) / f.samplerate\n\n# Apply to all audio files\ntqdm.pandas()\ndata_path[\"duration\"] = data_path[\"filepath\"].progress_apply(get_duration)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T10:08:35.056780Z","iopub.execute_input":"2025-05-13T10:08:35.057089Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot the distribution\nplt.figure(figsize=(10, 5))\nsns.histplot(data_path[\"duration\"], bins=50, kde=True, color=\"teal\")\nplt.title(\"Distribution of Audio Durations\")\nplt.xlabel(\"Duration (seconds)\")\nplt.ylabel(\"Number of Files\")\nplt.grid(True, linestyle=\"--\", alpha=0.6)\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Waveplots - Waveplots let us know the loudness of the audio at a given time.\n# Spectograms - A spectrogram is a visual representation of the spectrum of frequencies of sound or other signals \n# as they vary with time. It’s a representation of frequencies changing with respect to time for given audio/music signals.\n\ndef create_waveplot(sample_file, e=None):\n    data, sr = librosa.load(sample_file)\n\n    plt.figure(figsize=(10, 3))\n    plt.title('Waveplot for {} class'.format(e), size=15)\n    librosa.display.waveshow(data, sr=sr)\n    plt.show()\n\ndef create_spectrogram(sample_file, bird_class=None):\n    data, sr = librosa.load(sample_file)\n\n    # stft function converts the data into short term fourier transform\n    X = librosa.stft(data)\n    Xdb = librosa.amplitude_to_db(abs(X))\n    plt.figure(figsize=(12, 3))\n    if bird_class is not None:\n        plt.title('Spectrogram for {} class'.format(bird_class), size=15)\n    librosa.display.specshow(Xdb, sr=sr, x_axis='time', y_axis='hz')   \n    plt.colorbar()\n    plt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nsample_file , bird = data_path.sample(1)[['filepath','class_name']].values[0]\ncreate_waveplot(sample_file, bird)\ncreate_spectrogram(sample_file, bird)\nAudio(sample_file)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i, row in data_path.sample(5).iterrows():\n    print(f\"Sample {i+1} - Primary Label: {row['primary_label']}\")\n    create_spectrogram(row['filepath'])\n    Audio(row['filepath'])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_path['duration'].sort_values(ascending=True)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preprocess_audio(path, sr=CONFIG.SR, duration=CONFIG.DURATION, n_mels=CONFIG.N_MELS,\n                     fmin=CONFIG.FMIN, fmax=CONFIG.FMAX, hop_length=CONFIG.HOP_LENGTH):\n    \n    y, sr = librosa.load(path, sr=sr, duration=duration)\n    \n    # Pad if audio is too short\n    expected_length = sr * duration\n    if len(y) < expected_length:\n        y = np.pad(y, (0, expected_length - len(y)))\n    else:\n        y = y[:expected_length]\n    \n    # Create mel spectrogram\n    mel = librosa.feature.melspectrogram(y=y, sr=sr, n_mels=n_mels,\n                                         fmin=fmin, fmax=fmax, hop_length=hop_length)\n    mel_db = librosa.power_to_db(mel, ref=np.max)\n\n    # Normalize to 0-1\n    mel_db -= mel_db.min()\n    mel_db /= mel_db.max()\n\n    return mel_db.astype(np.float32)  # shape: [n_mels, time]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def set_seed(seed=42):\n    \"\"\"Set seed for reproducibility.\"\"\"\n    random.seed(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n\nset_seed(CONFIG.SEED)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class BirdClefDataset(Dataset):\n    def __init__(self, df, label2id, transform=None):\n        self.df = df.reset_index(drop=True)\n        self.label2id = label2id\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        filepath = row[\"filepath\"]\n        label = row[\"primary_label\"]\n\n        # Preprocess audio\n        mel = preprocess_audio(filepath)  # shape: [n_mels, time]\n\n        if self.transform:\n            mel = self.transform(mel)\n\n        # Convert to tensor and add channel dimension\n        mel = torch.tensor(mel).unsqueeze(0)  # shape: [1, n_mels, time]\n\n        # Convert label to index\n        label_idx = self.label2id[label]\n\n        return mel, torch.tensor(label_idx, dtype=torch.long)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create label-to-index and index-to-label mappings\nunique_labels = sorted(data_path[\"primary_label\"].unique())\nlabel2id = {label: idx for idx, label in enumerate(unique_labels)}\nid2label = {idx: label for label, idx in label2id.items()}","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df = data_path.sample(5).reset_index(drop=True)\n\n# Instantiate the dataset\ndataset = BirdClefDataset(sample_df, label2id=label2id)\n\n# Test the first sample\nmel_tensor, label = dataset[1]\n\nprint(f\"Mel spectrogram shape: {mel_tensor.shape}\")  # Expected: [1, n_mels, time]\nprint(f\"Label index: {label} → {list(label2id.keys())[list(label2id.values()).index(label.item())]}\")\nplt.figure(figsize=(10,4))\nplt.imshow(mel_tensor.squeeze(0).numpy(), aspect='auto', origin='lower')\nplt.title('mel spectogram tensor')\n\nplt.xlabel('time')\nplt.ylabel('mel bins')\n\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class BirdCLEFModel(nn.Module):\n    def __init__(self, model_name=CONFIG.MODEL_NAME, num_classes=CONFIG.NUM_CLASSES, pretrained=CONFIG.PRETRAINED):\n        super(BirdCLEFModel, self).__init__()\n        \n        # Use timm backbone with no classifier, include pooling\n        self.backbone = timm.create_model(\n            model_name,\n            pretrained=pretrained,\n            in_chans=1,\n            num_classes=0,\n            global_pool='avg'  # this handles pooling internally\n        )\n        \n        # Add classifier\n        self.classifier = nn.Linear(self.backbone.num_features, num_classes)\n\n    def forward(self, x):\n        x = self.backbone(x)      # shape: (B, features)\n        x = self.classifier(x)    # shape: (B, num_classes)\n        return x","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create sample dataset and dataloader\nsample_df = data_path.sample(8).reset_index(drop=True)\ndataset = BirdClefDataset(sample_df, label2id=label2id)\ndataloader = DataLoader(dataset, batch_size=4, shuffle=False)\n\n# Instantiate model\nmodel = BirdCLEFModel().to(CONFIG.DEVICE)\n\n# Get a batch of data\nbatch = next(iter(dataloader))\ninputs, targets = batch\ninputs = inputs.to(CONFIG.DEVICE)\n\n# Forward pass\noutputs = model(inputs)\n\n# Display shapes\nprint(f\"Input shape      : {inputs.shape}\")   # [B, 1, 128, time_steps]\nprint(f\"Output shape     : {outputs.shape}\")  # [B, num_classes]\nprint(f\"Target labels     : {targets}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_one_epoch(model, dataloader, optimizer, criterion):\n    model.train()\n    total_loss = 0\n    all_preds, all_labels = [], []\n\n    for inputs, labels in tqdm(dataloader, desc=\"Training\", leave=False):\n        inputs, labels = inputs.to(CONFIG.DEVICE), labels.to(CONFIG.DEVICE)\n\n        optimizer.zero_grad()\n        outputs = model(inputs)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n\n        total_loss += loss.item()\n        preds = torch.argmax(outputs, dim=1)\n        all_preds.extend(preds.cpu().numpy())\n        all_labels.extend(labels.cpu().numpy())\n\n    acc = accuracy_score(all_labels, all_preds)\n    return total_loss / len(dataloader), acc\n\n\ndef validate_one_epoch(model, dataloader, criterion):\n    model.eval()\n    total_loss = 0\n    all_preds, all_labels = [], []\n\n    with torch.no_grad():\n        for inputs, labels in tqdm(dataloader, desc=\"Validating\", leave=False):\n            inputs, labels = inputs.to(CONFIG.DEVICE), labels.to(CONFIG.DEVICE)\n            outputs = model(inputs)\n            loss = criterion(outputs, labels)\n\n            total_loss += loss.item()\n            preds = torch.argmax(outputs, dim=1)\n            all_preds.extend(preds.cpu().numpy())\n            all_labels.extend(labels.cpu().numpy())\n\n    acc = accuracy_score(all_labels, all_preds)\n    return total_loss / len(dataloader), acc","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Training function with checkpoint saving\ndef train_model(model, train_loader, val_loader, epochs=CONFIG.EPOCHS, lr=CONFIG.LR, resume=False, checkpoint_path=None):\n    optimizer = torch.optim.Adam(model.parameters(), lr=lr)\n    criterion = nn.CrossEntropyLoss()\n\n    best_val_acc = 0.0\n    start_epoch = 0\n\n    # Load from checkpoint if resuming\n    if resume and checkpoint_path is not None:\n        checkpoint = torch.load(checkpoint_path)\n        model.load_state_dict(checkpoint['model_state_dict'])\n        optimizer.load_state_dict(checkpoint['optimizer_state_dict'])\n        start_epoch = checkpoint['epoch'] + 1\n        best_val_acc = checkpoint.get('val_acc', 0.0)\n        print(f\"Resumed from checkpoint: {checkpoint_path} | Starting at epoch {start_epoch + 1}\")\n\n    for epoch in range(start_epoch, epochs):\n        print(f\"\\n Epoch {epoch+1}/{epochs}\")\n\n        start = time.time()\n        train_loss, train_acc = train_one_epoch(model, train_loader, optimizer, criterion)\n        val_loss, val_acc = validate_one_epoch(model, val_loader, criterion)\n        end = time.time()\n\n        print(f\" Train Loss: {train_loss:.4f} | Accuracy: {train_acc:.4f}\")\n        print(f\" Val   Loss: {val_loss:.4f} | Accuracy: {val_acc:.4f}\")\n        print(f\" Time: {(end - start):.2f}s\")\n\n        # Save checkpoint every epoch\n        torch.save({\n            'epoch': epoch,\n            'model_state_dict': model.state_dict(),\n            'optimizer_state_dict': optimizer.state_dict(),\n            'val_acc': val_acc\n        }, f\"/kaggle/working/checkpoint_epoch_{epoch+1}.pth\")\n\n        # Save best model and label2id mapping\n        if val_acc > best_val_acc:\n            best_val_acc = val_acc\n\n            # Save model\n            torch.save(model.state_dict(), \"/kaggle/working/best_model.pth\")\n            print(\"Best model saved.\")\n\n            # Save label mapping\n            with open(\"/kaggle/working/label2id.json\", \"w\") as f:\n                json.dump(label2id, f)\n            print(\"label2id mapping saved.\")\n\n    print(f\"\\n Best Validation Accuracy: {best_val_acc:.4f}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Spliting training data\ntrain_df, val_df = train_test_split(data_path, test_size=0.2, stratify=data_path['primary_label'], random_state=CONFIG.SEED)\n\ntrain_dataset = BirdClefDataset(train_df.reset_index(drop=True), label2id)\nval_dataset = BirdClefDataset(val_df.reset_index(drop=True), label2id)\n\ntrain_loader = DataLoader(train_dataset, batch_size=CONFIG.BATCH_SIZE, shuffle=True, num_workers=CONFIG.NUM_WORKERS)\nval_loader = DataLoader(val_dataset, batch_size=CONFIG.BATCH_SIZE, shuffle=False, num_workers=CONFIG.NUM_WORKERS)\n\n# Initialize model\nmodel = BirdCLEFModel().to(CONFIG.DEVICE)\n\n# Launch training (can resume later with resume=True)\ntrain_model(model, train_loader, val_loader, resume=False)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}