{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30839,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Import the required modules","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nimport sys\nimport cv2\nimport math\nimport numpy as np\nimport pandas as pd\nfrom glob import glob\nfrom tqdm.notebook import tqdm\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import KFold\nimport librosa\nfrom scipy import signal as sci_signal\nfrom sklearn.model_selection import train_test_split\n\nimport torch\nfrom torch import nn\nfrom torchvision.models import efficientnet\n\nfrom torch.utils.data import DataLoader\n\nimport albumentations as albu\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2.Set up the configuration","metadata":{}},{"cell_type":"code","source":"class config:\n    SEED = 2024  # random seed\n    DEVICE = 'cuda'  # device to be used\n    MIXED_PRECISION = False  # whether to use mixed-16 precision\n    OUTPUT_DIR = '/kaggle/working/'  # output folder\n    \n    # == data config ==\n    DATA_ROOT = '/kaggle/input/birdclef-2024'  # root folder\n    PREPROCESSED_DATA_ROOT = '/kaggle/input/birdclef24-spectrograms-via-cupy'\n    LOAD_DATA = True  # whether to load data from pre-processed dataset\n    FS = 32000  # sample rate\n    N_FFT = 1095  # n FFT of Spec.\n    WIN_SIZE = 412  # WIN_SIZE of Spec.\n    WIN_LAP = 100  # overlap of Spec.\n    LR_MAX = 3e-4  # Maximum learning rate\n    N_STEPS = 20 * len(train_loader)  # Total steps = epochs * batches per epoch\n    N_CLASSES = len(label_list)  # Number of classes in your dataset\n    MIN_FREQ = 40  # min frequency\n    MAX_FREQ = 15000  # max frequency\n    USE_XYMASKING = True ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. preprocessing","metadata":{}},{"cell_type":"code","source":"# labels\nlabel_list = sorted(os.listdir(os.path.join(config.DATA_ROOT, 'train_audio')))\nlabel_id_list = list(range(len(label_list)))\nlabel2id = dict(zip(label_list, label_id_list))\nid2label = dict(zip(label_id_list, label_list))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"metadata_df = pd.read_csv(f'{config.DATA_ROOT}/train_metadata.csv')\nmetadata_df.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# OOG to SPECTORGRAM \n","metadata":{}},{"cell_type":"code","source":"def oog2spec_via_cupy(audio_data):\n    \n    import cupy as cp\n    from cupyx.scipy import signal as cupy_signal\n    \n    audio_data = cp.array(audio_data)\n    \n    # handles NaNs\n    mean_signal = cp.nanmean(audio_data)\n    audio_data = cp.nan_to_num(audio_data, nan=mean_signal) if cp.isnan(audio_data).mean() < 1 else cp.zeros_like(audio_data)\n    \n    # to spec.\n    frequencies, times, spec_data = cupy_signal.spectrogram(\n        audio_data, \n        fs=config.FS, \n        nfft=config.N_FFT, \n        nperseg=config.WIN_SIZE, \n        noverlap=config.WIN_LAP, \n        window='hann'\n    )\n    \n    # Filter frequency range\n    valid_freq = (frequencies >= config.MIN_FREQ) & (frequencies <= config.MAX_FREQ)\n    spec_data = spec_data[valid_freq, :]\n    \n    # Log\n    spec_data = cp.log10(spec_data + 1e-20)\n    \n    # min/max normalize\n    spec_data = spec_data - spec_data.min()\n    spec_data = spec_data / spec_data.max()\n    \n    return spec_data.get()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if config.LOAD_DATA:\n    print('load from file')\n    all_bird_data = np.load(f'{config.PREPROCESSED_DATA_ROOT}/spec_center_5sec_256_256.npy', allow_pickle=True).item()\nelse:\n    all_bird_data = dict()\n    for i, row_metadata in tqdm(train_df.iterrows()):\n\n        # load ogg\n        audio_data, _ = librosa.load(row_metadata.filepath, sr=config.FS)\n\n        # crop\n        n_copy = math.ceil(5 * config.FS / len(audio_data))\n        if n_copy > 1: audio_data = np.concatenate([audio_data]*n_copy)\n\n        start_idx = int(len(audio_data) / 2 - 2.5 * config.FS)\n        end_idx = int(start_idx + 5.0 * config.FS)\n        input_audio = audio_data[start_idx:end_idx]\n\n        # ogg to spec.\n        input_spec = oog2spec_via_cupy(input_audio)\n        \n        input_spec = cv2.resize(input_spec, (256, 256), interpolation=cv2.INTER_AREA)\n\n        all_bird_data[row_metadata.samplename] = input_spec.astype(np.float32)\n\n    # save to file\n    np.save(os.path.join(config.OUTPUT_DIR, f'spec_center_5sec_256_256.npy'), all_bird_data)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_transforms(_type):\n    \n    if _type == 'train':\n        return albu.Compose([\n            albu.HorizontalFlip(0.5),\n            albu.XYMasking(\n                p=0.3,\n                num_masks_x=(1, 3),\n                num_masks_y=(1, 3),\n                mask_x_length=(1, 10),\n                mask_y_length=(1, 20),\n            ) if config.USE_XYMASKING else albu.NoOp()\n        ])\n    elif _type == 'valid':\n        return albu.Compose([])\ndef show_batch(ds, row=3, col=3):\n    fig = plt.figure(figsize=(10, 10))\n    img_index = np.random.randint(0, len(ds)-1, row*col)\n    \n    for i in range(len(img_index)):\n        img, label = dummy_dataset[img_index[i]]\n        \n        if isinstance(img, torch.Tensor):\n            img = img.detach().numpy()\n        \n        ax = fig.add_subplot(row, col, i + 1, xticks=[], yticks=[])\n        ax.imshow(img, cmap='jet')\n        ax.set_title(f'ID: {img_index[i]}; Target: {label}')\n    \n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# dataset and dataloader\n","metadata":{}},{"cell_type":"code","source":"class BirdDataset(torch.utils.data.Dataset):\n    \n    def __init__(\n        self,\n        metadata,\n        augmentation=None,\n        mode='train'\n    ):\n        super().__init__()\n        self.metadata = metadata\n        self.augmentation = augmentation\n        self.mode = mode\n    \n    def __len__(self):\n        return len(self.metadata)\n    \n    def __getitem__(self, index):\n        \n        row_metadata = self.metadata.iloc[index]\n        \n        # Load spec. data (image)\n        input_spec = all_bird_data[row_metadata.samplename]\n        \n        # Ensure input_spec is a numpy array or tensor\n        input_spec = np.array(input_spec)\n\n        # Augmentation\n        if self.augmentation is not None:\n            input_spec = self.augmentation(image=input_spec)['image']\n        resize = albu.Resize(224, 224)\n        input_spec = resize(image=input_spec)['image']\n        # Check if input_spec is 2D (grayscale) or 3D (RGB)\n        if input_spec.ndim == 2:  # Grayscale image (Height x Width)\n            input_spec = np.expand_dims(input_spec, axis=-1)  # Convert to (Height, Width, 1)\n\n        # Now, ensure it's a 3-channel image (if needed)\n        if input_spec.shape[-1] == 1:  # If single channel (grayscale)\n            input_spec = np.repeat(input_spec, 3, axis=-1)  # Convert to 3 channels (RGB)\n        \n        # Convert to tensor\n        input_spec = torch.tensor(input_spec, dtype=torch.float32)\n\n        # Convert to the format (C, H, W)\n        input_spec = input_spec.permute(2, 0, 1)  # Change to (3, H, W)\n\n        # Target label\n        target = row_metadata.target\n        \n        return input_spec, torch.tensor(target, dtype=torch.long)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_metadata_split, valid_metadata_split = train_test_split(\n    train_df, test_size=0.2, random_state=42, stratify=train_df['primary_label']\n)\ntrain_dataset = BirdDataset(metadata=train_metadata_split, augmentation=get_transforms('train'), mode='train')\nvalid_dataset = BirdDataset(metadata=valid_metadata_split, augmentation=get_transforms('valid'), mode='valid')\n\n# Create DataLoaders for both train and validation\ntrain_loader = DataLoader(train_dataset, batch_size=32, shuffle=True, num_workers=4)\nvalid_loader = DataLoader(valid_dataset, batch_size=32, shuffle=False, num_workers=4)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for inputs, labels in train_loader:\n    print(inputs.size())  # Should now print: torch.Size([32, 1, 256, 256])\n    break\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ACC = torchmetrics.Accuracy(task='multiclass', num_classes=CONFIG.N_CLASSES).cuda()\nROC_AUC = torchmetrics.AUROC(task='multiclass', num_classes=CONFIG.N_CLASSES).cuda()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# model \n","metadata":{}},{"cell_type":"code","source":"class EfficientNetModel(nn.Module):\n    def __init__(self, num_classes):\n        super(EfficientNetModel, self).__init__()\n        \n        # Load pre-trained EfficientNet\n        self.efficientnet = models.efficientnet_b0(weights='IMAGENET1K_V1')\n        \n        # Modify the final fully connected layer for your number of classes\n        self.efficientnet.classifier[1] = nn.Linear(self.efficientnet.classifier[1].in_features, num_classes)\n    \n    def forward(self, x):\n        return self.efficientnet(x)\nmodel = EfficientNetModel(num_classes=CONFIG.N_CLASSES)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nif torch.cuda.device_count() > 1:\n    print(f\"Using {torch.cuda.device_count()} GPUs!\")\n    model = nn.DataParallel(model)  # Wrap model in DataParallel for multi-GPU usage\n\nmodel.to(device)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# training ","metadata":{}},{"cell_type":"code","source":"criterion = nn.CrossEntropyLoss()\noptimizer = optim.Adam(model.parameters(), lr=1e-3)\n\n# Define the OneCycleLR scheduler\nscheduler = torch.optim.lr_scheduler.OneCycleLR(\n    optimizer=optimizer,\n    max_lr=CONFIG.LR_MAX,\n    total_steps=CONFIG.N_STEPS,\n    pct_start=0.10,  # 10% of the total steps for increasing LR\n    anneal_strategy='cos',  # Cosine annealing\n    div_factor=1e3,  # Initial LR is LR_MAX/div_factor\n    final_div_factor=1e4,  # Final LR is LR_MAX/final_div_factor\n)\n\n# Training loop\nepochs = 20\nfor epoch in range(epochs):\n    model.train()  # Set model to training mode\n    train_loss = 0\n    ACC.reset()  # Reset metrics before each epoch\n    ROC_AUC.reset()\n\n    pbar = tqdm(train_loader, total=len(train_loader), desc=f'Epoch {epoch+1}/{epochs}')\n    \n    for step, (inputs, labels) in enumerate(pbar):\n        inputs, labels = inputs.to(device), labels.to(device)\n\n        optimizer.zero_grad()  # Clear gradients\n        \n        # Forward pass\n        outputs = model(inputs)\n        \n        # Compute loss\n        loss = criterion(outputs, labels)\n        loss.backward()  # Backpropagation\n        optimizer.step()  # Update weights\n        \n        # Update the learning rate\n        scheduler.step()\n        \n        # Track loss\n        train_loss += loss.item()\n\n        # Convert logits to probabilities\n        probs = torch.softmax(outputs, dim=1)  \n\n        # Update Accuracy and ROC AUC\n        ACC.update(probs, labels)\n        ROC_AUC.update(probs, labels)\n\n    # Compute final ACC and ROC AUC for epoch\n    acc_value = ACC.compute().item()\n    roc_auc_value = ROC_AUC.compute().item()\n\n    # Update progress bar and print metrics\n    pbar.set_postfix(loss=train_loss / len(train_loader), acc=acc_value, roc_auc=roc_auc_value)\n    print(f\"Epoch {epoch+1}/{epochs} - Loss: {train_loss/len(train_loader):.4f}, ACC: {acc_value:.4f}, ROC AUC: {roc_auc_value:.4f}\")\n    print(f\"Learning Rate: {scheduler.get_last_lr()[0]:.6f}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# save the model for inference notebook","metadata":{}},{"cell_type":"code","source":"# Assuming `model` is your trained model\ntorch.save(model.state_dict(), '/kaggle/working/my_model.pth')\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}