{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":21669,"databundleVersionId":1692278,"sourceType":"competition"}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import torch.nn as nn\nimport numpy as np\nimport copy\nimport warnings\nimport torch\nimport librosa\nimport csv\nimport random\nimport os\nimport pandas as pd\n\nfrom skimage.transform import resize\nfrom skimage.filters import gaussian\nfrom skimage import exposure, util\nfrom torchvision.models import resnet50\nfrom torch.utils.data import Dataset, DataLoader\nfrom tqdm import tqdm\nfrom sklearn.model_selection import KFold\nfrom concurrent.futures import ThreadPoolExecutor\n\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-13T16:37:53.076653Z","iopub.execute_input":"2025-12-13T16:37:53.077317Z","iopub.status.idle":"2025-12-13T16:38:04.296495Z","shell.execute_reply.started":"2025-12-13T16:37:53.077289Z","shell.execute_reply":"2025-12-13T16:38:04.295631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Global hyperparameters and constants\nLABELS = 24  # Number of species to classify\nSR = 48000  # Sampling rate\nLENGTH = 10 * SR  # 10 seconds of audio\nF_MIN = 24000  # Initial minimum frequency (will be updated)\nF_MAX = 0  # Initial maximum frequency (will be updated)\nLEARNING_RATE = 2e-4\nEPOCHS = 20\nN_FOLD = 5  # Number of folds for cross-validation\n\ndevice = 'cuda:0' if torch.cuda.is_available() else 'cpu'\nprint(f\"Using device: {device}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T16:38:29.661155Z","iopub.execute_input":"2025-12-13T16:38:29.661411Z","iopub.status.idle":"2025-12-13T16:38:29.753333Z","shell.execute_reply.started":"2025-12-13T16:38:29.661393Z","shell.execute_reply":"2025-12-13T16:38:29.752412Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class AudioAugmentations:\n    \"\"\"Class containing various audio spectrogram augmentation techniques\"\"\"\n    \n    def __init__(self):\n        # List of available augmentation functions\n        self.augs = [self.add_noise, self.contrast_stretch, self.h_flip, self.v_flip]\n    \n    def h_flip(self, image):\n        return np.stack([image[:, ::-1]] * 3)\n    \n    def v_flip(self, image):\n        return np.stack([image[::-1, :]] * 3)\n    \n    def add_noise(self, image):\n        noise_img = util.random_noise(image)\n        return np.stack([noise_img] * 3)\n    \n    def contrast_stretch(self, image):\n        contrast_img = exposure.rescale_intensity(image)\n        return np.stack([contrast_img] * 3)\n    \n    def apply_random_augmentation(self, image):\n        aug_func = random.choice(self.augs)\n        return aug_func(image)\n\n\ndef spec_to_image(spec):\n    \"\"\"Convert spectrogram to normalized image format suitable for ResNet\"\"\"\n    # Resize to dimensions compatible with ResNet input\n    spec = resize(spec, (224, 400))\n    \n    # Normalize the spectrogram\n    eps = 1e-6\n    mean = spec.mean()\n    std = spec.std()\n    spec_norm = (spec - mean) / (std + eps)\n    \n    # Scale to 0-255 range for image representation\n    spec_min, spec_max = spec_norm.min(), spec_norm.max()\n    spec_scaled = 255 * (spec_norm - spec_min) / (spec_max - spec_min)\n    spec_scaled = spec_scaled.astype(np.uint8)\n    \n    return spec_scaled\n\n\ndef get_model():\n    \"\"\"Initialize and configure ResNet50 model for our classification task\"\"\"\n    model = resnet50(pretrained=True)  # Load pre-trained ResNet50\n    \n    # Replace the final fully connected layer for our 24-class classification\n    num_ftrs = model.fc.in_features\n    model.fc = nn.Linear(num_ftrs, LABELS)\n    \n    return model.to(device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T16:39:15.097363Z","iopub.execute_input":"2025-12-13T16:39:15.097903Z","iopub.status.idle":"2025-12-13T16:39:15.105680Z","shell.execute_reply.started":"2025-12-13T16:39:15.097878Z","shell.execute_reply":"2025-12-13T16:39:15.104906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load training data and determine frequency range\ndata = pd.read_csv(\"/kaggle/input/rfcx-species-audio-detection/train_tp.csv\")\n\n# Find the minimum and maximum frequency bounds from the data\nfor i in range(len(data)):\n    current_f_min = float(data.iloc[i]['f_min'])\n    current_f_max = float(data.iloc[i]['f_max'])\n    \n    if F_MIN > current_f_min:\n        F_MIN = current_f_min\n    if F_MAX < current_f_max:\n        F_MAX = current_f_max\n\n# Apply safety margins to frequency bounds\nF_MIN = int(F_MIN * 0.9)\nF_MAX = int(F_MAX * 1.1)\n\nprint(f\"Frequency range: {F_MIN}Hz to {F_MAX}Hz\")\n\n# Extract labels and recording IDs\nlabel_list = data['species_id'].tolist()\ndata_list = data['recording_id'].tolist()\n\n# Dictionary to cache processed spectrograms\naudio_data = {}\n\n\ndef process_audio(i):\n    \"\"\"Process a single audio file into a spectrogram\"\"\"\n    recording_id = data_list[i]\n    species_id = label_list[i]\n    \n    # Load audio file\n    wav, sr = librosa.load(\n        f'/kaggle/input/rfcx-species-audio-detection/train/{recording_id}.flac', \n        sr=None\n    )\n    \n    # Extract the segment containing the species call\n    t_min = int(data.at[i, 't_min'] * sr)\n    t_max = int(data.at[i, 't_max'] * sr)\n    \n    # Center the segment and extract 10 seconds\n    center = np.round((t_min + t_max) / 2)\n    beginning = max(center - LENGTH // 2, 0)\n    ending = min(beginning + LENGTH, len(wav))\n    \n    # Adjust beginning if segment is too short\n    if ending - beginning < LENGTH:\n        beginning = ending - LENGTH\n    \n    audio_slice = wav[int(beginning):int(ending)]\n    \n    # Generate mel spectrogram\n    spec = librosa.feature.melspectrogram(\n        y=audio_slice, \n        sr=sr, \n        fmin=F_MIN, \n        fmax=F_MAX\n    )\n    spec_db = librosa.power_to_db(spec, top_db=80)\n    \n    # Convert to image format\n    img = spec_to_image(spec_db)\n    \n    return recording_id, img","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T16:40:28.528197Z","iopub.execute_input":"2025-12-13T16:40:28.528473Z","iopub.status.idle":"2025-12-13T16:40:28.658717Z","shell.execute_reply.started":"2025-12-13T16:40:28.528454Z","shell.execute_reply":"2025-12-13T16:40:28.657991Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Process all audio files in parallel\nprint(\"Processing audio files...\")\nwith ThreadPoolExecutor() as executor:\n    results = list(tqdm(\n        executor.map(process_audio, range(len(data))), \n        total=len(data)\n    ))\n\n# Store processed spectrograms in dictionary\nfor recording_id, img in results:\n    audio_data[recording_id] = img","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T16:41:15.597801Z","iopub.execute_input":"2025-12-13T16:41:15.598199Z","iopub.status.idle":"2025-12-13T16:42:28.882626Z","shell.execute_reply.started":"2025-12-13T16:41:15.598175Z","shell.execute_reply":"2025-12-13T16:42:28.881967Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class AudioData(Dataset):\n    \"\"\"Custom PyTorch Dataset for audio spectrogram data\"\"\"\n    \n    def __init__(self, X, y, data_type, augmentations=None):\n        self.X = X  # List of recording IDs\n        self.y = y  # List of labels\n        self.data_type = data_type  # \"train\" or \"valid\"\n        self.audio_data = audio_data  # Cache of processed spectrograms\n        self.augmentations = augmentations  # Augmentation object\n    \n    def __len__(self):\n        return len(self.X)\n    \n    def __getitem__(self, idx):\n        recording_id = self.X[idx]\n        label = self.y[idx]\n        \n        # Retrieve preprocessed spectrogram\n        img = self.audio_data[recording_id]\n        \n        # Apply augmentations only during training\n        if self.data_type == \"train\" and self.augmentations:\n            img = self.augmentations.apply_random_augmentation(img)\n        else:\n            # Convert to 3-channel format (RGB-like for ResNet)\n            img = np.stack((img, img, img))\n        \n        # Convert to tensor\n        img_tensor = torch.tensor(img, dtype=torch.float32)\n        label_tensor = torch.tensor(label, dtype=torch.long)\n        \n        return img_tensor, label_tensor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T16:42:54.125919Z","iopub.execute_input":"2025-12-13T16:42:54.126656Z","iopub.status.idle":"2025-12-13T16:42:54.132001Z","shell.execute_reply.started":"2025-12-13T16:42:54.126634Z","shell.execute_reply":"2025-12-13T16:42:54.131302Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize loss function and augmenter\nloss_fn = nn.CrossEntropyLoss()\naudio_augmenter = AudioAugmentations()\n\n\ndef train(model, loss_fn, train_loader, valid_loader, optimizer, scheduler):\n    \"\"\"Training loop with validation\"\"\"\n    best_model_wts = copy.deepcopy(model.state_dict())\n    best_acc = 0.0\n    train_losses = []\n    valid_losses = []\n    \n    for epoch in tqdm(range(1, EPOCHS + 1)):\n        # Training phase\n        model.train()\n        batch_losses = []\n        \n        for batch_idx, data_batch in enumerate(train_loader):\n            x, y = data_batch\n            x = x.to(device, dtype=torch.float32)\n            y = y.to(device, dtype=torch.long)\n            \n            optimizer.zero_grad()\n            y_hat = model(x)\n            loss = loss_fn(y_hat, y)\n            loss.backward()\n            optimizer.step()\n            \n            batch_losses.append(loss.item())\n        \n        train_losses.append(batch_losses)\n        \n        # Validation phase\n        model.eval()\n        batch_losses = []\n        trace_y = []\n        trace_yhat = []\n        \n        with torch.no_grad():\n            for batch_idx, data_batch in enumerate(valid_loader):\n                x, y = data_batch\n                x = x.to(device, dtype=torch.float32)\n                y = y.to(device, dtype=torch.long)\n                \n                y_hat = model(x)\n                loss = loss_fn(y_hat, y)\n                \n                trace_y.append(y.cpu().detach().numpy())\n                trace_yhat.append(y_hat.cpu().detach().numpy())\n                batch_losses.append(loss.item())\n        \n        valid_losses.append(batch_losses)\n        \n        # Calculate validation accuracy\n        trace_y = np.concatenate(trace_y)\n        trace_yhat = np.concatenate(trace_yhat)\n        accuracy = np.mean(trace_yhat.argmax(axis=1) == trace_y)\n        \n        print(f\"Epoch {epoch}: \"\n              f\"Train Loss = {np.mean(train_losses[-1]):.5f}, \"\n              f\"Val Loss = {np.mean(valid_losses[-1]):.5f}, \"\n              f\"Val Accuracy = {accuracy:.5f}\")\n        \n        # Update learning rate\n        scheduler.step(np.mean(valid_losses[-1]))\n        \n        # Save best model\n        if accuracy > best_acc:\n            best_acc = accuracy\n            best_model_wts = copy.deepcopy(model.state_dict())\n    \n    # Load best model weights\n    model.load_state_dict(best_model_wts)\n    return model\n\n\n# K-fold cross-validation\nskf = KFold(n_splits=N_FOLD, shuffle=True, random_state=563)\n\nfor fold_id, (train_index, val_index) in enumerate(skf.split(data_list, label_list)):\n    print(f\"\\n{'='*50}\")\n    print(f\"Training Fold {fold_id}\")\n    print(f\"{'='*50}\")\n    \n    # Split data for current fold\n    X_train = np.take(data_list, train_index)\n    y_train = np.take(label_list, train_index, axis=0)\n    X_val = np.take(data_list, val_index)\n    y_val = np.take(label_list, val_index, axis=0)\n    \n    # Create datasets and dataloaders\n    train_data = AudioData(X_train, y_train, \"train\", augmentations=audio_augmenter)\n    valid_data = AudioData(X_val, y_val, \"valid\")\n    \n    train_loader = DataLoader(train_data, batch_size=8, shuffle=True, drop_last=True)\n    valid_loader = DataLoader(valid_data, batch_size=8, shuffle=True, drop_last=True)\n    \n    # Initialize model, optimizer, and scheduler\n    model = get_model()\n    optimizer = torch.optim.Adam(model.parameters(), lr=LEARNING_RATE)\n    scheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(\n        optimizer, 'min', patience=3\n    )\n    \n    # Train the model\n    model = train(model, loss_fn, train_loader, valid_loader, optimizer, scheduler)\n    \n    # Save model weights\n    torch.save(model.state_dict(), f\"./model{fold_id}.pt\")\n    \n    # Clean up to free memory\n    del train_data, valid_data, train_loader, valid_loader, model\n    del X_train, X_val, y_train, y_val\n    torch.cuda.empty_cache()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T16:42:56.603286Z","iopub.execute_input":"2025-12-13T16:42:56.603532Z","iopub.status.idle":"2025-12-13T17:22:39.259109Z","shell.execute_reply.started":"2025-12-13T16:42:56.603517Z","shell.execute_reply":"2025-12-13T17:22:39.258294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_test_file(filename):\n    \"\"\"Load and segment test audio file into spectrograms\"\"\"\n    filepath = '/kaggle/input/rfcx-species-audio-detection/test/' + filename\n    wav, sr = librosa.load(filepath, sr=None)\n    \n    # Calculate number of 10-second segments\n    segments = len(wav) / LENGTH\n    segments = int(np.ceil(segments))\n    \n    mel_array = []\n    \n    for i in range(segments):\n        # Extract 10-second segment\n        if (i + 1) * LENGTH > len(wav):\n            audio_slice = wav[len(wav) - LENGTH:len(wav)]\n        else:\n            audio_slice = wav[i * LENGTH:(i + 1) * LENGTH]\n        \n        # Generate spectrogram\n        spec = librosa.feature.melspectrogram(\n            y=audio_slice, \n            sr=sr, \n            fmin=F_MIN, \n            fmax=F_MAX\n        )\n        spec_db = librosa.power_to_db(spec, top_db=80)\n        \n        # Convert to image format\n        img = spec_to_image(spec_db)\n        mel_spec = np.stack((img, img, img))  # 3-channel format\n        mel_array.append(mel_spec)\n    \n    return np.array(mel_array)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T17:24:25.727116Z","iopub.execute_input":"2025-12-13T17:24:25.727853Z","iopub.status.idle":"2025-12-13T17:24:25.733529Z","shell.execute_reply.started":"2025-12-13T17:24:25.727829Z","shell.execute_reply":"2025-12-13T17:24:25.732637Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load all trained models for ensemble prediction\nprint(\"\\nLoading models for ensemble prediction...\")\nmembers = []\n\nfor i in range(N_FOLD):\n    print(f\"Loading model from fold {i}\")\n    model = get_model()\n    model.load_state_dict(torch.load(f'./model{i}.pt'))\n    model.eval()\n    members.append(model)\n\n# Clean up model files\nfor i in range(N_FOLD):\n    os.remove(f'./model{i}.pt')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T17:24:29.859424Z","iopub.execute_input":"2025-12-13T17:24:29.859689Z","iopub.status.idle":"2025-12-13T17:24:32.966633Z","shell.execute_reply.started":"2025-12-13T17:24:29.859668Z","shell.execute_reply":"2025-12-13T17:24:32.965754Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_and_predict(test_file, members):\n    \"\"\"Load a test file and generate predictions using ensemble\"\"\"\n    # Load and preprocess test data\n    data = load_test_file(test_file)\n    data = torch.tensor(data).float()\n    \n    if torch.cuda.is_available():\n        data = data.cuda()\n    \n    # Collect predictions from all models\n    output_list = []\n    \n    for model in members:\n        with torch.no_grad():\n            output = model(data)\n            # Take max along segments dimension\n            maxed_output = torch.max(output, dim=0)[0]\n            maxed_output = maxed_output.cpu().detach()\n            output_list.append(maxed_output)\n    \n    # Ensemble by averaging predictions\n    avg_maxed_output = torch.mean(torch.stack(output_list), dim=0)\n    \n    # Format results\n    file_id = test_file.split('.')[0]\n    return [file_id] + [out.item() for out in avg_maxed_output]\n\n\ndef save_submission(predictions, output_file='submission.csv'):\n    \"\"\"Save predictions to CSV file in competition format\"\"\"\n    with open(output_file, 'w', newline='') as csvfile:\n        submission_writer = csv.writer(csvfile, delimiter=',')\n        \n        # Write header\n        header = ['recording_id'] + [f's{i}' for i in range(LABELS)]\n        submission_writer.writerow(header)\n        \n        # Write predictions\n        for pred in predictions:\n            submission_writer.writerow(pred)\n    \n    print(f\"Submission saved to {output_file}\")\n\n\ndef generate_predictions(test_files, members):\n    \"\"\"Generate predictions for all test files\"\"\"\n    predictions = []\n    \n    print(f\"Processing {len(test_files)} test files...\")\n    \n    # Process test files in parallel\n    with ThreadPoolExecutor(max_workers=4) as executor:\n        futures = [\n            executor.submit(load_and_predict, test_file, members)\n            for test_file in test_files\n        ]\n        \n        for future in tqdm(futures, total=len(test_files)):\n            predictions.append(future.result())\n    \n    # Save submission file\n    save_submission(predictions)\n    \n    return predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T17:24:39.122136Z","iopub.execute_input":"2025-12-13T17:24:39.122414Z","iopub.status.idle":"2025-12-13T17:24:39.130787Z","shell.execute_reply.started":"2025-12-13T17:24:39.122394Z","shell.execute_reply":"2025-12-13T17:24:39.130227Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generate predictions on test set\ntest_files = os.listdir('/kaggle/input/rfcx-species-audio-detection/test/')\n\n# Move models to GPU if available\nif torch.cuda.is_available():\n    members = [m.cuda() for m in members]\n\n# Generate final predictions\npredictions = generate_predictions(test_files, members)\n\nprint(\"\\nPipeline completed successfully!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T17:24:48.175791Z","iopub.execute_input":"2025-12-13T17:24:48.176389Z","iopub.status.idle":"2025-12-13T17:32:18.492931Z","shell.execute_reply.started":"2025-12-13T17:24:48.176365Z","shell.execute_reply":"2025-12-13T17:32:18.492225Z"}},"outputs":[],"execution_count":null}]}