{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":91844,"databundleVersionId":11361821}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Import necessary libraries\nimport os\nimport pandas as pd\nimport numpy as np\nimport librosa\nimport librosa.display\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nfrom tqdm.notebook import tqdm # Or standard tqdm if not in notebook\nimport time\nimport random\nimport cv2\n\n\nimport torch\nfrom torch.utils.data import Dataset, DataLoader\nimport torchaudio.transforms as T # For SpecAugment","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T01:56:36.331500Z","iopub.execute_input":"2025-05-12T01:56:36.331885Z","iopub.status.idle":"2025-05-12T01:56:46.777400Z","shell.execute_reply.started":"2025-05-12T01:56:36.331845Z","shell.execute_reply":"2025-05-12T01:56:46.775987Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"CSV_PATH = \"/kaggle/input/birdclef-2025/train.csv\"\nAUDIO_BASE_PATH = \"/kaggle/input/birdclef-2025/train_audio/\"\n# Audio Processing Parameters\nSAMPLE_RATE = 32000\nDURATION = 5\nN_MELS = 128 # EfficientNet works well with larger image sizes, 128 or 224 are common\nFMIN = 20\nFMAX = 16000\nHOP_LENGTH = 512\nN_FFT = 2048\nTARGET_SHAPE = 224\n\n\nVALIDATION_SPLIT = 0.2\nRANDOM_SEED = 42\nNUM_WORKERS = 2 # os.cpu_count()\nLABEL_SMOOTHING = 0.1 # Set to 0 to disable\nPRELOAD_DATA_IN_RAM = True # Set to True if you have enough RAM and want faster epochs after initial load\n\n# --- Set Seed for Reproducibility ---\ndef seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    if torch.cuda.is_available():\n        torch.cuda.manual_seed(seed)\n        torch.cuda.manual_seed_all(seed)\n\nseed_everything(RANDOM_SEED)\n\n# --- Determine Device ---\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")\n\n# --- Load the training metadata ---\ntry:\n    data = pd.read_csv(CSV_PATH)\n    print(f\"Successfully loaded {CSV_PATH}\")\n    print(f\"Data shape: {data.shape}\")\nexcept FileNotFoundError:\n    print(f\"Error: Could not find {CSV_PATH}\")\n    exit()\n\n# Construct full audio file paths\ndata['full_path'] = data['filename'].apply(lambda x: os.path.join(AUDIO_BASE_PATH, x))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T01:57:02.699883Z","iopub.execute_input":"2025-05-12T01:57:02.700276Z","iopub.status.idle":"2025-05-12T01:57:03.020050Z","shell.execute_reply.started":"2025-05-12T01:57:02.700247Z","shell.execute_reply":"2025-05-12T01:57:03.018400Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 2. Label Encoding ---\nencoder = LabelEncoder()\ndata['label_encoded'] = encoder.fit_transform(data['primary_label'])\nNUM_CLASSES = len(encoder.classes_)\nprint(f\"\\nNumber of unique classes: {NUM_CLASSES}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T01:57:04.703735Z","iopub.execute_input":"2025-05-12T01:57:04.704104Z","iopub.status.idle":"2025-05-12T01:57:04.717432Z","shell.execute_reply.started":"2025-05-12T01:57:04.704077Z","shell.execute_reply":"2025-05-12T01:57:04.716115Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 3. Train/Validation Split ---\ntrain_df, val_df = train_test_split(\n    data,\n    test_size=VALIDATION_SPLIT,\n    random_state=RANDOM_SEED,\n    stratify=data['label_encoded'] # Crucial for imbalanced datasets\n)\nprint(f\"\\nTraining set size: {len(train_df)}\")\nprint(f\"Validation set size: {len(val_df)}\")\n\n# Calculate expected spectrogram shape (for reference and padding)\n# For librosa.stft, frame length is N_FFT. Number of frames is roughly len(y) / hop_length.\n# For melspectrogram, the time dimension is ceil(samples / hop_length)\n# After padding/truncating audio to DURATION * SAMPLE_RATE:\nTARGET_LENGTH_SAMPLES = int(SAMPLE_RATE * DURATION)\nN_FRAMES = int(np.ceil(TARGET_LENGTH_SAMPLES / HOP_LENGTH)) + 1 # Librosa can add a frame due to centering\nprint(f\"Expected Spectrogram Shape (H, W) after processing: ({N_MELS}, {N_FRAMES})\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T01:57:05.917546Z","iopub.execute_input":"2025-05-12T01:57:05.917892Z","iopub.status.idle":"2025-05-12T01:57:05.963584Z","shell.execute_reply.started":"2025-05-12T01:57:05.917856Z","shell.execute_reply":"2025-05-12T01:57:05.962476Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 4. Audio Preprocessing & Augmentation Function ---\ndef load_and_preprocess_audio(\n    file_path, target_sr=SAMPLE_RATE, duration_samples=TARGET_LENGTH_SAMPLES,\n    n_mels=N_MELS, hop_length=HOP_LENGTH, n_fft=N_FFT,\n    fmin=FMIN, fmax=FMAX, is_train=False):\n    \"\"\"Loads, augments (if train), pads/truncates, and computes Mel spectrogram.\"\"\"\n    try:\n        y, sr = librosa.load(file_path, sr=None) # Load native SR\n        if sr != target_sr:\n            y = librosa.resample(y, orig_sr=sr, target_sr=target_sr)\n\n        # Pad or truncate audio to target_length_samples\n        if len(y) < duration_samples:\n            padding = duration_samples - len(y)\n            offset = random.randint(0, padding) if is_train else padding // 2 # Pad randomly for train\n            y = np.pad(y, (offset, padding - offset), mode='constant')\n        elif len(y) > duration_samples:\n            start = random.randint(0, len(y) - duration_samples) if is_train else 0 # Random crop for train\n            y = y[start : start + duration_samples]\n\n        mel_spec = librosa.feature.melspectrogram(\n            y=y, sr=target_sr, n_fft=n_fft, hop_length=hop_length,\n            n_mels=n_mels, fmin=fmin, fmax=fmax, window='hann' # Hann window is common\n        )\n        mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max)\n\n        # Normalize to [0, 1] (or standardize if preferred)\n        min_val = np.min(mel_spec_db)\n        max_val = np.max(mel_spec_db)\n        if max_val > min_val:\n            mel_spec_db = (mel_spec_db - min_val) / (max_val - min_val)\n        else: # Handle silent clips\n            mel_spec_db = np.zeros_like(mel_spec_db)\n\n        # Ensure consistent shape (especially time dimension) due to potential minor librosa variations\n        if mel_spec_db.shape[1] < N_FRAMES:\n            pad_width = N_FRAMES - mel_spec_db.shape[1]\n            mel_spec_db = np.pad(mel_spec_db, ((0,0), (0, pad_width)), mode='constant', constant_values=0.)\n        elif mel_spec_db.shape[1] > N_FRAMES:\n            mel_spec_db = mel_spec_db[:, :N_FRAMES]\n\n        mel_spec_resized = cv2.resize(mel_spec_db, (TARGET_SHAPE, TARGET_SHAPE), interpolation=cv2.INTER_LINEAR)\n        return mel_spec_resized\n    except Exception as e:\n        print(f\"Error processing {file_path}: {e}\")\n        return np.zeros((n_mels, N_FRAMES)) # Return zeros on error, with expected shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T01:57:09.388196Z","iopub.execute_input":"2025-05-12T01:57:09.388510Z","iopub.status.idle":"2025-05-12T01:57:09.399290Z","shell.execute_reply.started":"2025-05-12T01:57:09.388487Z","shell.execute_reply":"2025-05-12T01:57:09.397842Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 5. Create PyTorch Dataset ---+\nclass BirdSoundDataset(Dataset):\n    def __init__(self, dataframe, is_train=False, preload_in_ram=PRELOAD_DATA_IN_RAM,\n                 apply_spec_augment=True):\n        self.dataframe = dataframe\n        self.is_train = is_train\n        self.preload_in_ram = preload_in_ram\n        self.apply_spec_augment = apply_spec_augment\n\n        if self.is_train and self.apply_spec_augment:\n            # SpecAugment: Applied on the Mel spectrogram tensor\n            self.spec_augmenter =torch.nn.Sequential(\n                    T.FrequencyMasking(freq_mask_param=N_MELS // 8),\n                    T.TimeMasking(time_mask_param=int(N_FRAMES * 0.1))\n                )\n            # self.spec_augmenter = nn.Sequential( # Alternative definition\n            #     T.FrequencyMasking(freq_mask_param=N_MELS // 8), # Mask up to 1/8th of mel bands\n            #     T.TimeMasking(time_mask_param=int(N_FRAMES * 0.1)) # Mask up to 10% of time steps\n            # )\n        else:\n            self.spec_augmenter = None\n\n        if self.preload_in_ram:\n            self.spectrograms = []\n            self.labels = []\n            print(f\"Preloading {len(dataframe)} audio files into RAM for {'training' if is_train else 'validation'}...\")\n            for idx in tqdm(range(len(dataframe))):\n                row = self.dataframe.iloc[idx]\n                file_path = row['full_path']\n                label = row['label_encoded']\n                # For preloading, we don't apply instance-specific augmentations like SpecAugment here,\n                # as they should be random for each epoch. Audio-level augs are applied during loading.\n                spectrogram = load_and_preprocess_audio(file_path, is_train=self.is_train) # is_train for audio augs\n                spectrogram_tensor = torch.tensor(spectrogram, dtype=torch.float32).unsqueeze(0)\n                spectrogram_tensor_3channel = spectrogram_tensor.repeat(3, 1, 1)\n                self.spectrograms.append(spectrogram_tensor_3channel)\n                self.labels.append(torch.tensor(label, dtype=torch.long))\n            print(f\"All {len(self.spectrograms)} spectrograms loaded into memory!\")\n\n    def __len__(self):\n        return len(self.dataframe)\n\n    def __getitem__(self, idx):\n        if torch.is_tensor(idx):\n            idx = idx.tolist()\n\n        if self.preload_in_ram:\n            spectrogram_3channel = self.spectrograms[idx]\n            label = self.labels[idx]\n        else:\n            row = self.dataframe.iloc[idx]\n            file_path = row['full_path']\n            label_val = row['label_encoded']\n\n            # Load and process audio (includes audio-level augmentations if is_train)\n            spectrogram = load_and_preprocess_audio(file_path, is_train=self.is_train)\n            spectrogram_tensor = torch.tensor(spectrogram, dtype=torch.float32).unsqueeze(0) # (1, H, W)\n            spectrogram_3channel = spectrogram_tensor.repeat(3, 1, 1) # (3, H, W)\n            label = torch.tensor(label_val, dtype=torch.long)\n\n        # Apply SpecAugment (if training and enabled)\n        if self.is_train and self.spec_augmenter and random.random() < APPLY_AUGMENTATION_PROB:\n            # Ensure spec_augmenter is on the same device if it contains parameters,\n            # or apply before moving tensor to GPU in training loop.\n            # For torchaudio.transforms, they are typically stateless or handle device internally.\n            try:\n                spectrogram_3channel = self.spec_augmenter(spectrogram_3channel)\n            except Exception as e: # Catch potential errors with SpecAugment on edge cases\n                # print(f\"SpecAugment error for sample {idx}, shape {spectrogram_3channel.shape}: {e}\")\n                pass # Skip SpecAugment for this sample if it errors\n\n        return spectrogram_3channel, label\n\ntrain_data = BirdSoundDataset(train_df, is_train=True, preload_in_ram=PRELOAD_DATA_IN_RAM)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T01:57:09.452016Z","iopub.execute_input":"2025-05-12T01:57:09.452361Z","iopub.status.idle":"2025-05-12T02:24:04.566402Z","shell.execute_reply.started":"2025-05-12T01:57:09.452336Z","shell.execute_reply":"2025-05-12T02:24:04.564835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"torch.save(train_data, \"/kaggle/working/train_dataset.pt\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T02:27:06.533574Z","iopub.execute_input":"2025-05-12T02:27:06.535522Z","iopub.status.idle":"2025-05-12T02:27:58.291404Z","shell.execute_reply.started":"2025-05-12T02:27:06.535458Z","shell.execute_reply":"2025-05-12T02:27:58.288471Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_data = BirdSoundDataset(val_df, is_train=False, preload_in_ram=PRELOAD_DATA_IN_RAM, apply_spec_augment=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T02:30:14.854263Z","iopub.execute_input":"2025-05-12T02:30:14.854746Z","iopub.status.idle":"2025-05-12T02:36:46.018305Z","shell.execute_reply.started":"2025-05-12T02:30:14.854714Z","shell.execute_reply":"2025-05-12T02:36:46.016543Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"torch.save(val_data, \"/kaggle/working/val_dataset.pt\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T02:36:54.441534Z","iopub.execute_input":"2025-05-12T02:36:54.441923Z","iopub.status.idle":"2025-05-12T02:37:04.121690Z","shell.execute_reply.started":"2025-05-12T02:36:54.441884Z","shell.execute_reply":"2025-05-12T02:37:04.120253Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}