{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-03T22:26:09.811054Z","iopub.execute_input":"2025-05-03T22:26:09.811526Z","iopub.status.idle":"2025-05-03T22:26:09.817002Z","shell.execute_reply.started":"2025-05-03T22:26:09.811489Z","shell.execute_reply":"2025-05-03T22:26:09.816059Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\ndir_path = \"/kaggle/input/birdclef-2025\"\ntrain_csv = os.path.join(dir_path, \"train.csv\")\ntrain_df = pd.read_csv(train_csv)\n\ntrain_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T22:32:25.169129Z","iopub.execute_input":"2025-05-03T22:32:25.169641Z","iopub.status.idle":"2025-05-03T22:32:25.472938Z","shell.execute_reply.started":"2025-05-03T22:32:25.169563Z","shell.execute_reply":"2025-05-03T22:32:25.471766Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.astype('object').describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T22:32:28.074563Z","iopub.execute_input":"2025-05-03T22:32:28.075542Z","iopub.status.idle":"2025-05-03T22:32:28.234485Z","shell.execute_reply.started":"2025-05-03T22:32:28.075487Z","shell.execute_reply":"2025-05-03T22:32:28.233329Z"}},"outputs":[],"execution_count":null},{"cell_type":"raw","source":"# Create train test split\n# Understand what each collection number means\n# Create module for data processing -> Convert audio files to spectograms \n# Use simplest classification model as baseline\n# Get results against test set\n\n\n# Train Audio -> 1 min audio. Labelled by collection directory\n# Train Soundscapes -> Unlabelled 1 min audio with site, date and local time\n# Test soundscapes -> Will be populated when submitting model predictions\n# Train csv -> Metadata. Provides species name, location,etc. for each collection.\n","metadata":{}},{"cell_type":"code","source":"# Data Cleaning and Processing\n# Remove human audio from train audio and soundscapes\n# Convert audio files to spectograms\n# Use a small CNN as baseline, trained on train_audio\n# Get results on ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T22:32:30.264261Z","iopub.execute_input":"2025-05-03T22:32:30.264644Z","iopub.status.idle":"2025-05-03T22:32:30.26954Z","shell.execute_reply.started":"2025-05-03T22:32:30.264619Z","shell.execute_reply":"2025-05-03T22:32:30.268417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport torch\nfrom torch.utils.data import Dataset, DataLoader\nimport librosa\nimport numpy as np\n\n\nclass AudioSpectogramDataset(Dataset):\n    def __init__(self, root_dir, classes, sample_rate=22050, transform=None):\n        self.root_dir = root_dir\n        self.sample_rate = sample_rate\n        self.classes = classes\n        self.transform = transform\n        self.audio_length = 60 # Number of frames in the audio clip\n\n        # Initialize list of files for each class\n        self.file_lists = []\n        self.class_index = {index: class_name for index, class_name in enumerate(self.classes)}\n        \n        # Populate the file lists\n        for root, dirs, files in os.walk(self.root_dir):\n            for dir in dirs:\n                class_dir = os.path.join(root, dir)\n                for audio_file in os.listdir(class_dir):\n                    self.file_lists.append((os.path.join(class_dir, audio_file), dir))\n                # self.file_lists[dir] = [os.path.join(class_dir, f) for f in os.listdir(class_dir)]\n\n    def __len__(self):\n        return len(self.classes)\n\n\n    # Define a function to convert an audio file to Mel spectrogram\n    def _get_mel_spectrogram(self, audio_file):\n        # Load the audio file using torchaudio\n        audio, sr = librosa.load(audio_file)\n\n        # Resize the audio to specified length\n        audio = np.copy(audio)\n        audio.resize(self.audio_length * self.sample_rate)\n        \n        return librosa.feature.melspectrogram(y=audio, sr=sr)\n\n    def __getitem__(self, index):\n        # Get the file name and corresponding class\n        file_path, class_name = self.file_lists[index]\n        # class_files = self.file_lists[self.classes[idx]]\n        # file_name = class_files[np.random.randint(0, len(class_files))]\n\n        y = self._get_mel_spectrogram(file_path)\n        \n        # Perform any required audio processing or transformation\n        if self.transform is not None:\n            y = self.transform(y)\n        \n        # Return the processed audio\n        return y, class_name","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T23:04:02.293396Z","iopub.execute_input":"2025-05-03T23:04:02.293768Z","iopub.status.idle":"2025-05-03T23:04:02.306759Z","shell.execute_reply.started":"2025-05-03T23:04:02.293745Z","shell.execute_reply":"2025-05-03T23:04:02.305514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_audio_path=os.path.join(dir_path, \"train_audio\")\nclasses = train_df[\"primary_label\"].unique()\nspectrogram_ds = AudioSpectogramDataset(root_dir=train_audio_path, classes=classes)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T23:04:25.238444Z","iopub.execute_input":"2025-05-03T23:04:25.239634Z","iopub.status.idle":"2025-05-03T23:04:42.840741Z","shell.execute_reply.started":"2025-05-03T23:04:25.239594Z","shell.execute_reply":"2025-05-03T23:04:42.839504Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample, sr = librosa.load(os.path.join(dir_path, \"train_audio\", \"1139490\", \"CSA36385.ogg\"))\nsample.shape\n\nspec = np.abs(librosa.stft(sample, hop_length=512))\nspec = librosa.amplitude_to_db(spec, ref=np.max)\n\n# librosa.get_duration(y=sample)\nspectrogram_ds[0][0].shape\n\n# spec.shape, sample.shape\n\n# MAKE SPECTOGRAM SHAPE SAME FOR ALL AUDIO FILES\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T23:04:42.843248Z","iopub.execute_input":"2025-05-03T23:04:42.843813Z","iopub.status.idle":"2025-05-03T23:04:43.33189Z","shell.execute_reply.started":"2025-05-03T23:04:42.843776Z","shell.execute_reply":"2025-05-03T23:04:43.330748Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert to log scale\nmel_spectrogram, _ = spectrogram_ds[10000]\nlog_mel_spectrogram = librosa.power_to_db(mel_spectrogram, ref=np.max)\n\n\n# librosa.display.specshow(mel_spectrogram, y_axis='mel', fmax=1500, x_axis='time');\n# Plot mel-spectrogram\nplt.figure(figsize=(12, 6))\nplt.imshow(log_mel_spectrogram, aspect='auto', origin='lower',\n           cmap='inferno', interpolation='nearest')\nplt.xlabel('Time')\nplt.ylabel('Mel Frequency Bin')\nplt.title('Mel Spectrogram of Audio File')\n\n# Display frequency bins on the x-axis\nplt.xticks(np.arange(128))\nplt.yticks(range(mel_spectrogram.shape[0]))\n\nplt.colorbar(format='%+02.0f dB')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T23:04:43.332913Z","iopub.execute_input":"2025-05-03T23:04:43.333166Z","iopub.status.idle":"2025-05-03T23:04:44.829074Z","shell.execute_reply.started":"2025-05-03T23:04:43.333147Z","shell.execute_reply":"2025-05-03T23:04:44.827766Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a Data Loader\nbatch_size = 32\ndata_loader = DataLoader(spectrogram_ds, batch_size=batch_size, shuffle=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T23:04:44.831239Z","iopub.execute_input":"2025-05-03T23:04:44.832088Z","iopub.status.idle":"2025-05-03T23:04:44.842088Z","shell.execute_reply.started":"2025-05-03T23:04:44.832047Z","shell.execute_reply":"2025-05-03T23:04:44.84087Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define device\ndevice = torch.device('cpu')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T23:06:35.323815Z","iopub.execute_input":"2025-05-03T23:06:35.324177Z","iopub.status.idle":"2025-05-03T23:06:35.329627Z","shell.execute_reply.started":"2025-05-03T23:06:35.324151Z","shell.execute_reply":"2025-05-03T23:06:35.328536Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torchvision import transforms, models\nimport torch.nn as nn\nimport torch.optim as optim\n\nclass CNN(nn.Module):\n    def __init__(self, num_channels=128, num_features=2584, num_classes=206):\n        super(CNN, self).__init__()\n\n        self.mel_channels = num_channels\n        self.window_size = num_features\n        \n        # Define the model architecture\n        self.conv1 = nn.Conv2d(in_channels=1, out_channels=64, kernel_size=7)\n        self.bn1 = nn.BatchNorm2d(num_features=64)\n        self.relu1 = nn.ReLU()\n        \n        self.conv2 = nn.Conv2d(in_channels=64, out_channels=32, kernel_size=3)\n        self.bn2 = nn.BatchNorm2d(num_features=32)\n        self.relu2 = nn.ReLU()\n        \n        self.fc = nn.Linear(in_features=num_features, out_features=num_classes)\n\n    def forward(self, x):\n        # Apply the convolutional layers\n        x = self.conv1(x)\n        x = self.bn1(x)\n        x = self.relu1(x)\n        \n        x = self.conv2(x)\n        x = self.bn2(x)\n        x = self.relu2(x)\n        \n        # Apply the fully connected layer\n        x = x.view(-1, self.window_size)\n        x = self.fc(x)\n        \n        return x\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T23:23:18.047538Z","iopub.execute_input":"2025-05-03T23:23:18.047913Z","iopub.status.idle":"2025-05-03T23:23:18.05712Z","shell.execute_reply.started":"2025-05-03T23:23:18.04788Z","shell.execute_reply":"2025-05-03T23:23:18.055879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize model\nmodel = CNN().to(device)\ncriterion = nn.CrossEntropyLoss()\noptimizer = optim.Adam(model.parameters(), lr=0.001)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T23:23:19.672456Z","iopub.execute_input":"2025-05-03T23:23:19.672912Z","iopub.status.idle":"2025-05-03T23:23:19.687618Z","shell.execute_reply.started":"2025-05-03T23:23:19.672883Z","shell.execute_reply":"2025-05-03T23:23:19.68623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the model\ndevice = torch.device('cpu')\nmodel.to(device)\n\nfor epoch in range(10):  # loop over the dataset multiple times\n\n    running_loss = 0.0\n    for batch_index, batch in enumerate(data_loader):\n        mel_spectrograms, labels = batch\n        mel_spectrograms = mel_spectrograms.unsqueeze(1).to(device)\n        # labels = labels.to(device)\n        \n        \n        # Zero the parameter gradients\n        optimizer.zero_grad()\n        \n        # Forward + backward + optimize\n        outputs = model(mel_spectrograms)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n\n        # Print statistics\n        running_loss += loss.item()\n        if batch_index % 100 == 0:\n            print('[%d, %5d] loss: %.3f' %\n                  (epoch+1, batch_index+1, running_loss/(i+1)))\n            writer.add_scalar('loss', running_loss/(batch_index+1), epoch)\n\n    # reset the loss\n    running_loss = 0.0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T23:23:28.406763Z","iopub.execute_input":"2025-05-03T23:23:28.407069Z","iopub.status.idle":"2025-05-03T23:23:52.449181Z","shell.execute_reply.started":"2025-05-03T23:23:28.407049Z","shell.execute_reply":"2025-05-03T23:23:52.447688Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save the model\ntorch.save(model.state_dict(), os.path.join())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}