{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"## I learned how to make this dataset object in this Youtube video:\n## https://www.youtube.com/watch?v=88FFnqt5MNI&list=PL-wATfeyAMNoirN4idjev6aRu8ISZYVWm&index=4\n## Huge props to Valerio Velardo for his informative and great videos about audio signal proccesing!","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-14T18:04:58.181167Z","iopub.execute_input":"2024-04-14T18:04:58.181597Z","iopub.status.idle":"2024-04-14T18:04:58.186633Z","shell.execute_reply.started":"2024-04-14T18:04:58.181563Z","shell.execute_reply":"2024-04-14T18:04:58.185434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install torchsummary","metadata":{"execution":{"iopub.status.busy":"2024-04-27T14:54:39.808886Z","iopub.execute_input":"2024-04-27T14:54:39.809693Z","iopub.status.idle":"2024-04-27T14:54:53.191931Z","shell.execute_reply.started":"2024-04-27T14:54:39.809663Z","shell.execute_reply":"2024-04-27T14:54:53.190860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\n\nimport matplotlib.pyplot as plt\nimport matplotlib\n\nimport IPython.display as ipd\nimport os\nfrom scipy.signal import butter, sosfilt\n\nimport torchaudio\nimport torch\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch import nn\n\nfrom torchsummary import summary","metadata":{"execution":{"iopub.status.busy":"2024-04-27T14:54:56.583651Z","iopub.execute_input":"2024-04-27T14:54:56.584451Z","iopub.status.idle":"2024-04-27T14:54:56.595208Z","shell.execute_reply.started":"2024-04-27T14:54:56.584410Z","shell.execute_reply":"2024-04-27T14:54:56.594131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_audio = '/kaggle/input/birdclef-2024/train_audio'\ntrain_meta = '/kaggle/input/birdclef-2024/train_metadata.csv'\nTax = '/kaggle/input/birdclef-2024/eBird_Taxonomy_v2021.csv'","metadata":{"execution":{"iopub.status.busy":"2024-04-27T14:55:00.002251Z","iopub.execute_input":"2024-04-27T14:55:00.003009Z","iopub.status.idle":"2024-04-27T14:55:00.007080Z","shell.execute_reply.started":"2024-04-27T14:55:00.002973Z","shell.execute_reply":"2024-04-27T14:55:00.006084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(train_meta)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-27T14:55:02.356109Z","iopub.execute_input":"2024-04-27T14:55:02.356492Z","iopub.status.idle":"2024-04-27T14:55:02.578686Z","shell.execute_reply.started":"2024-04-27T14:55:02.356459Z","shell.execute_reply":"2024-04-27T14:55:02.577733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BirdDataset(Dataset):\n    def __init__(self, train_meta, audio_dir, transformation, target_sample_rate, audio_length):\n        self.meta = pd.read_csv(train_meta)\n        self.audio_dir = audio_dir\n        self.transformation = transformation\n        self.sample_rate = target_sample_rate\n        self.audio_length = audio_length\n\n        # Create a mapping from labels to indices\n        self.label_to_index = {label: idx for idx, label in enumerate(self.meta['primary_label'].unique())}\n\n    def __len__(self):\n    ## How to calculate len(number of samples)\n        return len(self.meta)\n        \n    def __getitem__(self, index):\n        audio_sample_path = self._get_audio_sample_path(index)\n        label = self._get_audio_sample_label(index)\n        signal, sr = torchaudio.load(audio_sample_path)\n        signal = self._cut_if_necessary(signal)\n        signal = self._right_pad_if_necessary(signal)\n        signal = self.transformation(signal)\n\n        # Convert label to a numeric index\n        label_index = self.label_to_index[label]\n        label_tensor = torch.tensor(label_index, dtype=torch.long)\n\n        return signal, label_tensor\n\n    \n    def _cut_if_necessary(self, signal):\n        if signal.shape[1] > self.audio_length:\n            signal = signal[:, :self.audio_length]\n        return signal\n    \n    def _right_pad_if_necessary(self, signal):\n        length = signal.shape[1]\n        if length < self.audio_length:\n            num_missing_samples = self.audio_length - length  # Ensure 'length' is spelled correctly\n            padding_array = (0, num_missing_samples)\n            signal = torch.nn.functional.pad(signal, padding_array, \"constant\", 0)  # Also added mode and value for clarity\n        return signal\n\n            \n            \n    \n    def _get_audio_sample_path(self, index):\n        # Use the actual filename column from your DataFrame\n        filename = self.meta.iloc[index]['filename']\n        path = os.path.join(self.audio_dir, filename)\n        return path\n\n    \n    def _get_audio_sample_label(self, index):\n        # Use the actual label column from your DataFrame\n        return self.meta.iloc[index]['primary_label']\n","metadata":{"execution":{"iopub.status.busy":"2024-04-27T14:55:05.642992Z","iopub.execute_input":"2024-04-27T14:55:05.643777Z","iopub.status.idle":"2024-04-27T14:55:05.655934Z","shell.execute_reply.started":"2024-04-27T14:55:05.643746Z","shell.execute_reply":"2024-04-27T14:55:05.654904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_rate = 32000\naudio_len = sample_rate * 5 #5 sec audio sample\nmel_spec = torchaudio.transforms.MelSpectrogram(sample_rate = sample_rate,\n                                               n_fft = 1024,\n                                               hop_length = 512,\n                                               n_mels = 64)\n\ntrain_data = BirdDataset(train_meta,\n                          train_audio,\n                          mel_spec,\n                          sample_rate,\n                         audio_len)  \nsignal, label = train_data[0]\nprint(f\"There are {len(train_data)} samples in the dataset\")\nprint(f\"data_sample[0] label is: {label}\")\nprint(signal.shape)\nprint(signal.type())\n","metadata":{"execution":{"iopub.status.busy":"2024-04-27T14:55:09.152970Z","iopub.execute_input":"2024-04-27T14:55:09.153314Z","iopub.status.idle":"2024-04-27T14:55:09.575746Z","shell.execute_reply.started":"2024-04-27T14:55:09.153288Z","shell.execute_reply":"2024-04-27T14:55:09.574660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CNNNetwork(nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.conv1 = nn.Sequential(nn.Conv2d(\n            in_channels=1,   # Initial number of channels in the input (assuming grayscale images)\n            out_channels=16, # Output channels from the first conv layer\n            kernel_size=3,\n            stride=1,\n            padding=2),\n            nn.ReLU(),\n            nn.MaxPool2d(kernel_size=2))\n\n        self.conv2 = nn.Sequential(nn.Conv2d(\n            in_channels=16,  # Input channels should match output channels of the previous layer\n            out_channels=32, # Output channels from the second conv layer\n            kernel_size=3,\n            stride=1,\n            padding=2),\n            nn.ReLU(),\n            nn.MaxPool2d(kernel_size=2))\n\n        self.conv3 = nn.Sequential(nn.Conv2d(\n            in_channels=32,  # Input channels should match output channels of the previous layer\n            out_channels=64, # Output channels from the third conv layer\n            kernel_size=3,\n            stride=1,\n            padding=2),\n            nn.ReLU(),\n            nn.MaxPool2d(kernel_size=2))\n\n        self.conv4 = nn.Sequential(nn.Conv2d(\n            in_channels=64,  # Input channels should match output channels of the previous layer\n            out_channels=128,# Output channels from the fourth conv layer\n            kernel_size=3,\n            stride=1,\n            padding=2),\n            nn.ReLU(),\n            nn.MaxPool2d(kernel_size=2))\n\n        self.flatten = nn.Flatten()\n        self.linear = nn.Linear(13440, 182) # Ensure the input features to linear layer are correctly calculated based on the output size of conv4\n        self.softmax = nn.Softmax(dim=1)\n\n    def forward(self, input_data):\n        x = self.conv1(input_data)\n        x = self.conv2(x)  # Input from conv1\n        x = self.conv3(x)  # Input from conv2\n        x = self.conv4(x)  # Input from conv3\n        x = self.flatten(x)\n        logits = self.linear(x)\n        pred = self.softmax(logits)\n        return pred\n","metadata":{"execution":{"iopub.status.busy":"2024-04-27T14:55:13.488151Z","iopub.execute_input":"2024-04-27T14:55:13.488805Z","iopub.status.idle":"2024-04-27T14:55:13.499786Z","shell.execute_reply.started":"2024-04-27T14:55:13.488777Z","shell.execute_reply":"2024-04-27T14:55:13.498900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## cpu ot gpu??\nif torch.cuda.is_available():\n    device = \"cuda\"\n    \nelse:\n    device = \"cpu\"\n    \n    \nprint(device)","metadata":{"execution":{"iopub.status.busy":"2024-04-27T14:55:17.402519Z","iopub.execute_input":"2024-04-27T14:55:17.403234Z","iopub.status.idle":"2024-04-27T14:55:17.436448Z","shell.execute_reply.started":"2024-04-27T14:55:17.403201Z","shell.execute_reply":"2024-04-27T14:55:17.435394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cnn = CNNNetwork().to(device)\nsummary(cnn, (1, 64, 313))","metadata":{"execution":{"iopub.status.busy":"2024-04-27T14:55:20.418042Z","iopub.execute_input":"2024-04-27T14:55:20.418664Z","iopub.status.idle":"2024-04-27T14:55:21.232242Z","shell.execute_reply.started":"2024-04-27T14:55:20.418632Z","shell.execute_reply":"2024-04-27T14:55:21.231257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_data_loader(train_data, batch_size):\n    train_dataloader = DataLoader(train_data, batch_size=batch_size, shuffle=True)\n    return train_dataloader\n\n\ndef train_one_epoch(model, data_loader, loss_fn, optimiser, device):\n    model.train()\n    for batch in data_loader:\n        inputs, targets = batch  # This should normally suffice if batch is correctly formatted\n        inputs, targets = inputs.to(device), targets.to(device)\n        \n        predictions = model(inputs)\n        loss = loss_fn(predictions, targets)\n        optimiser.zero_grad()\n        loss.backward()\n        optimiser.step()\n\n\n    print(\"Loss : \", loss.item())\n\n     \ndef model_train(model, data_loader, loss_fn, optimiser, epochs):\n    for i in range(epochs):\n        print(f\"Epoch {i+1}\")\n        train_one_epoch(model, data_loader, loss_fn, optimiser, device)\n        print(\"======================================================================\")\n    \n    print(\"Training is DONE!\")\n    ","metadata":{"execution":{"iopub.status.busy":"2024-04-27T14:58:25.526084Z","iopub.execute_input":"2024-04-27T14:58:25.527019Z","iopub.status.idle":"2024-04-27T14:58:25.534942Z","shell.execute_reply.started":"2024-04-27T14:58:25.526984Z","shell.execute_reply":"2024-04-27T14:58:25.534032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 128\nEPOCHS = 10\nLEARNING_RATE = 0.001\ntrain_dataloader = create_data_loader(train_data, BATCH_SIZE)\nloss_fn = nn.CrossEntropyLoss()\noptimizer = torch.optim.Adam(cnn.parameters(),\n                            lr = LEARNING_RATE)\n#model_train(cnn, train_dataloader, loss_fn, optimizer, EPOCHS)","metadata":{"execution":{"iopub.status.busy":"2024-04-27T15:00:54.425435Z","iopub.execute_input":"2024-04-27T15:00:54.425896Z","iopub.status.idle":"2024-04-27T15:00:54.432613Z","shell.execute_reply.started":"2024-04-27T15:00:54.425838Z","shell.execute_reply":"2024-04-27T15:00:54.431493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}