{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":8119992,"sourceType":"datasetVersion","datasetId":4797902}],"dockerImageVersionId":30683,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\n# Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-14T22:25:26.623796Z","iopub.execute_input":"2024-04-14T22:25:26.624137Z","iopub.status.idle":"2024-04-14T22:25:31.256679Z","shell.execute_reply.started":"2024-04-14T22:25:26.624109Z","shell.execute_reply":"2024-04-14T22:25:31.25569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\nimport numpy as np\nimport random\n\ndef compute_spectral_contrast(audio_file, sr=22050, hop_length=512):\n    \"\"\"\n    Computes the spectral contrast of an audio file with an increased number of subbands.\n\n    Args:\n        audio_file (str): Path to the audio file.\n        sr (int): Sampling rate of the audio file.\n        hop_length (int): Number of samples between successive frames.\n        n_bands (int): Number of frequency subbands used in the spectral contrast calculation.\n\n    Returns:\n        np.ndarray: Spectral contrast matrix.\n    \"\"\"\n    # Load the audio file\n    y, sr = librosa.load(audio_file, sr=sr)\n    \n    total_samples = len(y)\n    duration = 5.94*2\n    samples_per_clip = int(sr * duration)\n\n    if total_samples < samples_per_clip:\n        raise ValueError(\"The audio file is shorter than the requested crop duration.\")\n\n    # Choose a random start point for the clip\n    start = random.randint(0, total_samples - samples_per_clip)\n    \n    # Crop the clip from the audio\n    clip = y[start:start + samples_per_clip]\n\n    \n    # Compute the spectral contrast\n    S = librosa.stft(clip)\n#     contrast = librosa.feature.spectral_contrast(S=S, sr=sr, hop_length=hop_length)\n    \n    return np.abs(S)\n\n# Example usage\naudio_path = '/kaggle/input/birdclef-2024/train_audio/asbfly/XC134896.ogg'\nspectral_contrast_vectors = compute_spectral_contrast(audio_path)\n\nprint(\"Spectral Contrast Vectors Shape:\", spectral_contrast_vectors.shape)\nprint(\"Spectral Contrast Vectors:\", spectral_contrast_vectors)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-14T22:25:31.258859Z","iopub.execute_input":"2024-04-14T22:25:31.259391Z","iopub.status.idle":"2024-04-14T22:25:40.979491Z","shell.execute_reply.started":"2024-04-14T22:25:31.259343Z","shell.execute_reply":"2024-04-14T22:25:40.978535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport librosa\nimport numpy as np\nimport concurrent.futures\nfrom tqdm import tqdm\n\ndef compute_and_save_spectral_contrast(file_path, input_dir, output_dir, sr=22050, hop_length=512):\n    \"\"\"\n    Computes the spectral contrast of an audio file and saves it in 16-bit precision.\n\n    Args:\n        file_path (str): Full path to the audio file.\n        input_dir (str): Root input directory containing the audio files.\n        output_dir (str): Root output directory where results will be saved.\n        sr (int): Sampling rate of the audio file.\n        hop_length (int): Number of samples between successive frames.\n    \"\"\"\n    try:\n        # Load the audio file\n        y, _ = librosa.load(file_path, sr=sr)\n        \n        # Compute the spectral contrast\n#         S = np.abs(librosa.stft(y))\n        contrast = librosa.feature.melspectrogram(y=y, sr=sr, hop_length=hop_length).astype(np.float16)\n        \n        # Prepare the save path\n        relative_path = os.path.relpath(os.path.dirname(file_path), input_dir)\n        save_dir = os.path.join(output_dir, relative_path)\n        os.makedirs(save_dir, exist_ok=True)\n        save_path = os.path.join(save_dir, os.path.splitext(os.path.basename(file_path))[0] + '.npy')\n        \n        # Save the spectral contrast in 16-bit precision\n        np.save(save_path, contrast)\n    except Exception as e:\n        print(f\"Failed to process {file_path}: {e}\")\n\ndef process_audio_files(input_dir, output_dir):\n    \"\"\"\n    Processes all audio files in the input directory using parallel processing,\n    computes their spectral contrast, and saves the results in the output directory\n    with the same structure.\n    \n    Args:\n        input_dir (str): The root directory to search for audio files.\n        output_dir (str): The root directory to save the spectral contrast data.\n    \"\"\"\n    files = [os.path.join(dp, f) for dp, dn, filenames in os.walk(input_dir) \n             for f in filenames if f.endswith(('.wav', '.mp3', '.ogg'))]\n    total_files = len(files)\n    \n    with concurrent.futures.ProcessPoolExecutor() as executor:\n        tasks = [executor.submit(compute_and_save_spectral_contrast, file, input_dir, output_dir)\n                 for file in files]\n        for task in tqdm(concurrent.futures.as_completed(tasks), total=total_files):\n            pass  # tqdm will update the progress as tasks complete\n\n# Example usage\ninput_dir = '/kaggle/input/birdclef-2024/train_audio'\noutput_dir = '/kaggle/working/output/contrast'\n\nprint('saving')\n#process_audio_files(input_dir, output_dir)\nprint('done')","metadata":{"execution":{"iopub.status.busy":"2024-04-14T22:25:40.980821Z","iopub.execute_input":"2024-04-14T22:25:40.981242Z","iopub.status.idle":"2024-04-14T22:25:40.994439Z","shell.execute_reply.started":"2024-04-14T22:25:40.981215Z","shell.execute_reply":"2024-04-14T22:25:40.993222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport torch\nimport librosa\nimport numpy as np\nimport random\nfrom torch.utils.data import Dataset, DataLoader\n\n# from torch.utils.data import Dataset, DataLoader\n# import torch\n# import librosa\n# import numpy as np\n# import random\n# import os\n\nclass BirdClefDataset(Dataset):\n    def __init__(self, root_dir, sr=22050, hop_length=512):\n        \"\"\"\n        Args:\n            root_dir (string): Directory with all the categories and audio files.\n            sr (int): Sampling rate for audio files.\n            hop_length (int): Hop length for spectral contrast calculation.\n        \"\"\"\n        self.root_dir = root_dir\n        self.sr = sr\n        self.hop_length = hop_length\n        self.files = []\n        self.category_labels = {}\n        # Walk through the directory to list all audio files and assign labels\n        for dirname, _, filenames in os.walk(root_dir):\n            category = dirname.split('/')[-1]\n            if category not in self.category_labels:\n                self.category_labels[category] = len(self.category_labels)\n            for filename in filenames:\n                if filename.endswith('.npy'):\n                    self.files.append((os.path.join(dirname, filename), self.category_labels[category]))\n\n    def __len__(self):\n        return len(self.files)\n\n    def __getitem__(self, idx):\n        if idx == len(self.files):\n            idx = 0\n        audio_file, label = self.files[idx]\n        try:\n            spectral_contrast = self.compute_spectral_contrast(audio_file, self.sr, self.hop_length)\n        except:\n            return self.__getitem__(idx + 1)\n        spectral_contrast_tensor = torch.tensor(spectral_contrast, dtype=torch.float32)\n        return spectral_contrast_tensor, label\n\n    def compute_spectral_contrast(self, audio_file, sr, hop_length):\n#         y, sr = librosa.load(audio_file, sr=sr)\n        y = np.load(audio_file).T\n        total_samples = len(y)\n#         print(y.shape)\n        duration = 5.94*2\n        samples_per_clip = 512#int(sr * duration)\n        if total_samples < samples_per_clip:\n            raise ValueError(\"The audio file is shorter than the requested crop duration.\")\n        start = random.randint(0, total_samples - samples_per_clip)\n        clip = y[start:start + samples_per_clip]\n        #S = np.abs(librosa.stft(clip))\n        #contrast = librosa.feature.spectral_contrast(S=S, sr=sr, hop_length=hop_length)\n        # Compute the Mel Spectrogram\n#         mel_spec = librosa.feature.melspectrogram(y=clip, sr=sr, hop_length=hop_length)\n\n        # Transpose to align time axis\n#         contrast = contrast.T\n#         mel_spec = mel_spec.T\n\n        # Concatenate the spectral contrast and Mel Spectrogram along the feature axis\n        #features = np.concatenate((contrast, mel_spec), axis=1)\n        return clip#features\n\n    @property\n    def num_classes(self):\n        return len(self.category_labels)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-14T22:25:40.9959Z","iopub.execute_input":"2024-04-14T22:25:40.996249Z","iopub.status.idle":"2024-04-14T22:25:41.015262Z","shell.execute_reply.started":"2024-04-14T22:25:40.996218Z","shell.execute_reply":"2024-04-14T22:25:41.014393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = BirdClefDataset(root_dir='/kaggle/input/birdcalls-mel/output/contrast')\ndataloader = DataLoader(dataset, batch_size=64, shuffle=True, num_workers=4)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T22:25:41.01783Z","iopub.execute_input":"2024-04-14T22:25:41.018117Z","iopub.status.idle":"2024-04-14T22:25:45.596192Z","shell.execute_reply.started":"2024-04-14T22:25:41.018085Z","shell.execute_reply":"2024-04-14T22:25:45.595451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.num_classes","metadata":{"execution":{"iopub.status.busy":"2024-04-14T22:25:45.59711Z","iopub.execute_input":"2024-04-14T22:25:45.597378Z","iopub.status.idle":"2024-04-14T22:25:45.603752Z","shell.execute_reply.started":"2024-04-14T22:25:45.597355Z","shell.execute_reply":"2024-04-14T22:25:45.602727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for batch in dataloader:\n    spectral_contrasts, categories = batch\n    print(spectral_contrasts.shape, categories)\n#     (model(spectral_contrasts).shape)\n    break  # Just show the first bat","metadata":{"execution":{"iopub.status.busy":"2024-04-14T22:25:45.604764Z","iopub.execute_input":"2024-04-14T22:25:45.605049Z","iopub.status.idle":"2024-04-14T22:25:47.179057Z","shell.execute_reply.started":"2024-04-14T22:25:45.605019Z","shell.execute_reply":"2024-04-14T22:25:47.177928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\nclass NewGELU(nn.Module):\n    \"\"\"Careful there are a few versions of GeLU, this one is the exact one used by OpenAI\"\"\"\n    def forward(self, input):\n        return 0.5 * input * (1.0 + torch.tanh(math.sqrt(2.0 / math.pi) * (input + 0.044715 * torch.pow(input, 3.0))))\n\nclass SelfAttention(nn.Module):\n\n    def __init__(self, n_embd = 256, n_head = 8):\n        super().__init__()\n        assert n_embd % n_head == 0\n        self.n_head = n_head\n        self.n_embd = n_embd\n        # key, query, value projections for all heads, but in a batch\n        self.c_attn = nn.Linear(n_embd, 3 * n_embd)\n        # output projection\n        self.c_proj = nn.Linear(n_embd, n_embd)\n        # regularization\n        \n#         # not really a 'bias', more of a mask, but following the OpenAI/HF naming though\n#         self.register_buffer(\"bias\", torch.tril(torch.ones(config.block_size, config.block_size))\n#                                      .view(1, 1, config.block_size, config.block_size))\n\n    def forward(self, x):\n        B, T, C = x.size() # batch size, sequence length, embedding dimensionality (n_embd)\n        # calculate query, key, values for all heads in batch and move head forward to be the batch dim\n        qkv = self.c_attn(x)\n        q, k, v = qkv.split(self.n_embd, dim=2)\n        k = k.view(B, T, self.n_head, C // self.n_head).transpose(1, 2) # (B, nh, T, hs)\n        q = q.view(B, T, self.n_head, C // self.n_head).transpose(1, 2) # (B, nh, T, hs)\n        v = v.view(B, T, self.n_head, C // self.n_head).transpose(1, 2) # (B, nh, T, hs)\n        # manual implementation of attention\n        att = (q @ k.transpose(-2, -1)) * (1.0 / math.sqrt(k.size(-1)))\n#         att = att.masked_fill(self.bias[:,:,:T,:T] == 0, float('-inf'))\n        att = F.softmax(att, dim=-1)\n        y = att @ v # (B, nh, T, T) x (B, nh, T, hs) -> (B, nh, T, hs)\n        y = y.transpose(1, 2).contiguous().view(B, T, C) # re-assemble all head outputs side by side\n        # output projection\n        y = self.c_proj(y)\n        return y\n\nclass MLP1(nn.Module):\n\n    def __init__(self, n_inp = 256, n_embd = 256):\n        super().__init__()\n        self.c_fc    = nn.Linear(n_inp, 2 * n_embd)\n        self.gelu    = NewGELU()\n        self.c_proj  = nn.Linear(2 * n_embd, 2*n_embd)\n        self.gelu1    = NewGELU()\n        self.c_proj1  = nn.Linear(2 * n_embd, n_embd)\n\n    def forward(self, x):\n        x = self.c_fc(x)\n        x = self.gelu(x)\n        x = self.c_proj(x)\n        x = self.gelu1(x)\n        x = self.c_proj1(x)\n        return x\n    \n\nclass MLP(nn.Module):\n\n    def __init__(self, n_inp = 256, n_embd = 256):\n        super().__init__()\n        self.c_fc    = nn.Linear(n_inp, 4 * n_embd)\n        self.gelu    = NewGELU()\n        self.c_proj  = nn.Linear(4 * n_embd, n_embd)\n        \n    def forward(self, x):\n        x = self.c_fc(x)\n        x = self.gelu(x)\n        x = self.c_proj(x)\n        return x\n\n# class Block(nn.Module):\n\n#     def __init__(self, n_inp = 512):\n#         super().__init__()\n#         self.ln_1 = nn.LayerNorm(n_inp)\n#         self.attn1 = SelfAttention(512, 8)\n# #         self.ln_2 = nn.LayerNorm(512)\n#         self.mlp = MLP(n_inp, 512)\n# #         self.ln_3 = nn.LayerNorm(512)\n#         self.attn2 = SelfAttention(512, 8)\n#         self.attn3 = SelfAttention(512, 8)\n\n#     def forward(self, x):\n# #         x = self.mlp(x)\n# #         print(x.shape)\n#         B, N, C = x.shape\n#         S = 8\n#         x = x.view(-1, S, C)\n#         x = self.attn1(self.ln_1(x))\n#         x = self.attn2(x)\n#         x = self.attn3(x)\n#         output = torch.mean(x, dim = 1)\n# #         print(output.shape)\n#         output = output.view(B, -1, C)\n# #         print(output.shape)\n# #         print('-----')\n#         return output\n\n\nclass Block(nn.Module):\n    def __init__(self, n_inp=512):\n        super().__init__()\n        self.ln_1 = nn.LayerNorm(n_inp)\n        self.attn1 = SelfAttention(512, 8)\n        self.ln_2 = nn.LayerNorm(n_inp)\n        self.mlp1 = MLP(512, 512)\n        self.attn2 = SelfAttention(512, 8)\n        self.ln_3 = nn.LayerNorm(n_inp)\n        self.mlp2 = MLP(512, 512)\n        self.attn3 = SelfAttention(512, 8)\n        self.pooler = nn.Parameter(torch.randn(n_inp))\n        self.ln_4 = nn.LayerNorm(n_inp)\n        self.ln_5 = nn.LayerNorm(n_inp)\n        self.ln_6 = nn.LayerNorm(n_inp)\n        self.mlp3 = MLP1(512, 512)\n\n    def forward(self, x):\n        B, N, C = x.shape\n        S = 8\n        x = x.view(-1, S, C)\n\n        # Reshape pooler to concatenate it to each sample in x\n        # Expand pooler to match the batch size and sequence length, and unsqueeze to add a singleton dimension for concatenation\n        pooler = self.pooler.expand(B * (N // S), 1, C)\n\n        # Concatenate pooler to each sample in x along the second dimension (sequence dimension)\n        x = torch.cat([x, pooler], dim=1)\n\n        # Process through layers\n#         x = self.ln_1(x)\n        x = x + self.attn1(self.ln_1(x))\n        x = x + self.mlp1(self.ln_2(x))\n        x = x + self.attn2(self.ln_3(x))\n        x = x + self.mlp2(self.ln_4(x))\n        x = x + self.attn3(self.ln_5(x))\n\n        # Get the output from the pooler-enhanced position (the last in the sequence)\n        output = self.mlp3(self.ln_6(x[:, -1, :]))\n#         x = x + self.mlp2(self.ln_3(x))\n        output = output.view(B, -1, C)\n        return output","metadata":{"execution":{"iopub.status.busy":"2024-04-14T22:25:47.181589Z","iopub.execute_input":"2024-04-14T22:25:47.181902Z","iopub.status.idle":"2024-04-14T22:25:47.208137Z","shell.execute_reply.started":"2024-04-14T22:25:47.181874Z","shell.execute_reply":"2024-04-14T22:25:47.20732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Classifier(nn.Module):\n    def __init__(self, input_dim = 7, classes = 183):\n        \"\"\"\n        Initialize the model with two layers of multi-head attention.\n\n        Args:\n            embed_dim (int): Size of each embedding vector (must be divisible by num_heads).\n            num_heads (int): Number of heads in the multi-head attention models.\n        \"\"\"\n        super(Classifier, self).__init__()\n        self.mlp = MLP1(128, 512)\n        self.layer1 = Block()\n        self.layer2 = Block()\n        self.layer3 = Block()\n        self.fcn = nn.Sequential(\n                nn.Linear(512, 512),  # First linear layer from 256 to 512 nodes\n                torch.nn.SiLU(),            # ReLU activation for non-linearity\n                nn.Linear(512, 512),  # Second linear layer from 512 to 256 nodes\n                torch.nn.SiLU(),            # Another ReLU activation\n                nn.Linear(512, 183)\n        )\n        \n    def forward(self, x):\n        \"\"\"\n        Forward pass for the model which applies two layers of attention\n        and returns the mean of all outputs.\n\n        Args:\n            x (Tensor): Input tensor of shape (S, N, E) where S is the sequence length,\n                        N is the batch size, and E is the embedding dimension.\n\n        Returns:\n            Tensor: A tensor of shape (N, E) representing the mean output of the attention layers.\n        \"\"\"\n#         print(x.shape)\n        x = self.mlp(x)\n        x = self.layer1(x)\n#         print(x.shape)\n        x = self.layer2(x)\n#         print(x.shape)\n        x = self.layer3(x)\n#         print(x.shape)\n        x = x.squeeze(1)\n#         print(x.shape)\n        out = self.fcn(x)\n        return out\n\n# # Example usage\n# B, N, S, C = 4, 64, 8, 128  # Batch size B, sequence blocks N, 8 vectors per block, C features\n# input_tensor = torch.rand(B * N, S, C)  # Simulate input tensor\n\n# # Split the tensor into blocks (we assume B * N is divisible by S for simplicity)\n# blocks = input_tensor.view(-1, S, C)  # This reshapes the tensor to have B * N / S blocks of shape (S, C)\n\n# Initialize the model\n# model = BlockAttention(input_dim = 7, embed_dim=256, num_heads=8)\n\n# # Apply the model to each block and collect results\n# outputs = torch.stack([model(block.unsqueeze(1)) for block in blocks])\n\n# # Reshape outputs to match the expected output shape (B, N, C)\n# outputs = outputs.view(B, N, C)\n\n# print(\"Output Tensor Shape:\", outputs.shape)\n# print(\"Output Tensor:\", outputs)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-14T22:25:47.209495Z","iopub.execute_input":"2024-04-14T22:25:47.209829Z","iopub.status.idle":"2024-04-14T22:25:47.225358Z","shell.execute_reply.started":"2024-04-14T22:25:47.209798Z","shell.execute_reply":"2024-04-14T22:25:47.224599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch.optim as optim","metadata":{"execution":{"iopub.status.busy":"2024-04-14T22:25:47.22646Z","iopub.execute_input":"2024-04-14T22:25:47.226746Z","iopub.status.idle":"2024-04-14T22:25:47.239005Z","shell.execute_reply.started":"2024-04-14T22:25:47.226724Z","shell.execute_reply":"2024-04-14T22:25:47.238146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# Loss Function\nloss_function = nn.CrossEntropyLoss()\n\ndevice = \"cuda\"\nmodel = Classifier().to(device)\n# Device configuration\n\n\noptimizer = optim.AdamW(model.parameters(), lr=0.0001)\n# model","metadata":{"execution":{"iopub.status.busy":"2024-04-14T22:28:50.871138Z","iopub.execute_input":"2024-04-14T22:28:50.871772Z","iopub.status.idle":"2024-04-14T22:28:51.226352Z","shell.execute_reply.started":"2024-04-14T22:28:50.871737Z","shell.execute_reply":"2024-04-14T22:28:51.225593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save_path = 'saved_models'\nos.makedirs(save_path, exist_ok=True)\nepochs = 100","metadata":{"execution":{"iopub.status.busy":"2024-04-14T22:28:51.894232Z","iopub.execute_input":"2024-04-14T22:28:51.894928Z","iopub.status.idle":"2024-04-14T22:28:51.899921Z","shell.execute_reply.started":"2024-04-14T22:28:51.894894Z","shell.execute_reply":"2024-04-14T22:28:51.898668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nL = len(dataloader)//10\nprint(len(dataloader))","metadata":{"execution":{"iopub.status.busy":"2024-04-14T22:28:52.488142Z","iopub.execute_input":"2024-04-14T22:28:52.488523Z","iopub.status.idle":"2024-04-14T22:28:52.493923Z","shell.execute_reply.started":"2024-04-14T22:28:52.488493Z","shell.execute_reply":"2024-04-14T22:28:52.49311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model.train()  # Set the model to training mode\nfor epoch in (range(epochs)):\n    total_loss = 0\n    i = 0\n    for inputs, labels in (dataloader):\n        inputs, labels = inputs.to(device), labels.to(device)\n\n        # Forward pass\n        outputs = model(inputs)\n        loss = loss_function(outputs, labels)\n\n        # Backward pass and optimization\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n\n        total_loss += loss.item()\n        i += 1\n        if i%L == 0:\n            print('Loss', i, total_loss/(i))\n\n    # Print average loss for the epoch\n    average_loss = total_loss / len(dataloader)\n    print(f\"Epoch {epoch+1}/{epochs}, Loss: {average_loss:.4f}\")\n\n    # Save the model checkpoint\n    if (epoch + 1) % 10 == 0:  # Save every 10 epochs\n        torch.save(model.state_dict(), os.path.join(save_path, f\"model_epoch_{epoch+1}.pth\"))\n\n# Path where the model will be saved\n","metadata":{"execution":{"iopub.status.busy":"2024-04-14T22:28:52.983814Z","iopub.execute_input":"2024-04-14T22:28:52.984414Z","iopub.status.idle":"2024-04-14T22:38:13.12735Z","shell.execute_reply.started":"2024-04-14T22:28:52.984385Z","shell.execute_reply":"2024-04-14T22:38:13.125685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image","metadata":{"execution":{"iopub.status.busy":"2024-04-14T22:38:13.128673Z","iopub.status.idle":"2024-04-14T22:38:13.128995Z","shell.execute_reply.started":"2024-04-14T22:38:13.128838Z","shell.execute_reply":"2024-04-14T22:38:13.128851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"im = Image.open('/kaggle/input/birdclef-2024-mel-spectrograms/train_images/asbfly/XC374520_00.png')","metadata":{"execution":{"iopub.status.busy":"2024-04-14T22:38:13.129871Z","iopub.status.idle":"2024-04-14T22:38:13.130196Z","shell.execute_reply.started":"2024-04-14T22:38:13.130037Z","shell.execute_reply":"2024-04-14T22:38:13.130051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"im","metadata":{"execution":{"iopub.status.busy":"2024-04-14T22:28:35.806538Z","iopub.status.idle":"2024-04-14T22:28:35.80699Z","shell.execute_reply.started":"2024-04-14T22:28:35.806751Z","shell.execute_reply":"2024-04-14T22:28:35.80677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.array(im)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T22:28:35.808473Z","iopub.status.idle":"2024-04-14T22:28:35.808793Z","shell.execute_reply.started":"2024-04-14T22:28:35.808646Z","shell.execute_reply":"2024-04-14T22:28:35.808658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport librosa\nimport numpy as np\nimport concurrent.futures\nfrom tqdm import tqdm\n\ndef compute_and_save_spectral_contrast(file_path, input_dir, output_dir, sr=22050, hop_length=512):\n    \"\"\"\n    Computes the spectral contrast of an audio file and saves it in 16-bit precision.\n\n    Args:\n        file_path (str): Full path to the audio file.\n        input_dir (str): Root input directory containing the audio files.\n        output_dir (str): Root output directory where results will be saved.\n        sr (int): Sampling rate of the audio file.\n        hop_length (int): Number of samples between successive frames.\n    \"\"\"\n    try:\n        # Load the audio file\n        y, _ = librosa.load(file_path, sr=sr)\n        \n        # Compute the spectral contrast\n#         S = np.abs(librosa.stft(y))\n        contrast = librosa.feature.melspectrogram(y=y, sr=sr, hop_length=hop_length).astype(np.float16)\n        \n        # Prepare the save path\n        relative_path = os.path.relpath(os.path.dirname(file_path), input_dir)\n        save_dir = os.path.join(output_dir, relative_path)\n        os.makedirs(save_dir, exist_ok=True)\n        save_path = os.path.join(save_dir, os.path.splitext(os.path.basename(file_path))[0] + '.npy')\n        \n        # Save the spectral contrast in 16-bit precision\n        np.save(save_path, contrast)\n    except Exception as e:\n        print(f\"Failed to process {file_path}: {e}\")\n\ndef process_audio_files(input_dir, output_dir):\n    \"\"\"\n    Processes all audio files in the input directory using parallel processing,\n    computes their spectral contrast, and saves the results in the output directory\n    with the same structure.\n    \n    Args:\n        input_dir (str): The root directory to search for audio files.\n        output_dir (str): The root directory to save the spectral contrast data.\n    \"\"\"\n    files = [os.path.join(dp, f) for dp, dn, filenames in os.walk(input_dir) \n             for f in filenames if f.endswith(('.wav', '.mp3', '.ogg'))]\n    total_files = len(files)\n    \n    with concurrent.futures.ProcessPoolExecutor() as executor:\n        tasks = [executor.submit(compute_and_save_spectral_contrast, file, input_dir, output_dir)\n                 for file in files]\n        for task in tqdm(concurrent.futures.as_completed(tasks), total=total_files):\n            pass  # tqdm will update the progress as tasks complete\n\n# Example usage\ninput_dir = '/kaggle/input/birdclef-2024/train_audio'\noutput_dir = '/kaggle/working/output/contrast'\n\nprocess_audio_files(input_dir, output_dir)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-14T22:28:35.810385Z","iopub.status.idle":"2024-04-14T22:28:35.810694Z","shell.execute_reply.started":"2024-04-14T22:28:35.81054Z","shell.execute_reply":"2024-04-14T22:28:35.810553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport zipfile\n\ndef zip_folder(folder_path, output_path):\n    \"\"\"Zip the contents of an entire folder (with that folder included in the archive). \n    Excludes hidden files and folders.\"\"\"\n    with zipfile.ZipFile(output_path, 'w', zipfile.ZIP_DEFLATED) as zipf:\n        len_dir_path = len(folder_path)\n        for root, _, files in os.walk(folder_path):\n            for file in files:\n                # Create a zip file excluding hidden files and folders\n                if not file.startswith('.'):\n                    file_path = os.path.join(root, file)\n                    zipf.write(file_path, file_path[len_dir_path:])\n\n# Example usage: Zipping the '/kaggle/working/my_folder' directory to 'my_folder.zip'\nzip_folder('/kaggle/working/output/contrast', '/kaggle/working/contrast.zip')","metadata":{"execution":{"iopub.status.busy":"2024-04-14T22:28:35.811655Z","iopub.status.idle":"2024-04-14T22:28:35.811947Z","shell.execute_reply.started":"2024-04-14T22:28:35.8118Z","shell.execute_reply":"2024-04-14T22:28:35.811813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nos.cpu_count()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T22:28:35.81356Z","iopub.status.idle":"2024-04-14T22:28:35.814122Z","shell.execute_reply.started":"2024-04-14T22:28:35.813904Z","shell.execute_reply":"2024-04-14T22:28:35.813927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}