{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":12361093,"sourceType":"datasetVersion","datasetId":7793403}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport librosa\nfrom torch.utils.data import DataLoader,Dataset\nimport cv2\nimport numpy as np\nimport torchaudio\nimport os\nimport torchaudio\nimport torch\nimport torch.optim as optim\nimport torchaudio.transforms as transforms","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-07T03:26:51.503590Z","iopub.execute_input":"2025-07-07T03:26:51.504316Z","iopub.status.idle":"2025-07-07T03:26:53.309860Z","shell.execute_reply.started":"2025-07-07T03:26:51.504288Z","shell.execute_reply":"2025-07-07T03:26:53.308874Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\ntorch.set_num_threads(1)\n\nfrom IPython.display import Audio\nfrom pprint import pprint\n# download example\ntorch.hub.download_url_to_file('https://models.silero.ai/vad_models/en.wav', 'en_example.wav')\n\nmodel, utils = torch.hub.load(repo_or_dir='snakers4/silero-vad',\n                              model='silero_vad',\n                              force_reload=True)\n\n(get_speech_timestamps,_, read_audio,*_) = utils\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T03:26:53.311230Z","iopub.execute_input":"2025-07-07T03:26:53.311749Z","iopub.status.idle":"2025-07-07T03:26:55.153984Z","shell.execute_reply.started":"2025-07-07T03:26:53.311717Z","shell.execute_reply":"2025-07-07T03:26:55.153422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_human_voice(wav, model, sampling_rate=16000, min_duration_sec=1.0):\n    speech_timestamps = get_speech_timestamps(wav, model, sampling_rate=sampling_rate)\n\n    if not speech_timestamps:\n        return None\n\n    min_samples = int(min_duration_sec * sampling_rate)\n    speech_clips = []\n\n    for segment in speech_timestamps:\n        start = segment['start']\n        end = segment['end']\n        if end - start >= min_samples:\n            clip = wav[start:end]\n            speech_clips.append(clip)\n\n    return speech_clips, sampling_rate","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T03:26:56.155213Z","iopub.execute_input":"2025-07-07T03:26:56.155946Z","iopub.status.idle":"2025-07-07T03:26:56.160359Z","shell.execute_reply.started":"2025-07-07T03:26:56.155910Z","shell.execute_reply":"2025-07-07T03:26:56.159678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_segments(mask, min_duration=5):\n    segments = []\n    start = None\n    for i, val in enumerate(mask):\n        if val and start is None:\n            start = i\n        elif not val and start is not None:\n            if i - start >= min_duration:\n                segments.append((start, i))\n            start = None\n    if start is not None and len(mask) - start >= min_duration:\n        segments.append((start, len(mask)))\n    return segments\n\ndef extract_bird_clip(audio_path, threshold=0.05, desired_duration=5.0):\n    waveform, sample_rate = torchaudio.load(audio_path)\n    waveform = torchaudio.functional.highpass_biquad(waveform, sample_rate, cutoff_freq=1000)\n\n    if waveform.size(0) > 1:\n        waveform = waveform.mean(dim=0, keepdim=True)\n\n    frame_size = int(0.02 * sample_rate)\n    hop_size = int(0.01 * sample_rate)\n    frames = waveform.unfold(1, frame_size, hop_size)\n    energies = frames.pow(2).mean(dim=-1).squeeze()\n    energies = (energies - energies.min()) / (energies.max() - energies.min() + 1e-6)\n\n    speech_mask = energies > threshold\n    segments = get_segments(speech_mask, min_duration=5)\n\n    if not segments:\n        return None, None\n\n    # Choose the longest segment\n    start_frame, end_frame = max(segments, key=lambda x: x[1] - x[0])\n    start_sample = start_frame * hop_size\n    end_sample = end_frame * hop_size\n\n    # Get 5-sec clip centered around bird sound (2s before, 3s after start)\n    clip_start = start_sample - int(2.0 * sample_rate)\n    clip_end = start_sample + int(3.0 * sample_rate)\n\n    # Ensure bounds are valid\n    if clip_start<0:\n        clip_start=0\n        clip_end = int(5.0*sample_rate)\n    if clip_end > waveform.shape[1]:\n        clip_end = waveform.shape[1]\n        clip_start = clip_end-int(5.0*sample_rate)\n\n    bird_clip = waveform[:, clip_start:clip_end]\n    return bird_clip, sample_rate","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T03:26:57.364439Z","iopub.execute_input":"2025-07-07T03:26:57.365024Z","iopub.status.idle":"2025-07-07T03:26:57.373376Z","shell.execute_reply.started":"2025-07-07T03:26:57.365001Z","shell.execute_reply":"2025-07-07T03:26:57.372557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# testing human voice\n\naudio_path = \"/kaggle/input/new-preprocessed/kaggle/working/curr_preprocessed_data/1192948/CSA36366.wav\"\nwav = read_audio(audio_path, sampling_rate=16000)\noutput= extract_human_voice(wav, model)\nclips=None\nif output is None:\n    print(\"nhi hai human voice\")\nelse:\n    clips,sr = output\n# Play the first speech clip\nif clips:\n    display(Audio(clips[0].numpy(), rate=sr))\nelse:\n    print(\"No human voice found.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T03:27:05.458404Z","iopub.execute_input":"2025-07-07T03:27:05.459011Z","iopub.status.idle":"2025-07-07T03:27:05.579822Z","shell.execute_reply.started":"2025-07-07T03:27:05.458987Z","shell.execute_reply":"2025-07-07T03:27:05.579153Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# #testing bird voice\n\n# import IPython.display\n\n# clip, sr = extract_bird_clip(\"/kaggle/input/birdclef-2025/train_audio/41663/iNat1025799.ogg\")\n# if clip is not None:\n#     IPython.display.display(IPython.display.Audio(clip.numpy(), rate=sr))\n# else:\n#     print(\"No bird sound found.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T03:27:06.452130Z","iopub.execute_input":"2025-07-07T03:27:06.452788Z","iopub.status.idle":"2025-07-07T03:27:06.456221Z","shell.execute_reply.started":"2025-07-07T03:27:06.452763Z","shell.execute_reply":"2025-07-07T03:27:06.455487Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\nprint(f\"Using device: {device}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T03:27:07.468562Z","iopub.execute_input":"2025-07-07T03:27:07.469176Z","iopub.status.idle":"2025-07-07T03:27:07.491540Z","shell.execute_reply.started":"2025-07-07T03:27:07.469139Z","shell.execute_reply":"2025-07-07T03:27:07.490693Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Preprocessing:\n    def __init__(self, input_audio_folder, output_audio_folder):\n        self.input_audio_folder = input_audio_folder\n        self.output_audio_folder = output_audio_folder\n\n    def __call__(self, extract_bird_clip, extract_human_voice):\n        for subfolder in os.listdir(self.input_audio_folder):\n            audio_folder = os.path.join(self.input_audio_folder, subfolder)\n            if not os.path.isdir(audio_folder): \n                continue\n\n            for audio in os.listdir(audio_folder):\n                full_audio_path = os.path.join(audio_folder, audio)\n\n                bird_clip, bird_sample_rate = extract_bird_clip(full_audio_path)\n                if bird_clip is None:\n                    continue\n\n                human_voice_result = extract_human_voice(bird_clip, model, bird_sample_rate)\n                if human_voice_result is not None:\n                    human_clips, _ = human_voice_result\n                    if human_clips: \n                        print(f\"Skipped {audio}: human voice detected.\")\n                        continue\n                        \n                output_path = os.path.join(self.output_audio_folder, subfolder)\n                os.makedirs(output_path, exist_ok=True)\n                \n                output_file_path = os.path.join(output_path, os.path.splitext(audio)[0] + \".wav\")\n                torchaudio.save(output_file_path, bird_clip, bird_sample_rate)\n                print(f\"Saved: {output_file_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T03:27:09.165358Z","iopub.execute_input":"2025-07-07T03:27:09.165697Z","iopub.status.idle":"2025-07-07T03:27:09.171709Z","shell.execute_reply.started":"2025-07-07T03:27:09.165676Z","shell.execute_reply":"2025-07-07T03:27:09.171044Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# output_audio_folder = \"/kaggle/working/curr_preprocessed_data\"\n# # input_audio_folders = [\n# #     \"/kaggle/input/birdclef-2022/train_audio\",\n# #     \"/kaggle/input/birdclef-2023/train_audio\",\n# #     \"/kaggle/input/birdclef-2024/train_audio\"\n# # ]\n\n# input_audio_folders = [\"/kaggle/input/birdclef-2025/train_audio\"]\n\n\n# for input_folder in input_audio_folders:\n#     preprocess = Preprocessing(input_folder, output_audio_folder)\n#     preprocess(extract_bird_clip, extract_human_voice)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T03:22:45.983289Z","iopub.execute_input":"2025-07-07T03:22:45.983886Z","iopub.status.idle":"2025-07-07T03:22:45.987165Z","shell.execute_reply.started":"2025-07-07T03:22:45.983861Z","shell.execute_reply":"2025-07-07T03:22:45.986453Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import shutil\n# shutil.copy(\"/kaggle/outputs/preprocessed_data.zip\", \"/kaggle/working/preprocessed_data.zip\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T03:22:46.428361Z","iopub.execute_input":"2025-07-07T03:22:46.428658Z","iopub.status.idle":"2025-07-07T03:22:46.431815Z","shell.execute_reply.started":"2025-07-07T03:22:46.428637Z","shell.execute_reply":"2025-07-07T03:22:46.431249Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def apply_effect(self,waveform, sample_rate):\n#     effector = torchaudio.io.AudioEffector(effect=\"atempo=1.5\",\"aecho=0.8:0.88:60:0.4\",\"vibrato:2\")\n#     return effector.apply(waveform, sample_rate)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T03:22:46.926438Z","iopub.execute_input":"2025-07-07T03:22:46.927107Z","iopub.status.idle":"2025-07-07T03:22:46.930097Z","shell.execute_reply.started":"2025-07-07T03:22:46.927087Z","shell.execute_reply":"2025-07-07T03:22:46.929505Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install efficientnet_pytorch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T03:27:15.434919Z","iopub.execute_input":"2025-07-07T03:27:15.435216Z","iopub.status.idle":"2025-07-07T03:27:18.448215Z","shell.execute_reply.started":"2025-07-07T03:27:15.435197Z","shell.execute_reply":"2025-07-07T03:27:18.447475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from efficientnet_pytorch import EfficientNet\nefficient_model = EfficientNet.from_pretrained('efficientnet-b2')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T03:24:44.309315Z","iopub.execute_input":"2025-07-07T03:24:44.309951Z","iopub.status.idle":"2025-07-07T03:24:45.146645Z","shell.execute_reply.started":"2025-07-07T03:24:44.309924Z","shell.execute_reply":"2025-07-07T03:24:45.145968Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(efficient_model)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T03:47:14.054765Z","iopub.execute_input":"2025-07-06T03:47:14.055034Z","iopub.status.idle":"2025-07-06T03:47:14.058139Z","shell.execute_reply.started":"2025-07-06T03:47:14.055011Z","shell.execute_reply":"2025-07-06T03:47:14.057524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"efficient_model._fc = nn.Linear(in_features=1408, out_features=206, bias=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T03:24:47.213092Z","iopub.execute_input":"2025-07-07T03:24:47.213381Z","iopub.status.idle":"2025-07-07T03:24:47.220341Z","shell.execute_reply.started":"2025-07-07T03:24:47.213360Z","shell.execute_reply":"2025-07-07T03:24:47.219783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(efficient_model)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T03:47:14.075187Z","iopub.execute_input":"2025-07-06T03:47:14.075852Z","iopub.status.idle":"2025-07-06T03:47:14.085514Z","shell.execute_reply.started":"2025-07-06T03:47:14.075827Z","shell.execute_reply":"2025-07-06T03:47:14.084804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"efficient_model.to(device)\nif torch.cuda.device_count() > 1:\n    print(f\"Using {torch.cuda.device_count()} GPUs with DataParallel.\")\n    efficient_model = nn.DataParallel(efficient_model)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T03:47:14.086252Z","iopub.execute_input":"2025-07-06T03:47:14.086485Z","iopub.status.idle":"2025-07-06T03:47:14.225542Z","shell.execute_reply.started":"2025-07-06T03:47:14.086464Z","shell.execute_reply":"2025-07-06T03:47:14.224977Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\n\nno_of_audios = []\nfolder_names = []\n\ninput_folder = \"/kaggle/input/new-preprocessed/kaggle/working/curr_preprocessed_data\"\n\nfor subfolder in sorted(os.listdir(input_folder)):\n    audio_folder = os.path.join(input_folder, subfolder)\n    if not os.path.isdir(audio_folder):\n        continue\n    folder_names.append(subfolder)\n    no_of_audios.append(len(os.listdir(audio_folder)))\n\n# Plot bar chart\nplt.figure(figsize=(28, 8))\nplt.bar(folder_names, no_of_audios)\nplt.xticks(rotation=90)\nplt.xlabel(\"Folder (Bird Class)\")\nplt.ylabel(\"Number of Audio Files\")\nplt.title(\"Number of Audio Files per Subfolder\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T03:47:14.226337Z","iopub.execute_input":"2025-07-06T03:47:14.226880Z","iopub.status.idle":"2025-07-06T03:47:15.750731Z","shell.execute_reply.started":"2025-07-06T03:47:14.226851Z","shell.execute_reply":"2025-07-06T03:47:15.750086Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"min_val = 1e6\nmean = 0\nmax_val = 1e-4\n\nfor i in no_of_audios:\n    min_val = min(min_val,i)\n    max_val = max(max_val,i)\n    mean = mean + i\n\nprint(f\"min val : {min_val}\")\nprint(f\"max val : {max_val}\")\nprint(f\"mean : {mean/len(no_of_audios)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T03:47:15.751595Z","iopub.execute_input":"2025-07-06T03:47:15.752085Z","iopub.status.idle":"2025-07-06T03:47:15.757351Z","shell.execute_reply.started":"2025-07-06T03:47:15.752061Z","shell.execute_reply":"2025-07-06T03:47:15.756568Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CustomClass(Dataset):\n    def __init__(self, input_folder, fixed_waveform):\n        super(CustomClass, self).__init__()\n        self.input_folder = input_folder\n        self.fixed_waveform = fixed_waveform  # should be a torch.Tensor\n        self.audio_path = []\n        self.audio_label = []\n        self.label_to_idx = {}\n        self.idx_to_label = {}\n\n        for i, subfolder in enumerate(os.listdir(self.input_folder)):\n            audio_folder = os.path.join(self.input_folder, subfolder)\n            if not os.path.isdir(audio_folder):\n                continue\n            for audio in os.listdir(audio_folder):\n                full_audio_path = os.path.join(audio_folder, audio)\n                self.audio_path.append(full_audio_path)\n                self.audio_label.append(subfolder)\n            self.label_to_idx[subfolder] = i\n            self.idx_to_label[i] = subfolder\n\n    def __len__(self):\n        return len(self.audio_path)\n\n    def convert_to_mel_spec(self, waveform, sample_rate):\n        transform = transforms.MelSpectrogram(sample_rate)\n        mel_spectrogram = transform(waveform)\n        if mel_spectrogram.shape[0] != 3:\n            mel_spectrogram = mel_spectrogram.repeat(3, 1, 1)\n        return mel_spectrogram\n\n    def __getitem__(self, idx):\n        while True:\n            path = self.audio_path[idx]\n            waveform_np, sample_rate = librosa.load(path, sr=None, mono=True)\n            waveform = torch.from_numpy(waveform_np)\n            if waveform.shape == self.fixed_waveform.shape:\n                if torch.allclose(waveform, self.fixed_waveform, atol=1e-5):\n                    idx = torch.randint(0, len(self.audio_path), (1,)).item()\n                    continue\n            break\n\n        if waveform.shape[-1] < int(sample_rate*5):\n            padding = int(5*sample_rate-waveform.shape[-1])\n            noise = torch.randn(padding)*0.005\n            waveform = torch.cat([waveform,noise],dim=-1)\n        string_label = self.audio_label[idx]\n        encoded_label = self.label_to_idx[string_label]\n        full_label = torch.zeros(206)\n        full_label[encoded_label] = 1\n        mel_spectrogram = self.convert_to_mel_spec(waveform, sample_rate)\n        return mel_spectrogram, full_label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T03:27:28.534968Z","iopub.execute_input":"2025-07-07T03:27:28.535685Z","iopub.status.idle":"2025-07-07T03:27:28.547333Z","shell.execute_reply.started":"2025-07-07T03:27:28.535652Z","shell.execute_reply":"2025-07-07T03:27:28.546686Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"waveform,sr = torchaudio.load(\"/kaggle/input/new-preprocessed/kaggle/working/curr_preprocessed_data/1139490/CSA36385.wav\")\ninput_folder = \"/kaggle/input/new-preprocessed/kaggle/working/curr_preprocessed_data\"\noptimizer = optim.Adam(efficient_model.parameters(),lr = 1e-5);\ncriterion = nn.BCEWithLogitsLoss()\ndataset = CustomClass(input_folder,waveform)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T03:27:35.137024Z","iopub.execute_input":"2025-07-07T03:27:35.137756Z","iopub.status.idle":"2025-07-07T03:27:35.140794Z","shell.execute_reply.started":"2025-07-07T03:27:35.137734Z","shell.execute_reply":"2025-07-07T03:27:35.140138Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nskf = StratifiedKFold(n_splits=5,shuffle=True,random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T03:47:16.941685Z","iopub.execute_input":"2025-07-06T03:47:16.941995Z","iopub.status.idle":"2025-07-06T03:47:17.399339Z","shell.execute_reply.started":"2025-07-06T03:47:16.941977Z","shell.execute_reply":"2025-07-06T03:47:17.398735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_to_idx = dataset.label_to_idx\nidx_to_label = dataset.idx_to_label\nx = dataset.audio_path\ny = [label_to_idx[i] for i in dataset.audio_label]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T03:47:17.400058Z","iopub.execute_input":"2025-07-06T03:47:17.400370Z","iopub.status.idle":"2025-07-06T03:47:17.405075Z","shell.execute_reply.started":"2025-07-06T03:47:17.400353Z","shell.execute_reply":"2025-07-06T03:47:17.404563Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torch.utils.data import DataLoader, Subset\nimport matplotlib.pyplot as plt\n\nEpoch = 10\nall_train_losses = []\nall_valid_losses = []\n\nfor fold, (train_idx, val_idx) in enumerate(skf.split(x, y)):\n    print(f\"Fold: {fold}\")\n    train_dataset = Subset(dataset, train_idx)\n    val_dataset = Subset(dataset, val_idx)\n\n    train_dataloader = DataLoader(train_dataset, batch_size=4, shuffle=True)\n    valid_dataloader = DataLoader(val_dataset, batch_size=4, shuffle=False)\n\n    train_loss_list = []\n    valid_loss_list = []\n\n    best_loss = float('inf')\n    best_model_state = None\n\n    for epoch in range(Epoch):\n        efficient_model.train()\n        train_loss = 0.0\n\n        for i, (inputs, labels) in enumerate(train_dataloader):\n            inputs = inputs.to(device)\n            labels = labels.to(device)\n            \n\n            optimizer.zero_grad()\n            outputs = efficient_model(inputs)\n            loss = criterion(outputs, labels)\n\n            loss.backward()\n            optimizer.step()\n\n            train_loss += loss.item()\n\n        efficient_model.eval()\n        valid_loss = 0.0\n\n        with torch.no_grad():\n            for j, (vinput, vlabel) in enumerate(valid_dataloader):\n                vinput = vinput.to(device)\n                vlabel = vlabel.to(device)\n\n                voutput = efficient_model(vinput)\n                vloss = criterion(voutput, vlabel)\n\n                valid_loss += vloss.item()\n\n        avg_train_loss = train_loss / (i + 1)\n        avg_valid_loss = valid_loss / (j + 1)\n\n        train_loss_list.append(avg_train_loss)\n        valid_loss_list.append(avg_valid_loss)\n\n        print(f\"Epoch {epoch+1} | Train Loss: {avg_train_loss:.4f} | Val Loss: {avg_valid_loss:.4f}\")\n\n        if avg_valid_loss < best_loss:\n            best_loss = avg_valid_loss\n            best_model_state = efficient_model.state_dict()\n\n    # Save best model for this fold\n    torch.save(best_model_state, f\"best_model_fold{fold}.pt\")\n\n    all_train_losses.append(train_loss_list)\n    all_valid_losses.append(valid_loss_list)\n\n    # Plot train and val loss for this fold\n    plt.figure(figsize=(6, 4))\n    plt.plot(train_loss_list, label=\"Train Loss\")\n    plt.plot(valid_loss_list, label=\"Valid Loss\")\n    plt.title(f\"Fold :{fold} | Best_loss : {best_loss}\")\n    plt.xlabel(\"Epoch\")\n    plt.ylabel(\"Loss\")\n    plt.legend()\n    plt.grid(True)\n    plt.tight_layout()\n    plt.savefig(f\"loss_curve_fold{fold}.png\")  # Optional: save to file\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T03:47:17.405921Z","iopub.execute_input":"2025-07-06T03:47:17.406220Z","iopub.status.idle":"2025-07-06T12:24:30.324903Z","shell.execute_reply.started":"2025-07-06T03:47:17.406202Z","shell.execute_reply":"2025-07-06T12:24:30.324046Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## efficient-v4\n","metadata":{}},{"cell_type":"code","source":"from efficientnet_pytorch import EfficientNet\nefficient_b4 = EfficientNet.from_pretrained('efficientnet-b4')\nefficient_b4._fc = nn.Linear(in_features=1792, out_features=206, bias=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T03:27:43.441300Z","iopub.execute_input":"2025-07-07T03:27:43.442281Z","iopub.status.idle":"2025-07-07T03:27:43.797850Z","shell.execute_reply.started":"2025-07-07T03:27:43.442248Z","shell.execute_reply":"2025-07-07T03:27:43.797010Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"efficient_b4.to(device)\nif torch.cuda.device_count() > 1:\n    print(f\"Using {torch.cuda.device_count()} GPUs with DataParallel.\")\n    efficient_b4 = nn.DataParallel(efficient_b4)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T03:27:47.300652Z","iopub.execute_input":"2025-07-07T03:27:47.300921Z","iopub.status.idle":"2025-07-07T03:27:47.447996Z","shell.execute_reply.started":"2025-07-07T03:27:47.300904Z","shell.execute_reply":"2025-07-07T03:27:47.447271Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"waveform,sr = torchaudio.load(\"/kaggle/input/new-preprocessed/kaggle/working/curr_preprocessed_data/1139490/CSA36385.wav\")\ninput_folder = \"/kaggle/input/new-preprocessed/kaggle/working/curr_preprocessed_data\"\noptimizer = optim.Adam(efficient_b4.parameters(),lr = 1e-5);\ncriterion = nn.BCEWithLogitsLoss()\ndataset = CustomClass(input_folder,waveform)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T03:27:53.733019Z","iopub.execute_input":"2025-07-07T03:27:53.733284Z","iopub.status.idle":"2025-07-07T03:27:55.166191Z","shell.execute_reply.started":"2025-07-07T03:27:53.733268Z","shell.execute_reply":"2025-07-07T03:27:55.165596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nskf = StratifiedKFold(n_splits=5,shuffle=True,random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T03:27:56.582932Z","iopub.execute_input":"2025-07-07T03:27:56.583305Z","iopub.status.idle":"2025-07-07T03:27:57.022883Z","shell.execute_reply.started":"2025-07-07T03:27:56.583285Z","shell.execute_reply":"2025-07-07T03:27:57.022320Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_to_idx = dataset.label_to_idx\nidx_to_label = dataset.idx_to_label\nx = dataset.audio_path\ny = [label_to_idx[i] for i in dataset.audio_label]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T03:27:58.497376Z","iopub.execute_input":"2025-07-07T03:27:58.498208Z","iopub.status.idle":"2025-07-07T03:27:58.502953Z","shell.execute_reply.started":"2025-07-07T03:27:58.498181Z","shell.execute_reply":"2025-07-07T03:27:58.502260Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torch.utils.data import DataLoader, Subset\nimport matplotlib.pyplot as plt\n\nEpoch = 10\nall_train_losses = []\nall_valid_losses = []\n\nfor fold, (train_idx, val_idx) in enumerate(skf.split(x, y)):\n    print(f\"Fold: {fold}\")\n    train_dataset = Subset(dataset, train_idx)\n    val_dataset = Subset(dataset, val_idx)\n\n    train_dataloader = DataLoader(train_dataset, batch_size=4,shuffle=True)\n    valid_dataloader = DataLoader(val_dataset, batch_size=4,shuffle=False)\n\n    train_loss_list = []\n    valid_loss_list = []\n\n    best_loss = float('inf')\n    best_model_state = None\n\n    for epoch in range(Epoch):\n        efficient_b4.train()\n        train_loss = 0.0\n\n        for i, (inputs, labels) in enumerate(train_dataloader):\n            inputs = inputs.to(device)\n            labels = labels.to(device)\n            \n\n            optimizer.zero_grad()\n            outputs = efficient_b4(inputs)\n            loss = criterion(outputs, labels)\n\n            loss.backward()\n            optimizer.step()\n\n            train_loss += loss.item()\n\n        efficient_b4.eval()\n        valid_loss = 0.0\n\n        with torch.no_grad():\n            for j, (vinput, vlabel) in enumerate(valid_dataloader):\n                vinput = vinput.to(device)\n                vlabel = vlabel.to(device)\n\n                voutput = efficient_b4(vinput)\n                vloss = criterion(voutput, vlabel)\n\n                valid_loss += vloss.item()\n\n        avg_train_loss = train_loss / (i + 1)\n        avg_valid_loss = valid_loss / (j + 1)\n\n        train_loss_list.append(avg_train_loss)\n        valid_loss_list.append(avg_valid_loss)\n\n        print(f\"Epoch {epoch+1} | Train Loss: {avg_train_loss:.4f} | Val Loss: {avg_valid_loss:.4f}\")\n\n        if avg_valid_loss < best_loss:\n            best_loss = avg_valid_loss\n            best_model_state = efficient_b4.state_dict()\n\n    # Save best model for this fold\n    torch.save(best_model_state, f\"best_model_fold{fold}.pt\")\n\n    all_train_losses.append(train_loss_list)\n    all_valid_losses.append(valid_loss_list)\n\n    # Plot train and val loss for this fold\n    plt.figure(figsize=(6, 4))\n    plt.plot(train_loss_list, label=\"Train Loss\")\n    plt.plot(valid_loss_list, label=\"Valid Loss\")\n    plt.title(f\"Fold :{fold} | Best_loss : {best_loss}\")\n    plt.xlabel(\"Epoch\")\n    plt.ylabel(\"Loss\")\n    plt.legend()\n    plt.grid(True)\n    plt.tight_layout()\n    plt.savefig(f\"loss_curve_fold{fold}.png\")  # Optional: save to file\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T03:28:02.647892Z","iopub.execute_input":"2025-07-07T03:28:02.648205Z","execution_failed":"2025-07-07T10:55:04.181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}