{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import librosa\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport torch\nimport os\nfrom pathlib import Path\nimport shutil\nimport soundfile as sf\nimport torchaudio\nimport random\nimport IPython.display as ipd\nimport warnings\n\n# Ignore all warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-19T16:35:40.376965Z","iopub.execute_input":"2025-05-19T16:35:40.377323Z","iopub.status.idle":"2025-05-19T16:35:43.493543Z","shell.execute_reply.started":"2025-05-19T16:35:40.377302Z","shell.execute_reply":"2025-05-19T16:35:43.492392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class DatasetEnhancer:\n    def __init__(self, input_dir='/kaggle/input/birdclef-2025', output_dir='/kaggle/working/enhanced-dataset',\n                sample_size=1.0, random_seed=42):\n        \"\"\"\n        Initialize the dataset enhancer.\n        \n        Args:\n            input_dir: Directory containing original dataset\n            output_dir: Directory to save enhanced dataset\n            sample_size: Float between 0 and 1, representing the portion of the dataset to use\n            random_seed: Random seed for reproducibility when sampling\n        \"\"\"\n        self.input_dir = input_dir\n        self.output_dir = output_dir\n        self.train_audio_dir = os.path.join(input_dir, 'train_audio')\n        self.output_train_audio_dir = os.path.join(output_dir, 'train_audio')\n        self.chunk_len = 0.1  # Chunk length in seconds\n        self.sample_size = min(max(sample_size, 0.0), 1.0)  # Ensure between 0 and 1\n        self.random_seed = random_seed\n        \n        # Set random seed for reproducibility\n        random.seed(self.random_seed)\n        np.random.seed(self.random_seed)\n        \n        # Sampled dataframes - will be populated when needed\n        self.sampled_train_df = None\n        self.sampled_test_df = None\n        \n        # Initialize VAD model\n        self._init_vad_model()\n        \n        # Create output directories if they don't exist\n        os.makedirs(self.output_dir, exist_ok=True)\n        os.makedirs(self.output_train_audio_dir, exist_ok=True)\n    \n    def _init_vad_model(self):\n        \"\"\"Initialize the Silero Voice Activity Detection model\"\"\"\n        torch.set_num_threads(1)\n        self.model, (self.get_speech_timestamps, _, _, _, _) = torch.hub.load(\n            repo_or_dir='snakers4/silero-vad', \n            model='silero_vad'\n        )\n    \n    def visualize_audio_with_voice_detection(self, audio_path, title=None):\n        \"\"\"\n        Visualize an audio file showing the audio power and voice detection segments.\n        \n        Args:\n            audio_path: Path to the audio file\n            title: Optional title for the plot\n            \n        Returns:\n            Matplotlib figure\n        \"\"\"\n        # Load the audio file\n        wav, sr = librosa.load(audio_path)\n        \n        # Calculate the sound power\n        power = wav ** 2\n        \n        # Split the data into chunks and sum the energy in every chunk\n        chunk = int(self.chunk_len * sr)\n        \n        pad = int(np.ceil(len(power) / chunk) * chunk - len(power))\n        power = np.pad(power, (0, pad))\n        power = power.reshape((-1, chunk)).sum(axis=1)\n        \n        # Detect speech segments\n        speech_timestamps = self.get_speech_timestamps(torch.Tensor(wav), self.model)\n        segmentation = np.zeros_like(wav)\n        for st in speech_timestamps:\n            segmentation[st['start']: st['end']] = 20\n        \n        # Create plot\n        fig = plt.figure(figsize=(24, 3))\n        \n        # Set title\n        if title is None:\n            filename = os.path.basename(audio_path)\n            fig.suptitle(f'Audio: {filename}')\n        else:\n            fig.suptitle(title)\n        \n        # Plot power in blue\n        t = np.arange(len(power)) * self.chunk_len\n        plt.plot(t, 10 * np.log10(power), 'b', label='Audio Power (dB)')\n        \n        # Plot voice segments in red\n        t = np.arange(len(segmentation)) / sr\n        plt.plot(t, segmentation, 'r', label='Voice Detection')\n        \n        plt.xlabel('Time (s)')\n        plt.ylabel('Amplitude (dB) / Voice Detection')\n        plt.legend()\n        \n        # Return the figure (doesn't show it yet)\n        return fig\n    \n    def find_and_visualize_audio_sample(self, dataset_names, author='Fabio A. Sarria-S'):\n        \"\"\"\n        Find an audio sample and visualize it with voice detection.\n        \n        Args:\n            dataset_names: List of datasets to search for samples\n            author: Author name to search for (default: Fabio A. Sarria-S)\n            \n        Returns:\n            Audio display object and matplotlib figure\n        \"\"\"\n        # Find a sample file\n        sample_path = None\n        sample_author = None\n        sample_filename = None\n        \n        for dataset_name in dataset_names:\n            if dataset_name.lower() == 'train':\n                csv_path = os.path.join(self.input_dir, 'train.csv')\n                audio_dir = self.train_audio_dir\n            elif dataset_name.lower() == 'test':\n                csv_path = os.path.join(self.input_dir, 'test.csv')\n                audio_dir = os.path.join(self.input_dir, 'test_audio')\n            else:\n                continue\n                \n            if os.path.exists(csv_path):\n                df = pd.read_csv(csv_path)\n                \n                # Try to find a file by the specified author\n                if 'author' in df.columns:\n                    author_df = df[df.author == author] if author in df.author.values else None\n                    \n                    if author_df is not None and len(author_df) > 0:\n                        sample_row = author_df.iloc[0]\n                        sample_filename = sample_row.filename\n                        sample_author = author\n                        sample_path = os.path.join(audio_dir, sample_filename)\n                        if os.path.exists(sample_path):\n                            break\n                \n                # If no file by the author, use any file\n                if sample_path is None and len(df) > 0:\n                    sample_row = df.iloc[0]\n                    sample_filename = sample_row.filename\n                    sample_author = sample_row.author if 'author' in df.columns else 'Unknown'\n                    sample_path = os.path.join(audio_dir, sample_filename)\n                    if os.path.exists(sample_path):\n                        break\n        \n        if sample_path is None:\n            print(\"No suitable audio sample found.\")\n            return None, None\n        \n        # Visualize the sample\n        title = f'{sample_filename} by {sample_author}'\n        fig = self.visualize_audio_with_voice_detection(sample_path, title)\n        \n        # Create audio player\n        audio = ipd.Audio(sample_path)\n        \n        # Return both the audio player and the figure\n        return audio, fig\n    \n    def _sample_dataframe(self, df, sample_size):\n        \"\"\"\n        Sample a portion of a dataframe, trying to maintain class distribution.\n        \n        Args:\n            df: DataFrame to sample from\n            sample_size: Float between 0 and 1, representing the portion to sample\n            \n        Returns:\n            Sampled DataFrame\n        \"\"\"\n        if sample_size >= 1.0:\n            return df  # Return the entire dataframe\n            \n        # If we have primary_label column, try to maintain class distribution\n        if 'primary_label' in df.columns:\n            sampled_df = df.groupby('primary_label', group_keys=False).apply(\n                lambda x: x.sample(max(1, int(len(x) * sample_size)), random_state=self.random_seed)\n            )\n            \n            # If sampling resulted in an empty dataframe, just sample randomly\n            if len(sampled_df) == 0:\n                sampled_df = df.sample(max(1, int(len(df) * sample_size)), random_state=self.random_seed)\n                \n            return sampled_df\n        else:\n            # Simple random sampling if no class information\n            return df.sample(max(1, int(len(df) * sample_size)), random_state=self.random_seed)\n    \n    def _get_sampled_train_df(self):\n        \"\"\"Get a sampled train dataframe\"\"\"\n        if self.sampled_train_df is not None:\n            return self.sampled_train_df\n            \n        train_csv_path = os.path.join(self.input_dir, 'train.csv')\n        if os.path.exists(train_csv_path):\n            train_df = pd.read_csv(train_csv_path)\n            self.sampled_train_df = self._sample_dataframe(train_df, self.sample_size)\n            return self.sampled_train_df\n        return None\n    \n    def _get_sampled_test_df(self):\n        \"\"\"Get a sampled test dataframe\"\"\"\n        if self.sampled_test_df is not None:\n            return self.sampled_test_df\n            \n        test_csv_path = os.path.join(self.input_dir, 'test.csv')\n        if os.path.exists(test_csv_path):\n            test_df = pd.read_csv(test_csv_path)\n            self.sampled_test_df = self._sample_dataframe(test_df, self.sample_size)\n            return self.sampled_test_df\n        return None\n    \n    def remove_human_voice(self, dataset_names=None, use_previous_output=False):\n        \"\"\"\n        Remove human voice from recordings and create enhanced dataset.\n        This is the first enhancement method.\n        \n        Args:\n            dataset_names: List of strings with dataset names to process (e.g., ['train', 'test']).\n                          If None, all available datasets will be processed.\n            use_previous_output: If True, use the output of the previous enhancement as input.\n        \n        Returns:\n            Path to enhanced dataset and sample file path\n        \"\"\"\n        # Update input directory if using previous output\n        if use_previous_output and os.path.exists(self.output_dir):\n            original_input_dir = self.input_dir\n            self.input_dir = self.output_dir\n            self.train_audio_dir = os.path.join(self.input_dir, 'train_audio')\n            \n            # Create a new output directory for this enhancement\n            self.output_dir = os.path.join(os.path.dirname(self.output_dir), 'voice_removed_dataset')\n            self.output_train_audio_dir = os.path.join(self.output_dir, 'train_audio')\n            \n            # Create output directories\n            os.makedirs(self.output_dir, exist_ok=True)\n            os.makedirs(self.output_train_audio_dir, exist_ok=True)\n        \n        # Determine which datasets to process\n        if dataset_names is None:\n            # Default: process all available datasets\n            dataset_names = []\n            if os.path.exists(os.path.join(self.input_dir, 'train.csv')):\n                dataset_names.append('train')\n            if os.path.exists(os.path.join(self.input_dir, 'test.csv')):\n                dataset_names.append('test')\n        \n        # Process selected datasets\n        for dataset_name in dataset_names:\n            if dataset_name.lower() == 'train':\n                self._process_train_data(process_fabio_files=True)\n            elif dataset_name.lower() == 'test':\n                test_csv_path = os.path.join(self.input_dir, 'test.csv')\n                if os.path.exists(test_csv_path):\n                    self._process_test_data(test_csv_path)\n        \n        if self.sample_size < 1.0:\n            print(f\"Enhanced dataset created at {self.output_dir} for {', '.join(dataset_names)} (using {self.sample_size*100:.1f}% sample)\")\n        else:\n            print(f\"Enhanced dataset created at {self.output_dir} for {', '.join(dataset_names)}\")\n        \n        # Play a sample audio file to verify\n        # First make sure our processed files have been written to disk\n        if os.path.exists(self.output_dir):\n            print(f\"Output directory exists: {self.output_dir}\")\n            \n            # Print directory structure for debugging\n            print(\"\\nDirectory structure of output folder:\")\n            depth = 0\n            for root, dirs, files in os.walk(self.output_dir):\n                depth += 1\n                if depth > 3:  # Limit depth to avoid too much output\n                    continue\n                    \n                level = root.replace(self.output_dir, '').count(os.sep)\n                indent = ' ' * 4 * level\n                print(f\"{indent}{os.path.basename(root) or 'root'}/\")\n                sub_indent = ' ' * 4 * (level + 1)\n                num_files = len(files)\n                if num_files > 0:\n                    print(f\"{sub_indent}{num_files} files (showing up to 5)\")\n                    for i, f in enumerate(files[:5]):\n                        print(f\"{sub_indent}- {f}\")\n            \n            # Check each dataset directory manually\n            for dataset_name in dataset_names:\n                if dataset_name.lower() == 'train':\n                    audio_dir = os.path.join(self.output_dir, 'train_audio')\n                    if os.path.exists(audio_dir):\n                        print(f\"Checking {audio_dir} for samples\")\n                        # Try to find any audio file\n                        for root, dirs, files in os.walk(audio_dir):\n                            for file in files:\n                                if file.lower().endswith(('.wav', '.mp3', '.ogg')):\n                                    sample_path = os.path.join(root, file)\n                                    print(f\"Found audio file in train_audio: {sample_path}\")\n                                    return self.output_dir, sample_path\n                elif dataset_name.lower() == 'test':\n                    audio_dir = os.path.join(self.output_dir, 'test_audio')\n                    if os.path.exists(audio_dir):\n                        print(f\"Checking {audio_dir} for samples\")\n                        # Try to find any audio file\n                        for root, dirs, files in os.walk(audio_dir):\n                            for file in files:\n                                if file.lower().endswith(('.wav', '.mp3', '.ogg')):\n                                    sample_path = os.path.join(root, file)\n                                    print(f\"Found audio file in test_audio: {sample_path}\")\n                                    return self.output_dir, sample_path\n            \n            # Try to find a processed sample file\n            sample_path = self._get_sample_audio_path(dataset_names)\n            if sample_path:\n                print(f\"Found sample file from the voice-removed dataset: {sample_path}\")\n                return self.output_dir, sample_path\n            else:\n                print(\"Could not find a sample file. Searching for any audio file...\")\n                # Last resort: search for any audio file in the output directory\n                for root, dirs, files in os.walk(self.output_dir):\n                    for file in files:\n                        if file.lower().endswith(('.wav', '.mp3', '.ogg')):\n                            sample_path = os.path.join(root, file)\n                            print(f\"Found audio file via last resort search: {sample_path}\")\n                            return self.output_dir, sample_path\n        \n        print(\"WARNING: No sample audio file found.\")\n        return self.output_dir, None\n    \n    def create_fixed_duration_clips(self, duration_seconds=5, selection_method='random', \n                                   sample_rate=32000, dataset_names=None, use_previous_output=False):\n        \"\"\"\n        Create a dataset with fixed-duration audio clips from each recording.\n        \n        Args:\n            duration_seconds: Length of each audio clip in seconds\n            selection_method: How to select the segment ('start', 'end', 'random')\n            sample_rate: Sample rate for the output audio files\n            dataset_names: List of strings with dataset names to process (e.g., ['train', 'test']).\n                          If None, all available datasets will be processed.\n            use_previous_output: If True, use the output of the previous enhancement as input.\n            \n        Returns:\n            Path to enhanced dataset and sample file path\n        \"\"\"\n        # Update input directory if using previous output\n        if use_previous_output and os.path.exists(self.output_dir):\n            original_input_dir = self.input_dir\n            self.input_dir = self.output_dir\n            self.train_audio_dir = os.path.join(self.input_dir, 'train_audio')\n        \n        # Set output directory specific to this enhancement\n        clip_output_dir = os.path.join(os.path.dirname(self.output_dir), f'fixed_duration_{duration_seconds}sec')\n        \n        # Determine which datasets to process\n        if dataset_names is None:\n            # Default: process all available datasets\n            dataset_names = []\n            if os.path.exists(os.path.join(self.input_dir, 'train.csv')):\n                dataset_names.append('train')\n            if os.path.exists(os.path.join(self.input_dir, 'test.csv')):\n                dataset_names.append('test')\n                \n        # Calculate minimum segment length in samples\n        min_segment_samples = int(duration_seconds * sample_rate)\n        \n        # Process selected datasets\n        for dataset_name in dataset_names:\n            if dataset_name.lower() == 'train':\n                # Process training data\n                clip_train_audio_dir = os.path.join(clip_output_dir, 'train_audio')\n                os.makedirs(clip_train_audio_dir, exist_ok=True)\n                \n                train_csv_path = os.path.join(self.input_dir, 'train.csv')\n                if os.path.exists(train_csv_path):\n                    self._process_fixed_duration_train(\n                        train_csv_path, \n                        clip_output_dir,\n                        clip_train_audio_dir,\n                        duration_seconds, \n                        selection_method, \n                        sample_rate, \n                        min_segment_samples\n                    )\n            \n            elif dataset_name.lower() == 'test':\n                # Process test data\n                test_csv_path = os.path.join(self.input_dir, 'test.csv')\n                if os.path.exists(test_csv_path):\n                    clip_test_audio_dir = os.path.join(clip_output_dir, 'test_audio')\n                    os.makedirs(clip_test_audio_dir, exist_ok=True)\n                    \n                    self._process_fixed_duration_test(\n                        test_csv_path, \n                        clip_output_dir,\n                        clip_test_audio_dir,\n                        duration_seconds, \n                        selection_method, \n                        sample_rate, \n                        min_segment_samples\n                    )\n        \n        if self.sample_size < 1.0:\n            print(f\"Fixed-duration ({duration_seconds}s) dataset created at {clip_output_dir} for {', '.join(dataset_names)} (using {self.sample_size*100:.1f}% sample)\")\n        else:\n            print(f\"Fixed-duration ({duration_seconds}s) dataset created at {clip_output_dir} for {', '.join(dataset_names)}\")\n        \n        # Update object state to use the new output directory for further enhancements\n        self.output_dir = clip_output_dir\n        self.output_train_audio_dir = os.path.join(clip_output_dir, 'train_audio')\n        \n        # Play a sample audio file to verify\n        sample_path = self._get_sample_audio_path(dataset_names, clip_output_dir)\n        if sample_path:\n            print(f\"Playing a sample file from the fixed-duration dataset: {sample_path}\")\n            return clip_output_dir, sample_path\n            \n        return clip_output_dir, None\n    \n    def _get_sample_audio_path(self, dataset_names, output_dir=None):\n        \"\"\"Get a sample audio file path to play for verification\"\"\"\n        if output_dir is None:\n            output_dir = self.output_dir\n            \n        print(f\"Looking for sample files in: {output_dir}\")\n        \n        # First try a direct search for any audio file in the output directory\n        for root, dirs, files in os.walk(output_dir):\n            for file in files:\n                if file.endswith(('.wav', '.mp3', '.ogg')):\n                    sample_path = os.path.join(root, file)\n                    print(f\"Found sample file via direct search: {sample_path}\")\n                    return sample_path\n        \n        # If that didn't work, try to find a sample via CSV\n        for dataset_name in dataset_names:\n            if dataset_name.lower() == 'train':\n                audio_dir = os.path.join(output_dir, 'train_audio')\n                csv_path = os.path.join(output_dir, 'train.csv')\n            elif dataset_name.lower() == 'test':\n                audio_dir = os.path.join(output_dir, 'test_audio')\n                csv_path = os.path.join(output_dir, 'test.csv')\n            else:\n                continue\n                \n            print(f\"Checking {dataset_name} dataset - CSV: {csv_path}, Audio dir: {audio_dir}\")\n                \n            # Check if the CSV exists\n            if os.path.exists(csv_path):\n                try:\n                    df = pd.read_csv(csv_path)\n                    \n                    # First try Fabio's files (since they had human voice)\n                    if 'author' in df.columns:\n                        fabio_df = df[df.author == 'Fabio A. Sarria-S']\n                        \n                        if len(fabio_df) > 0:\n                            # Try first 5 files from Fabio\n                            for idx, row in fabio_df.head(5).iterrows():\n                                sample_filename = row.filename\n                                sample_path = os.path.join(audio_dir, sample_filename)\n                                if os.path.exists(sample_path):\n                                    print(f\"Found Fabio's sample file: {sample_path}\")\n                                    return sample_path\n                    \n                    # If no Fabio files or author column doesn't exist, try any file\n                    if len(df) > 0:\n                        # Try first 10 files from the dataset\n                        for idx, row in df.head(10).iterrows():\n                            sample_filename = row.filename\n                            sample_path = os.path.join(audio_dir, sample_filename)\n                            if os.path.exists(sample_path):\n                                print(f\"Found sample file: {sample_path}\")\n                                return sample_path\n                except Exception as e:\n                    print(f\"Error reading CSV {csv_path}: {str(e)}\")\n                        \n            # If CSV approach didn't work, try directory listing\n            if os.path.exists(audio_dir):\n                print(f\"Trying direct directory listing of {audio_dir}\")\n                # List all subdirectories\n                subdirs = [d for d in os.listdir(audio_dir) if os.path.isdir(os.path.join(audio_dir, d))]\n                \n                if subdirs:\n                    # Get first subdir\n                    subdir = subdirs[0]\n                    subdir_path = os.path.join(audio_dir, subdir)\n                    # List files in that subdir\n                    try:\n                        files = os.listdir(subdir_path)\n                        for file in files:\n                            if file.endswith(('.wav', '.mp3', '.ogg')):\n                                sample_path = os.path.join(subdir_path, file)\n                                print(f\"Found sample via directory listing: {sample_path}\")\n                                return sample_path\n                    except Exception as e:\n                        print(f\"Error listing directory {subdir_path}: {str(e)}\")\n                \n        print(\"No suitable audio sample found after exhaustive search.\")\n        return None\n    \n    def play_audio_sample(self, audio_path):\n        \"\"\"Play an audio sample using IPython.display\"\"\"\n        try:\n            return ipd.Audio(audio_path)\n        except Exception as e:\n            print(f\"Error playing audio file: {str(e)}\")\n            return None\n    \n    def _process_fixed_duration_train(self, train_csv_path, output_dir, output_audio_dir, \n                                     duration_seconds, selection_method, sample_rate, min_segment_samples):\n        \"\"\"\n        Process training data to create fixed-duration clips.\n        \n        Args:\n            train_csv_path: Path to training CSV\n            output_dir: Base output directory\n            output_audio_dir: Directory to save processed audio\n            duration_seconds: Length of each clip in seconds\n            selection_method: How to select the segment ('start', 'end', 'random')\n            sample_rate: Sample rate for output audio\n            min_segment_samples: Minimum segment length in samples\n        \"\"\"\n        # Load training data - use sampled version if sample_size < 1\n        if self.sample_size < 1.0:\n            train_df = self._get_sampled_train_df()\n            if train_df is None:\n                train_df = pd.read_csv(train_csv_path)\n        else:\n            train_df = pd.read_csv(train_csv_path)\n        \n        # Copy the CSV file (full CSV, not just the sample)\n        full_train_df = pd.read_csv(train_csv_path)\n        full_train_df.to_csv(os.path.join(output_dir, 'train.csv'), index=False)\n        \n        print(f\"Processing {len(train_df)} training files to create {duration_seconds}s clips ({selection_method})\")\n        \n        # Process each file\n        for idx, row in train_df.iterrows():\n            input_path = os.path.join(self.train_audio_dir, row.filename)\n            output_path = os.path.join(output_audio_dir, row.filename)\n            \n            # Create output directory if it doesn't exist\n            os.makedirs(os.path.dirname(output_path), exist_ok=True)\n            \n            # Create fixed-duration clip\n            self._create_fixed_duration_clip(\n                input_path, \n                output_path, \n                selection_method, \n                sample_rate, \n                min_segment_samples\n            )\n    \n    def _process_fixed_duration_test(self, test_csv_path, output_dir, output_audio_dir, \n                                   duration_seconds, selection_method, sample_rate, min_segment_samples):\n        \"\"\"\n        Process test data to create fixed-duration clips.\n        \n        Args:\n            test_csv_path: Path to test CSV\n            output_dir: Base output directory\n            output_audio_dir: Directory to save processed audio\n            duration_seconds: Length of each clip in seconds\n            selection_method: How to select the segment ('start', 'end', 'random')\n            sample_rate: Sample rate for output audio\n            min_segment_samples: Minimum segment length in samples\n        \"\"\"\n        # Load test data - use sampled version if sample_size < 1\n        if self.sample_size < 1.0:\n            test_df = self._get_sampled_test_df()\n            if test_df is None:\n                test_df = pd.read_csv(test_csv_path)\n        else:\n            test_df = pd.read_csv(test_csv_path)\n        \n        # Copy the CSV file (full CSV, not just the sample)\n        full_test_df = pd.read_csv(test_csv_path)\n        full_test_df.to_csv(os.path.join(output_dir, 'test.csv'), index=False)\n        \n        print(f\"Processing {len(test_df)} test files to create {duration_seconds}s clips ({selection_method})\")\n        \n        # Process each file\n        input_test_audio_dir = os.path.join(self.input_dir, 'test_audio')\n        for idx, row in test_df.iterrows():\n            input_path = os.path.join(input_test_audio_dir, row.filename)\n            output_path = os.path.join(output_audio_dir, row.filename)\n            \n            # Create output directory if it doesn't exist\n            os.makedirs(os.path.dirname(output_path), exist_ok=True)\n            \n            # Create fixed-duration clip\n            self._create_fixed_duration_clip(\n                input_path, \n                output_path, \n                selection_method, \n                sample_rate, \n                min_segment_samples\n            )\n    \n    def _create_fixed_duration_clip(self, input_path, output_path, selection_method, sample_rate, min_segment_samples):\n        \"\"\"\n        Create a fixed-duration clip from an audio file.\n        \n        Args:\n            input_path: Path to input audio file\n            output_path: Path to save output audio file\n            selection_method: How to select the segment ('start', 'end', 'random')\n            sample_rate: Sample rate for output audio\n            min_segment_samples: Minimum segment length in samples\n        \"\"\"\n        # Check if output file is OGG format\n        is_ogg = output_path.lower().endswith('.ogg')\n        \n        try:\n            # For OGG files, use librosa+soundfile approach directly to avoid torchaudio codec issues\n            if is_ogg:\n                wav, sr = librosa.load(input_path, sr=sample_rate)\n                \n                # Select segment based on method\n                if len(wav) <= min_segment_samples:\n                    # If audio is shorter than required duration, pad with zeros\n                    wav = np.pad(wav, (0, min_segment_samples - len(wav)))\n                    segment = wav[:min_segment_samples]\n                else:\n                    if selection_method == 'start':\n                        # Take from the beginning\n                        segment = wav[:min_segment_samples]\n                    elif selection_method == 'end':\n                        # Take from the end\n                        segment = wav[-min_segment_samples:]\n                    elif selection_method == 'random':\n                        # Take a random segment\n                        max_start = len(wav) - min_segment_samples\n                        start_idx = random.randint(0, max_start)\n                        segment = wav[start_idx:start_idx + min_segment_samples]\n                    else:\n                        # Default to beginning\n                        segment = wav[:min_segment_samples]\n                \n                # Save using soundfile\n                sf.write(output_path, segment, sample_rate)\n                return\n            \n            # For non-OGG files, use the torchaudio approach\n            sig, orig_sr = torchaudio.load(input_path)\n            \n            # Resample if necessary\n            if orig_sr != sample_rate:\n                resampler = torchaudio.transforms.Resample(orig_sr, sample_rate)\n                sig = resampler(sig)\n            \n            # Get audio length in samples\n            audio_length = sig.shape[1]\n            \n            # If audio is shorter than required duration, pad with zeros\n            if audio_length <= min_segment_samples:\n                sig = torch.cat([sig, torch.zeros(1, min_segment_samples - audio_length)], dim=1)\n                segment = sig[:, :min_segment_samples]\n            else:\n                # Select segment based on method\n                if selection_method == 'start':\n                    # Take from the beginning\n                    segment = sig[:, :min_segment_samples]\n                elif selection_method == 'end':\n                    # Take from the end\n                    segment = sig[:, -min_segment_samples:]\n                elif selection_method == 'random':\n                    # Take a random segment\n                    max_start = audio_length - min_segment_samples\n                    start_idx = random.randint(0, max_start)\n                    segment = sig[:, start_idx:start_idx + min_segment_samples]\n                else:\n                    # Default to beginning\n                    segment = sig[:, :min_segment_samples]\n            \n            # Save the segment\n            torchaudio.save(output_path, segment, sample_rate)\n            \n        except Exception as e:\n            print(f\"Error processing {input_path}: {str(e)}\")\n            # If there's an error, try using librosa as fallback\n            try:\n                wav, sr = librosa.load(input_path, sr=sample_rate)\n                wav = wav[:min_segment_samples]  # Take first segment as fallback\n                if len(wav) < min_segment_samples:\n                    wav = np.pad(wav, (0, min_segment_samples - len(wav)))\n                sf.write(output_path, wav, sample_rate)\n            except Exception as e2:\n                print(f\"Failed to process {input_path} with fallback method: {str(e2)}\")\n                # If we can't process it at all, create a silent audio file\n                try:\n                    silence = np.zeros(min_segment_samples)\n                    sf.write(output_path, silence, sample_rate)\n                    print(f\"Created silent audio for {output_path}\")\n                except:\n                    print(f\"Failed to create even silent audio for {output_path}\")\n    \n    def _process_audio_file(self, input_path, output_path):\n        \"\"\"\n        Process an audio file to remove human voice segments.\n        \n        Args:\n            input_path: Path to input audio file\n            output_path: Path to save processed audio file\n        \"\"\"\n        # Load the audio file\n        wav, sr = librosa.load(input_path)\n        \n        # Detect speech segments using Silero VAD\n        speech_timestamps = self.get_speech_timestamps(torch.Tensor(wav), self.model)\n        \n        if not speech_timestamps:\n            # No speech detected, just copy the file\n            sf.write(output_path, wav, sr)\n            return\n        \n        # Create an array marking sections to keep (non-speech segments)\n        keep_mask = np.ones_like(wav, dtype=bool)\n        \n        # Mark speech segments (with small buffers around them) for removal\n        buffer_samples = int(0.2 * sr)  # 200ms buffer\n        for ts in speech_timestamps:\n            start = max(0, ts['start'] - buffer_samples)\n            end = min(len(wav), ts['end'] + buffer_samples)\n            keep_mask[start:end] = False\n        \n        # Keep only non-speech segments\n        cleaned_audio = wav[keep_mask]\n        \n        # If we removed everything or almost everything, keep a portion of the original\n        if len(cleaned_audio) < 0.1 * len(wav):\n            # Save the first and last third of the audio without speech segments\n            third = len(wav) // 3\n            keep_sections = np.concatenate([wav[:third], wav[-third:]])\n            sf.write(output_path, keep_sections, sr)\n        else:\n            # Save the cleaned audio\n            sf.write(output_path, cleaned_audio, sr)\n    \n    def _process_train_data(self, process_fabio_files=True):\n        \"\"\"\n        Process train dataset - copy CSV and audio files.\n        For Fabio's recordings, remove human voice if process_fabio_files is True.\n        \n        Args:\n            process_fabio_files: Whether to process Fabio's files to remove human voice\n        \"\"\"\n        # Load train CSV - use sampled version if sample_size < 1\n        train_csv_path = os.path.join(self.input_dir, 'train.csv')\n        \n        if self.sample_size < 1.0:\n            train_df = self._get_sampled_train_df()\n            if train_df is None:\n                train_df = pd.read_csv(train_csv_path)\n                \n            # Copy the full CSV file (not just the sampled portion)\n            full_train_df = pd.read_csv(train_csv_path)\n            full_train_df.to_csv(os.path.join(self.output_dir, 'train.csv'), index=False)\n        else:\n            train_df = pd.read_csv(train_csv_path)\n            # Copy the original CSV\n            train_df.to_csv(os.path.join(self.output_dir, 'train.csv'), index=False)\n        \n        # Process files by author\n        fabio_df = train_df[train_df.author == 'Fabio A. Sarria-S'].copy() if 'author' in train_df.columns else None\n        \n        # Process Fabio's recordings - remove human voice\n        if fabio_df is not None and process_fabio_files and len(fabio_df) > 0:\n            print(f'Processing {len(fabio_df)} recordings by Fabio A. Sarria-S to remove human voice')\n            for idx, rec in fabio_df.iterrows():\n                filename = rec.filename\n                input_path = os.path.join(self.train_audio_dir, filename)\n                output_path = os.path.join(self.output_train_audio_dir, filename)\n                \n                # Create output directory if it doesn't exist\n                os.makedirs(os.path.dirname(output_path), exist_ok=True)\n                \n                # Process audio to remove human voice\n                self._process_audio_file(input_path, output_path)\n        \n        # Copy other authors' files or all files if author info not available\n        if fabio_df is not None:\n            non_fabio_df = train_df[train_df.author != 'Fabio A. Sarria-S'].copy()\n            print(f'Copying {len(non_fabio_df)} recordings by other authors')\n            for idx, rec in non_fabio_df.iterrows():\n                filename = rec.filename\n                input_path = os.path.join(self.train_audio_dir, filename)\n                output_path = os.path.join(self.output_train_audio_dir, filename)\n                \n                # Create output directory if it doesn't exist\n                os.makedirs(os.path.dirname(output_path), exist_ok=True)\n                \n                # Just copy the file if it's not by Fabio or if we're not processing Fabio's files\n                if not os.path.exists(output_path):\n                    shutil.copy2(input_path, output_path)\n        else:\n            # Process all files if we don't have author information\n            print(f'Processing all {len(train_df)} recordings (author info not available)')\n            for idx, rec in train_df.iterrows():\n                filename = rec.filename\n                input_path = os.path.join(self.train_audio_dir, filename)\n                output_path = os.path.join(self.output_train_audio_dir, filename)\n                \n                # Create output directory if it doesn't exist\n                os.makedirs(os.path.dirname(output_path), exist_ok=True)\n                \n                # Process all files to remove human voice\n                self._process_audio_file(input_path, output_path)\n        \n        return os.path.join(self.output_dir, 'train.csv')\n    \n    def _process_test_data(self, test_csv_path):\n        \"\"\"\n        Process test dataset if it exists\n        \n        Args:\n            test_csv_path: Path to test CSV file\n        \n        Returns:\n            Path to processed test CSV\n        \"\"\"\n        # Load test CSV - use sampled version if sample_size < 1\n        if self.sample_size < 1.0:\n            test_df = self._get_sampled_test_df()\n            if test_df is None:\n                test_df = pd.read_csv(test_csv_path)\n                \n            # Copy the full CSV file (not just the sampled portion)\n            full_test_df = pd.read_csv(test_csv_path)\n            full_test_df.to_csv(os.path.join(self.output_dir, 'test.csv'), index=False)\n        else:\n            test_df = pd.read_csv(test_csv_path)\n            # Copy the original CSV\n            test_df.to_csv(os.path.join(self.output_dir, 'test.csv'), index=False)\n        \n        # Create test audio directory\n        output_test_audio_dir = os.path.join(self.output_dir, 'test_audio')\n        os.makedirs(output_test_audio_dir, exist_ok=True)\n        \n        # Copy test audio files\n        input_test_audio_dir = os.path.join(self.input_dir, 'test_audio')\n        if os.path.exists(input_test_audio_dir):\n            # Process all test files to remove human voice\n            print(f'Processing {len(test_df)} test recordings to remove human voice')\n            for idx, row in test_df.iterrows():\n                filename = row.filename\n                input_path = os.path.join(input_test_audio_dir, filename)\n                output_path = os.path.join(output_test_audio_dir, filename)\n                \n                # Create output directory if it doesn't exist\n                os.makedirs(os.path.dirname(output_path), exist_ok=True)\n                \n                # Process audio to remove human voice\n                self._process_audio_file(input_path, output_path)\n                \n        return os.path.join(self.output_dir, 'test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T16:35:43.622316Z","iopub.execute_input":"2025-05-19T16:35:43.622720Z","iopub.status.idle":"2025-05-19T16:35:43.698337Z","shell.execute_reply.started":"2025-05-19T16:35:43.622684Z","shell.execute_reply":"2025-05-19T16:35:43.697449Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Example usage - Pipeline that follows the steps\n# if __name__ == \"__main__\":\n#     # Step 1: Initialize the enhancer with input and output directories\n#     # and use only 1% of the data for processing\n#     enhancer = DatasetEnhancer(\n#         input_dir='/kaggle/input/birdclef-2025',\n#         output_dir='/kaggle/working/enhanced-dataset',\n#         sample_size=0.01  # Use only 1% of the data\n#     )\n#     \n#     # Visualize a sample before processing to see voice detection\n#     audio, fig = enhancer.find_and_visualize_audio_sample(['train'])\n#     if fig:\n#         display(fig)\n#     if audio:\n#         display(audio)\n#     \n#     # Step 2: Remove human voices from both train and test datasets\n#     voice_removed_dir, voice_removed_sample = enhancer.remove_human_voice(\n#         dataset_names=['train', 'test']\n#     )\n#     \n#     # Play a sample after removing human voice\n#     if voice_removed_sample:\n#         print(\"Playing a sample audio after removing human voice:\")\n#         display(enhancer.play_audio_sample(voice_removed_sample))\n#         \n#         # Visualize the voice-removed sample\n#         fig = enhancer.visualize_audio_with_voice_detection(\n#             voice_removed_sample, \n#             \"Voice-removed audio sample\"\n#         )\n#         display(fig)\n#     \n#     # Step 3: Create 5-second clips from the human-voice-removed dataset\n#     clips_dir, clip_sample = enhancer.create_fixed_duration_clips(\n#         duration_seconds=5,\n#         selection_method='start',\n#         sample_rate=32000,\n#         dataset_names=['train', 'test'],\n#         use_previous_output=True  # Use the output from the voice removal step\n#     )\n#     \n#     # Play a sample after creating 5-second clips\n#     if clip_sample:\n#         print(\"Playing a sample audio after creating 5-second clips:\")\n#         display(enhancer.play_audio_sample(clip_sample))\n#         \n#         # Visualize the 5-second clip\n#         fig = enhancer.visualize_audio_with_voice_detection(\n#             clip_sample,\n#             \"5-second clip sample\"\n#         )\n#         display(fig)\n#         \n#     print(\"Enhancement pipeline completed successfully!\")\n#     print(f\"Voice-removed dataset: {voice_removed_dir}\")\n#     print(f\"Fixed-duration clips dataset: {clips_dir}\")\n#     print(f\"Only processed {enhancer.sample_size*100:.1f}% of the data for demonstration purposes.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 1: Initialize the enhancer with input and output directories\n# and use only 1% of the data for processing\nenhancer = DatasetEnhancer(\n    input_dir='/kaggle/input/birdclef-2025',\n    output_dir='/kaggle/working/enhanced-dataset',\n    sample_size=1  # Use only 100% of the data\n)\n\n# Visualize a sample before processing to see voice detection\naudio, fig = enhancer.find_and_visualize_audio_sample(['train'])\nif fig:\n    display(fig)\nif audio:\n    display(audio)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T16:35:49.959293Z","iopub.execute_input":"2025-05-19T16:35:49.959665Z","iopub.status.idle":"2025-05-19T16:36:15.232187Z","shell.execute_reply.started":"2025-05-19T16:35:49.959637Z","shell.execute_reply":"2025-05-19T16:36:15.231071Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 2: Remove human voices from both train and test datasets\nvoice_removed_dir, voice_removed_sample = enhancer.remove_human_voice(\n    dataset_names=['train', 'test']\n)\n\n# Play a sample after removing human voice\nif voice_removed_sample:\n    print(\"Playing a sample audio after removing human voice:\")\n    display(enhancer.play_audio_sample(voice_removed_sample))\n    \n    # Visualize the voice-removed sample\n    fig = enhancer.visualize_audio_with_voice_detection(\n        voice_removed_sample, \n        \"Voice-removed audio sample\"\n    )\n    display(fig)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T16:36:15.233743Z","iopub.execute_input":"2025-05-19T16:36:15.234320Z","iopub.status.idle":"2025-05-19T16:45:09.943449Z","shell.execute_reply.started":"2025-05-19T16:36:15.234281Z","shell.execute_reply":"2025-05-19T16:45:09.941840Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 3: Create 5-second clips from the human-voice-removed dataset\nclips_dir, clip_sample = enhancer.create_fixed_duration_clips(\n    duration_seconds=5,\n    selection_method='start',\n    sample_rate=32000,\n    dataset_names=['train', 'test'],\n    use_previous_output=True  # Use the output from the voice removal step\n)\n\n# Play a sample after creating 5-second clips\nif clip_sample:\n    print(\"Playing a sample audio after creating 5-second clips:\")\n    display(enhancer.play_audio_sample(clip_sample))\n    \n    # Visualize the 5-second clip\n    fig = enhancer.visualize_audio_with_voice_detection(\n        clip_sample,\n        \"5-second clip sample\"\n    )\n    display(fig)\n    \nprint(\"Enhancement pipeline completed successfully!\")\nprint(f\"Voice-removed dataset: {voice_removed_dir}\")\nprint(f\"Fixed-duration clips dataset: {clips_dir}\")\nprint(f\"Only processed {enhancer.sample_size*100:.1f}% of the data for demonstration purposes.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T16:45:09.946183Z","iopub.execute_input":"2025-05-19T16:45:09.946642Z","iopub.status.idle":"2025-05-19T17:26:38.041520Z","shell.execute_reply.started":"2025-05-19T16:45:09.946568Z","shell.execute_reply":"2025-05-19T17:26:38.039651Z"}},"outputs":[],"execution_count":null}]}