{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This Python 3 script scans audio recordings across multiple directories to detect the presence of human speech within individual files and segments. \n\nIn some recordings, speech appears as narration, typically near the end, while in others it may be faint and masked by ambient noise. \nIdentifying these segments can be valuable for further analysis.\nClassification relies on the Audio Spectrogram Transformer (AST) model, pre-trained on the AudioSet database. \nThe script systematically traverses designated training folders, resampling each recording to 16 kHz (as required by the AST model). \nEach file is then segmented into 10-second chunks with a 50% overlap, yielding approximate temporal boundaries for detected speech.\nBecause speech detection is essentially a binary task, there is an inherent trade-off between missed detections and false alarms. \nTo mitigate false alarms, the detector only flags speech when it is the top-ranked class, and its probability exceeds 0.5. \n\nLowering the probability threshold can reveal more speech segments, but many will be of lower quality or contain various vocalizations. We recommend retaining these lower-quality segments rather than discarding them.\n\nThe code is slow but accurate.\n\nAn optional debug mode allows users to view the transformer's top five predictions for each audio chunk, providing insight into the model’s classification process.\n\nDid you Know that:\n1. There are many recordings that include speech in the training set? Some with singing! \n2. There are recordings both in the training set and in the train_soundscapes with sound of chainsaw(?)\n","metadata":{}},{"cell_type":"code","source":"import os\nimport torch\n#import numpy as np\nimport librosa\nfrom transformers import ASTFeatureExtractor, ASTForAudioClassification\n\nspeech_class = [\n    \"Speech\",\n    \"Female speech, woman speaking\",\n    \"Hubbub, speech noise, speech babble\",\n    \"Male speech, man speaking\",\n    \"Narration, monologue\",\n    \"Child speech, kid speaking\",\n    \"Children playing\",\n    \"Children shouting\",\n    \"Cheering\",\n    \"Crowd\",\n    \"Female singing\",\n    \"Male singing\",\n    \"Singing\"\n]\n\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\ndef get_top_k_predictions(probs, id2label, k=3):\n    topk = torch.topk(probs, k)\n    labels = [id2label[idx.item()] for idx in topk.indices]\n    probabilities = topk.values.tolist()\n    return list(zip(labels, probabilities))\n\ndef classify_audio_chunk(chunk, feature_extractor, model):\n    inputs = feature_extractor(chunk, sampling_rate=16000, return_tensors=\"pt\").to(device)\n    with torch.no_grad():\n        outputs = model(**inputs)\n        logits = outputs.logits\n        probabilities = torch.softmax(logits, dim=-1)\n    return probabilities\n\ndef format_time(seconds):\n    h = int(seconds // 3600)\n    m = int((seconds % 3600) // 60)\n    s = int(seconds % 60)\n    return f\"{h:02d}:{m:02d}:{s:02d}\"\n\ndef process_ogg_files(\n    root_folder,\n    model_name=\"MIT/ast-finetuned-audioset-10-10-0.4593\",\n    chunk_duration_s=10.0,\n    overlap_fraction=0.5,\n    specific_folder=None,\n    debug=False\n):\n    \"\"\"\n    Recursively search for .ogg files under 'root_folder' (optionally only in\n    'specific_folder'). Resample to 16kHz mono, run chunk-based AST classification,\n    and print the detected speech segments with total speech durations.\n    \n    If debug=True, prints per-chunk classification results (top-k).\n    Any file with total_speech_seconds == 0 is not printed at all.\n\n    Additionally: \n    - The final chunk is skipped if it is shorter than 5 seconds.\n    \"\"\"\n\n    feature_extractor = ASTFeatureExtractor.from_pretrained(model_name)\n    model = ASTForAudioClassification.from_pretrained(model_name).to(device)\n    id2label = model.config.id2label\n    \n    sample_rate = 16000\n    chunk_size = int(chunk_duration_s * sample_rate)  # e.g. 10s => 160000 samples\n    overlap = int(overlap_fraction * chunk_size)      # e.g. 50% => 80000\n    step_size = chunk_size - overlap                  # e.g. 80000\n\n    for dirpath, dirnames, filenames in os.walk(root_folder):\n\n        if specific_folder is not None:\n            if os.path.basename(dirpath) != specific_folder:\n                continue\n\n        for filename in filenames:\n            if filename.lower().endswith(\".ogg\"):\n                file_path = os.path.join(dirpath, filename)\n\n                audio_data, sr = librosa.load(file_path, sr=None, mono=True)\n                if sr != sample_rate:\n                    audio_data = librosa.resample(audio_data, orig_sr=sr, target_sr=sample_rate)\n                    sr = sample_rate\n                \n                n_samples = len(audio_data)\n\n                segments = []\n                in_speech = False\n                current_seg_start_s = 0.0\n\n                chunk_start = 0\n                chunk_index = 1\n\n                while chunk_start < n_samples:\n                    chunk_end = min(chunk_start + chunk_size, n_samples)\n                    if chunk_end <= chunk_start:\n                        break\n\n                    chunk_duration_s = (chunk_end - chunk_start) / sample_rate\n                    # If the final chunk is shorter than 5s, skip it\n                    if chunk_duration_s < 5:\n                        chunk_start += step_size\n                        chunk_index += 1\n                        continue\n\n                    chunk = audio_data[chunk_start:chunk_end]\n                    chunk_start_s = chunk_start / sample_rate\n                    chunk_end_s   = chunk_end   / sample_rate\n\n                    # Classify chunk\n                    probabilities = classify_audio_chunk(chunk, feature_extractor, model)\n                    top_k = get_top_k_predictions(probabilities[0], id2label, k=1) # We are conservative so that we avoid false alarms\n\n                    # Debug output\n                    if debug:\n                        print(f\"DEBUG: {filename} | Chunk {chunk_index} | Start={format_time(chunk_start_s)} - {format_time(chunk_end_s)}\")\n                        for label, prob in top_k:\n                            print(f\"  {label}: {prob:.4f}\")\n                        print()\n\n                    # Check if chunk has speech\n                    chunk_has_speech = any(label in speech_class for label, _ in top_k)\n\n                    if chunk_has_speech:\n                        if not in_speech:\n                            in_speech = True\n                            current_seg_start_s = chunk_start_s\n                    else:\n                        if in_speech:\n                            # Close segment at the start of this chunk\n                            if chunk_start_s > current_seg_start_s:\n                                segments.append((current_seg_start_s, chunk_start_s))\n                            in_speech = False\n\n                    chunk_start += step_size\n                    chunk_index += 1\n\n                # If we ended while in speech, close at the file end\n                if in_speech:\n                    file_end_s = n_samples / sample_rate\n                    if file_end_s > current_seg_start_s:\n                        segments.append((current_seg_start_s, file_end_s))\n                    in_speech = False\n\n                # Filter out zero-length segments\n                segments = [(start, end) for (start, end) in segments if end > start]\n\n                # Merge overlapping or consecutive segments\n                merged_segments = []\n                for seg in segments:\n                    if not merged_segments:\n                        merged_segments.append(seg)\n                    else:\n                        prev_start, prev_end = merged_segments[-1]\n                        curr_start, curr_end = seg\n                        if curr_start <= prev_end:\n                            merged_segments[-1] = (prev_start, max(prev_end, curr_end))\n                        else:\n                            merged_segments.append(seg)\n\n                total_speech_seconds = sum(end - start for (start, end) in merged_segments)\n\n                # Print only if total_speech_seconds > 0\n                if total_speech_seconds > 0:\n                    folder_name = os.path.basename(dirpath)\n                    print(f\"Folder: {folder_name}, File: {filename}\")\n                    for idx, (start_s, end_s) in enumerate(merged_segments, start=1):\n                        print(f\"  Segment #{idx}: {format_time(start_s)} - {format_time(end_s)}\")\n                    print(f\"  Total Speech: {format_time(total_speech_seconds)}\\n\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T13:28:15.887078Z","iopub.execute_input":"2025-03-23T13:28:15.887382Z","iopub.status.idle":"2025-03-23T13:28:45.846250Z","shell.execute_reply.started":"2025-03-23T13:28:15.887351Z","shell.execute_reply":"2025-03-23T13:28:45.845578Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Example usage\nif __name__ == \"__main__\":\n    process_ogg_files(\n        root_folder=\"/kaggle/input\",\n        #specific_folder=\"24322\",\n        debug=False\n    )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T13:28:45.847293Z","iopub.execute_input":"2025-03-23T13:28:45.847798Z","execution_failed":"2025-03-23T13:31:53.028Z"}},"outputs":[],"execution_count":null}]}