{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":73047,"databundleVersionId":8149390,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#-----------------------------\n# imports\n#-----------------------------\nimport os \nimport torch\n#import torchaudio\nimport pandas as pd \nimport numpy as np\nimport librosa\nimport matplotlib.pyplot as plt\n\nfrom tqdm import tqdm\nfrom glob import glob\nfrom IPython.display import display,Audio\nfrom pprint import pprint\nfrom itertools import zip_longest\ntqdm.pandas()\n#-----------------------------\n# globals\n#-----------------------------\n\nDUR_MIN=15\nDUR_MAX=25\nDUR_THRESH=10\nSAMPLING_RATE = 16000\nUSE_ONNX = False # change this to True if you want to test onnx model\nmodel, utils = torch.hub.load(repo_or_dir='snakers4/silero-vad',\n                              model='silero_vad',\n                              force_reload=True,\n                              onnx=USE_ONNX)\n\n(get_speech_timestamps,\n save_audio,\n read_audio,\n VADIterator,\n collect_chunks) = utils","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-05-05T07:56:31.527553Z","iopub.execute_input":"2024-05-05T07:56:31.528758Z","iopub.status.idle":"2024-05-05T07:56:39.219519Z","shell.execute_reply.started":"2024-05-05T07:56:31.528701Z","shell.execute_reply":"2024-05-05T07:56:39.218470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#-----------------------------\n# functions\n#-----------------------------\n\n\ndistrict = 'Nilphamari_2'\n\n\ndef create_dir(base,ext):\n    _path=os.path.join(base,ext)\n    if not os.path.exists(_path):\n        os.makedirs(_path)\n    return _path\n    \nbig_audio_path=create_dir(os.getcwd(), f\"vad_chunks/{district}/big_audios\")\nsmall_audio_path=create_dir(os.getcwd(), f\"vad_chunks/{district}/small_audios\")\ndata_path=create_dir(os.getcwd(), f\"vad_chunks/{district}/audios\")\n\ndef load_data(path):\n    \"\"\"loads a wav\"\"\"\n    wave,sampling_rate= librosa.load(path, sr=SAMPLING_RATE, mono=True)\n    duration = librosa.get_duration(y=wave, sr=sampling_rate)\n    wave=np.trim_zeros(wave)\n    return duration, wave\n    \ndef normalize_signal(signal):\n    \"\"\"Normailize signal to [-1, 1] range\"\"\"\n    gain = 1.0 / (np.max(np.abs(signal)) + 1e-9)\n    return signal * gain\n\n\ndef split_by_duration(speech_timestamps):\n    # calculate duration\n    stamps=[]\n    for ts in speech_timestamps:\n        dur=(ts[\"end\"]-ts[\"start\"])/SAMPLING_RATE\n        ts[\"duration\"]=dur\n        stamps.append(ts)\n    \n    # split by big audios\n    data_split=[]\n    big_audios=[]\n    temp=[]\n    for ts in stamps:\n        dur=ts[\"duration\"]\n        if dur<DUR_MAX:\n            temp.append(ts)\n        else:\n            data_split.append(temp)\n            big_audios.append(ts)\n            temp=[]\n    if len(temp)>1:\n        data_split.append(temp)\n    return data_split,big_audios\n\ndef sequence_stamps(stamps):\n    sequence=[]\n    ts_2=None\n    for idx in range(len(stamps)-1):\n        ts_1=stamps[idx]\n        ts_1[\"type\"]=\"voice\"\n        ts_2=stamps[idx+1]\n        ts_2[\"type\"]=\"voice\"\n        \n        ns={}\n        ns[\"start\"]=ts_1[\"end\"]\n        ns[\"end\"]=ts_2[\"start\"]\n        dur=(ns[\"end\"]-ns[\"start\"])/SAMPLING_RATE\n        ns[\"duration\"]=dur\n        ns[\"type\"]=\"noise\"\n        \n        sequence.append(ts_1)\n        sequence.append(ns)\n    if ts_2 is not None:\n        sequence.append(ts_2)\n    return sequence\n        \ndef create_audio_stamps(seq):\n    data=[]\n    audio=[]\n    dur=0\n    for s in seq:\n        dur+=s[\"duration\"]\n        if dur>=DUR_MIN and dur<=DUR_MAX:\n            audio.append(s)\n            data.append(audio)\n            audio=[]\n            dur=0\n        elif dur<DUR_MIN:\n            audio.append(s)\n        elif dur>DUR_MAX:\n            data.append(audio)\n            audio=[]\n            audio.append(s)\n            dur=s[\"duration\"]\n    if len(audio)>1:\n        data.append(audio)\n    return data\n    \n    \nDATA_DICTS=[]\ndef crop_data(audio_path):\n    # load and normalize\n    wave=normalize_signal(load_data(audio_path))\n    wav=torch.tensor(wave)\n    #-------noise----------------\n    # get speech stamps\n    speech_timestamps = get_speech_timestamps(wav, model, sampling_rate=SAMPLING_RATE)\n    data_splits,big_audios = split_by_duration(speech_timestamps)\n    # create sequences of from data_splits\n    crops=[]\n    for ds in data_splits:\n        seq=sequence_stamps(ds)\n        crops+=create_audio_stamps(seq)\n        \n    iden=os.path.basename(audio_path).split(\".\")[0]\n    \n    # save big audios\n    for idx,ba in enumerate(big_audios):\n        audio=collect_chunks([ba],wav)\n        audio_path=os.path.join(big_audio_path,f\"{iden}_big_audio_{idx}.wav\")\n        save_audio(audio_path,audio)\n        DATA_DICTS.append({\"path\":audio_path,\n                           \"length\":(ba[\"end\"]-ba[\"start\"])/SAMPLING_RATE,\n                           \"classification\":\"big\"})\n    # save audios\n    sidx=0\n    idx=0\n    for crop in tqdm(crops):\n        try:\n            audio=collect_chunks(crop,wav)\n            dur=(crop[-1][\"end\"]-crop[0][\"start\"])/SAMPLING_RATE\n            if dur<DUR_THRESH:\n                audio_path=os.path.join(small_audio_path,f\"{iden}_small_audio_{sidx}.wav\")\n                save_audio(audio_path,audio)\n                sidx+=1\n                DATA_DICTS.append({\"path\":audio_path,\n                                   \"length\":dur,\n                                   \"classification\":\"small\"})\n            else:\n                audio_path=os.path.join(data_path,f\"{iden}_audio_{idx}.wav\")\n                save_audio(audio_path,audio)\n                idx+=1\n                DATA_DICTS.append({\"path\":audio_path,\n                                   \"length\":dur,\n                                   \"classification\":\"good\"})\n        except Exception as e:\n            print(iden,crop)","metadata":{"execution":{"iopub.status.busy":"2024-05-05T08:05:01.093468Z","iopub.execute_input":"2024-05-05T08:05:01.093925Z","iopub.status.idle":"2024-05-05T08:05:01.123971Z","shell.execute_reply.started":"2024-05-05T08:05:01.093893Z","shell.execute_reply":"2024-05-05T08:05:01.122635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-05-05T07:57:35.492259Z","iopub.execute_input":"2024-05-05T07:57:35.492668Z","iopub.status.idle":"2024-05-05T07:57:36.524322Z","shell.execute_reply.started":"2024-05-05T07:57:35.492630Z","shell.execute_reply":"2024-05-05T07:57:36.523119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\n\nwavs=[wav for wav in tqdm(glob(\"/kaggle/input/ben10/ben10/16_kHz_train_audio/*.wav\"))]\nmotherdf = []\nfor wavTemp in tqdm(wavs):\n    duration, signal = load_data(wavTemp)\n    wave=normalize_signal(signal)\n    wav=torch.tensor(wave)\n    #-------noise----------------\n    # get speech stamps\n    speech_timestamps = get_speech_timestamps(wav, model, sampling_rate=SAMPLING_RATE)\n    data_splits,big_audios = split_by_duration(speech_timestamps)\n    try:\n        df = pd.DataFrame({'name': wavTemp.split('/')[-1],'voice': [np.sum([x['duration'] for x in data_splits[0]])], 'total':[duration]})\n        motherdf.append(df)\n    except:\n        continue\n#    print('HHHHHHHHHHHHHHHHHHHHHHHHHHHHHHHH')\nmotherdf = pd.concat(motherdf)\n\nmotherdf['silence'] = motherdf['total'] - motherdf['voice']\nmotherdf.to_csv('VAD Analysis Train.csv')","metadata":{"execution":{"iopub.status.busy":"2024-05-05T08:11:36.488068Z","iopub.execute_input":"2024-05-05T08:11:36.488508Z","iopub.status.idle":"2024-05-05T08:12:14.965763Z","shell.execute_reply.started":"2024-05-05T08:11:36.488469Z","shell.execute_reply":"2024-05-05T08:12:14.964340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"motherdf","metadata":{"execution":{"iopub.status.busy":"2024-05-05T08:15:01.458842Z","iopub.execute_input":"2024-05-05T08:15:01.459270Z","iopub.status.idle":"2024-05-05T08:15:01.481593Z","shell.execute_reply.started":"2024-05-05T08:15:01.459239Z","shell.execute_reply":"2024-05-05T08:15:01.480323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\n\nwavs=[wav for wav in tqdm(glob(\"/kaggle/input/ben10/ben10/16_kHz_valid_audio/*.wav\"))]\nmotherdf = []\nfor wavTemp in tqdm(wavs):\n    duration, signal = load_data(wavTemp)\n    wave=normalize_signal(signal)\n    wav=torch.tensor(wave)\n    #-------noise----------------\n    # get speech stamps\n    speech_timestamps = get_speech_timestamps(wav, model, sampling_rate=SAMPLING_RATE)\n    data_splits,big_audios = split_by_duration(speech_timestamps)\n    try:\n        df = pd.DataFrame({'name': wavTemp.split('/')[-1],'voice': [np.sum([x['duration'] for x in data_splits[0]])], 'total':[duration]})\n        motherdf.append(df)\n    except:\n        continue\n#    print('HHHHHHHHHHHHHHHHHHHHHHHHHHHHHHHH')\nmotherdf = pd.concat(motherdf)\n\nmotherdf['silence'] = motherdf['total'] - motherdf['voice']\nmotherdf.to_csv('VAD Analysis Val.csv')","metadata":{"execution":{"iopub.status.busy":"2024-05-05T08:13:16.144208Z","iopub.execute_input":"2024-05-05T08:13:16.145083Z","iopub.status.idle":"2024-05-05T08:13:16.151894Z","shell.execute_reply.started":"2024-05-05T08:13:16.145048Z","shell.execute_reply":"2024-05-05T08:13:16.150566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-05-05T08:13:21.632651Z","iopub.execute_input":"2024-05-05T08:13:21.633079Z","iopub.status.idle":"2024-05-05T08:13:21.654922Z","shell.execute_reply.started":"2024-05-05T08:13:21.633049Z","shell.execute_reply":"2024-05-05T08:13:21.653549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}