{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":52324,"databundleVersionId":6229904,"sourceType":"competition"},{"sourceId":6337437,"sourceType":"datasetVersion","datasetId":3648344},{"sourceId":6339378,"sourceType":"datasetVersion","datasetId":3649700},{"sourceId":7724140,"sourceType":"datasetVersion","datasetId":4512367}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\n# import soundfile as sf\nfrom pydub import AudioSegment\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\n# from mutagen.mp3 import MP3\n# import ffmpeg\n# from concurrent.futures import ThreadPoolExecutor","metadata":{"execution":{"iopub.status.busy":"2024-03-02T15:44:58.657430Z","iopub.execute_input":"2024-03-02T15:44:58.657771Z","iopub.status.idle":"2024-03-02T15:44:59.729519Z","shell.execute_reply.started":"2024-03-02T15:44:58.657744Z","shell.execute_reply":"2024-03-02T15:44:59.728656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_qa = pd.read_csv(\"/kaggle/input/bengaliai-speech-train-nisqa/NISQA_wavfiles.csv\",sep=\",\")\ndf_qa.rename(columns={'deg':'id'},inplace=True) ## rename to match other dfs\ndf_qa['id'] = df_qa['id'].apply(lambda x:x.split('.')[0])  ## remove .wav\ndf_qa.drop(columns='model', inplace=True)\ndf_qa.info()","metadata":{"execution":{"iopub.status.busy":"2024-03-02T15:44:59.731596Z","iopub.execute_input":"2024-03-02T15:44:59.732179Z","iopub.status.idle":"2024-03-02T15:45:02.836732Z","shell.execute_reply.started":"2024-03-02T15:44:59.732145Z","shell.execute_reply":"2024-03-02T15:45:02.835349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_dur = pd.read_csv('/kaggle/input/bengaliai-speech-train-duration/train_meta.csv', sep=',')\ndf_dur.info()","metadata":{"execution":{"iopub.status.busy":"2024-03-02T15:45:02.838060Z","iopub.execute_input":"2024-03-02T15:45:02.838415Z","iopub.status.idle":"2024-03-02T15:45:12.048709Z","shell.execute_reply.started":"2024-03-02T15:45:02.838374Z","shell.execute_reply":"2024-03-02T15:45:12.047658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_dur.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-02T15:45:12.051122Z","iopub.execute_input":"2024-03-02T15:45:12.051447Z","iopub.status.idle":"2024-03-02T15:45:12.070081Z","shell.execute_reply.started":"2024-03-02T15:45:12.051419Z","shell.execute_reply":"2024-03-02T15:45:12.069109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Listen to lowest overall quality samples\n\npath = \"/kaggle/input/bengaliai-speech/train_mp3s/\"\nfile = \"e9edd1290895\" \n\n## 4d7232a0b2df sounds like some sampling rate issue post recording\n## cebe4a39953b is music playing in the background\n## e818cefce6a5 is empty but somehow bypassed our algorithmic checks (!)\n# possibly due to a divide by zero error\n\nprint(file)\ndisplay(AudioSegment.from_file(path+file+'.mp3'))\ndf_dur[df_dur['id']==file]","metadata":{"execution":{"iopub.status.busy":"2024-03-02T15:45:12.071428Z","iopub.execute_input":"2024-03-02T15:45:12.072409Z","iopub.status.idle":"2024-03-02T15:45:13.034264Z","shell.execute_reply.started":"2024-03-02T15:45:12.072356Z","shell.execute_reply":"2024-03-02T15:45:13.033104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Noise Level Analysis\nDetermination of acceptable level of speech quality ","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport ipywidgets as widgets","metadata":{"execution":{"iopub.status.busy":"2024-03-02T15:45:13.035960Z","iopub.execute_input":"2024-03-02T15:45:13.036685Z","iopub.status.idle":"2024-03-02T15:45:13.121872Z","shell.execute_reply.started":"2024-03-02T15:45:13.036644Z","shell.execute_reply":"2024-03-02T15:45:13.120981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_qa.describe()","metadata":{"execution":{"iopub.status.busy":"2024-03-02T15:45:13.123315Z","iopub.execute_input":"2024-03-02T15:45:13.123621Z","iopub.status.idle":"2024-03-02T15:45:13.325434Z","shell.execute_reply.started":"2024-03-02T15:45:13.123583Z","shell.execute_reply":"2024-03-02T15:45:13.324431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.merge(df_dur, df_qa, on='id', how='inner')\ndf.info()","metadata":{"execution":{"iopub.status.busy":"2024-03-02T15:45:13.326516Z","iopub.execute_input":"2024-03-02T15:45:13.326824Z","iopub.status.idle":"2024-03-02T15:45:14.929412Z","shell.execute_reply.started":"2024-03-02T15:45:13.326798Z","shell.execute_reply":"2024-03-02T15:45:14.928477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define function to select random ID based on specified range of values\ndef get_sample_audio(mos_range, noi_range, dis_range, col_range, loud_range):\n    df_sel = df.query(\n        f'mos_pred >= {mos_range[0]} & mos_pred <= {mos_range[1]} & ' +\n        f'noi_pred >= {noi_range[0]} & noi_pred <= {noi_range[1]} & ' +\n        f'dis_pred >= {dis_range[0]} & dis_pred <= {dis_range[1]} & ' +\n        f'col_pred >= {col_range[0]} & col_pred <= {col_range[1]} & ' +\n        f'loud_pred >= {loud_range[0]} & loud_pred <= {loud_range[1]}'\n    )\n    \n    df_sel = df_sel.sort_values(by=['mos_pred', 'noi_pred', 'dis_pred', 'col_pred', 'loud_pred'], ascending=True)\n    \n    audio_path = \"/kaggle/input/bengaliai-speech/train_mp3s/\"\n    \n    if not df_sel.empty:\n        file = df_sel['id'].iloc[0]\n        print(f\"Lowest quality sample id from given range: {file}\")\n        print(df_sel[['id', 'sentence', 'duration', 'mos_pred', 'noi_pred', 'dis_pred', 'col_pred', 'loud_pred']].set_index('id').loc[file])\n        print()\n        display(AudioSegment.from_file(audio_path+file+'.mp3'))\n        print(f\"Total number of samples within range: {len(df_sel)}\")\n    else:\n        print(\"No IDs found within specified range\")\n# Define interactive widgets\nrange_sliders = [\n    widgets.FloatRangeSlider(\n        value=[df[col].min(), df[col].max()],\n        min=df[col].min(),\n        max=df[col].max(),\n        step=0.1,\n        description=f'{col.upper()} Range:',\n        readout_format='.1f',\n        layout={'width': '80%'},  # Adjust slider length here\n    ) for col in ['mos_pred', 'noi_pred', 'dis_pred', 'col_pred', 'loud_pred']\n]\n\n# Define output widget\noutput = widgets.Output()\n\n# Define function to update output when sliders change\ndef on_change(change):\n    with output:\n        output.clear_output()\n        ranges = [slider.value for slider in range_sliders]\n        get_sample_audio(*ranges)\n\nfor slider in range_sliders:\n    slider.observe(on_change, 'value')\n\n# Display widgets\nwidgets.VBox(range_sliders + [output])","metadata":{"execution":{"iopub.status.busy":"2024-03-02T15:45:14.931222Z","iopub.execute_input":"2024-03-02T15:45:14.931676Z","iopub.status.idle":"2024-03-02T15:45:15.006110Z","shell.execute_reply.started":"2024-03-02T15:45:14.931637Z","shell.execute_reply":"2024-03-02T15:45:15.005128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df[df['mos_pred']>2.6]\ndf.info()","metadata":{"execution":{"iopub.status.busy":"2024-03-02T15:45:15.010226Z","iopub.execute_input":"2024-03-02T15:45:15.010861Z","iopub.status.idle":"2024-03-02T15:45:15.303667Z","shell.execute_reply.started":"2024-03-02T15:45:15.010822Z","shell.execute_reply":"2024-03-02T15:45:15.302633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Speaker Data Analysis","metadata":{}},{"cell_type":"code","source":"df_group = df.groupby('client_id').agg(\n    mos_pred_min=('mos_pred', 'min'),\n    mos_pred_max=('mos_pred', 'max'),\n    mos_pred_mean=('mos_pred', 'mean'),\n    noi_pred_min=('noi_pred', 'min'),\n    noi_pred_max=('noi_pred', 'max'),\n    noi_pred_mean=('noi_pred', 'mean'),\n    dis_pred_min=('dis_pred', 'min'),\n    dis_pred_max=('dis_pred', 'max'),\n    dis_pred_mean=('dis_pred', 'mean'),\n    col_pred_min=('col_pred', 'min'),\n    col_pred_max=('col_pred', 'max'),\n    col_pred_mean=('col_pred', 'mean'),\n    loud_pred_min=('loud_pred', 'min'),\n    loud_pred_max=('loud_pred', 'max'),\n    loud_pred_mean=('loud_pred', 'mean'),\n    duration_sum=('duration', 'sum'),\n    count=('client_id', 'count')\n)\n\ndf_group.info()","metadata":{"execution":{"iopub.status.busy":"2024-03-02T15:45:15.304832Z","iopub.execute_input":"2024-03-02T15:45:15.305149Z","iopub.status.idle":"2024-03-02T15:45:15.606418Z","shell.execute_reply.started":"2024-03-02T15:45:15.305123Z","shell.execute_reply":"2024-03-02T15:45:15.605520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_group.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-02T15:45:15.607571Z","iopub.execute_input":"2024-03-02T15:45:15.607857Z","iopub.status.idle":"2024-03-02T15:45:15.629962Z","shell.execute_reply.started":"2024-03-02T15:45:15.607834Z","shell.execute_reply":"2024-03-02T15:45:15.628990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# speakers with total duration at least 30 mins and total sentence count at least 400 (VCTK config)\ndf_crop = df_group[(df_group['duration_sum'] >= 1800)&(df_group['count'] >= 400)]\n\nplt.figure(figsize=(10, 6))\nplt.hist(df_crop['duration_sum'], bins=30, color='skyblue', edgecolor='black')\nplt.hist(df_crop['count'], bins=30, color='orange', edgecolor='black')\nplt.title('Distribution of Audio Duration')\nplt.xlabel('Duration (seconds)')\nplt.ylabel('Frequency')\nplt.grid(True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-02T15:45:15.631324Z","iopub.execute_input":"2024-03-02T15:45:15.631703Z","iopub.status.idle":"2024-03-02T15:45:16.046365Z","shell.execute_reply.started":"2024-03-02T15:45:15.631671Z","shell.execute_reply":"2024-03-02T15:45:16.045342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_crop.info()","metadata":{"execution":{"iopub.status.busy":"2024-03-02T15:45:16.047696Z","iopub.execute_input":"2024-03-02T15:45:16.048069Z","iopub.status.idle":"2024-03-02T15:45:16.060546Z","shell.execute_reply.started":"2024-03-02T15:45:16.048042Z","shell.execute_reply":"2024-03-02T15:45:16.059337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Denoising","metadata":{}},{"cell_type":"code","source":"!pip install cleanunet -q","metadata":{"execution":{"iopub.status.busy":"2024-03-02T15:45:16.062086Z","iopub.execute_input":"2024-03-02T15:45:16.062491Z","iopub.status.idle":"2024-03-02T15:45:42.424994Z","shell.execute_reply.started":"2024-03-02T15:45:16.062454Z","shell.execute_reply":"2024-03-02T15:45:42.423568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir denoised_audio","metadata":{"execution":{"iopub.status.busy":"2024-03-02T15:45:42.426566Z","iopub.execute_input":"2024-03-02T15:45:42.426922Z","iopub.status.idle":"2024-03-02T15:45:43.465745Z","shell.execute_reply.started":"2024-03-02T15:45:42.426894Z","shell.execute_reply":"2024-03-02T15:45:43.464306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torchaudio\nfrom cleanunet import CleanUNet\n\nfile_root = \"/kaggle/input/bengaliai-speech/train_mp3s/\"\nnet = CleanUNet.from_pretrained(varient='high', device='cuda')","metadata":{"execution":{"iopub.status.busy":"2024-03-02T15:45:43.467568Z","iopub.execute_input":"2024-03-02T15:45:43.468021Z","iopub.status.idle":"2024-03-02T15:45:55.042246Z","shell.execute_reply.started":"2024-03-02T15:45:43.467982Z","shell.execute_reply":"2024-03-02T15:45:55.041241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_id = '0ff51e50fe7f'\naud, sr = torchaudio.load(file_root+file_id+'.mp3')\ndenoised_aud = net(aud.to('cuda'))[0]\n\n# Save the denoised audio\ntorchaudio.save(f'denoised_audio/{file_id}.mp3', denoised_aud.to('cpu'), sr)\n\n\nprint(\"Before Denoising:\")\ndisplay(AudioSegment.from_file(file_root + file_id + '.mp3'))\nprint(\"After Denoising:\")\ndisplay(AudioSegment.from_file(f'denoised_audio/{file_id}.mp3'))","metadata":{"execution":{"iopub.status.busy":"2024-03-02T15:48:11.929417Z","iopub.execute_input":"2024-03-02T15:48:11.930079Z","iopub.status.idle":"2024-03-02T15:48:12.592272Z","shell.execute_reply.started":"2024-03-02T15:48:11.930044Z","shell.execute_reply":"2024-03-02T15:48:12.591066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}