{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":52324,"databundleVersionId":6229904,"sourceType":"competition"},{"sourceId":146760769,"sourceType":"kernelVersion"}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Exploratory notebook 1 \n## Bengali speech recognition\n### AUDIO \n\nThis notebook is part of the exploratory work for the Kaggle competition Bengali.AI. \n\nThe goal is to transcribe audio from bengali speakers into written form. \nThis notebook will focus on the audio part. The second notebook on the text part. \n","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd \nimport soundfile as sf\nfrom pathlib import Path\nimport IPython\nimport librosa\nimport numpy as np \nimport matplotlib.pyplot as plt\nfrom pydub import AudioSegment","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:34:45.475890Z","iopub.execute_input":"2023-12-08T09:34:45.476404Z","iopub.status.idle":"2023-12-08T09:34:45.976194Z","shell.execute_reply.started":"2023-12-08T09:34:45.476340Z","shell.execute_reply":"2023-12-08T09:34:45.974622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root         = Path(\"../input/bengaliai-speech\")\naudio_folder = root / \"train_mp3s\"\ndf = pd.read_csv(root / \"train.csv\")\nprint(\"Data shape : \", df.shape)\ndf.head()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-08T09:34:45.978310Z","iopub.execute_input":"2023-12-08T09:34:45.978937Z","iopub.status.idle":"2023-12-08T09:34:51.997376Z","shell.execute_reply.started":"2023-12-08T09:34:45.978892Z","shell.execute_reply":"2023-12-08T09:34:51.996120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The training database is composed of more than 900.000 row. \n\nEach row corresponds to an audio id that can me found in the train_mp3s folder as \"id\".mp3 file name. \n\nThis folder takes **23G in memory**. \nSentences are written in bengali alphabet. \n\nLet's take a sample for exploratory work. ","metadata":{}},{"cell_type":"code","source":"for i in range(4) : \n    id = df.iloc[i][\"id\"]\n    audio_path = audio_folder / f\"{id}.mp3\"\n    audio , sr = sf.read(audio_path)\n    print(\"Path : \", audio_path, \"sample rate : \", sr, \" Hz\")\n    print(\"Transcript : \", df.iloc[i][\"sentence\"])\n    plt.plot(audio, label=f\"{id}.mp3\" )\n    plt.xlabel(\"sample\")\n    plt.ylim([-1,1])\n    \n    IPython.display.display(IPython.display.Audio(audio_path))\nplt.title(\"Audio waveforms\")\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:34:51.998730Z","iopub.execute_input":"2023-12-08T09:34:51.999164Z","iopub.status.idle":"2023-12-08T09:34:52.860617Z","shell.execute_reply.started":"2023-12-08T09:34:51.999119Z","shell.execute_reply":"2023-12-08T09:34:52.859281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"First of all, each audio file does not have the same length. Speaker are easer male of female. \n\nThe sound level is different for each audio, some are more noisy then others.\n\nSome transcription also have punctuation. \n\nLet's analyse **10%** of the dataset to ease computation.\n","metadata":{}},{"cell_type":"code","source":"df = df.sample(int(df.shape[0] * 0.1))","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:34:52.862985Z","iopub.execute_input":"2023-12-08T09:34:52.863424Z","iopub.status.idle":"2023-12-08T09:34:52.945916Z","shell.execute_reply.started":"2023-12-08T09:34:52.863376Z","shell.execute_reply":"2023-12-08T09:34:52.944877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"full_path\"] = df.apply(lambda row : audio_folder / f\"{row.id}.mp3\", axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:34:52.947054Z","iopub.execute_input":"2023-12-08T09:34:52.947349Z","iopub.status.idle":"2023-12-08T09:34:54.830649Z","shell.execute_reply.started":"2023-12-08T09:34:52.947324Z","shell.execute_reply":"2023-12-08T09:34:54.829439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_audio_length(row) : \n    return librosa.get_duration(path=row.full_path)\n\ndf[\"duration\"] = df.apply(get_audio_length, axis=1)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:34:54.832187Z","iopub.execute_input":"2023-12-08T09:34:54.832562Z","iopub.status.idle":"2023-12-08T09:55:07.692912Z","shell.execute_reply.started":"2023-12-08T09:34:54.832523Z","shell.execute_reply":"2023-12-08T09:55:07.691870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Total duration of 10% of the dataset: \",str(int( df[\"duration\"].sum() / 3600)), \"h\" ) \nprint(\"Total duration estimation : \", str(int(df[\"duration\"].sum() / 3600)*10) , \"h\") ","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:55:07.694228Z","iopub.execute_input":"2023-12-08T09:55:07.695472Z","iopub.status.idle":"2023-12-08T09:55:07.704864Z","shell.execute_reply.started":"2023-12-08T09:55:07.695428Z","shell.execute_reply":"2023-12-08T09:55:07.703589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig , (ax1,ax2) = plt.subplots(1,2,figsize=(12,5))\ndf.hist( column=[\"duration\"], edgecolor='black', bins=20, ax=ax1)\ndf.boxplot(column=[\"duration\"],ax=ax2)\nax1.set_title(\"Duration of audio files (s) on 10%\\ of the dataset\")\nax2.set_title(\"Duration of audio files (s) on 10%\\ of the dataset\")\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:55:07.706790Z","iopub.execute_input":"2023-12-08T09:55:07.707659Z","iopub.status.idle":"2023-12-08T09:55:08.259689Z","shell.execute_reply.started":"2023-12-08T09:55:07.707602Z","shell.execute_reply":"2023-12-08T09:55:08.258427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"duration\"].describe()","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:55:08.260848Z","iopub.execute_input":"2023-12-08T09:55:08.261148Z","iopub.status.idle":"2023-12-08T09:55:08.278218Z","shell.execute_reply.started":"2023-12-08T09:55:08.261123Z","shell.execute_reply":"2023-12-08T09:55:08.277149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_sampling_rate(row) : \n    return librosa.get_samplerate(path=row.full_path)\n\ndf[\"fs\"] = df.apply(get_sampling_rate, axis=1)\ndf[\"fs\"].unique()","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:55:08.281426Z","iopub.execute_input":"2023-12-08T09:55:08.281772Z","iopub.status.idle":"2023-12-08T09:56:57.437182Z","shell.execute_reply.started":"2023-12-08T09:55:08.281743Z","shell.execute_reply":"2023-12-08T09:56:57.436067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The only sample rate of all files is **32kHz** . ","metadata":{}},{"cell_type":"markdown","source":"## And the sound level ? \n### What are the different sound level of all audio files : \nIn digital audio we use **dBFS (decibels relative to full scale)** as reference to  measure audio level. \n\ndBFS is expressed as a negative number relative to the maximum level available in a digital system.\n\nAs further evaluation are long, we will take only 0.1% of the dataset","metadata":{}},{"cell_type":"code","source":"df = df.sample(int(df.shape[0] * 0.01))","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:56:57.438663Z","iopub.execute_input":"2023-12-08T09:56:57.439898Z","iopub.status.idle":"2023-12-08T09:56:57.495826Z","shell.execute_reply.started":"2023-12-08T09:56:57.439859Z","shell.execute_reply":"2023-12-08T09:56:57.494649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:56:57.497105Z","iopub.execute_input":"2023-12-08T09:56:57.497525Z","iopub.status.idle":"2023-12-08T09:56:57.503884Z","shell.execute_reply.started":"2023-12-08T09:56:57.497488Z","shell.execute_reply":"2023-12-08T09:56:57.502720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_dBFs(row): \n    sound = AudioSegment.from_file(row.full_path, format=\"mp3\")\n    return  sound.dBFS\n\ndf[\"dBFS\"] = df.apply(get_dBFs, axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:56:57.505635Z","iopub.execute_input":"2023-12-08T09:56:57.506315Z","iopub.status.idle":"2023-12-08T09:59:33.033432Z","shell.execute_reply.started":"2023-12-08T09:56:57.506271Z","shell.execute_reply":"2023-12-08T09:59:33.032050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.hist( column=[\"dBFS\"], edgecolor='black', bins=20)\nplt.title(\"Sound Level of audio files (dBFS)\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:59:33.034830Z","iopub.execute_input":"2023-12-08T09:59:33.035140Z","iopub.status.idle":"2023-12-08T09:59:33.309412Z","shell.execute_reply.started":"2023-12-08T09:59:33.035113Z","shell.execute_reply":"2023-12-08T09:59:33.308228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IPython.display.display(IPython.display.Audio(df[df[\"dBFS\"] < -50].iloc[0][\"full_path\"]))","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:59:33.311218Z","iopub.execute_input":"2023-12-08T09:59:33.311687Z","iopub.status.idle":"2023-12-08T09:59:33.322641Z","shell.execute_reply.started":"2023-12-08T09:59:33.311647Z","shell.execute_reply":"2023-12-08T09:59:33.321617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It's clear that we do not ear anything. ","metadata":{}},{"cell_type":"code","source":"IPython.display.display(IPython.display.Audio(df[df[\"dBFS\"] > -10].iloc[0][\"full_path\"]))","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:59:33.324179Z","iopub.execute_input":"2023-12-08T09:59:33.324873Z","iopub.status.idle":"2023-12-08T09:59:33.336395Z","shell.execute_reply.started":"2023-12-08T09:59:33.324833Z","shell.execute_reply":"2023-12-08T09:59:33.335225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It's very loud and the sound is saturating. ","metadata":{}},{"cell_type":"markdown","source":"## Male or female ? \nLet's use alefiury/wav2vec2-large-xlsr-53-gender-recognition-librispeech to label our data extract.","metadata":{}},{"cell_type":"code","source":"import os\nfrom typing import List, Optional, Union, Dict\n\nimport tqdm\nimport torch\nimport torchaudio\nimport numpy as np\nimport pandas as pd\nfrom torch import nn\nfrom torch.utils.data import DataLoader\nfrom torch.nn import functional as F\nfrom transformers import (\n    AutoFeatureExtractor,\n    AutoModelForAudioClassification,\n    Wav2Vec2Processor\n)\n\n\nclass CustomDataset(torch.utils.data.Dataset):\n    def __init__(\n        self,\n        dataset: List,\n        basedir: Optional[str] = None,\n        sampling_rate: int = 16000,\n        max_audio_len: int = 5,\n    ):\n        self.dataset = dataset\n        self.basedir = basedir\n\n        self.sampling_rate = sampling_rate\n        self.max_audio_len = max_audio_len\n\n    def __len__(self):\n        \"\"\"\n        Return the length of the dataset\n        \"\"\"\n        return len(self.dataset)\n\n    def _cutorpad(self, audio: np.ndarray) -> np.ndarray:\n        \"\"\"\n        Cut or pad audio to the wished length\n        \"\"\"\n        effective_length = self.sampling_rate * self.max_audio_len\n        len_audio = len(audio)\n\n        # If audio length is bigger than wished audio length\n        if len_audio > effective_length:\n            audio = audio[:effective_length]\n\n        # Expand one dimension related to the channel dimension\n        return audio\n\n\n    def __getitem__(self, index) -> torch.Tensor:\n        \"\"\"\n        Return the audio and the sampling rate\n        \"\"\"\n        if self.basedir is None:\n            filepath = self.dataset[index]\n        else:\n            filepath = os.path.join(self.basedir, self.dataset[index])\n\n        speech_array, sr = torchaudio.load(filepath)\n\n        # Transform to mono\n        if speech_array.shape[0] > 1:\n            speech_array = torch.mean(speech_array, dim=0, keepdim=True)\n\n        if sr != self.sampling_rate:\n            transform = torchaudio.transforms.Resample(sr, self.sampling_rate)\n            speech_array = transform(speech_array)\n            sr = self.sampling_rate\n\n        speech_array = speech_array.squeeze().numpy()\n\n        # Cut or pad audio\n        speech_array = self._cutorpad(speech_array)\n\n        return speech_array\n\nclass CollateFunc:\n    def __init__(\n        self,\n        processor: Wav2Vec2Processor,\n        max_length: Optional[int] = None,\n        padding: Union[bool, str] = True,\n        pad_to_multiple_of: Optional[int] = None,\n        sampling_rate: int = 16000,\n    ):\n        self.padding = padding\n        self.processor = processor\n        self.max_length = max_length\n        self.sampling_rate = sampling_rate\n        self.pad_to_multiple_of = pad_to_multiple_of\n\n    def __call__(self, batch: List):\n        input_features = []\n\n        for audio in batch:\n            input_tensor = self.processor(audio, sampling_rate=self.sampling_rate).input_values\n            input_tensor = np.squeeze(input_tensor)\n            input_features.append({\"input_values\": input_tensor})\n\n        batch = self.processor.pad(\n            input_features,\n            padding=self.padding,\n            max_length=self.max_length,\n            pad_to_multiple_of=self.pad_to_multiple_of,\n            return_tensors=\"pt\",\n        )\n\n        return batch\n\n\ndef predict(test_dataloader, model, device: torch.device):\n    \"\"\"\n    Predict the class of the audio\n    \"\"\"\n    model.to(device)\n    model.eval()\n    preds = []\n\n    with torch.no_grad():\n        for batch in tqdm.tqdm(test_dataloader):\n            input_values, attention_mask = batch['input_values'].to(device), batch['attention_mask'].to(device)\n\n            logits = model(input_values, attention_mask=attention_mask).logits\n            scores = F.softmax(logits, dim=-1)\n\n            pred = torch.argmax(scores, dim=1).cpu().detach().numpy()\n\n            preds.extend(pred)\n\n    return preds\n\n\ndef get_gender(model_name_or_path: str, audio_paths: List[str], label2id: Dict, id2label: Dict, device: torch.device):\n    num_labels = 2\n\n    feature_extractor = AutoFeatureExtractor.from_pretrained(model_name_or_path)\n    model = AutoModelForAudioClassification.from_pretrained(\n        pretrained_model_name_or_path=model_name_or_path,\n        num_labels=num_labels,\n        label2id=label2id,\n        id2label=id2label,\n    )\n\n    test_dataset = CustomDataset(audio_paths)\n    data_collator = CollateFunc(\n        processor=feature_extractor,\n        padding=True,\n        sampling_rate=16000,\n    )\n\n    test_dataloader = DataLoader(\n        dataset=test_dataset,\n        batch_size=16,\n        collate_fn=data_collator,\n        shuffle=False,\n        num_workers=10\n    )\n\n    preds = predict(test_dataloader=test_dataloader, model=model, device=device)\n\n    return preds\n\n\ndef get_gender_list(audio_paths) : \n    model_name_or_path = \"alefiury/wav2vec2-large-xlsr-53-gender-recognition-librispeech\"\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n    label2id = {\n        \"female\": 0,\n        \"male\": 1\n    }\n\n    id2label = {\n        0: \"female\",\n        1: \"male\"\n    }\n\n    num_labels = 2\n\n    preds = get_gender(model_name_or_path, audio_paths, label2id, id2label, device)\n    \n    return preds ","metadata":{"execution":{"iopub.status.busy":"2023-12-08T10:00:44.328828Z","iopub.execute_input":"2023-12-08T10:00:44.329330Z","iopub.status.idle":"2023-12-08T10:00:50.564411Z","shell.execute_reply.started":"2023-12-08T10:00:44.329288Z","shell.execute_reply":"2023-12-08T10:00:50.563080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths = df[\"full_path\"].to_list()\npreds = get_gender_list(paths)","metadata":{"execution":{"iopub.status.busy":"2023-12-08T10:01:03.170872Z","iopub.execute_input":"2023-12-08T10:01:03.172072Z","iopub.status.idle":"2023-12-08T10:36:49.768373Z","shell.execute_reply.started":"2023-12-08T10:01:03.172008Z","shell.execute_reply":"2023-12-08T10:36:49.766436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Number of female : {len(preds) - np.sum(preds)} over {len(preds)}, {(len(preds) - np.sum(preds))/len(preds) *100} % \")\nprint(f\"Number of male : {np.sum(preds)} over {len(preds)}, {(np.sum(preds))/len(preds) *100} % \")","metadata":{"execution":{"iopub.status.busy":"2023-12-08T10:58:24.062761Z","iopub.execute_input":"2023-12-08T10:58:24.066072Z","iopub.status.idle":"2023-12-08T10:58:24.077482Z","shell.execute_reply.started":"2023-12-08T10:58:24.065995Z","shell.execute_reply":"2023-12-08T10:58:24.076150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The data seems to be balanced.","metadata":{}},{"cell_type":"markdown","source":"## Conclusion \nThe audio data contains lot of noise of different nature and are of variable length.\n\nVery long audio will damage training and need to be set apart (first clean then considered if time). \n\nAudio below -50dBFS and over -5dBFS needs to be taken also apart.","metadata":{}}]}