{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#Bangladesh University of Engineering and Technology (BUET)\n\n![](https://scientificbangladesh.com/wp-content/uploads/2020/01/maxresdefault.jpg)scientificbangladesh.com","metadata":{}},{"cell_type":"markdown","source":"\"Bengali is the fifth most spoken of all native languages all over the world. But so far very little work has been done on Bengali Speech Transcription. Considering the large audience and far-reaching opportunities, there’s significant business and educational interest in developing AI that can be used in 𝐁𝐞𝐧𝐠𝐚𝐥𝐢 𝐀𝐒𝐑 (Automatic Speech Recognition). The authors, at the 𝐃𝐞𝐩𝐚𝐫𝐭𝐦𝐞𝐧𝐭 𝐨𝐟 𝐂𝐨𝐦𝐩𝐮𝐭𝐞𝐫 𝐒𝐜𝐢𝐞𝐧𝐜𝐞 𝐚𝐧𝐝 𝐄𝐧𝐠𝐢𝐧𝐞𝐞𝐫𝐢𝐧𝐠, 𝐁𝐔𝐄𝐓, in partnership with 𝐁𝐞𝐧𝐠𝐚𝐥𝐢.𝐀𝐈, are glad to present the very first Bengali ASR competition of its kind, 𝐃𝐋 𝐒𝐩𝐫𝐢𝐧𝐭, as a part of BUET CSE Fest 2022.\"\n\nhttps://www.kaggle.com/competitions/dlsprint","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:16:58.356240Z","iopub.execute_input":"2022-08-12T21:16:58.357333Z","iopub.status.idle":"2022-08-12T21:16:58.388003Z","shell.execute_reply.started":"2022-08-12T21:16:58.357125Z","shell.execute_reply":"2022-08-12T21:16:58.386955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"songs = []\ncwd = '/kaggle/input/dlsprint/validation_files/'\n\nfor dirname, _, filenames in os.walk('/kaggle/input/dlsprint/validation_files/'):\n    for filename in filenames:\n        #print(os.path.join(dirname, filename))\n        songs.append(filename)\n#data = pd.read_csv('/kaggle/input/us-election-2020-presidential-debates/us_election_2020_1st_presidential_debate.csv')\nsongs.pop(0)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:17:03.723788Z","iopub.execute_input":"2022-08-12T21:17:03.724303Z","iopub.status.idle":"2022-08-12T21:17:12.913704Z","shell.execute_reply.started":"2022-08-12T21:17:03.724255Z","shell.execute_reply":"2022-08-12T21:17:12.912688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! pip install pydub","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-12T21:19:15.779584Z","iopub.execute_input":"2022-08-12T21:19:15.780574Z","iopub.status.idle":"2022-08-12T21:19:28.020878Z","shell.execute_reply.started":"2022-08-12T21:19:15.780533Z","shell.execute_reply":"2022-08-12T21:19:28.019333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pydub import AudioSegment\nimport IPython\n\n# We will listen to this file:\n# 213_1p5_Pr_mc_AKGC417L.wav\nfile = '/kaggle/input/dlsprint/validation_files/common_voice_bn_31522735.mp3'\nprint(cwd+songs[0])\nIPython.display.Audio(cwd+songs[2])","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:19:43.129081Z","iopub.execute_input":"2022-08-12T21:19:43.129715Z","iopub.status.idle":"2022-08-12T21:19:43.170228Z","shell.execute_reply.started":"2022-08-12T21:19:43.129670Z","shell.execute_reply":"2022-08-12T21:19:43.169491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Converting MP3 into WAVs","metadata":{}},{"cell_type":"code","source":"# https://www.kaggle.com/rakibilly/extract-audio-starter\nimport subprocess\nimport glob\nimport os\nfrom pathlib import Path\nimport shutil\nfrom zipfile import ZipFile","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:19:49.489782Z","iopub.execute_input":"2022-08-12T21:19:49.490826Z","iopub.status.idle":"2022-08-12T21:19:49.496704Z","shell.execute_reply.started":"2022-08-12T21:19:49.490758Z","shell.execute_reply":"2022-08-12T21:19:49.495487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#ADD this file https://www.kaggle.com/rakibilly/ffmpeg-static-build so that the code below works.\n\nWithout this file we can't move on (Brett Smith) ffmpeg-git-20191209-amd64-static/mode","metadata":{}},{"cell_type":"code","source":"! tar xvf ../input/ffmpeg-static-build/ffmpeg-git-amd64-static.tar.xz","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-12T21:20:03.024198Z","iopub.execute_input":"2022-08-12T21:20:03.024658Z","iopub.status.idle":"2022-08-12T21:20:08.425354Z","shell.execute_reply.started":"2022-08-12T21:20:03.024619Z","shell.execute_reply":"2022-08-12T21:20:08.424207Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert MP3s to WAV for easy conversion to numpy arrays:\noutput_format = 'wav'  # can also use aac, wav, etc\noutput_dir = Path(f\"{output_format}s\")\nPath(output_dir).mkdir(exist_ok=True, parents=True)\n\n#Only do first 50 because notebook memory limitations...\nfor song in songs[:50]:\n    file = cwd+song\n    file_name = song.replace(\".mp3\",\"\")\n    command = f\"../working/ffmpeg-git-20191209-amd64-static/ffmpeg -i {file} -ab 192000 -ac 2 -ar 44100 -vn {output_dir/file_name}.{output_format}\"\n    subprocess.call(command, shell=True)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-12T21:20:38.463967Z","iopub.execute_input":"2022-08-12T21:20:38.464377Z","iopub.status.idle":"2022-08-12T21:20:40.227733Z","shell.execute_reply.started":"2022-08-12T21:20:38.464342Z","shell.execute_reply":"2022-08-12T21:20:40.226186Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.io.wavfile import read, write\n#a = read(\"adios.wav\")\nwavs = []\nnp_arrays = []\nfor dirname, _, filenames in os.walk('/kaggle/working/wavs/'):\n    for filename in filenames:\n        wav_file = dirname+filename\n        #print(wav_file)\n        wavs.append(wav_file)\n        try:\n            fs, io_file = read(wav_file)\n        except ValueError:\n            continue\n        data = np.array(io_file,dtype=float)\n        wav_info= {\n            'name': filename,\n            'fs' : fs,\n            'left': data[:,0],\n            'right': data[:,1]\n        }\n        \n        np_arrays.append(wav_info)\n\nprint(\"Succesfully converted: \"+str(len(np_arrays)))","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:20:54.314656Z","iopub.execute_input":"2022-08-12T21:20:54.315060Z","iopub.status.idle":"2022-08-12T21:20:54.570158Z","shell.execute_reply.started":"2022-08-12T21:20:54.315027Z","shell.execute_reply":"2022-08-12T21:20:54.569015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Plot & Play\n\nTo avoid ERROR we must add another file: ffmpeg-git-20191209-amd64-static/mode\n\nWhich was already added lines above!","metadata":{}},{"cell_type":"code","source":"#Ignore warnings\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:21:03.151144Z","iopub.execute_input":"2022-08-12T21:21:03.151606Z","iopub.status.idle":"2022-08-12T21:21:03.157635Z","shell.execute_reply.started":"2022-08-12T21:21:03.151565Z","shell.execute_reply":"2022-08-12T21:21:03.156376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy import signal\nfrom scipy.fft import fftshift\nimport matplotlib.pyplot as plt\n\nsong_data = np_arrays[26]\nstart = 0\nend = 10\n\nif end != None:\n    wav = song_data['left'][fs*start:fs*end]\nelse:\n    wav = song_data['left'][fs*start:]\nfs = song_data['fs']\nplt.specgram(wav,Fs=fs)\nplt.ylim(top=15000)\nprint(song_data['name'].replace(\".wav\",\"\"))\nplt.show() \n\nIPython.display.Audio(wav, rate=fs)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:21:08.805867Z","iopub.execute_input":"2022-08-12T21:21:08.806404Z","iopub.status.idle":"2022-08-12T21:21:09.726546Z","shell.execute_reply.started":"2022-08-12T21:21:08.806358Z","shell.execute_reply":"2022-08-12T21:21:09.725509Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Denoise","metadata":{}},{"cell_type":"code","source":"! pip install pyyawt","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-12T21:21:16.671966Z","iopub.execute_input":"2022-08-12T21:21:16.672374Z","iopub.status.idle":"2022-08-12T21:21:45.726271Z","shell.execute_reply.started":"2022-08-12T21:21:16.672341Z","shell.execute_reply":"2022-08-12T21:21:45.725219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load a noisy signal\n# Phylloscopus-collybita-171141\n\nsong_data = np_arrays[4]\nstart = 1\nend = 12\n\nif end != None:\n    wav = song_data['left'][fs*start:fs*end]\nelse:\n    wav = song_data['left'][fs*start:]\nfs = song_data['fs']\nplt.specgram(wav,Fs=fs)\nplt.ylim(top=15000)\nprint(song_data['name'].replace(\".wav\",\"\"))\nplt.show() \n\nIPython.display.Audio(wav, rate=fs)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:22:00.849806Z","iopub.execute_input":"2022-08-12T21:22:00.850614Z","iopub.status.idle":"2022-08-12T21:22:01.167025Z","shell.execute_reply.started":"2022-08-12T21:22:00.850568Z","shell.execute_reply":"2022-08-12T21:22:01.166178Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport pywt\nimport pyyawt\n\nstds = []\nmeans = []\ndecomps = []\nthrs = []\nwavelets = pywt.wavedec(wav, 'db5', level=10)\n\nfor i, wavelet in enumerate(wavelets):\n    thrs.append(pyyawt.thselect(wavelet, 'heursure'))\n    stds.append(wavelet.std(0))\n    means.append(wavelet.mean(0))\n    decomps.append(wavelet)\n    \n    #ax[i+1,0].plot(wavelet)\n    #ax[i+1,0].plot(wavelet)\n    #sns.distplot(wavelet, ax=ax[i+1,1], hist=False, vertical=True)\n\nthresholded = []\n\nfig, ax = plt.subplots(len(wavelets), figsize=(20,20))\n\n\nfor i, decomp in enumerate(decomps):\n    thresh =((np.amax(decomp)-means[i])*thrs[i])\n    print(thrs[i], np.amax(decomp), thresh)\n    thresholded.append(pywt.threshold(decomp, thresh, 'soft'))\n    ax[i].plot(wavelets[i])\n    ax[i].plot(thresholded[i])\n\nprint(\"Denoised: \"+song_data['name'].replace(\".wav\",\"\"))\nreconstructed = pywt.waverec(thresholded, 'db5')\nplt.specgram(reconstructed,Fs=fs)\nplt.show()\nIPython.display.Audio(reconstructed, rate=fs)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:22:09.901182Z","iopub.execute_input":"2022-08-12T21:22:09.901731Z","iopub.status.idle":"2022-08-12T21:22:11.587595Z","shell.execute_reply.started":"2022-08-12T21:22:09.901684Z","shell.execute_reply":"2022-08-12T21:22:11.586352Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Above mp3 codes by Brett Smith https://www.kaggle.com/bretts/plot-spectrogram-play-audio","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\n\nfrom IPython.display import Markdown, display\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom plotly.subplots import make_subplots\nfrom wordcloud import WordCloud\nimport re","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:22:24.850525Z","iopub.execute_input":"2022-08-12T21:22:24.851806Z","iopub.status.idle":"2022-08-12T21:22:24.928476Z","shell.execute_reply.started":"2022-08-12T21:22:24.851749Z","shell.execute_reply":"2022-08-12T21:22:24.927535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#DF is train file to make easier to make the charts","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/dlsprint/train.csv\", delimiter=',', encoding='utf8')\ndf.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:22:32.326065Z","iopub.execute_input":"2022-08-12T21:22:32.326485Z","iopub.status.idle":"2022-08-12T21:22:34.491705Z","shell.execute_reply.started":"2022-08-12T21:22:32.326426Z","shell.execute_reply":"2022-08-12T21:22:34.490631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val = pd.read_csv(\"/kaggle/input/dlsprint/validation.csv\", delimiter=',', encoding='utf8')\nval.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:22:43.955418Z","iopub.execute_input":"2022-08-12T21:22:43.957072Z","iopub.status.idle":"2022-08-12T21:22:44.097275Z","shell.execute_reply.started":"2022-08-12T21:22:43.957003Z","shell.execute_reply":"2022-08-12T21:22:44.096211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\n!pip install langdetect # Language Detection\n!pip install bnlp_toolkit # For Bangla Word Cloud\n!wget https://www.omicronlab.com/download/fonts/kalpurush.ttf # Bangla Font For the Word Cloud","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:22:49.409961Z","iopub.execute_input":"2022-08-12T21:22:49.410389Z","iopub.status.idle":"2022-08-12T21:23:20.437573Z","shell.execute_reply.started":"2022-08-12T21:22:49.410352Z","shell.execute_reply":"2022-08-12T21:23:20.435646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def printmd(string):\n  display(Markdown(string))\n\nfrom langdetect import detect\nimport unicodedata\nimport html","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:23:38.373818Z","iopub.execute_input":"2022-08-12T21:23:38.375125Z","iopub.status.idle":"2022-08-12T21:23:38.393377Z","shell.execute_reply.started":"2022-08-12T21:23:38.375066Z","shell.execute_reply":"2022-08-12T21:23:38.392470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Code by Tanjima Nasreen Jenia  https://www.kaggle.com/tanjimanasreenjenia/bangladeshi-restaurants-analytics/notebook\n\nfrom bnlp.corpus import stopwords, punctuations\nregex = r\"[\\u0980-\\u09FF]+\" \ndata = df\n\nwc = WordCloud(background_color = 'green',\n                      height =2000,\n                      width = 2000,\n                      colormap = 'Reds',\n                      font_path=\"./kalpurush.ttf\"\n                     ).generate(str(df[\"sentence\"]))                                                                                                                        \n\nplt.rcParams['figure.figsize'] = (12,12)\nplt.imshow(wc, interpolation=\"bilinear\")\nplt.axis('off')\nplt.show()\nresult = wc.to_file(\"Bangla_word_cloud.png\")\nprintmd(\"Bengali Sentences Transcripts\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-12T21:23:44.284630Z","iopub.execute_input":"2022-08-12T21:23:44.285045Z","iopub.status.idle":"2022-08-12T21:23:55.233667Z","shell.execute_reply.started":"2022-08-12T21:23:44.285011Z","shell.execute_reply":"2022-08-12T21:23:55.232229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Upvotes and Downvotes? For the speech, sentence or accent? Or they simply don't like the person?","metadata":{}},{"cell_type":"code","source":"#Code by Puru Behl https://www.kaggle.com/accountstatus/mt-cars-data-analysis\n\nsns.distplot(df['up_votes'])\nplt.axvline(df['up_votes'].values.mean(), color='red', linestyle='dashed', linewidth=1)\nplt.title('Upvotes Distribution');","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-12T21:24:01.422596Z","iopub.execute_input":"2022-08-12T21:24:01.423022Z","iopub.status.idle":"2022-08-12T21:24:02.625555Z","shell.execute_reply.started":"2022-08-12T21:24:01.422988Z","shell.execute_reply":"2022-08-12T21:24:02.624481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Code by Firat Gonen https://www.kaggle.com/frtgnn/elo-eda-lgbm/notebook \n\nplt.figure(figsize=(10, 6))\nplt.title('Downvotes Distribution')\nsns.despine()\nsns.set_context(\"notebook\", font_scale=1.5, rc={\"lines.linewidth\": 2.5})\n\nsns.distplot(df['down_votes'], hist=True, rug=False,norm_hist=True);","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-12T21:24:14.568615Z","iopub.execute_input":"2022-08-12T21:24:14.569493Z","iopub.status.idle":"2022-08-12T21:24:15.650251Z","shell.execute_reply.started":"2022-08-12T21:24:14.569448Z","shell.execute_reply":"2022-08-12T21:24:15.649401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Bengali Accents 257?? ","metadata":{}},{"cell_type":"code","source":"df[\"accents\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:24:33.717218Z","iopub.execute_input":"2022-08-12T21:24:33.717679Z","iopub.status.idle":"2022-08-12T21:24:33.734786Z","shell.execute_reply.started":"2022-08-12T21:24:33.717635Z","shell.execute_reply":"2022-08-12T21:24:33.733595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#I didn't include the words in Bengali since it resulted in squares on the piechart.","metadata":{}},{"cell_type":"code","source":"labels = 'normal begali clear accent', 'Bangladeshi', 'Narsingdi dialect', 'Native', 'Khati Bangali,Native speaker'\nsizes = [337, 149, 136, 79, 56]  #must have same number labels, sizes and explode\nexplode = (0, 0.2, 0, 0, 0)  # only \"explode\" the 2nd slice \n\nfig1, ax1 = plt.subplots(figsize=(10,10))\nax1.pie(sizes, explode=explode, labels=labels, autopct='%1.1f%%',\n        shadow=True, startangle=90)\nax1.axis('equal')  # Equal aspect ratio ensures that pie is drawn as a circle.\n\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-12T21:24:42.326883Z","iopub.execute_input":"2022-08-12T21:24:42.327336Z","iopub.status.idle":"2022-08-12T21:24:42.576639Z","shell.execute_reply.started":"2022-08-12T21:24:42.327291Z","shell.execute_reply":"2022-08-12T21:24:42.575510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.graph_objects as go\nimport plotly.offline as py\nimport plotly.express as px","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:32:22.635271Z","iopub.execute_input":"2022-08-12T21:32:22.635716Z","iopub.status.idle":"2022-08-12T21:32:23.821850Z","shell.execute_reply.started":"2022-08-12T21:32:22.635674Z","shell.execute_reply":"2022-08-12T21:32:23.820756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.bar(df[['gender','down_votes', 'accents']].sort_values('gender', ascending=False), \n                        y = \"down_votes\", x= \"gender\", color='accents', template='ggplot2')\nfig.update_xaxes(tickangle=45, tickfont=dict(family='Rockwell', color='crimson', size=14))\nfig.update_layout(title_text=\"Bengali speech Downvotes by Gender\")\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:48:45.020623Z","iopub.execute_input":"2022-08-12T21:48:45.021047Z","iopub.status.idle":"2022-08-12T21:48:46.348407Z","shell.execute_reply.started":"2022-08-12T21:48:45.021014Z","shell.execute_reply":"2022-08-12T21:48:46.346963Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Though I don't understand Bangla speech, that competition is a great opportunity to learn things in non-english language and much more.","metadata":{}},{"cell_type":"markdown","source":"#Acknowledgements:\n\nBrett Smith https://www.kaggle.com/bretts/plot-spectrogram-play-audio\n\nEBLICT Project, BCC, ICT Division, Bangladesh and IEEE Computer Society BUET Student Branch Chapter are jointly hosting the event with the organizers. The event is also supported by Incepta Solutions Limited.","metadata":{}},{"cell_type":"markdown","source":"![](https://cse.buet.ac.bd/assets/img/cse_flag.jpg)https://cse.buet.ac.bd/","metadata":{}}]}