{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":256618,"sourceType":"datasetVersion","datasetId":107620}],"dockerImageVersionId":30162,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 🔊 Working with Audio in Python\n\n<img src=\"https://miro.medium.com/max/1100/1*Zx9QAMPzxhama9O4q9xWXg.jpeg\" width=\"600\"/>\n\nThis notebook is intended to be an introduction for anyone interested in using python to interperate audio data.\n\nPlease watch the youtube video that discusses the contents of this notebook if you want to learn more!\n- [Video Link](https://www.youtube.com/watch?v=ZqpSb5p1xQo)\n- [Youtube Channel](https://www.youtube.com/channel/UCxladMszXan-jfgzyeIMyvw)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pylab as plt\nimport seaborn as sns\n\nfrom glob import glob\n\nimport librosa\nimport librosa.display\nimport IPython.display as ipd\n\nfrom itertools import cycle\n\nsns.set_theme(style=\"white\", palette=None)\ncolor_pal = plt.rcParams[\"axes.prop_cycle\"].by_key()[\"color\"]\ncolor_cycle = cycle(plt.rcParams[\"axes.prop_cycle\"].by_key()[\"color\"])","metadata":{"execution":{"iopub.status.busy":"2024-05-11T13:17:04.047928Z","iopub.execute_input":"2024-05-11T13:17:04.048232Z","iopub.status.idle":"2024-05-11T13:17:04.056652Z","shell.execute_reply.started":"2024-05-11T13:17:04.048199Z","shell.execute_reply":"2024-05-11T13:17:04.055666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Terms to know for Audio in Digital Form:\n\n## Frequency (Hz)\n- Frequency describes the differences of wave lengths.\n- We interperate frequency has high and low pitches.\n\n<img src=\"https://uploads-cdn.omnicalculator.com/images/britannica-wave-frequency.jpg\" width=\"400\"/>\n\n## Intensity (db / power)\n- Intensity describes the amplitude (height) of the wave.\n\n<img src=\"https://ars.els-cdn.com/content/image/3-s2.0-B9780124722804500162-f13-15-9780124722804.gif\" width=\"400\"/>\n\n## Sample Rate\n- Sample rate is specific to how the computer reads in the audio file.\n- Think of it as the \"resolution\" of the audio.\n\n<img src=\"https://www.headphonesty.com/wp-content/uploads/2019/07/Sample-Rate-Bit-Depth-and-Bit-Rate.jpeg\" width=\"400\"/>\n","metadata":{}},{"cell_type":"markdown","source":"# Reading in Audio Files\nThere are many types of audio files: `mp3`, `wav`, `m4a`, `flac`, `ogg`","metadata":{}},{"cell_type":"code","source":"audio_files = glob('../input/ravdess-emotional-speech-audio/*/*.wav')","metadata":{"execution":{"iopub.status.busy":"2024-05-11T13:18:12.960381Z","iopub.execute_input":"2024-05-11T13:18:12.960737Z","iopub.status.idle":"2024-05-11T13:18:12.985822Z","shell.execute_reply.started":"2024-05-11T13:18:12.960699Z","shell.execute_reply":"2024-05-11T13:18:12.984698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Play audio file\nipd.Audio(audio_files[0])","metadata":{"execution":{"iopub.status.busy":"2024-05-11T13:19:30.268648Z","iopub.execute_input":"2024-05-11T13:19:30.269025Z","iopub.status.idle":"2024-05-11T13:19:30.301966Z","shell.execute_reply.started":"2024-05-11T13:19:30.268956Z","shell.execute_reply":"2024-05-11T13:19:30.301088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y, sr = librosa.load(audio_files[0])     #y = sample(raw data of the audio file) and sr = sampling rate\nprint(f'y: {y[:10]}') # so the audio file is just a large numpy array\nprint(f'shape y: {y.shape}')\nprint(f'sr: {sr}')","metadata":{"execution":{"iopub.status.busy":"2024-05-11T13:20:41.838163Z","iopub.execute_input":"2024-05-11T13:20:41.838551Z","iopub.status.idle":"2024-05-11T13:20:42.855579Z","shell.execute_reply.started":"2024-05-11T13:20:41.838510Z","shell.execute_reply":"2024-05-11T13:20:42.854442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.Series(y).plot(figsize=(10, 5),\n                  lw=1,\n                  title='Raw Audio Example',\n                 color=color_pal[0])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-11T13:23:03.223776Z","iopub.execute_input":"2024-05-11T13:23:03.224145Z","iopub.status.idle":"2024-05-11T13:23:03.629188Z","shell.execute_reply.started":"2024-05-11T13:23:03.224108Z","shell.execute_reply":"2024-05-11T13:23:03.628111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Trimming leading/lagging silence\ny_trimmed, _ = librosa.effects.trim(y, top_db=20) #20 is the top db that it trims out, default is 60dB\n\npd.Series(y_trimmed).plot(figsize=(10, 5),\n                  lw=1,\n                  title='Raw Audio Trimmed Example',\n                 color=color_pal[1])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-11T13:24:51.642465Z","iopub.execute_input":"2024-05-11T13:24:51.642949Z","iopub.status.idle":"2024-05-11T13:24:52.126805Z","shell.execute_reply.started":"2024-05-11T13:24:51.642915Z","shell.execute_reply":"2024-05-11T13:24:52.125918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#for zooming in\npd.Series(y[30000:30500]).plot(figsize=(10, 5),\n                  lw=1,\n                  title='Raw Audio Zoomed In Example',\n                 color=color_pal[2])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-11T13:26:18.409932Z","iopub.execute_input":"2024-05-11T13:26:18.410342Z","iopub.status.idle":"2024-05-11T13:26:18.686878Z","shell.execute_reply.started":"2024-05-11T13:26:18.410294Z","shell.execute_reply":"2024-05-11T13:26:18.686033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Spectogram","metadata":{}},{"cell_type":"code","source":"D = librosa.stft(y)      #short time fourier transform\nS_db = librosa.amplitude_to_db(np.abs(D), ref=np.max) #convert amplitudes to decibels\nS_db.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-11T13:28:16.705760Z","iopub.execute_input":"2024-05-11T13:28:16.706545Z","iopub.status.idle":"2024-05-11T13:28:16.722562Z","shell.execute_reply.started":"2024-05-11T13:28:16.706497Z","shell.execute_reply":"2024-05-11T13:28:16.721451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the transformed audio data (this is the sort of data u can feed in a ML model)\nfig, ax = plt.subplots(figsize=(10, 5))\nimg = librosa.display.specshow(S_db,\n                              x_axis='time',\n                              y_axis='log',\n                              ax=ax)\nax.set_title('Spectogram Example', fontsize=20)\nfig.colorbar(img, ax=ax, format=f'%0.2f')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-11T13:29:18.128162Z","iopub.execute_input":"2024-05-11T13:29:18.128637Z","iopub.status.idle":"2024-05-11T13:29:18.639716Z","shell.execute_reply.started":"2024-05-11T13:29:18.128603Z","shell.execute_reply":"2024-05-11T13:29:18.638851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Mel Spectogram","metadata":{}},{"cell_type":"markdown","source":"A mel spectrogram (melodic). we will use this to express the frequencies that we can hear in audio","metadata":{}},{"cell_type":"code","source":"S = librosa.feature.melspectrogram(y=y,\n                                   sr=sr,\n                                   n_mels=128 * 2,)\nS_db_mel = librosa.amplitude_to_db(S, ref=np.max)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T13:31:36.892480Z","iopub.execute_input":"2024-05-11T13:31:36.892877Z","iopub.status.idle":"2024-05-11T13:31:36.918620Z","shell.execute_reply.started":"2024-05-11T13:31:36.892828Z","shell.execute_reply":"2024-05-11T13:31:36.917910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(10, 5))\n# Plot the mel spectogram\nimg = librosa.display.specshow(S_db_mel,\n                              x_axis='time',\n                              y_axis='log',\n                              ax=ax)\nax.set_title('Mel Spectogram Example', fontsize=20)\nfig.colorbar(img, ax=ax, format=f'%0.2f')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-11T13:33:06.674849Z","iopub.execute_input":"2024-05-11T13:33:06.675616Z","iopub.status.idle":"2024-05-11T13:33:07.108809Z","shell.execute_reply.started":"2024-05-11T13:33:06.675573Z","shell.execute_reply":"2024-05-11T13:33:07.107803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}