{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Notebook Goal\nThe next step for me is to develop a pipeline that will convert the data given in the model testing to a usable format for a convolutional neural network. The most overt hurdle within that is changing a .ogg file into a spectrogram. This notebook will serve as a testing ground for my solution.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"### Loading in an Audio Recording","metadata":{}},{"cell_type":"code","source":"# loading in dependancies\nimport os\nimport tensorflow_io as tfio\nimport tensorflow as tf\nimport numpy as np\nimport pandas as pd\nfrom matplotlib import pyplot as plt\nimport librosa","metadata":{"execution":{"iopub.status.busy":"2023-03-08T18:58:45.236845Z","iopub.execute_input":"2023-03-08T18:58:45.237291Z","iopub.status.idle":"2023-03-08T18:58:45.244444Z","shell.execute_reply.started":"2023-03-08T18:58:45.237250Z","shell.execute_reply":"2023-03-08T18:58:45.242302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# declaring audio var & loading into memory\ntest_file = os.path.join('/kaggle', 'input', 'birdclef-2023', 'train_audio', 'abethr1', 'XC128013.ogg')\ntest_file # verifying results","metadata":{"execution":{"iopub.status.busy":"2023-03-08T18:58:45.248306Z","iopub.execute_input":"2023-03-08T18:58:45.249186Z","iopub.status.idle":"2023-03-08T18:58:45.262301Z","shell.execute_reply.started":"2023-03-08T18:58:45.249133Z","shell.execute_reply":"2023-03-08T18:58:45.260759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Building a Dataloading Function","metadata":{}},{"cell_type":"code","source":"# will convert an .ogg file to a waveform\ndef ogg_to_wave(filename):\n    ogg, sample_rate = librosa.load(filename)\n    int16 = (ogg * 32767).astype(np.int16)\n    return int16, sample_rate","metadata":{"execution":{"iopub.status.busy":"2023-03-08T18:58:45.264104Z","iopub.execute_input":"2023-03-08T18:58:45.265845Z","iopub.status.idle":"2023-03-08T18:58:45.277987Z","shell.execute_reply.started":"2023-03-08T18:58:45.265773Z","shell.execute_reply":"2023-03-08T18:58:45.276879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plotting out waveform\nwave_test, wave_test_sample_rate = ogg_to_wave(test_file)\nplt.plot(wave_test)","metadata":{"execution":{"iopub.status.busy":"2023-03-08T18:58:45.279286Z","iopub.execute_input":"2023-03-08T18:58:45.279660Z","iopub.status.idle":"2023-03-08T18:58:45.877759Z","shell.execute_reply.started":"2023-03-08T18:58:45.279618Z","shell.execute_reply":"2023-03-08T18:58:45.876379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Wave form has {len(wave_test)} points.\")\nprint(f\"Wave form sample rate is {wave_test_sample_rate}Hz.\")","metadata":{"execution":{"iopub.status.busy":"2023-03-08T18:58:45.880837Z","iopub.execute_input":"2023-03-08T18:58:45.881296Z","iopub.status.idle":"2023-03-08T18:58:45.887619Z","shell.execute_reply.started":"2023-03-08T18:58:45.881258Z","shell.execute_reply":"2023-03-08T18:58:45.886147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy import signal\n#will take in created waveform and create a spectogram matrix for neural network input layer\ndef wave_to_spec(waveform, sample_rate):\n    freqs, time, spectrogram = signal.spectrogram(waveform, sample_rate, window='hann', nperseg=256, noverlap=128)\n    return freqs, time, spectrogram","metadata":{"execution":{"iopub.status.busy":"2023-03-08T18:58:45.888923Z","iopub.execute_input":"2023-03-08T18:58:45.889236Z","iopub.status.idle":"2023-03-08T18:58:45.901513Z","shell.execute_reply.started":"2023-03-08T18:58:45.889206Z","shell.execute_reply":"2023-03-08T18:58:45.900178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#creating spectrogram\nfrequencies, times, spectrogram = wave_to_spec(wave_test, wave_test_sample_rate)","metadata":{"execution":{"iopub.status.busy":"2023-03-08T18:58:45.903835Z","iopub.execute_input":"2023-03-08T18:58:45.904352Z","iopub.status.idle":"2023-03-08T18:58:45.970422Z","shell.execute_reply.started":"2023-03-08T18:58:45.904299Z","shell.execute_reply":"2023-03-08T18:58:45.969024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.pcolormesh(times, frequencies, 10*np.log10(spectrogram), cmap='viridis')\nplt.xlabel(\"Time(s)\")\nplt.ylabel(\"Frequency (Hz)\")\nplt.title(\"Audio Spectrogram Representation\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-08T18:58:45.972441Z","iopub.execute_input":"2023-03-08T18:58:45.974553Z","iopub.status.idle":"2023-03-08T18:58:46.865775Z","shell.execute_reply.started":"2023-03-08T18:58:45.974494Z","shell.execute_reply":"2023-03-08T18:58:46.864358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#function assembly for preprocessing data\ndef audio_preprocessing(filepath):\n    waveform, sample_rate = ogg_to_wave(filepath)\n    freqs, time, spectrogram = wave_to_spec(waveform, sample_rate)\n    return freqs, time, spectrogram","metadata":{"execution":{"iopub.status.busy":"2023-03-08T18:58:46.867788Z","iopub.execute_input":"2023-03-08T18:58:46.868250Z","iopub.status.idle":"2023-03-08T18:58:46.876188Z","shell.execute_reply.started":"2023-03-08T18:58:46.868196Z","shell.execute_reply":"2023-03-08T18:58:46.874185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading in Full Dataset\nFor the purposes of this notebook, I'm going to populate an array of 'audiofile' objects to test a start-to-finish audio preprocessing pipeline that will only need slight tweaking during actual project development.","metadata":{}},{"cell_type":"code","source":"# array that will store audiofile objects\naudio_files = []","metadata":{"execution":{"iopub.status.busy":"2023-03-08T18:58:46.877673Z","iopub.execute_input":"2023-03-08T18:58:46.878014Z","iopub.status.idle":"2023-03-08T18:58:46.886555Z","shell.execute_reply.started":"2023-03-08T18:58:46.877981Z","shell.execute_reply":"2023-03-08T18:58:46.885338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# audiofile class\nclass AudioFile:\n    def __init__(self, filename):\n        self.filename = filename\n        self.freqs, self.time, self.spectrogram = audio_preprocessing(filename)\n    # method to output objects spectrogram to terminal\n    def plot(self):\n        plt.pcolormesh(self.time, self.freqs, 10*np.log10(self.spectrogram), cmap='viridis')\n        plt.xlabel(\"Time(s)\")\n        plt.ylabel(\"Frequency (Hz)\")\n        plt.title(f\"Spectrogram for {self.filename[40:-1]}\")\n        plt.show()\n        ","metadata":{"execution":{"iopub.status.busy":"2023-03-08T18:58:46.888233Z","iopub.execute_input":"2023-03-08T18:58:46.888701Z","iopub.status.idle":"2023-03-08T18:58:46.899046Z","shell.execute_reply.started":"2023-03-08T18:58:46.888664Z","shell.execute_reply":"2023-03-08T18:58:46.897826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# loading in small subset of training metadata\ntrain_metadata = pd.read_csv('/kaggle/input/birdclef-2023/train_metadata.csv')\ntrain_metadata = train_metadata.head(10)\ntrain_metadata.info()","metadata":{"execution":{"iopub.status.busy":"2023-03-08T18:58:46.900294Z","iopub.execute_input":"2023-03-08T18:58:46.900651Z","iopub.status.idle":"2023-03-08T18:58:46.985880Z","shell.execute_reply.started":"2023-03-08T18:58:46.900618Z","shell.execute_reply":"2023-03-08T18:58:46.984608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# looping through dataframe and creating audio objects tied to audio in 'filename' column\nfor key, value in train_metadata['filename'].iteritems():\n    audio_files.append(AudioFile(\"/kaggle/input/birdclef-2023/train_audio/\" + value))","metadata":{"execution":{"iopub.status.busy":"2023-03-08T18:58:46.989379Z","iopub.execute_input":"2023-03-08T18:58:46.989733Z","iopub.status.idle":"2023-03-08T18:58:47.766199Z","shell.execute_reply.started":"2023-03-08T18:58:46.989698Z","shell.execute_reply":"2023-03-08T18:58:47.764889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plotting out all spectrograms in array\nfor file in audio_files:\n    file.plot()","metadata":{"execution":{"iopub.status.busy":"2023-03-08T18:58:47.768090Z","iopub.execute_input":"2023-03-08T18:58:47.768566Z","iopub.status.idle":"2023-03-08T18:58:54.193641Z","shell.execute_reply.started":"2023-03-08T18:58:47.768516Z","shell.execute_reply":"2023-03-08T18:58:54.192475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Conclusion\nI now have a working data preprocessing pipline from '.ogg' -> spectrogram for the neural network. This pipeline will be used to convert training & testing audio files (coupled with their bird species metadata) into useable 'objects'.\n\nWhat did you think of the notebook? Is there anything you would have changed? What things did you enjoy?\n\nI hope you enjoyed! As always if you have any suggestions on how to improve my own process(es) do not hesitate to comment!\n\nCheers","metadata":{}}]}