{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        os.path.join(dirname, filename)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-14T12:49:26.685238Z","iopub.execute_input":"2023-04-14T12:49:26.685825Z","iopub.status.idle":"2023-04-14T12:49:31.775309Z","shell.execute_reply.started":"2023-04-14T12:49:26.685791Z","shell.execute_reply":"2023-04-14T12:49:31.773919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Going through data","metadata":{}},{"cell_type":"code","source":"#train_metatdata.csv\ntrain_metadata = pd.read_csv('/kaggle/input/birdclef-2023/train_metadata.csv')\ntrain_metadata.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:31.782065Z","iopub.execute_input":"2023-04-14T12:49:31.785523Z","iopub.status.idle":"2023-04-14T12:49:31.934591Z","shell.execute_reply.started":"2023-04-14T12:49:31.785485Z","shell.execute_reply":"2023-04-14T12:49:31.933437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Finding amount of missing data\nnull_values = train_metadata.isna().sum()\nprint(null_values)","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:31.938059Z","iopub.execute_input":"2023-04-14T12:49:31.938404Z","iopub.status.idle":"2023-04-14T12:49:31.952283Z","shell.execute_reply.started":"2023-04-14T12:49:31.938372Z","shell.execute_reply":"2023-04-14T12:49:31.951031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#adding full filepath column to metadata dataframe\n#This can help us to access files of audio easily \ntrain_metadata['filename'] = train_metadata['filename'].apply(lambda x : \"/kaggle/input/birdclef-2023/train_audio/\" + x)","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:31.955707Z","iopub.execute_input":"2023-04-14T12:49:31.956075Z","iopub.status.idle":"2023-04-14T12:49:31.969992Z","shell.execute_reply.started":"2023-04-14T12:49:31.956039Z","shell.execute_reply":"2023-04-14T12:49:31.96893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_metadata_2 = train_metadata.drop(['secondary_labels', 'type', 'latitude', 'longitude',\n       'scientific_name', 'common_name', 'author', 'license', 'rating', 'url'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:31.971933Z","iopub.execute_input":"2023-04-14T12:49:31.972402Z","iopub.status.idle":"2023-04-14T12:49:31.981766Z","shell.execute_reply.started":"2023-04-14T12:49:31.972368Z","shell.execute_reply":"2023-04-14T12:49:31.980684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Preprocessing","metadata":{}},{"cell_type":"code","source":"import librosa\nfrom scipy import signal\nfrom matplotlib import pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:31.983328Z","iopub.execute_input":"2023-04-14T12:49:31.984069Z","iopub.status.idle":"2023-04-14T12:49:32.35655Z","shell.execute_reply.started":"2023-04-14T12:49:31.984034Z","shell.execute_reply":"2023-04-14T12:49:32.355514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"__Librosa__ is powerful Python library built to work with audio and perform analysis on it. It is the starting point towards working with audio data at scale for a wide range of applications such as detecting voice from a person to finding personal characteristics from an audio.\n\n__SciPy__ is a scientific computation library that uses NumPy underneath.SciPy stands for Scientific Python.","metadata":{}},{"cell_type":"code","source":"#To convert an .ogg file to a waveform\ndef ogg_to_wave(filename):\n    ogg, sample_rate = librosa.load(filename)\n    int_16 = (ogg * 32767).astype(np.int16)\n    return int_16, sample_rate","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:32.357996Z","iopub.execute_input":"2023-04-14T12:49:32.358461Z","iopub.status.idle":"2023-04-14T12:49:32.365478Z","shell.execute_reply.started":"2023-04-14T12:49:32.358424Z","shell.execute_reply":"2023-04-14T12:49:32.364125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"__ogg__ is a NumPy array that contains audio samples with float values between -1 and 1, representing the _amplitude_ of the audio signal at different points in time.\n\n__(ogg * 32767)__ scales the audio samples so that they fall within the range of -32767 to 32767, which is the range of values that can be represented by a _16-bit_ signed integer.\n\n__astype(np.int16)__ converts the scaled audio samples to 16-bit signed integer data type. This conversion is necessary because most audio processing libraries and tools work with audio data represented as 16-bit integers rather than floating point values.\n\nThe resulting __int_16__ array contains the audio samples in the 16-bit integer format that can be used for further audio processing and analysis.","metadata":{}},{"cell_type":"code","source":"#Take in created waveform and create a spectogram matrix for neural network input layer\ndef wave_to_spec(waveform, sample_rate):\n    freq, time, spectrogram = signal.spectrogram(waveform, sample_rate, window='hann', nperseg=256, noverlap=128)\n    return freq, time, spectrogram","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:32.367039Z","iopub.execute_input":"2023-04-14T12:49:32.367529Z","iopub.status.idle":"2023-04-14T12:49:32.375461Z","shell.execute_reply.started":"2023-04-14T12:49:32.367494Z","shell.execute_reply":"2023-04-14T12:49:32.374404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"A __spectrogram__ is a 2D representation of a sound signal that shows how the frequency content of a signal changes over time. It is essentially a graph of the signal's frequency spectrum, with time on the x-axis, frequency on the y-axis, and the magnitude of each frequency component represented by a color or brightness value.\n\nA spectrogram matrix is a matrix representation of a spectrogram, where each element of the matrix represents the amplitude or energy of a frequency component at a particular time. Spectrogram matrices are commonly used as input data for machine learning models that process audio signals.\n\nThe function uses the __spectrogram()__ function from the __scipy.signal__ library to calculate the spectrogram. The spectrogram() function calculates the short-term Fourier transform of the waveform and returns the frequency, time, and spectrogram arrays.\n\nThe __window__ parameter specifies the type of window function to be used. Here, the __\"hann\"__ window is used. The __nperseg__ parameter specifies the length of each segment of the signal to be used for calculating the _Fourier transform_. The __noverlap__ parameter specifies the number of samples of overlap between the segments.\n\nThe function returns the frequency, time, and spectrogram arrays. The spectrogram array is a 2D matrix that contains the magnitude of the frequency components at each time segment of the waveform. It is commonly used as an input to neural networks for audio processing tasks such as speech recognition or music genre classification.","metadata":{}},{"cell_type":"code","source":"#function assembly for preprocessing data\ndef audio_preprocessing(filepath):\n    waveform, sample_rate = ogg_to_wave(filepath)\n    freq, time, spectrogram = wave_to_spec(waveform, sample_rate)\n    return freq, time, spectrogram","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:32.377489Z","iopub.execute_input":"2023-04-14T12:49:32.377987Z","iopub.status.idle":"2023-04-14T12:49:32.384454Z","shell.execute_reply.started":"2023-04-14T12:49:32.377953Z","shell.execute_reply":"2023-04-14T12:49:32.383416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Till now, we created a pipeline that takes filepath and return its spectrogram","metadata":{}},{"cell_type":"code","source":"test = train_metadata_2.drop_duplicates(subset=[\"primary_label\"], keep=False)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:32.389278Z","iopub.execute_input":"2023-04-14T12:49:32.389618Z","iopub.status.idle":"2023-04-14T12:49:32.404704Z","shell.execute_reply.started":"2023-04-14T12:49:32.389593Z","shell.execute_reply":"2023-04-14T12:49:32.403624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#replacing filename with spectrogram representation\ntest['spectrogram'] = test['filename'].apply(lambda x : audio_preprocessing(x)[2])\ntest.drop('filename', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:32.406919Z","iopub.execute_input":"2023-04-14T12:49:32.4076Z","iopub.status.idle":"2023-04-14T12:49:41.005049Z","shell.execute_reply.started":"2023-04-14T12:49:32.407567Z","shell.execute_reply":"2023-04-14T12:49:41.003846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The code __metadata.drop_duplicates(subset=[\"primary_label\"], keep=False)__ drops duplicate rows from the metadata DataFrame based on the \"primary_label\" column.\n\nThe __keep=False__ parameter indicates that all duplicates should be dropped, rather than keeping the first or last occurrence.","metadata":{}},{"cell_type":"code","source":"test.reset_index(inplace=True)\n\n#one-hot encoding species column\nspecies = pd.get_dummies(test['primary_label']).astype('float64')\n\n#dropping primary_label column (will be replaced with OH encoded columns)\ntest = test.drop(['primary_label'], axis=1)\ntest = test.join(species)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:41.006707Z","iopub.execute_input":"2023-04-14T12:49:41.007958Z","iopub.status.idle":"2023-04-14T12:49:41.497538Z","shell.execute_reply.started":"2023-04-14T12:49:41.007908Z","shell.execute_reply":"2023-04-14T12:49:41.496586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Combine all functions to create a single function 'preprocessing'","metadata":{}},{"cell_type":"code","source":"#will take in .csv filepath (formatting expected to reflect that described earlier)\ndef preprocessing(data):\n    \n    #creating dataframe\n    df = pd.read_csv(data).head(5)\n    \n    #creating spectrogram column\n    file_prefix = \"/kaggle/input/birdclef-2023/train_audio/\"\n    df['input'] = df['filename'].apply(lambda x : audio_preprocessing(file_prefix + x)[2])\n    \n    #dropping other columns\n    df = df[['input', 'primary_label']]\n    \n    #one-hot encoding primary_label column\n    species = pd.get_dummies(df['primary_label']).astype('float64')\n    df = df.drop('primary_label', axis=1)\n    df = df.join(species)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:41.500651Z","iopub.execute_input":"2023-04-14T12:49:41.500946Z","iopub.status.idle":"2023-04-14T12:49:41.507377Z","shell.execute_reply.started":"2023-04-14T12:49:41.500919Z","shell.execute_reply":"2023-04-14T12:49:41.506183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Creation of array for audiofile","metadata":{}},{"cell_type":"code","source":"# array that will store audiofile objects\naudio_files = []\n\n# audiofile class\nclass AudioFile:\n    def __init__(self, filename):\n        self.filename = filename\n        self.freqs, self.time, self.spectrogram = audio_preprocessing(filename)\n        \n    # method to output objects spectrogram to terminal\n    def plot(self):\n        plt.pcolormesh(self.time, self.freqs, 10*np.log10(self.spectrogram), cmap='viridis')\n        plt.xlabel(\"Time(s)\")\n        plt.ylabel(\"Frequency (Hz)\")\n        plt.title(f\"Spectrogram for {self.filename[40:-1]}\")\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:41.508975Z","iopub.execute_input":"2023-04-14T12:49:41.509626Z","iopub.status.idle":"2023-04-14T12:49:41.518102Z","shell.execute_reply.started":"2023-04-14T12:49:41.509588Z","shell.execute_reply":"2023-04-14T12:49:41.517048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Loading in small subset of training metadata\ntrain_metadata_3 = pd.read_csv('/kaggle/input/birdclef-2023/train_metadata.csv')\ntrain_metadata_3 = train_metadata_3.head(10)\ntrain_metadata_3.info() ","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:41.519599Z","iopub.execute_input":"2023-04-14T12:49:41.520077Z","iopub.status.idle":"2023-04-14T12:49:41.586418Z","shell.execute_reply.started":"2023-04-14T12:49:41.520043Z","shell.execute_reply":"2023-04-14T12:49:41.585277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Looping through dataframe and creating audio objects tied to audio in 'filename' column\nfor key, value in train_metadata_3['filename'].iteritems():\n    audio_files.append(AudioFile(\"/kaggle/input/birdclef-2023/train_audio/\" + value))","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:41.588275Z","iopub.execute_input":"2023-04-14T12:49:41.588644Z","iopub.status.idle":"2023-04-14T12:49:42.318833Z","shell.execute_reply.started":"2023-04-14T12:49:41.588606Z","shell.execute_reply":"2023-04-14T12:49:42.317693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plotting out all spectrograms in array\nfor file in audio_files:\n    file.plot()","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:42.321059Z","iopub.execute_input":"2023-04-14T12:49:42.321865Z","iopub.status.idle":"2023-04-14T12:49:48.293014Z","shell.execute_reply.started":"2023-04-14T12:49:42.321823Z","shell.execute_reply":"2023-04-14T12:49:48.292087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now, we created a pipeline to convert all .ogg files into spectrogram. It can be useful to train and test audio files by converting them into class objects.","metadata":{}},{"cell_type":"code","source":"#filepath\n#Every row is being converted into tuple\nX = np.asarray(train_metadata['filename'])\n\n#One-hot encoding\ny = np.asarray(pd.get_dummies(train_metadata['primary_label']))","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:48.294721Z","iopub.execute_input":"2023-04-14T12:49:48.295395Z","iopub.status.idle":"2023-04-14T12:49:48.310563Z","shell.execute_reply.started":"2023-04-14T12:49:48.295359Z","shell.execute_reply":"2023-04-14T12:49:48.309581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#This function will take in a tuple (input, output) and return the same \n#(intput, output) tuple,but the input will have been coverted from a \n#filepath to a tensorflow waveform object.\n\ndef load_audio_file(file_path):\n    # Read audio file contents into a tensor\n    audio_binary = tf.io.read_file(file_path)\n    \n    # Decode audio binary into a numpy array\n    audio, sr = librosa.load(io.BytesIO(audio_binary.numpy()), sr=16000, mono=True)\n    \n    audio = audio * 32768.0\n    audio = tf.cast(audio, tf.int16)\n\n    return audio","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:48.311867Z","iopub.execute_input":"2023-04-14T12:49:48.312718Z","iopub.status.idle":"2023-04-14T12:49:48.319322Z","shell.execute_reply.started":"2023-04-14T12:49:48.312682Z","shell.execute_reply":"2023-04-14T12:49:48.318275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The __sr__ parameter is used to specify the desired sampling rate of the output waveform, which is set to _16000_ samples per second in this case. \n\nThe __mono__ parameter is set to _True_, which means that the output waveform will be converted to mono by averaging the samples across all channels, if the input audio has multiple channels.\n\nThe __BytesIO__ function from the io library is used to convert the binary data (audio_binary) into a stream of bytes that can be read by librosa.load. \n\nThe numpy() method is used to convert the binary data (audio_binary) from a TensorFlow tensor to a NumPy array, which is required by BytesIO.","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport io","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:48.320703Z","iopub.execute_input":"2023-04-14T12:49:48.321144Z","iopub.status.idle":"2023-04-14T12:49:55.121179Z","shell.execute_reply.started":"2023-04-14T12:49:48.321093Z","shell.execute_reply":"2023-04-14T12:49:55.120037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess(filepath, labels):\n    \n    # Loading in audio\n    waveform = tf.py_function(load_audio_file, [filepath], tf.int16)\n    \n    # Normalize waveform to [-1, 1]\n    waveform = tf.cast(waveform, tf.float32) / 32768.0\n    \n    # Getting first 3 seconds of clip\n    waveform = waveform[:48000]\n    \n    # Add zero padding\n    zero_padding = tf.zeros([48000] - tf.shape(waveform), dtype=tf.float32)\n    wav = tf.concat([zero_padding, waveform], 0)\n    \n    # Spectrogram\n    stft = tf.signal.stft(wav, frame_length=320, frame_step=32)\n    spectrogram = tf.abs(stft)\n    spectrogram = tf.expand_dims(spectrogram, axis=2)\n    \n    return spectrogram, labels","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:55.122801Z","iopub.execute_input":"2023-04-14T12:49:55.123544Z","iopub.status.idle":"2023-04-14T12:49:55.133292Z","shell.execute_reply.started":"2023-04-14T12:49:55.123514Z","shell.execute_reply":"2023-04-14T12:49:55.132188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\n\n# Generating random index to test\ntest_index = random.randint(0,len(X))\ntest_index","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:55.135128Z","iopub.execute_input":"2023-04-14T12:49:55.135672Z","iopub.status.idle":"2023-04-14T12:49:55.146618Z","shell.execute_reply.started":"2023-04-14T12:49:55.135634Z","shell.execute_reply":"2023-04-14T12:49:55.145125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Should return spectrogram and numpy array of classes\nX_function_test, y_function_test = preprocess(X[test_index], y[test_index])","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:55.149325Z","iopub.execute_input":"2023-04-14T12:49:55.150009Z","iopub.status.idle":"2023-04-14T12:49:57.665086Z","shell.execute_reply.started":"2023-04-14T12:49:55.149972Z","shell.execute_reply":"2023-04-14T12:49:57.664009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Spectrogram shape: {X_function_test.shape}')\nprint(f'Spectrogram type: {type(X_function_test)}')\nprint(f'Classes shape: {y_function_test.shape}')\nprint(f'Classes type: {type(y_function_test)}')","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:57.668208Z","iopub.execute_input":"2023-04-14T12:49:57.668884Z","iopub.status.idle":"2023-04-14T12:49:57.675521Z","shell.execute_reply.started":"2023-04-14T12:49:57.668843Z","shell.execute_reply":"2023-04-14T12:49:57.674261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Plot spectrogram","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(5, 5))\nplt.imshow(tf.transpose(X_function_test)[0], aspect='auto', origin='lower', cmap='viridis')\nplt.colorbar()\nplt.xlabel('Time')\nplt.ylabel('Frequency')\nplt.title('Bird Call Spectrogram (Downsampled to 16 kHz)')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:57.677321Z","iopub.execute_input":"2023-04-14T12:49:57.677771Z","iopub.status.idle":"2023-04-14T12:49:58.030087Z","shell.execute_reply.started":"2023-04-14T12:49:57.677733Z","shell.execute_reply":"2023-04-14T12:49:58.029122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Transform the data","metadata":{}},{"cell_type":"code","source":"# Creating dataset from 'train_metadata.csv' data\ndata = tf.data.Dataset.from_tensor_slices((X, y))","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:58.031517Z","iopub.execute_input":"2023-04-14T12:49:58.032587Z","iopub.status.idle":"2023-04-14T12:49:58.060455Z","shell.execute_reply.started":"2023-04-14T12:49:58.032551Z","shell.execute_reply":"2023-04-14T12:49:58.059511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Full data pipeline\ndata = data.map(preprocess, num_parallel_calls=tf.data.AUTOTUNE)\ndata = data.cache()\ndata = data.shuffle(buffer_size=50)\ndata = data.batch(batch_size=16)\ndata = data.prefetch(8)","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:58.06279Z","iopub.execute_input":"2023-04-14T12:49:58.063545Z","iopub.status.idle":"2023-04-14T12:49:58.260979Z","shell.execute_reply.started":"2023-04-14T12:49:58.063508Z","shell.execute_reply":"2023-04-14T12:49:58.259793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculating batch numbers for train and test\nbatches = len(data)\nprint(batches)\nprint(batches * (0.8))\nprint(batches - (batches * (0.8)))","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:58.266652Z","iopub.execute_input":"2023-04-14T12:49:58.26692Z","iopub.status.idle":"2023-04-14T12:49:58.27461Z","shell.execute_reply.started":"2023-04-14T12:49:58.266895Z","shell.execute_reply":"2023-04-14T12:49:58.273397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 847 Training batches\n# 212 Test batches\ntrain_1 = data.take(100)\n#train_2 = data.skip(200).take(200)\n#train_3 = data.skip(400).take(200)\ntest = data.skip(845).take(50)","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:58.276008Z","iopub.execute_input":"2023-04-14T12:49:58.277178Z","iopub.status.idle":"2023-04-14T12:49:58.287045Z","shell.execute_reply.started":"2023-04-14T12:49:58.277132Z","shell.execute_reply":"2023-04-14T12:49:58.285512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"verify data generator","metadata":{}},{"cell_type":"code","source":"# Should output ([batch_size], [element dimension]) for both input and output\nspectrograms, labels = train_1.as_numpy_iterator().next()\n\n# 16 examples of spectrograms\nspectrograms.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:49:58.288485Z","iopub.execute_input":"2023-04-14T12:49:58.288891Z","iopub.status.idle":"2023-04-14T12:50:02.052731Z","shell.execute_reply.started":"2023-04-14T12:49:58.288858Z","shell.execute_reply":"2023-04-14T12:50:02.051738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 16 examples of species label data\nlabels.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:50:02.05436Z","iopub.execute_input":"2023-04-14T12:50:02.054731Z","iopub.status.idle":"2023-04-14T12:50:02.063472Z","shell.execute_reply.started":"2023-04-14T12:50:02.054694Z","shell.execute_reply":"2023-04-14T12:50:02.062363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model Creation","metadata":{}},{"cell_type":"code","source":"# TensorFlow CNN Model Dependencies\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, Flatten, Dense, MaxPooling2D","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:50:02.064933Z","iopub.execute_input":"2023-04-14T12:50:02.065789Z","iopub.status.idle":"2023-04-14T12:50:02.076282Z","shell.execute_reply.started":"2023-04-14T12:50:02.065655Z","shell.execute_reply":"2023-04-14T12:50:02.075267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cnn_model.add(LeakyReLU(alpha=0.1))\nmodel = Sequential()\nmodel.add(Conv2D(16, (3, 3), activation='relu', input_shape=(1491, 257, 1)))\nmodel.add(MaxPooling2D(2, 2))\nmodel.add(Conv2D(8, (3, 3), activation='relu'))\nmodel.add(MaxPooling2D(2, 2))\nmodel.add(Flatten())\nmodel.add(Dense(264, activation='relu'))\nmodel.add(Dense(264, activation='softmax'))","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:50:02.078869Z","iopub.execute_input":"2023-04-14T12:50:02.0801Z","iopub.status.idle":"2023-04-14T12:50:02.170053Z","shell.execute_reply.started":"2023-04-14T12:50:02.080063Z","shell.execute_reply":"2023-04-14T12:50:02.169157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:50:02.171433Z","iopub.execute_input":"2023-04-14T12:50:02.17179Z","iopub.status.idle":"2023-04-14T12:50:02.196368Z","shell.execute_reply.started":"2023-04-14T12:50:02.171755Z","shell.execute_reply":"2023-04-14T12:50:02.19561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile('Adam', loss='CategoricalCrossentropy',\n              metrics=[tf.keras.metrics.CategoricalAccuracy()])","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:50:02.197325Z","iopub.execute_input":"2023-04-14T12:50:02.197719Z","iopub.status.idle":"2023-04-14T12:50:02.226922Z","shell.execute_reply.started":"2023-04-14T12:50:02.197685Z","shell.execute_reply":"2023-04-14T12:50:02.225875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''from tensorflow.keras.callbacks import EarlyStopping\n\n#Stops model if it is not steadily dropping loss val_loss function return\nearlyStop = EarlyStopping(monitor='val_loss', patience=3, restore_best_weights=True)'''","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:50:02.22819Z","iopub.execute_input":"2023-04-14T12:50:02.229149Z","iopub.status.idle":"2023-04-14T12:50:02.235788Z","shell.execute_reply.started":"2023-04-14T12:50:02.22909Z","shell.execute_reply":"2023-04-14T12:50:02.234666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fitting model to train_1\nhistory_train_1 = model.fit(train_1, epochs=10, \n                        validation_data=test)","metadata":{"execution":{"iopub.status.busy":"2023-04-14T12:50:02.237143Z","iopub.execute_input":"2023-04-14T12:50:02.238087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}