{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-23T06:25:12.090946Z","iopub.execute_input":"2023-05-23T06:25:12.091736Z","iopub.status.idle":"2023-05-23T06:25:12.550191Z","shell.execute_reply.started":"2023-05-23T06:25:12.091691Z","shell.execute_reply":"2023-05-23T06:25:12.549110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Going through data","metadata":{}},{"cell_type":"code","source":"#train_metatdata.csv\nimport pandas as pd\nimport numpy as np\ntrain_metadata = pd.read_csv('/kaggle/input/birdclef-2023/train_metadata.csv')\n\ntrain_metadata.head()\ntrain_metadata.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:25:12.552511Z","iopub.execute_input":"2023-05-23T06:25:12.552894Z","iopub.status.idle":"2023-05-23T06:25:12.624695Z","shell.execute_reply.started":"2023-05-23T06:25:12.552853Z","shell.execute_reply":"2023-05-23T06:25:12.623042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"groups = train_metadata.groupby('primary_label')\n\n# Initialize an empty list to store the sampled data\nsampled_data = []\n\n# Set the number of samples to select from each group\nsamples_per_group = 1\n\n# Sample the data from each group\nfor _, group in groups:\n    sampled_group = group.sample(samples_per_group)\n    sampled_data.append(sampled_group)\n\n# Concatenate the sampled data from each group\nsampled_data = pd.concat(sampled_data)\nsampled_data.shape\ntrain_metadata=sampled_data\n","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:25:12.626162Z","iopub.execute_input":"2023-05-23T06:25:12.626634Z","iopub.status.idle":"2023-05-23T06:25:12.791378Z","shell.execute_reply.started":"2023-05-23T06:25:12.626593Z","shell.execute_reply":"2023-05-23T06:25:12.790292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Finding amount of missing data\nnull_values = train_metadata.isna().sum()\nprint(null_values)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:25:12.794552Z","iopub.execute_input":"2023-05-23T06:25:12.794952Z","iopub.status.idle":"2023-05-23T06:25:12.804188Z","shell.execute_reply.started":"2023-05-23T06:25:12.794911Z","shell.execute_reply":"2023-05-23T06:25:12.802992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# No. of species present in dataset\nimport matplotlib.pyplot as plt\nspecies_frequency = train_metadata['primary_label'].value_counts()\nprint(len(species_frequency.values))\nplt.bar(species_frequency.index, species_frequency.values)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:25:12.806001Z","iopub.execute_input":"2023-05-23T06:25:12.807249Z","iopub.status.idle":"2023-05-23T06:26:05.243102Z","shell.execute_reply.started":"2023-05-23T06:25:12.807204Z","shell.execute_reply":"2023-05-23T06:26:05.241917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Bird_Taxonomy = pd.read_csv('/kaggle/input/birdclef-2023/eBird_Taxonomy_v2021.csv')\nBird_Taxonomy.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:05.244728Z","iopub.execute_input":"2023-05-23T06:26:05.245430Z","iopub.status.idle":"2023-05-23T06:26:05.327942Z","shell.execute_reply.started":"2023-05-23T06:26:05.245387Z","shell.execute_reply":"2023-05-23T06:26:05.326814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_values = Bird_Taxonomy.isna().sum()\nprint(null_values)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:05.329475Z","iopub.execute_input":"2023-05-23T06:26:05.330083Z","iopub.status.idle":"2023-05-23T06:26:05.356737Z","shell.execute_reply.started":"2023-05-23T06:26:05.330043Z","shell.execute_reply":"2023-05-23T06:26:05.355715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv('/kaggle/input/birdclef-2023/sample_submission.csv')\nsample_submission.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:05.361236Z","iopub.execute_input":"2023-05-23T06:26:05.363726Z","iopub.status.idle":"2023-05-23T06:26:05.410694Z","shell.execute_reply.started":"2023-05-23T06:26:05.363683Z","shell.execute_reply":"2023-05-23T06:26:05.409488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"One-Hot Encoding can be used","metadata":{}},{"cell_type":"code","source":"#list of all found species from metadata\nspecies = train_metadata['primary_label'].unique().tolist()\nlen(species)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:05.415479Z","iopub.execute_input":"2023-05-23T06:26:05.416160Z","iopub.status.idle":"2023-05-23T06:26:05.431077Z","shell.execute_reply.started":"2023-05-23T06:26:05.416115Z","shell.execute_reply":"2023-05-23T06:26:05.429741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#adding full filepath column to metadata dataframe\n#This can help us to access files of audio easily \ntrain_metadata['filename'] = train_metadata['filename'].apply(lambda x : \"/kaggle/input/birdclef-2023/train_audio/\" + x)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:05.439634Z","iopub.execute_input":"2023-05-23T06:26:05.439917Z","iopub.status.idle":"2023-05-23T06:26:05.450372Z","shell.execute_reply.started":"2023-05-23T06:26:05.439891Z","shell.execute_reply":"2023-05-23T06:26:05.448855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_metadata.columns","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:05.451957Z","iopub.execute_input":"2023-05-23T06:26:05.453667Z","iopub.status.idle":"2023-05-23T06:26:05.465242Z","shell.execute_reply.started":"2023-05-23T06:26:05.453621Z","shell.execute_reply":"2023-05-23T06:26:05.464205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_metadata_2 = train_metadata.drop(['secondary_labels', 'type', 'latitude', 'longitude',\n       'scientific_name', 'common_name', 'author', 'license', 'rating', 'url'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:05.466703Z","iopub.execute_input":"2023-05-23T06:26:05.468182Z","iopub.status.idle":"2023-05-23T06:26:05.481070Z","shell.execute_reply.started":"2023-05-23T06:26:05.468145Z","shell.execute_reply":"2023-05-23T06:26:05.480048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Preprocessing","metadata":{}},{"cell_type":"code","source":"import librosa\nfrom scipy import signal\nfrom matplotlib import pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:05.482454Z","iopub.execute_input":"2023-05-23T06:26:05.483407Z","iopub.status.idle":"2023-05-23T06:26:05.492699Z","shell.execute_reply.started":"2023-05-23T06:26:05.483369Z","shell.execute_reply":"2023-05-23T06:26:05.491740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ogg_to_wave(filename):\n    ogg, sample_rate = librosa.load(filename)\n    int_16 = (ogg * 32767).astype(np.int16)\n    return int_16,sample_rate","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:05.494066Z","iopub.execute_input":"2023-05-23T06:26:05.494888Z","iopub.status.idle":"2023-05-23T06:26:05.504828Z","shell.execute_reply.started":"2023-05-23T06:26:05.494850Z","shell.execute_reply":"2023-05-23T06:26:05.503765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"__Librosa__ is powerful Python library built to work with audio and perform analysis on it. It is the starting point towards working with audio data at scale for a wide range of applications such as detecting voice from a person to finding personal characteristics from an audio.\n\n__SciPy__ is a scientific computation library that uses NumPy underneath.SciPy stands for Scientific Python.","metadata":{}},{"cell_type":"markdown","source":"__ogg__ is a NumPy array that contains audio samples with float values between -1 and 1, representing the _amplitude_ of the audio signal at different points in time.\n\n__(ogg * 32767)__ scales the audio samples so that they fall within the range of -32767 to 32767, which is the range of values that can be represented by a _16-bit_ signed integer.\n\n__astype(np.int16)__ converts the scaled audio samples to 16-bit signed integer data type. This conversion is necessary because most audio processing libraries and tools work with audio data represented as 16-bit integers rather than floating point values.\n\nThe resulting __int_16__ array contains the audio samples in the 16-bit integer format that can be used for further audio processing and analysis.","metadata":{}},{"cell_type":"markdown","source":"A __spectrogram__ is a 2D representation of a sound signal that shows how the frequency content of a signal changes over time. It is essentially a graph of the signal's frequency spectrum, with time on the x-axis, frequency on the y-axis, and the magnitude of each frequency component represented by a color or brightness value.\n\nA spectrogram matrix is a matrix representation of a spectrogram, where each element of the matrix represents the amplitude or energy of a frequency component at a particular time. Spectrogram matrices are commonly used as input data for machine learning models that process audio signals.\n\nThe function uses the __spectrogram()__ function from the __scipy.signal__ library to calculate the spectrogram. The spectrogram() function calculates the short-term Fourier transform of the waveform and returns the frequency, time, and spectrogram arrays.\n\nThe __window__ parameter specifies the type of window function to be used. Here, the __\"hann\"__ window is used. The __nperseg__ parameter specifies the length of each segment of the signal to be used for calculating the _Fourier transform_. The __noverlap__ parameter specifies the number of samples of overlap between the segments.\n\nThe function returns the frequency, time, and spectrogram arrays. The spectrogram array is a 2D matrix that contains the magnitude of the frequency components at each time segment of the waveform. It is commonly used as an input to neural networks for audio processing tasks such as speech recognition or music genre classification.","metadata":{}},{"cell_type":"markdown","source":"The code __metadata.drop_duplicates(subset=[\"primary_label\"], keep=False)__ drops duplicate rows from the metadata DataFrame based on the \"primary_label\" column.\n\nThe __keep=False__ parameter indicates that all duplicates should be dropped, rather than keeping the first or last occurrence.","metadata":{}},{"cell_type":"code","source":"#Take in created waveform and create a spectogram matrix for neural network input layer\ndef wave_to_spec(waveform, sample_rate):\n    freq, time, spectrogram = signal.spectrogram(waveform, sample_rate, window='hann', nperseg=256, noverlap=128)\n    return freq, time,spectrogram","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:05.506164Z","iopub.execute_input":"2023-05-23T06:26:05.506953Z","iopub.status.idle":"2023-05-23T06:26:05.517012Z","shell.execute_reply.started":"2023-05-23T06:26:05.506912Z","shell.execute_reply":"2023-05-23T06:26:05.516030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def audio_preprocessing(filepath):\n    waveform, sample_rate = ogg_to_wave(filepath)\n    freq, time, spectrogram = wave_to_spec(waveform, sample_rate)\n    return freq, time,spectrogram","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:05.518454Z","iopub.execute_input":"2023-05-23T06:26:05.519317Z","iopub.status.idle":"2023-05-23T06:26:05.527056Z","shell.execute_reply.started":"2023-05-23T06:26:05.519220Z","shell.execute_reply":"2023-05-23T06:26:05.526033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Combine all functions to create a single function 'preprocessing'","metadata":{}},{"cell_type":"code","source":"#will take in .csv filepath (formatting expected to reflect that described earlier)\ndef preprocessing(data):\n    \n    #creating dataframe\n    df = pd.read_csv(data).head(5)\n    \n    #creating spectrogram column\n    file_prefix = \"/kaggle/input/birdclef-2023/train_audio/\"\n    df['input'] = df['filename'].apply(lambda x : audio_preprocessing(file_prefix + x)[2])\n    \n    #dropping other columns\n    df = df[['input', 'primary_label']]\n    \n    #one-hot encoding primary_label column\n    species = pd.get_dummies(df['primary_label']).astype('float64')\n    df = df.drop('primary_label', axis=1)\n    df = df.join(species)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:05.528340Z","iopub.execute_input":"2023-05-23T06:26:05.529123Z","iopub.status.idle":"2023-05-23T06:26:05.548610Z","shell.execute_reply.started":"2023-05-23T06:26:05.529086Z","shell.execute_reply":"2023-05-23T06:26:05.547605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check whether above function works or not","metadata":{}},{"cell_type":"markdown","source":"#### Creation of array for audiofile","metadata":{}},{"cell_type":"code","source":"# array that will store audiofile objects\naudio_files = []\n\n# audiofile class\nclass AudioFile:\n    def __init__(self, filename):\n        self.filename = filename\n        self.freqs, self.time, self.spectrogram = audio_preprocessing(filename)\n        \n    # method to output objects spectrogram to terminal\n    def plot(self):\n        plt.pcolormesh(self.time, self.freqs, 10*np.log10(self.spectrogram), cmap='viridis')\n        plt.xlabel(\"Time(s)\")\n        plt.ylabel(\"Frequency (Hz)\")\n        plt.title(f\"Spectrogram for {self.filename[40:-1]}\")\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:05.549606Z","iopub.execute_input":"2023-05-23T06:26:05.549932Z","iopub.status.idle":"2023-05-23T06:26:05.563137Z","shell.execute_reply.started":"2023-05-23T06:26:05.549899Z","shell.execute_reply":"2023-05-23T06:26:05.562033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Loading in small subset of training metadata\ntrain_metadata_3 = pd.read_csv('/kaggle/input/birdclef-2023/train_metadata.csv')\ntrain_metadata_3 = train_metadata_3.head(10)\ntrain_metadata_3.info() ","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:05.564380Z","iopub.execute_input":"2023-05-23T06:26:05.565207Z","iopub.status.idle":"2023-05-23T06:26:05.670900Z","shell.execute_reply.started":"2023-05-23T06:26:05.565171Z","shell.execute_reply":"2023-05-23T06:26:05.669839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\nfrom scipy import signal\nimport matplotlib as plt","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:05.675645Z","iopub.execute_input":"2023-05-23T06:26:05.678075Z","iopub.status.idle":"2023-05-23T06:26:05.684900Z","shell.execute_reply.started":"2023-05-23T06:26:05.678033Z","shell.execute_reply":"2023-05-23T06:26:05.683766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plotting out all spectrograms in array\nfor file in audio_files:\n    file.plot()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:05.690131Z","iopub.execute_input":"2023-05-23T06:26:05.692778Z","iopub.status.idle":"2023-05-23T06:26:05.699214Z","shell.execute_reply.started":"2023-05-23T06:26:05.692736Z","shell.execute_reply":"2023-05-23T06:26:05.698127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now, we created a pipeline to convert all .ogg files into spectrogram. It can be useful to train and test audio files by converting them into class objects.","metadata":{}},{"cell_type":"markdown","source":"> Interesting way of generating report from the data","metadata":{}},{"cell_type":"code","source":"#filepath\n#Every row is being converted into tuple\nX = np.asarray(train_metadata['filename'].apply(lambda x : x))\n\n\ny = np.asarray(pd.get_dummies(train_metadata['primary_label']))","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:05.702609Z","iopub.execute_input":"2023-05-23T06:26:05.703305Z","iopub.status.idle":"2023-05-23T06:26:05.717012Z","shell.execute_reply.started":"2023-05-23T06:26:05.703266Z","shell.execute_reply":"2023-05-23T06:26:05.715776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#This function will take in a tuple (input, output) and return the same \n#(intput, output) tuple,but the input will have been coverted from a \n#filepath to a tensorflow waveform object.\n\ndef load_audio_file(file_path):\n    # Read audio file contents into a tensor\n    audio_binary = tf.io.read_file(file_path)\n    \n    # Decode audio binary into a numpy array\n    audio, sr = librosa.load(io.BytesIO(audio_binary.numpy()), sr=44100, mono=True)\n    \n    audio = audio * 32768.0\n    audio = tf.cast(audio, tf.int16)\n\n    return audio","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:05.720523Z","iopub.execute_input":"2023-05-23T06:26:05.721160Z","iopub.status.idle":"2023-05-23T06:26:05.731645Z","shell.execute_reply.started":"2023-05-23T06:26:05.721122Z","shell.execute_reply":"2023-05-23T06:26:05.730438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The __sr__ parameter is used to specify the desired sampling rate of the output waveform, which is set to _16000_ samples per second in this case. \n\nThe __mono__ parameter is set to _True_, which means that the output waveform will be converted to mono by averaging the samples across all channels, if the input audio has multiple channels.\n\nThe __BytesIO__ function from the io library is used to convert the binary data (audio_binary) into a stream of bytes that can be read by librosa.load. \n\nThe numpy() method is used to convert the binary data (audio_binary) from a TensorFlow tensor to a NumPy array, which is required by BytesIO.","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport io","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:05.735202Z","iopub.execute_input":"2023-05-23T06:26:05.736852Z","iopub.status.idle":"2023-05-23T06:26:05.743668Z","shell.execute_reply.started":"2023-05-23T06:26:05.736813Z","shell.execute_reply":"2023-05-23T06:26:05.742677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X[0]","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:05.746719Z","iopub.execute_input":"2023-05-23T06:26:05.748035Z","iopub.status.idle":"2023-05-23T06:26:05.761572Z","shell.execute_reply.started":"2023-05-23T06:26:05.747996Z","shell.execute_reply":"2023-05-23T06:26:05.760365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Testing dataloading function\nX_test= load_audio_file(X[0])","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:05.763329Z","iopub.execute_input":"2023-05-23T06:26:05.764107Z","iopub.status.idle":"2023-05-23T06:26:05.901114Z","shell.execute_reply.started":"2023-05-23T06:26:05.764066Z","shell.execute_reply":"2023-05-23T06:26:05.900014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(type(X_test))\nprint(X_test.shape)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:05.906258Z","iopub.execute_input":"2023-05-23T06:26:05.908710Z","iopub.status.idle":"2023-05-23T06:26:05.919311Z","shell.execute_reply.started":"2023-05-23T06:26:05.908667Z","shell.execute_reply":"2023-05-23T06:26:05.918028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess(filepath, labels):\n    \n    # Loading in audio\n    waveform = tf.py_function(load_audio_file, [filepath], tf.int16)\n    \n    # Normalize waveform to [-1, 1]\n    waveform = tf.cast(waveform, tf.float32) / 32768.0\n    \n    # Getting first 3 seconds of clip\n    waveform = waveform[:48000]\n    \n    # Add zero padding\n    zero_padding = tf.zeros([48000] - tf.shape(waveform), dtype=tf.float32)\n    wav = tf.concat([zero_padding, waveform], 0)\n    \n    # Spectrogram\n    stft = tf.signal.stft(wav, frame_length=320, frame_step=32)\n    spectrogram = tf.abs(stft)\n    spectrogram = tf.expand_dims(spectrogram, axis=2)\n    \n    return spectrogram, labels","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:05.931887Z","iopub.execute_input":"2023-05-23T06:26:05.934275Z","iopub.status.idle":"2023-05-23T06:26:05.945159Z","shell.execute_reply.started":"2023-05-23T06:26:05.934236Z","shell.execute_reply":"2023-05-23T06:26:05.944004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess1(filepath):\n    \n    # Loading in audio\n    waveform = tf.py_function(load_audio_file, [filepath], tf.int16)\n    \n    # Normalize waveform to [-1, 1]\n    waveform = tf.cast(waveform, tf.float32) / 32768.0\n    \n    # Getting first 3 seconds of clip\n    waveform = waveform[:48000]\n    \n    # Add zero padding\n    zero_padding = tf.zeros([48000] - tf.shape(waveform), dtype=tf.float32)\n    wav = tf.concat([zero_padding, waveform], 0)\n    \n    # Spectrogram\n    stft = tf.signal.stft(wav, frame_length=320, frame_step=32)\n    spectrogram = tf.abs(stft)\n    spectrogram = tf.expand_dims(spectrogram, axis=2)\n    \n    return spectrogram","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:05.950161Z","iopub.execute_input":"2023-05-23T06:26:05.952807Z","iopub.status.idle":"2023-05-23T06:26:05.963460Z","shell.execute_reply.started":"2023-05-23T06:26:05.952768Z","shell.execute_reply":"2023-05-23T06:26:05.962424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\n\n# Generating random index to test\ntest_index = random.randint(0,len(X))\ntest_index","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:05.968592Z","iopub.execute_input":"2023-05-23T06:26:05.971054Z","iopub.status.idle":"2023-05-23T06:26:05.982053Z","shell.execute_reply.started":"2023-05-23T06:26:05.971015Z","shell.execute_reply":"2023-05-23T06:26:05.981013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Should return spectrogram and numpy array of classes\nX_function_test, y_function_test = preprocess(X[test_index], y[test_index])\nprint(f'Spectrogram is: {X_function_test} \\n')\nprint(f'Numpy array is: {y_function_test} ')","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:05.987115Z","iopub.execute_input":"2023-05-23T06:26:05.989189Z","iopub.status.idle":"2023-05-23T06:26:06.070935Z","shell.execute_reply.started":"2023-05-23T06:26:05.989151Z","shell.execute_reply":"2023-05-23T06:26:06.069953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Spectrogram shape: {X_function_test.shape}')\nprint(f'Spectrogram type: {type(X_function_test)}')\nprint(f'Classes shape: {y_function_test.shape}')\nprint(f'Classes type: {type(y_function_test)}')","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:06.075367Z","iopub.execute_input":"2023-05-23T06:26:06.077940Z","iopub.status.idle":"2023-05-23T06:26:06.089144Z","shell.execute_reply.started":"2023-05-23T06:26:06.077897Z","shell.execute_reply":"2023-05-23T06:26:06.088028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Plot spectrogram","metadata":{}},{"cell_type":"markdown","source":"#### Transform the data","metadata":{}},{"cell_type":"code","source":"# Creating dataset from 'train_metadata.csv' data\ndata = tf.data.Dataset.from_tensor_slices((X, y))","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:06.093535Z","iopub.execute_input":"2023-05-23T06:26:06.094263Z","iopub.status.idle":"2023-05-23T06:26:06.105266Z","shell.execute_reply.started":"2023-05-23T06:26:06.094223Z","shell.execute_reply":"2023-05-23T06:26:06.104026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Full data pipeline\ndata = data.map(preprocess, num_parallel_calls=tf.data.AUTOTUNE)\ndata = data.cache()\ndata = data.shuffle(buffer_size=100)\ndata = data.batch(batch_size=16)\ndata = data.prefetch(8)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:06.108295Z","iopub.execute_input":"2023-05-23T06:26:06.109552Z","iopub.status.idle":"2023-05-23T06:26:06.259194Z","shell.execute_reply.started":"2023-05-23T06:26:06.109515Z","shell.execute_reply":"2023-05-23T06:26:06.258153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This code snippet shows a data pipeline that preprocesses and batches data for training a machine learning model. Here is an explanation of each step:\n**\ndata.map(preprocess, num_parallel_calls=tf.data.AUTOTUNE)** - This step applies the preprocess function to each element in the dataset data. The num_parallel_calls=tf.data.AUTOTUNE parameter allows TensorFlow to dynamically determine the number of CPU cores to use for parallel processing, which can improve performance.\n\n**data.cache()**- This step caches the preprocessed data in memory or on disk, so that it can be quickly accessed during training without having to re-preprocess it.\n\n**data.shuffle(buffer_size=100)**- This step shuffles the data randomly, with a buffer size of 100. This is done to prevent the model from overfitting to the order of the data.\n\n**data.batch(batch_size=16)** - This step groups the preprocessed data into batches of size 16. Batching can improve training efficiency by allowing the model to process multiple examples at once.\n\n**data.prefetch(8)** - This step prefetches 8 batches of data, so that the next batch is ready to be processed as soon as the current batch is done. This can help minimize the amount of time the model spends waiting for data during training.","metadata":{}},{"cell_type":"code","source":"# 847 Training batches\n# 212 Test batches\ntrain_1 = data.take(264)\ntrain_2 = data.skip(264).take(200)\ntrain_3 = data.skip(264).take(200)\ntrain_4 = data.skip(600).take(247)\ntest = data.skip(847).take(212)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:06.260692Z","iopub.execute_input":"2023-05-23T06:26:06.261068Z","iopub.status.idle":"2023-05-23T06:26:06.274742Z","shell.execute_reply.started":"2023-05-23T06:26:06.261018Z","shell.execute_reply":"2023-05-23T06:26:06.273755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"verify data generator","metadata":{}},{"cell_type":"markdown","source":"### Model Creation","metadata":{}},{"cell_type":"code","source":"# TensorFlow CNN Model Dependencies\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, Flatten, Dense, MaxPooling2D","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:06.277989Z","iopub.execute_input":"2023-05-23T06:26:06.278289Z","iopub.status.idle":"2023-05-23T06:26:06.283900Z","shell.execute_reply.started":"2023-05-23T06:26:06.278260Z","shell.execute_reply":"2023-05-23T06:26:06.282322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cnn_model.add(LeakyReLU(alpha=0.1))\nmodel = Sequential()\nmodel.add(Conv2D(16, (3, 3), activation='relu', input_shape=(1491, 257, 1)))\nmodel.add(MaxPooling2D(2, 2))\nmodel.add(Conv2D(16, (3, 3), activation='relu'))\nmodel.add(MaxPooling2D(2, 2))\nmodel.add(Flatten())\nmodel.add(Dense(264, activation='relu'))\nmodel.add(Dense(264, activation='sigmoid'))","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:06.285553Z","iopub.execute_input":"2023-05-23T06:26:06.286130Z","iopub.status.idle":"2023-05-23T06:26:06.349855Z","shell.execute_reply.started":"2023-05-23T06:26:06.286090Z","shell.execute_reply":"2023-05-23T06:26:06.348910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:06.351124Z","iopub.execute_input":"2023-05-23T06:26:06.351461Z","iopub.status.idle":"2023-05-23T06:26:06.377331Z","shell.execute_reply.started":"2023-05-23T06:26:06.351424Z","shell.execute_reply":"2023-05-23T06:26:06.376548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile('Adam', loss='CategoricalCrossentropy',\n              metrics=[tf.keras.metrics.CategoricalAccuracy()])","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:06.378479Z","iopub.execute_input":"2023-05-23T06:26:06.379136Z","iopub.status.idle":"2023-05-23T06:26:06.404737Z","shell.execute_reply.started":"2023-05-23T06:26:06.379095Z","shell.execute_reply":"2023-05-23T06:26:06.403694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping\n\n#Stops model if it is not steadily dropping loss val_loss function return\nearlyStop = EarlyStopping(monitor='val_loss', patience=3, restore_best_weights=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:06.406046Z","iopub.execute_input":"2023-05-23T06:26:06.407024Z","iopub.status.idle":"2023-05-23T06:26:06.412190Z","shell.execute_reply.started":"2023-05-23T06:26:06.406983Z","shell.execute_reply":"2023-05-23T06:26:06.411019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fitting model to train_1\nhistory_train_1 = model.fit(train_1, epochs=5, \n                         callbacks=[earlyStop])","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:06.413528Z","iopub.execute_input":"2023-05-23T06:26:06.414430Z","iopub.status.idle":"2023-05-23T06:26:30.418578Z","shell.execute_reply.started":"2023-05-23T06:26:06.414401Z","shell.execute_reply":"2023-05-23T06:26:30.417354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = model.evaluate(train_1)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:30.420325Z","iopub.execute_input":"2023-05-23T06:26:30.420704Z","iopub.status.idle":"2023-05-23T06:26:31.192035Z","shell.execute_reply.started":"2023-05-23T06:26:30.420663Z","shell.execute_reply":"2023-05-23T06:26:31.190989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data=X[:15]","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:31.194924Z","iopub.execute_input":"2023-05-23T06:26:31.195992Z","iopub.status.idle":"2023-05-23T06:26:31.202268Z","shell.execute_reply.started":"2023-05-23T06:26:31.195935Z","shell.execute_reply":"2023-05-23T06:26:31.201168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test=np.append(test_data,'/kaggle/input/birdclef-2023/test_soundscapes/soundscape_29201.ogg')","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:31.204020Z","iopub.execute_input":"2023-05-23T06:26:31.204482Z","iopub.status.idle":"2023-05-23T06:26:31.212569Z","shell.execute_reply.started":"2023-05-23T06:26:31.204442Z","shell.execute_reply":"2023-05-23T06:26:31.211528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:31.214484Z","iopub.execute_input":"2023-05-23T06:26:31.214932Z","iopub.status.idle":"2023-05-23T06:26:31.222836Z","shell.execute_reply.started":"2023-05-23T06:26:31.214892Z","shell.execute_reply":"2023-05-23T06:26:31.221722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"l1=[]\nfor i in test:\n    sp=preprocess1(i)\n    l1.append(sp)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:31.224535Z","iopub.execute_input":"2023-05-23T06:26:31.225700Z","iopub.status.idle":"2023-05-23T06:26:33.600952Z","shell.execute_reply.started":"2023-05-23T06:26:31.225661Z","shell.execute_reply":"2023-05-23T06:26:33.599759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_tensor= tf.convert_to_tensor(l1)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:33.602581Z","iopub.execute_input":"2023-05-23T06:26:33.603000Z","iopub.status.idle":"2023-05-23T06:26:33.609082Z","shell.execute_reply.started":"2023-05-23T06:26:33.602933Z","shell.execute_reply":"2023-05-23T06:26:33.607907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_tensor","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:33.611062Z","iopub.execute_input":"2023-05-23T06:26:33.611759Z","iopub.status.idle":"2023-05-23T06:26:33.647173Z","shell.execute_reply.started":"2023-05-23T06:26:33.611720Z","shell.execute_reply":"2023-05-23T06:26:33.646035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(test_tensor)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:33.648959Z","iopub.execute_input":"2023-05-23T06:26:33.649357Z","iopub.status.idle":"2023-05-23T06:26:33.819108Z","shell.execute_reply.started":"2023-05-23T06:26:33.649318Z","shell.execute_reply":"2023-05-23T06:26:33.817962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nmain_pred=predictions[15]\nmain_pred=[0 if i <0.5 else 1 for i in main_pred]","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:33.822605Z","iopub.execute_input":"2023-05-23T06:26:33.822916Z","iopub.status.idle":"2023-05-23T06:26:33.828814Z","shell.execute_reply.started":"2023-05-23T06:26:33.822886Z","shell.execute_reply":"2023-05-23T06:26:33.827410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions[15]","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:33.830471Z","iopub.execute_input":"2023-05-23T06:26:33.831177Z","iopub.status.idle":"2023-05-23T06:26:33.850287Z","shell.execute_reply.started":"2023-05-23T06:26:33.831137Z","shell.execute_reply":"2023-05-23T06:26:33.848775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred1=predictions[15]\ndf=[0 if i <0.5 else 1 for i in pred1]","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:33.852290Z","iopub.execute_input":"2023-05-23T06:26:33.852696Z","iopub.status.idle":"2023-05-23T06:26:33.859586Z","shell.execute_reply.started":"2023-05-23T06:26:33.852651Z","shell.execute_reply":"2023-05-23T06:26:33.858040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=pd.DataFrame(df)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:33.861679Z","iopub.execute_input":"2023-05-23T06:26:33.862555Z","iopub.status.idle":"2023-05-23T06:26:33.870231Z","shell.execute_reply.started":"2023-05-23T06:26:33.862504Z","shell.execute_reply":"2023-05-23T06:26:33.868980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"index = df.loc[df[0] == 1].index.tolist()\nindex","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:33.871906Z","iopub.execute_input":"2023-05-23T06:26:33.872395Z","iopub.status.idle":"2023-05-23T06:26:33.890555Z","shell.execute_reply.started":"2023-05-23T06:26:33.872357Z","shell.execute_reply":"2023-05-23T06:26:33.889195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"main_pred=pd.DataFrame(main_pred)\nmain_pred=main_pred.T\n","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:33.892602Z","iopub.execute_input":"2023-05-23T06:26:33.893040Z","iopub.status.idle":"2023-05-23T06:26:33.899704Z","shell.execute_reply.started":"2023-05-23T06:26:33.893002Z","shell.execute_reply":"2023-05-23T06:26:33.898362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub = pd.read_csv(\"/kaggle/input/birdclef-2023/sample_submission.csv\")\nsample_sub.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:33.901939Z","iopub.execute_input":"2023-05-23T06:26:33.902461Z","iopub.status.idle":"2023-05-23T06:26:33.936914Z","shell.execute_reply.started":"2023-05-23T06:26:33.902424Z","shell.execute_reply":"2023-05-23T06:26:33.935742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub.columns","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:33.938858Z","iopub.execute_input":"2023-05-23T06:26:33.939285Z","iopub.status.idle":"2023-05-23T06:26:33.946848Z","shell.execute_reply.started":"2023-05-23T06:26:33.939240Z","shell.execute_reply":"2023-05-23T06:26:33.945465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(3):\n    for j in range(1,265):\n        if(sample_sub.iloc[i][j]==1):\n            sample_sub.iloc[i][j]=0","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:33.948888Z","iopub.execute_input":"2023-05-23T06:26:33.949902Z","iopub.status.idle":"2023-05-23T06:26:34.213983Z","shell.execute_reply.started":"2023-05-23T06:26:33.949864Z","shell.execute_reply":"2023-05-23T06:26:34.212847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(3):\n    for j in index:\n        sample_sub.iloc[i][j]=1\n        ","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:34.215609Z","iopub.execute_input":"2023-05-23T06:26:34.216045Z","iopub.status.idle":"2023-05-23T06:26:34.229948Z","shell.execute_reply.started":"2023-05-23T06:26:34.216004Z","shell.execute_reply":"2023-05-23T06:26:34.228872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"sample_sub.to_csv(\"submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T06:26:34.231569Z","iopub.execute_input":"2023-05-23T06:26:34.232372Z","iopub.status.idle":"2023-05-23T06:26:34.241900Z","shell.execute_reply.started":"2023-05-23T06:26:34.232237Z","shell.execute_reply":"2023-05-23T06:26:34.240742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}