{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Transfer Learning using Kaggle Models\n\nIn this notebook, I've demonstrated how to perform audio classification using a pre-trained model from [Kaggle models](https://www.kaggle.com/models), called [yamnet](https://www.kaggle.com/models/google/yamnet).","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-05T16:10:23.783972Z","iopub.execute_input":"2023-03-05T16:10:23.785225Z","iopub.status.idle":"2023-03-05T16:11:47.450363Z","shell.execute_reply.started":"2023-03-05T16:10:23.785162Z","shell.execute_reply":"2023-03-05T16:11:47.446180Z"},"_kg_hide-output":true}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"!pip install tensorflow_io==0.23.1\n!pip install tensorflow==2.7.1","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-03-20T15:55:09.292181Z","iopub.execute_input":"2023-03-20T15:55:09.292572Z","iopub.status.idle":"2023-03-20T15:56:32.837101Z","shell.execute_reply.started":"2023-03-20T15:55:09.292538Z","shell.execute_reply":"2023-03-20T15:56:32.835300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport tensorflow as tf\nimport tensorflow_hub as hub\nimport tensorflow_io as tfio\nimport os, random\nimport shutil\nfrom pydub import AudioSegment\nfrom glob import glob #2 List the files in a directory\nfrom pathlib import Path\nfrom IPython.display import display, Audio\nimport soundfile as sf","metadata":{"execution":{"iopub.status.busy":"2023-03-20T15:56:32.840099Z","iopub.execute_input":"2023-03-20T15:56:32.840492Z","iopub.status.idle":"2023-03-20T15:56:37.896103Z","shell.execute_reply.started":"2023-03-20T15:56:32.840447Z","shell.execute_reply":"2023-03-20T15:56:37.894711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load and Pre-process the dataset\nFor time and memory management, we'll be taking random sample of 15 birds, we'll also convert the audio files from ogg to wav because only wav can be used as input to the yamnet model.\n","metadata":{}},{"cell_type":"code","source":"ROOT = \"/kaggle/input/birdclef-2023/\"\ntrain_metadata = pd.read_csv(os.path.join(ROOT, 'train_metadata.csv'))[['primary_label', 'filename']]\ntrain_metadata['filepath'] = 'train_audio/' + train_metadata['filename']\ntrain_metadata.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-20T15:56:37.898279Z","iopub.execute_input":"2023-03-20T15:56:37.899659Z","iopub.status.idle":"2023-03-20T15:56:38.097215Z","shell.execute_reply.started":"2023-03-20T15:56:37.899602Z","shell.execute_reply":"2023-03-20T15:56:38.096086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Random sample of 15 birds\nclasses = set(random.sample(train_metadata['primary_label'].unique().tolist(), 15)) \nprint(classes)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T15:56:38.099697Z","iopub.execute_input":"2023-03-20T15:56:38.100007Z","iopub.status.idle":"2023-03-20T15:56:38.112763Z","shell.execute_reply.started":"2023-03-20T15:56:38.099979Z","shell.execute_reply":"2023-03-20T15:56:38.111291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_metadata = train_metadata[train_metadata.primary_label.apply(lambda x: x in classes)].reset_index(drop=True)\nkeys = set(train_metadata.primary_label)\nvalues = np.arange(0, len(keys))\ncode_dict = dict(zip(sorted(keys), values))\ntrain_metadata['label'] = train_metadata['primary_label'].apply(lambda x: code_dict[x])\ntrain_metadata.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-20T15:56:38.114422Z","iopub.execute_input":"2023-03-20T15:56:38.115067Z","iopub.status.idle":"2023-03-20T15:56:38.138578Z","shell.execute_reply.started":"2023-03-20T15:56:38.115035Z","shell.execute_reply":"2023-03-20T15:56:38.137593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classes_df = pd.DataFrame()\nclasses_df = train_metadata.filter(['primary_label','label'],axis=1)\nclasses_df = classes_df.drop_duplicates()\nclasses_df.reset_index(drop=True, inplace=True)\nclasses_df","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-03-20T15:56:38.139696Z","iopub.execute_input":"2023-03-20T15:56:38.139985Z","iopub.status.idle":"2023-03-20T15:56:38.157900Z","shell.execute_reply.started":"2023-03-20T15:56:38.139957Z","shell.execute_reply":"2023-03-20T15:56:38.156632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_list = []\n\nfor x in classes_df['label']:\n    train_sng_temp = train_metadata[train_metadata['label'] == x]\n    train_list.append(train_sng_temp)\nprint(train_list[0])","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-03-20T15:56:38.159511Z","iopub.execute_input":"2023-03-20T15:56:38.159913Z","iopub.status.idle":"2023-03-20T15:56:38.176838Z","shell.execute_reply.started":"2023-03-20T15:56:38.159871Z","shell.execute_reply":"2023-03-20T15:56:38.175551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATASET_ROOT = os.path.join(\"/kaggle/input/birdclef-2023/train_audio\")\nDATASET_AUDIO_PATH = os.path.join('./Data_Train/')","metadata":{"execution":{"iopub.status.busy":"2023-03-20T15:56:38.178731Z","iopub.execute_input":"2023-03-20T15:56:38.179481Z","iopub.status.idle":"2023-03-20T15:56:38.186862Z","shell.execute_reply.started":"2023-03-20T15:56:38.179433Z","shell.execute_reply":"2023-03-20T15:56:38.185944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for x in range(classes_df[classes_df.columns[1]].count()): \n    if os.path.exists(DATASET_AUDIO_PATH + \"/\" + classes_df['primary_label'][x]) is False:\n        os.makedirs(DATASET_AUDIO_PATH + \"/\" + classes_df['primary_label'][x])\n    for z in range(train_metadata.pivot_table(index = ['primary_label'], aggfunc ='size').min()):\n        data, samplerate = sf.read(\"/kaggle/input/birdclef-2023/train_audio/\" + str(train_list[x].iat[z,1])) \n        sf.write(DATASET_AUDIO_PATH +    str(train_list[x].iat[z,1])[:-4] + \".wav\",data, samplerate, subtype='PCM_16')","metadata":{"execution":{"iopub.status.busy":"2023-03-20T15:56:38.188264Z","iopub.execute_input":"2023-03-20T15:56:38.188548Z","iopub.status.idle":"2023-03-20T15:56:40.834420Z","shell.execute_reply.started":"2023-03-20T15:56:38.188520Z","shell.execute_reply":"2023-03-20T15:56:40.833212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_metadata.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-20T15:56:40.839944Z","iopub.execute_input":"2023-03-20T15:56:40.840322Z","iopub.status.idle":"2023-03-20T15:56:40.852330Z","shell.execute_reply.started":"2023-03-20T15:56:40.840285Z","shell.execute_reply":"2023-03-20T15:56:40.851151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for x in train_metadata.index:\n    train_metadata['new_filepath'] = DATASET_AUDIO_PATH + str(train_metadata['filename'][0])[:-4] + \".wav\"\ntrain_metadata.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-20T15:56:40.854077Z","iopub.execute_input":"2023-03-20T15:56:40.855341Z","iopub.status.idle":"2023-03-20T15:56:40.992090Z","shell.execute_reply.started":"2023-03-20T15:56:40.855296Z","shell.execute_reply":"2023-03-20T15:56:40.991030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filenames = train_metadata['new_filepath']\ntargets = train_metadata['label']\n\nmain_ds = tf.data.Dataset.from_tensor_slices((filenames, targets))\nmain_ds.element_spec","metadata":{"execution":{"iopub.status.busy":"2023-03-20T15:56:40.993832Z","iopub.execute_input":"2023-03-20T15:56:40.994177Z","iopub.status.idle":"2023-03-20T15:56:41.019424Z","shell.execute_reply.started":"2023-03-20T15:56:40.994145Z","shell.execute_reply":"2023-03-20T15:56:41.018667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Utility functions for loading audio files","metadata":{}},{"cell_type":"code","source":"@tf.function\ndef load_wav_16k_mono(filename):\n    \"\"\" Load a WAV file, convert it to a float tensor, resample to 16 kHz single-channel audio. \"\"\"\n    file_contents = tf.io.read_file(filename)\n    wav, sample_rate = tf.audio.decode_wav(\n          file_contents,\n          desired_channels=1)\n    wav = tf.squeeze(wav, axis=-1)\n    sample_rate = tf.cast(sample_rate, dtype=tf.int64)\n    wav = tfio.audio.resample(wav, rate_in=sample_rate, rate_out=16000)\n    return wav","metadata":{"execution":{"iopub.status.busy":"2023-03-20T15:56:41.020660Z","iopub.execute_input":"2023-03-20T15:56:41.021144Z","iopub.status.idle":"2023-03-20T15:56:41.027877Z","shell.execute_reply.started":"2023-03-20T15:56:41.021114Z","shell.execute_reply":"2023-03-20T15:56:41.026722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_wav_for_map(filename, label):\n    return load_wav_16k_mono(filename), label","metadata":{"execution":{"iopub.status.busy":"2023-03-20T15:56:41.029086Z","iopub.execute_input":"2023-03-20T15:56:41.029378Z","iopub.status.idle":"2023-03-20T15:56:41.038320Z","shell.execute_reply.started":"2023-03-20T15:56:41.029351Z","shell.execute_reply":"2023-03-20T15:56:41.037619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"main_ds = main_ds.map(load_wav_for_map)\nmain_ds.element_spec","metadata":{"execution":{"iopub.status.busy":"2023-03-20T15:56:41.039937Z","iopub.execute_input":"2023-03-20T15:56:41.040717Z","iopub.status.idle":"2023-03-20T15:56:41.921750Z","shell.execute_reply.started":"2023-03-20T15:56:41.040668Z","shell.execute_reply":"2023-03-20T15:56:41.920585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading the Model","metadata":{}},{"cell_type":"code","source":"yamnet_model_handle = 'https://kaggle.com/models/google/yamnet/frameworks/TensorFlow2/variations/yamnet/versions/1'\nyamnet_model = hub.load(yamnet_model_handle)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T15:56:41.923123Z","iopub.execute_input":"2023-03-20T15:56:41.923482Z","iopub.status.idle":"2023-03-20T15:56:47.217252Z","shell.execute_reply.started":"2023-03-20T15:56:41.923448Z","shell.execute_reply":"2023-03-20T15:56:47.215964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# applies the embedding extraction model to a wav data\ndef extract_embedding(wav_data, label):\n  ''' run YAMNet to extract embedding from the wav data '''\n  scores, embeddings, spectrogram = yamnet_model(wav_data)\n  num_embeddings = tf.shape(embeddings)[0]\n  return (embeddings,\n            tf.repeat(label, num_embeddings))\n\n# extract embedding\nmain_ds = main_ds.map(extract_embedding).unbatch()\nmain_ds.element_spec","metadata":{"execution":{"iopub.status.busy":"2023-03-20T15:56:47.219144Z","iopub.execute_input":"2023-03-20T15:56:47.219601Z","iopub.status.idle":"2023-03-20T15:56:47.468014Z","shell.execute_reply.started":"2023-03-20T15:56:47.219551Z","shell.execute_reply":"2023-03-20T15:56:47.466860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cached_ds = main_ds.cache()","metadata":{"execution":{"iopub.status.busy":"2023-03-20T15:56:47.469560Z","iopub.execute_input":"2023-03-20T15:56:47.469865Z","iopub.status.idle":"2023-03-20T15:56:47.474426Z","shell.execute_reply.started":"2023-03-20T15:56:47.469838Z","shell.execute_reply":"2023-03-20T15:56:47.473158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds = cached_ds.cache().shuffle(1000).batch(32).prefetch(tf.data.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T15:56:47.475701Z","iopub.execute_input":"2023-03-20T15:56:47.475956Z","iopub.status.idle":"2023-03-20T15:56:47.489634Z","shell.execute_reply.started":"2023-03-20T15:56:47.475931Z","shell.execute_reply":"2023-03-20T15:56:47.488713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds","metadata":{"execution":{"iopub.status.busy":"2023-03-20T15:56:47.491058Z","iopub.execute_input":"2023-03-20T15:56:47.491371Z","iopub.status.idle":"2023-03-20T15:56:47.501003Z","shell.execute_reply.started":"2023-03-20T15:56:47.491341Z","shell.execute_reply":"2023-03-20T15:56:47.500277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_model = tf.keras.Sequential([\n    tf.keras.layers.Input(shape=(1024), dtype=tf.float32,\n                          name='input_embedding'),\n    tf.keras.layers.Dense(512, activation='relu'),\n    tf.keras.layers.Dense(len(classes))\n], name='my_model')\n\nmy_model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-03-20T15:56:47.502348Z","iopub.execute_input":"2023-03-20T15:56:47.502673Z","iopub.status.idle":"2023-03-20T15:56:47.554280Z","shell.execute_reply.started":"2023-03-20T15:56:47.502638Z","shell.execute_reply":"2023-03-20T15:56:47.553101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_model.compile(loss=tf.keras.losses.SparseCategoricalCrossentropy(from_logits=True),\n                 optimizer=\"adam\",\n                 metrics=['accuracy'])\n\ncallback = tf.keras.callbacks.EarlyStopping(monitor='loss',\n                                            patience=3,\n                                            restore_best_weights=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T15:56:47.555896Z","iopub.execute_input":"2023-03-20T15:56:47.556263Z","iopub.status.idle":"2023-03-20T15:56:47.572810Z","shell.execute_reply.started":"2023-03-20T15:56:47.556221Z","shell.execute_reply":"2023-03-20T15:56:47.571712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEPS_PER_EPOCH = train_metadata.shape[0] // 32","metadata":{"execution":{"iopub.status.busy":"2023-03-20T15:56:47.574236Z","iopub.execute_input":"2023-03-20T15:56:47.574877Z","iopub.status.idle":"2023-03-20T15:56:47.580958Z","shell.execute_reply.started":"2023-03-20T15:56:47.574841Z","shell.execute_reply":"2023-03-20T15:56:47.579528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training the Model","metadata":{}},{"cell_type":"code","source":"history = my_model.fit(train_ds,\n                       steps_per_epoch = STEPS_PER_EPOCH,\n                       epochs=10)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T15:56:47.582457Z","iopub.execute_input":"2023-03-20T15:56:47.583186Z","iopub.status.idle":"2023-03-20T15:58:09.963244Z","shell.execute_reply.started":"2023-03-20T15:56:47.583121Z","shell.execute_reply":"2023-03-20T15:58:09.961962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Conclusion\n\nHere in this notebook, I've illustrated how [Kaggle models](https://www.kaggle.com/models) can be used to perform audio classification using a pre-trained model, called [yamnet](https://www.kaggle.com/models/google/yamnet), with an accuracy of more than 95%.\n\nNow, it's your turn to create some amazing transfer learning notebooks using [Kaggle Models](https://www.kaggle.com/models)\n\n## Useful resources which helped \n* https://www.kaggle.com/models/google/yamnet\n* https://colab.research.google.com/github/tensorflow/docs/blob/master/site/en/tutorials/audio/transfer_learning_audio.ipynb\n* https://www.kaggle.com/code/asisheriberto/convert-ogg-to-wav-and-predict/notebook","metadata":{}}]}