{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# RSNA TFWriter\n\nThe following is an approach to writing TFRecords for BirdCLEF.\n\n# Imports/Setup","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport os\nimport tensorflow as tf\nimport librosa\nimport tensorflow.keras as keras","metadata":{"execution":{"iopub.status.busy":"2023-03-21T17:40:09.862811Z","iopub.execute_input":"2023-03-21T17:40:09.863232Z","iopub.status.idle":"2023-03-21T17:40:09.869330Z","shell.execute_reply.started":"2023-03-21T17:40:09.863195Z","shell.execute_reply":"2023-03-21T17:40:09.867879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"taxonomy = '/kaggle/input/birdclef-2023/eBird_Taxonomy_v2021.csv'\nmetadata = '/kaggle/input/birdclef-2023/train_metadata.csv'","metadata":{"execution":{"iopub.status.busy":"2023-03-21T17:40:09.871319Z","iopub.execute_input":"2023-03-21T17:40:09.871984Z","iopub.status.idle":"2023-03-21T17:40:09.886024Z","shell.execute_reply.started":"2023-03-21T17:40:09.871867Z","shell.execute_reply":"2023-03-21T17:40:09.884585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dftax = pd.read_csv(taxonomy)\ndfmeta = pd.read_csv(metadata)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T17:40:09.888272Z","iopub.execute_input":"2023-03-21T17:40:09.888681Z","iopub.status.idle":"2023-03-21T17:40:10.000197Z","shell.execute_reply.started":"2023-03-21T17:40:09.888647Z","shell.execute_reply":"2023-03-21T17:40:09.998608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"im_size = [128, 384]","metadata":{"execution":{"iopub.status.busy":"2023-03-21T17:40:10.001360Z","iopub.execute_input":"2023-03-21T17:40:10.001765Z","iopub.status.idle":"2023-03-21T17:40:10.007797Z","shell.execute_reply.started":"2023-03-21T17:40:10.001681Z","shell.execute_reply":"2023-03-21T17:40:10.006304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exclude Some Labels    : (\n\nI have chosen to exclude any label with less than 10 instances. \n\nLater on I will import instances from **xeno-canto.org** to ensure every label has the bare minimum.","metadata":{}},{"cell_type":"code","source":"exclude_lst = dfmeta['primary_label'].value_counts()[\n    dfmeta['primary_label'].value_counts() < 10\n].index.to_list()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T17:40:10.009820Z","iopub.execute_input":"2023-03-21T17:40:10.010974Z","iopub.status.idle":"2023-03-21T17:40:10.030536Z","shell.execute_reply.started":"2023-03-21T17:40:10.010925Z","shell.execute_reply":"2023-03-21T17:40:10.029113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def include(row):\n    if row['primary_label'] in exclude_lst:\n        return False\n    return True","metadata":{"execution":{"iopub.status.busy":"2023-03-21T17:40:10.031708Z","iopub.execute_input":"2023-03-21T17:40:10.032056Z","iopub.status.idle":"2023-03-21T17:40:10.042518Z","shell.execute_reply.started":"2023-03-21T17:40:10.032025Z","shell.execute_reply":"2023-03-21T17:40:10.041494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"include = dfmeta.apply(include, axis=1)\ndfmeta_ex = dfmeta[include]","metadata":{"execution":{"iopub.status.busy":"2023-03-21T17:40:10.043941Z","iopub.execute_input":"2023-03-21T17:40:10.044333Z","iopub.status.idle":"2023-03-21T17:40:10.235921Z","shell.execute_reply.started":"2023-03-21T17:40:10.044298Z","shell.execute_reply":"2023-03-21T17:40:10.234858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Handle the Data\n\nSince we are dealing with such an inbalance of so many classes the easiest approach is to group the elements by label and then sample each group equally.","metadata":{}},{"cell_type":"code","source":"dfmeta_ex_g = dfmeta_ex.groupby(\"primary_label\")\ndft = dfmeta_ex_g.sample(frac=0.8, random_state=42)\ndfv = dfmeta_ex.drop(dft.index)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T17:40:10.237345Z","iopub.execute_input":"2023-03-21T17:40:10.238213Z","iopub.status.idle":"2023-03-21T17:40:10.375039Z","shell.execute_reply.started":"2023-03-21T17:40:10.238177Z","shell.execute_reply":"2023-03-21T17:40:10.373222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dft['primary_label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T17:40:10.376373Z","iopub.execute_input":"2023-03-21T17:40:10.376725Z","iopub.status.idle":"2023-03-21T17:40:10.385899Z","shell.execute_reply.started":"2023-03-21T17:40:10.376691Z","shell.execute_reply":"2023-03-21T17:40:10.384973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfv['primary_label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T17:40:10.389532Z","iopub.execute_input":"2023-03-21T17:40:10.389874Z","iopub.status.idle":"2023-03-21T17:40:10.402774Z","shell.execute_reply.started":"2023-03-21T17:40:10.389837Z","shell.execute_reply":"2023-03-21T17:40:10.401973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Just to make sure we got what we want...","metadata":{}},{"cell_type":"markdown","source":"# One Hot Encoding\nNext we need to setup one hot encoding. Create a tensorflow TextVect layer and predict.","metadata":{}},{"cell_type":"code","source":"unique_lst = dft['primary_label'].unique().tolist()\nv_len = len(unique_lst)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T17:40:10.404063Z","iopub.execute_input":"2023-03-21T17:40:10.404371Z","iopub.status.idle":"2023-03-21T17:40:10.428104Z","shell.execute_reply.started":"2023-03-21T17:40:10.404339Z","shell.execute_reply":"2023-03-21T17:40:10.426838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_vec =  keras.layers.TextVectorization(\n    max_tokens=v_len+1,\n    output_mode='multi_hot',\n    vocabulary=unique_lst\n)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T17:40:10.429460Z","iopub.execute_input":"2023-03-21T17:40:10.429855Z","iopub.status.idle":"2023-03-21T17:40:10.448509Z","shell.execute_reply.started":"2023-03-21T17:40:10.429801Z","shell.execute_reply":"2023-03-21T17:40:10.447421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"By default tensorflow has a [UNK] token. To deal with it we add a lambda layer taking everything after index 0.","metadata":{}},{"cell_type":"code","source":"model = keras.models.Sequential()\nmodel.add(keras.Input(shape=(1,), dtype=tf.string))\nmodel.add(text_vec)\nmodel.add(keras.layers.Lambda(lambda x: x[:, 1:]))","metadata":{"execution":{"iopub.status.busy":"2023-03-21T17:40:10.449620Z","iopub.execute_input":"2023-03-21T17:40:10.449943Z","iopub.status.idle":"2023-03-21T17:40:10.502417Z","shell.execute_reply.started":"2023-03-21T17:40:10.449889Z","shell.execute_reply":"2023-03-21T17:40:10.501035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# IO Functions\nThe first function we will use librosa to get the mel spec.\n\nThe second function will create a TFExample to write to a record.","metadata":{}},{"cell_type":"code","source":"def get_mel_spec(path):\n    data, sr = librosa.load(path)\n    hlen = int(data.shape[0] // (im_size[1] - 1))\n    mel = librosa.feature.melspectrogram(y=data, sr=sr, hop_length=hlen)\n    mel_db = librosa.power_to_db(mel, ref=np.max)\n\n    if mel_db.shape[1] > im_size[1]:\n        return mel_db[:, :im_size[1]]\n    return mel_db","metadata":{"execution":{"iopub.status.busy":"2023-03-21T17:40:10.504247Z","iopub.execute_input":"2023-03-21T17:40:10.504558Z","iopub.status.idle":"2023-03-21T17:40:10.511240Z","shell.execute_reply.started":"2023-03-21T17:40:10.504526Z","shell.execute_reply":"2023-03-21T17:40:10.509766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.train import BytesList, FloatList, Int64List\nfrom tensorflow.train import Feature, Features, Example\n\ndef get_example(image, label):\n    return Example(\n        features=Features(\n            feature={\n                'image': Feature(bytes_list=BytesList(value=[tf.io.serialize_tensor(image).numpy()])),\n                'label': Feature(bytes_list=BytesList(value=[tf.io.serialize_tensor(label).numpy()]))\n            }\n        )\n    )","metadata":{"execution":{"iopub.status.busy":"2023-03-21T17:40:10.513277Z","iopub.execute_input":"2023-03-21T17:40:10.513863Z","iopub.status.idle":"2023-03-21T17:40:10.523571Z","shell.execute_reply.started":"2023-03-21T17:40:10.513825Z","shell.execute_reply":"2023-03-21T17:40:10.522134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# TFWriter\nFinally, we create a set of paths, setup our writers/stack, and iterate through the dataset. At each iteration we choose the record based on how it will % into the n_shards (adding more randomness) and then compute the image and one-hot encoded label.\n\n> making sure to set verbose=0 so we can write in peace","metadata":{}},{"cell_type":"code","source":"from contextlib import ExitStack\n\ndef write_tfrecords(name, dataset, n_shards=50):\n    paths = [\"{}.tfrecord-{:02d}-of-{:02d}\".format(name, index, n_shards) for index in range(n_shards)]\n    \n    with ExitStack() as stack: \n        writers = [stack.enter_context(tf.io.TFRecordWriter(path)) for path in paths]\n        \n        for i, row in dataset.iterrows():\n            shard = i % n_shards\n            audio_path = '/kaggle/input/birdclef-2023/train_audio/' + row['filename']\n            img = get_mel_spec(audio_path)\n            label = model.predict([row['primary_label']], verbose=0).tolist()[0]\n            example = get_example(img, label)\n            writers[shard].write(example.SerializeToString())\n            \n    print('Done writing ' + name + '.')","metadata":{"execution":{"iopub.status.busy":"2023-03-21T17:40:10.525554Z","iopub.execute_input":"2023-03-21T17:40:10.526020Z","iopub.status.idle":"2023-03-21T17:40:10.537974Z","shell.execute_reply.started":"2023-03-21T17:40:10.525957Z","shell.execute_reply":"2023-03-21T17:40:10.536328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"write_tfrecords('train', dft)\nwrite_tfrecords('valid', dfv, 25)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T17:40:10.540145Z","iopub.execute_input":"2023-03-21T17:40:10.540863Z","iopub.status.idle":"2023-03-21T17:40:41.579247Z","shell.execute_reply.started":"2023-03-21T17:40:10.540821Z","shell.execute_reply":"2023-03-21T17:40:41.577728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Sanity check...","metadata":{}},{"cell_type":"markdown","source":"# TO BE CONTINUED...","metadata":{}}]}