{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# BirdCLEF TFWriter\n\nThe following is an approach to writing TFRecords for BirdCLEF.\n\n# Imports/Setup","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport os\nimport tensorflow as tf\nimport librosa\nimport tensorflow.keras as keras\nimport tensorflow_io as tfio\nimport requests","metadata":{"execution":{"iopub.status.busy":"2023-04-14T23:37:34.789271Z","iopub.execute_input":"2023-04-14T23:37:34.790268Z","iopub.status.idle":"2023-04-14T23:37:44.962187Z","shell.execute_reply.started":"2023-04-14T23:37:34.790217Z","shell.execute_reply":"2023-04-14T23:37:44.960891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"taxonomy = '/kaggle/input/birdclef-2023/eBird_Taxonomy_v2021.csv'\nmetadata = '/kaggle/input/birdclef-2023/train_metadata.csv'","metadata":{"execution":{"iopub.status.busy":"2023-04-14T23:37:44.964939Z","iopub.execute_input":"2023-04-14T23:37:44.965962Z","iopub.status.idle":"2023-04-14T23:37:44.972142Z","shell.execute_reply.started":"2023-04-14T23:37:44.965907Z","shell.execute_reply":"2023-04-14T23:37:44.970688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dftax = pd.read_csv(taxonomy)\ndfmeta = pd.read_csv(metadata)","metadata":{"execution":{"iopub.status.busy":"2023-04-14T23:37:44.973433Z","iopub.execute_input":"2023-04-14T23:37:44.974068Z","iopub.status.idle":"2023-04-14T23:37:45.188651Z","shell.execute_reply.started":"2023-04-14T23:37:44.974032Z","shell.execute_reply":"2023-04-14T23:37:45.187603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfmeta","metadata":{"execution":{"iopub.status.busy":"2023-04-14T23:37:45.192095Z","iopub.execute_input":"2023-04-14T23:37:45.192468Z","iopub.status.idle":"2023-04-14T23:37:45.240768Z","shell.execute_reply.started":"2023-04-14T23:37:45.192434Z","shell.execute_reply":"2023-04-14T23:37:45.239659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exclude Some Labels    : (\n\nI have chosen to exclude any label with less than 50 instances. \n\nLater on I will import instances from **xeno-canto.org** to ensure every label has the bare minimum.","metadata":{}},{"cell_type":"code","source":"exclude_lst = dfmeta['primary_label'].value_counts()[\n    dfmeta['primary_label'].value_counts() < 50\n].index.to_list()","metadata":{"execution":{"iopub.status.busy":"2023-04-14T23:37:45.241999Z","iopub.execute_input":"2023-04-14T23:37:45.242375Z","iopub.status.idle":"2023-04-14T23:37:45.258785Z","shell.execute_reply.started":"2023-04-14T23:37:45.242308Z","shell.execute_reply":"2023-04-14T23:37:45.257419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"name_exclude_lst = dfmeta['common_name'].value_counts()[\n    dfmeta['common_name'].value_counts() < 50\n].index.to_list()","metadata":{"execution":{"iopub.status.busy":"2023-04-14T23:37:45.260882Z","iopub.execute_input":"2023-04-14T23:37:45.261811Z","iopub.status.idle":"2023-04-14T23:37:45.271827Z","shell.execute_reply.started":"2023-04-14T23:37:45.261761Z","shell.execute_reply":"2023-04-14T23:37:45.270765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def include(row):\n    if row['primary_label'] in exclude_lst:\n        return False\n    return True","metadata":{"execution":{"iopub.status.busy":"2023-04-14T23:37:45.273185Z","iopub.execute_input":"2023-04-14T23:37:45.273655Z","iopub.status.idle":"2023-04-14T23:37:45.282899Z","shell.execute_reply.started":"2023-04-14T23:37:45.273601Z","shell.execute_reply":"2023-04-14T23:37:45.281873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"include = dfmeta.apply(include, axis=1)\ndfmeta_ex = dfmeta[include]","metadata":{"execution":{"iopub.status.busy":"2023-04-14T23:37:45.284664Z","iopub.execute_input":"2023-04-14T23:37:45.285349Z","iopub.status.idle":"2023-04-14T23:37:45.555350Z","shell.execute_reply.started":"2023-04-14T23:37:45.285286Z","shell.execute_reply":"2023-04-14T23:37:45.554346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Getting Extra Instances\n\nWe are going to use the **xeno-canto** API to query all the instances and hopefully fill the minimum requirement of 10 instances per label.","metadata":{}},{"cell_type":"code","source":"new_arr = []","metadata":{"execution":{"iopub.status.busy":"2023-04-14T17:30:38.039978Z","iopub.execute_input":"2023-04-14T17:30:38.040839Z","iopub.status.idle":"2023-04-14T17:30:38.046987Z","shell.execute_reply.started":"2023-04-14T17:30:38.040788Z","shell.execute_reply":"2023-04-14T17:30:38.045153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for name, label in zip(name_exclude_lst, exclude_lst):\n    \n    os.mkdir(label)\n\n    response = requests.get('https://xeno-canto.org/api/2/recordings?query=' + name)\n    resp_json = response.json()\n    \n    print('importing ... {}: {}'.format(name, label))\n    \n    for record in resp_json['recordings']:\n        bird_id = record['id']\n        filename = '/kaggle/working/' + label + '/' + bird_id + '.mp3'\n        with open(filename, 'wb') as f:\n            f.write(requests.get(record['file']).content)\n            \n        new_arr.append([label, bird_id, filename])","metadata":{"execution":{"iopub.status.busy":"2023-04-14T17:30:38.051339Z","iopub.execute_input":"2023-04-14T17:30:38.051740Z","iopub.status.idle":"2023-04-14T17:30:47.570769Z","shell.execute_reply.started":"2023-04-14T17:30:38.051701Z","shell.execute_reply":"2023-04-14T17:30:47.569209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_extra = pd.DataFrame(new_arr, columns=['primary_label', 'id', 'filename'])","metadata":{"execution":{"iopub.status.busy":"2023-04-14T17:30:47.571914Z","iopub.status.idle":"2023-04-14T17:30:47.573399Z","shell.execute_reply.started":"2023-04-14T17:30:47.573096Z","shell.execute_reply":"2023-04-14T17:30:47.573128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_id(row):\n    return row['url'].split('/')[-1]\n\ndfmeta['id'] = dfmeta.apply(get_id, axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-04-14T17:30:47.574678Z","iopub.status.idle":"2023-04-14T17:30:47.575482Z","shell.execute_reply.started":"2023-04-14T17:30:47.575235Z","shell.execute_reply":"2023-04-14T17:30:47.575265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfmeta = pd.concat([dfmeta, df_extra]).drop_duplicates(subset=['primary_label', 'id'])","metadata":{"execution":{"iopub.status.busy":"2023-04-14T17:30:47.577329Z","iopub.status.idle":"2023-04-14T17:30:47.578460Z","shell.execute_reply.started":"2023-04-14T17:30:47.578163Z","shell.execute_reply":"2023-04-14T17:30:47.578194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfmeta.index = [x for x in range(dfmeta.shape[0])]","metadata":{"execution":{"iopub.status.busy":"2023-04-14T17:30:47.579789Z","iopub.status.idle":"2023-04-14T17:30:47.581019Z","shell.execute_reply.started":"2023-04-14T17:30:47.580729Z","shell.execute_reply":"2023-04-14T17:30:47.580758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfmeta","metadata":{"execution":{"iopub.status.busy":"2023-04-14T17:30:47.582432Z","iopub.status.idle":"2023-04-14T17:30:47.582896Z","shell.execute_reply.started":"2023-04-14T17:30:47.582646Z","shell.execute_reply":"2023-04-14T17:30:47.582670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfprob = dfmeta['primary_label'].value_counts() / dfmeta.shape[0]\ndfprob = 264 / dfprob / 4472424.0\ndfprob","metadata":{"execution":{"iopub.status.busy":"2023-04-14T19:54:36.501017Z","iopub.execute_input":"2023-04-14T19:54:36.501437Z","iopub.status.idle":"2023-04-14T19:54:36.514828Z","shell.execute_reply.started":"2023-04-14T19:54:36.501403Z","shell.execute_reply":"2023-04-14T19:54:36.513456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Handle the Data\n\nSince we are dealing with such an inbalance of so many classes the easiest approach is to group the elements by label and then sample each group equally.","metadata":{}},{"cell_type":"code","source":"dfmeta_g = dfmeta.groupby(\"primary_label\")\ndft = dfmeta_g.sample(frac=0.8, random_state=42)\ndfv = dfmeta.drop(dft.index)","metadata":{"execution":{"iopub.status.busy":"2023-04-14T19:56:01.596802Z","iopub.execute_input":"2023-04-14T19:56:01.597211Z","iopub.status.idle":"2023-04-14T19:56:01.776587Z","shell.execute_reply.started":"2023-04-14T19:56:01.597176Z","shell.execute_reply":"2023-04-14T19:56:01.775258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dft['primary_label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-14T19:56:02.989287Z","iopub.execute_input":"2023-04-14T19:56:02.989727Z","iopub.status.idle":"2023-04-14T19:56:03.002930Z","shell.execute_reply.started":"2023-04-14T19:56:02.989691Z","shell.execute_reply":"2023-04-14T19:56:03.001344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dft['primary_label'].value_counts().to_csv('value_counts.csv')","metadata":{"execution":{"iopub.status.busy":"2023-04-14T19:51:50.329706Z","iopub.execute_input":"2023-04-14T19:51:50.330137Z","iopub.status.idle":"2023-04-14T19:51:50.344731Z","shell.execute_reply.started":"2023-04-14T19:51:50.330103Z","shell.execute_reply":"2023-04-14T19:51:50.343457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfv['primary_label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-14T19:56:05.599578Z","iopub.execute_input":"2023-04-14T19:56:05.600301Z","iopub.status.idle":"2023-04-14T19:56:05.614046Z","shell.execute_reply.started":"2023-04-14T19:56:05.600251Z","shell.execute_reply":"2023-04-14T19:56:05.612619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(set(dft['primary_label'].value_counts().keys().tolist()).intersection(set(dfv['primary_label'].value_counts().keys().tolist())))","metadata":{"execution":{"iopub.status.busy":"2023-04-14T19:56:07.920457Z","iopub.execute_input":"2023-04-14T19:56:07.920894Z","iopub.status.idle":"2023-04-14T19:56:07.933326Z","shell.execute_reply.started":"2023-04-14T19:56:07.920853Z","shell.execute_reply":"2023-04-14T19:56:07.932035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Just to make sure we got what we want...","metadata":{}},{"cell_type":"markdown","source":"# One Hot Encoding\nNext we need to setup one hot encoding. Create a tensorflow TextVect layer and predict.","metadata":{}},{"cell_type":"code","source":"unique_lst = dft['primary_label'].value_counts().index.tolist()\nv_len = len(unique_lst)","metadata":{"execution":{"iopub.status.busy":"2023-04-14T23:39:07.905258Z","iopub.execute_input":"2023-04-14T23:39:07.905726Z","iopub.status.idle":"2023-04-14T23:39:07.913713Z","shell.execute_reply.started":"2023-04-14T23:39:07.905685Z","shell.execute_reply":"2023-04-14T23:39:07.912349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(unique_lst).to_csv('unique_lst.csv')","metadata":{"execution":{"iopub.status.busy":"2023-04-14T23:39:27.011631Z","iopub.execute_input":"2023-04-14T23:39:27.012943Z","iopub.status.idle":"2023-04-14T23:39:27.024199Z","shell.execute_reply.started":"2023-04-14T23:39:27.012879Z","shell.execute_reply":"2023-04-14T23:39:27.023197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_vec =  keras.layers.TextVectorization(\n    max_tokens=v_len+1,\n    output_mode='multi_hot',\n    vocabulary=unique_lst\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-14T23:39:19.368608Z","iopub.execute_input":"2023-04-14T23:39:19.369497Z","iopub.status.idle":"2023-04-14T23:39:19.383375Z","shell.execute_reply.started":"2023-04-14T23:39:19.369449Z","shell.execute_reply":"2023-04-14T23:39:19.381887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"By default tensorflow has a [UNK] token. To deal with it we add a lambda layer taking everything after index 0.","metadata":{}},{"cell_type":"code","source":"model = keras.models.Sequential()\nmodel.add(keras.Input(shape=(1,), dtype=tf.string))\nmodel.add(text_vec)\nmodel.add(keras.layers.Lambda(lambda x: x[:, 1:]))","metadata":{"execution":{"iopub.status.busy":"2023-04-14T23:39:16.565594Z","iopub.execute_input":"2023-04-14T23:39:16.566607Z","iopub.status.idle":"2023-04-14T23:39:16.651965Z","shell.execute_reply.started":"2023-04-14T23:39:16.566562Z","shell.execute_reply":"2023-04-14T23:39:16.650598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# IO Functions\nThe function will create a TFExample to write to a record.","metadata":{}},{"cell_type":"code","source":"from tensorflow.train import BytesList, FloatList, Int64List\nfrom tensorflow.train import Feature, Features, Example\n\ndef get_example(audio_path, label):\n    audio, sr = librosa.load(audio_path, sr=None)\n    a = tf.convert_to_tensor(audio)\n    b = tf.expand_dims(a, axis=1)\n    \n    return Example(\n        features=Features(\n            feature={\n                'audio': Feature(bytes_list=BytesList(value=[tfio.audio.encode_mp3(b, sr).numpy()])),\n                'sr': Feature(int64_list=Int64List(value=[sr])),\n                'label': Feature(bytes_list=BytesList(value=[tf.io.serialize_tensor(label).numpy()]))\n            }\n        )\n    )","metadata":{"execution":{"iopub.status.busy":"2023-04-14T17:30:47.600960Z","iopub.status.idle":"2023-04-14T17:30:47.601627Z","shell.execute_reply.started":"2023-04-14T17:30:47.601406Z","shell.execute_reply":"2023-04-14T17:30:47.601430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# TFWriter\nFinally, we create a set of paths, setup our writers/stack, and iterate through the dataset. At each iteration we choose the record based on how it will % into the n_shards (adding more randomness) and then compute the image and one-hot encoded label.\n\n> making sure to set verbose=0 so we can write in peace","metadata":{}},{"cell_type":"code","source":"from contextlib import ExitStack\nfrom tqdm import tqdm\n\ndef write_tfrecords(name, dataset, n_shards=50):\n    paths = [\"{}.tfrecord-{:02d}-of-{:02d}\".format(name, index, n_shards) for index in range(n_shards)]\n    \n    with ExitStack() as stack: \n        writers = [stack.enter_context(tf.io.TFRecordWriter(path)) for path in paths]\n        \n        for i, row in tqdm(dataset.iterrows()):\n            shard = i % n_shards\n            if row['filename'][-3:] == 'ogg':\n                audio_path = '/kaggle/input/birdclef-2023/train_audio/' + row['filename']\n            else:\n                audio_path = row['filename']\n            label = model.predict([row['primary_label']], verbose=0).tolist()[0]\n            example = get_example(audio_path, label)\n            writers[shard].write(example.SerializeToString())\n            \n    print('Done writing ' + name + '.')","metadata":{"execution":{"iopub.status.busy":"2023-04-14T17:30:47.602885Z","iopub.status.idle":"2023-04-14T17:30:47.604345Z","shell.execute_reply.started":"2023-04-14T17:30:47.603876Z","shell.execute_reply":"2023-04-14T17:30:47.603933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"write_tfrecords('train', dft)\nwrite_tfrecords('valid', dfv, 25)","metadata":{"execution":{"iopub.status.busy":"2023-04-14T17:30:47.606124Z","iopub.status.idle":"2023-04-14T17:30:47.606945Z","shell.execute_reply.started":"2023-04-14T17:30:47.606662Z","shell.execute_reply":"2023-04-14T17:30:47.606692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Sanity check...","metadata":{}},{"cell_type":"markdown","source":"# Clean Up the Extra Recordings\nIterate through all the extra records and delete their references.","metadata":{}},{"cell_type":"code","source":"def delete_file(filepath):\n    print('deleting ' + filepath + ' from local')\n    os.remove('/kaggle/working/' + filepath)\n    \ndef clear_all_local():\n    for k in os.listdir('/kaggle/working/'):\n        if k == '.virtual_documents':\n            continue\n        delete_file(k)","metadata":{"execution":{"iopub.status.busy":"2023-04-14T17:30:47.608354Z","iopub.status.idle":"2023-04-14T17:30:47.609098Z","shell.execute_reply.started":"2023-04-14T17:30:47.608838Z","shell.execute_reply":"2023-04-14T17:30:47.608884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\n\nfor i in exclude_lst:\n    try:\n        shutil.rmtree('/kaggle/working/' + i)\n    except:\n        pass","metadata":{"execution":{"iopub.status.busy":"2023-04-14T17:30:47.610462Z","iopub.status.idle":"2023-04-14T17:30:47.611177Z","shell.execute_reply.started":"2023-04-14T17:30:47.610941Z","shell.execute_reply":"2023-04-14T17:30:47.610968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# TO BE CONTINUED...","metadata":{}}]}