{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_io as tfio\nimport os\nimport pandas as pd\nfrom IPython import display","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-10T23:05:52.591529Z","iopub.execute_input":"2023-05-10T23:05:52.591916Z","iopub.status.idle":"2023-05-10T23:05:52.597029Z","shell.execute_reply.started":"2023-05-10T23:05:52.591877Z","shell.execute_reply":"2023-05-10T23:05:52.596111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    competition = \"birdclef-2023\"\n    \n    frame_duration = 5 # seconde\n    sample_rate = 32_000\n    desired_sample_rate = 16_000\n    frame_length = frame_duration*desired_sample_rate\n    output = f\"{competition}/\"\n    train_dir = os.path.join(output, \"train\")\n    valid_dir = os.path.join(output, \"validation\")\n    \n    root = \"/kaggle/input/birdclef-2023\"\n    train_dir_inp = os.path.join(root, \"train_audio/\")\n    metada_file = os.path.join(root, \"train_metadata.csv\")\n    \n    records_size = 2024","metadata":{"execution":{"iopub.status.busy":"2023-05-10T23:05:52.598993Z","iopub.execute_input":"2023-05-10T23:05:52.599687Z","iopub.status.idle":"2023-05-10T23:05:52.611408Z","shell.execute_reply.started":"2023-05-10T23:05:52.599653Z","shell.execute_reply":"2023-05-10T23:05:52.610173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check GPU\nif tf.config.list_physical_devices('GPU'):\n    strategy = tf.distribute.MirroredStrategy()\n    device = \"GPU\"\n# Use CPU\nelse:\n    strategy = tf.distribute.get_strategy()\n    device = \"CPU\"\n\nprint(\"Number of accelerators \", strategy.num_replicas_in_sync,\"|\", device)","metadata":{"execution":{"iopub.status.busy":"2023-05-10T23:05:52.613880Z","iopub.execute_input":"2023-05-10T23:05:52.614502Z","iopub.status.idle":"2023-05-10T23:05:52.634185Z","shell.execute_reply.started":"2023-05-10T23:05:52.614468Z","shell.execute_reply":"2023-05-10T23:05:52.632959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def audio2frames(audio, label):\n    #----------------------\n    # Audio\n    #----------------------\n    frames = tf.signal.frame(\n        audio, \n        frame_length=Config.frame_length, \n        frame_step=Config.frame_length,\n        pad_end=True, \n        name=\"frames_audio\"\n    )\n    #----------------------\n    # Label\n    #----------------------\n    nlab = tf.shape(frames)[0]\n    label = tf.repeat(label, repeats=nlab)\n    return frames, label","metadata":{"execution":{"iopub.status.busy":"2023-05-10T23:05:52.636184Z","iopub.execute_input":"2023-05-10T23:05:52.636935Z","iopub.status.idle":"2023-05-10T23:05:52.643300Z","shell.execute_reply.started":"2023-05-10T23:05:52.636904Z","shell.execute_reply":"2023-05-10T23:05:52.642151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_file(path, label):\n    file = tf.io.read_file(path)\n    aud = tfio.audio.decode_vorbis(file)\n    aud = tfio.audio.resample(aud, rate_in=Config.sample_rate, rate_out=Config.desired_sample_rate)\n    aud = tf.squeeze(aud, axis=-1)\n    aud = tf.cast(aud, tf.float32)\n    frames, label = audio2frames(aud, label)\n    return frames, label ","metadata":{"execution":{"iopub.status.busy":"2023-05-10T23:05:52.645039Z","iopub.execute_input":"2023-05-10T23:05:52.646046Z","iopub.status.idle":"2023-05-10T23:05:52.654714Z","shell.execute_reply.started":"2023-05-10T23:05:52.645850Z","shell.execute_reply":"2023-05-10T23:05:52.652901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(Config.metada_file)\nencode = dict(map(reversed, dict(enumerate(df[\"primary_label\"].unique(), 1)).items()))\ndf[\"filepath\"] = df[\"filename\"].map(lambda x: os.path.join(Config.train_dir_inp, x))\ndf[\"label\"] = df[\"primary_label\"].replace(encode)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-10T23:05:52.658424Z","iopub.execute_input":"2023-05-10T23:05:52.659282Z","iopub.status.idle":"2023-05-10T23:05:53.860970Z","shell.execute_reply.started":"2023-05-10T23:05:52.659250Z","shell.execute_reply":"2023-05-10T23:05:53.859899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = df.sample(1)\npath = sample[\"filepath\"].values[0]\nlabel = sample[\"label\"].values[0]\nframe, label = read_file(path, label)\ndisplay.Audio(frame[0].numpy(), rate=Config.desired_sample_rate)","metadata":{"execution":{"iopub.status.busy":"2023-05-10T23:05:53.862910Z","iopub.execute_input":"2023-05-10T23:05:53.863723Z","iopub.status.idle":"2023-05-10T23:05:54.660417Z","shell.execute_reply.started":"2023-05-10T23:05:53.863684Z","shell.execute_reply":"2023-05-10T23:05:54.658524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def standardize_tensor(X, feature_range=(0,1), copy=True, clip=False):\n    X_min = tf.reduce_min(X, axis=0)\n    X_max = tf.reduce_max(X, axis=0)\n    X_std = (X - X_min) / (X_max - X_min)\n    X_scaled = X_std * (feature_range[1] - feature_range[0]) + feature_range[0]\n    if clip:\n        X_scaled = tf.clip_by_value(X_scaled, feature_range[0], feature_range[1])\n    if copy:\n        X_scaled = tf.identity(X_scaled)\n    return X_scaled\n","metadata":{"execution":{"iopub.status.busy":"2023-05-10T23:05:54.661854Z","iopub.execute_input":"2023-05-10T23:05:54.662897Z","iopub.status.idle":"2023-05-10T23:05:54.670328Z","shell.execute_reply.started":"2023-05-10T23:05:54.662849Z","shell.execute_reply":"2023-05-10T23:05:54.669397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm.notebook import tqdm\nimport tensorflow_hub as hub\n\nyamnetlayer = hub.KerasLayer('https://tfhub.dev/google/yamnet/1', trainable=False)\n\n\n\ndef extract_log_mel(frame):\n    frame = standardize_tensor(frame)\n    _, _, log_mel = yamnetlayer(frame)\n    return log_mel\n\ndef create_dataset(paths, labels):\n    with strategy.scope():\n        dataset = tf.data.Dataset.from_tensor_slices((paths, labels))\n        dataset = dataset.map(read_file, num_parallel_calls=tf.data.AUTOTUNE)\n        dataset = dataset.prefetch(buffer_size=tf.data.AUTOTUNE)\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2023-05-10T23:05:54.671746Z","iopub.execute_input":"2023-05-10T23:05:54.673058Z","iopub.status.idle":"2023-05-10T23:05:59.550624Z","shell.execute_reply.started":"2023-05-10T23:05:54.673024Z","shell.execute_reply":"2023-05-10T23:05:59.549634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lg_mel = extract_log_mel(standardize_tensor(frame[0], feature_range=[-1,1]))","metadata":{"execution":{"iopub.status.busy":"2023-05-10T23:05:59.553022Z","iopub.execute_input":"2023-05-10T23:05:59.553389Z","iopub.status.idle":"2023-05-10T23:06:04.407737Z","shell.execute_reply.started":"2023-05-10T23:05:59.553354Z","shell.execute_reply":"2023-05-10T23:06:04.406748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def _bytes_feature(value):\n    \"\"\"Returns a bytes_list from a string / byte.\"\"\"\n    if isinstance(value, type(tf.constant(0))): # if value ist tensor\n        value = value.numpy() # get value of tensor\n    return tf.train.Feature(bytes_list=tf.train.BytesList(value=[value]))\n\ndef _int64_feature(value):\n    \"\"\"Returns an int64_list from a bool / enum / int / uint.\"\"\"\n    return tf.train.Feature(int64_list=tf.train.Int64List(value=[value]))\n\ndef serialize_array(array):\n    array = tf.io.serialize_tensor(array)\n    return array\n\ndef create_example(frame, label):\n    log_mel = extract_log_mel(standardize_tensor(frame, feature_range=[-1, 1]))\n    feature = {\n        \"log_mel\": _bytes_feature(serialize_array(log_mel)),\n        \"label\": _int64_feature(label)\n    }\n    example = tf.train.Example(features=tf.train.Features(feature=feature))\n    return example","metadata":{"execution":{"iopub.status.busy":"2023-05-10T23:06:04.410930Z","iopub.execute_input":"2023-05-10T23:06:04.411306Z","iopub.status.idle":"2023-05-10T23:06:04.422961Z","shell.execute_reply.started":"2023-05-10T23:06:04.411261Z","shell.execute_reply":"2023-05-10T23:06:04.422031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def frame2records(paths, labels, path_records, record_size=2048, name='train'):\n    num_records = 0\n    count = 0\n    writer = None\n    dataset = create_dataset(paths, labels)\n    total = 0 \n    os.makedirs(path_records, exist_ok=True)\n    for fl, ll in tqdm(dataset):\n        ll = tf.cast(ll, tf.int16)\n        examples = map(create_example, tf.unstack(fl, axis=0), tf.unstack(ll, axis=0))\n        for example in examples:\n            if writer is None:\n                file = f\"{path_records}/file_{num_records}.tfrec\"\n                writer = tf.io.TFRecordWriter(file)\n            writer.write(example.SerializeToString())\n            count += 1\n            total += 1\n            if count > record_size:\n                writer.close()\n                num_records += 1\n                writer = None\n                count = 0\n                \n    if writer is not None:\n        writer.close()\n        num_records += 1\n        \n    print(\"| Write a metadata file --------------------------\")\n    dff = pd.DataFrame([[len(paths), record_size, total, Config.frame_duration, Config.desired_sample_rate]], \n                   columns=[\"Shape\", \"Records_size\", \"NumberOfRecords\", \"frame_duration\", \"sample_rate\"])\n    dff.to_csv(os.path.join(Config.output, f\"{name}.csv\"))\n    print(\"---------------------> Metadata saved with success\")\n    return None","metadata":{"execution":{"iopub.status.busy":"2023-05-10T23:06:04.427513Z","iopub.execute_input":"2023-05-10T23:06:04.428193Z","iopub.status.idle":"2023-05-10T23:06:04.439770Z","shell.execute_reply.started":"2023-05-10T23:06:04.428158Z","shell.execute_reply":"2023-05-10T23:06:04.438683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test \nsample = df.sample(10)\nfilepaths = sample[\"filepath\"]\nlabels = sample[\"label\"]\ndirectory = os.path.join(Config.output, \"test\")\nframe2records(filepaths, labels, path_records=directory, name='test', record_size=512)","metadata":{"execution":{"iopub.status.busy":"2023-05-10T23:06:04.441079Z","iopub.execute_input":"2023-05-10T23:06:04.441473Z","iopub.status.idle":"2023-05-10T23:06:06.552426Z","shell.execute_reply.started":"2023-05-10T23:06:04.441442Z","shell.execute_reply":"2023-05-10T23:06:06.551368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -r $Config.output","metadata":{"execution":{"iopub.status.busy":"2023-05-10T23:06:06.554480Z","iopub.execute_input":"2023-05-10T23:06:06.555270Z","iopub.status.idle":"2023-05-10T23:06:08.501040Z","shell.execute_reply.started":"2023-05-10T23:06:06.555233Z","shell.execute_reply":"2023-05-10T23:06:08.499747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ndef train_split_label_balanced(df, labels, test_size=0.3, random_state=1024, seuil=8):\n    # Like we observe before some class are less represented in the dataset \n    # So We want to define this function to split the data taking into account \n    # The weigth of the labels and we define a int seuil for the class that only get\n    # Get in the train_df, in this way the train set will complete will all label sample\n    df_train, df_valid = None, None \n    for lab in tqdm(labels):\n        dff = df[df[\"primary_label\"] == lab]\n        if dff.shape[0] <= seuil:\n            df_train = dff.copy() if df_train is None else pd.concat([df_train, dff.copy()])\n        else:\n            dff_train, dff_valid = train_test_split(dff, test_size=test_size, random_state=random_state)\n            df_train = dff_train if df_train is None else pd.concat([df_train, dff_train])\n            df_valid = dff_valid if df_valid is None else pd.concat([df_valid, dff_valid])\n    return df_train, df_valid","metadata":{"execution":{"iopub.status.busy":"2023-05-10T23:06:08.503424Z","iopub.execute_input":"2023-05-10T23:06:08.503848Z","iopub.status.idle":"2023-05-10T23:06:09.024813Z","shell.execute_reply.started":"2023-05-10T23:06:08.503797Z","shell.execute_reply":"2023-05-10T23:06:09.023914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.utils import shuffle\ndf = shuffle(df, random_state=1024)\n\ntrain_df, valid_df = train_split_label_balanced(df, labels=df[\"primary_label\"].unique(), test_size=0.3)","metadata":{"execution":{"iopub.status.busy":"2023-05-10T23:06:09.026527Z","iopub.execute_input":"2023-05-10T23:06:09.026924Z","iopub.status.idle":"2023-05-10T23:06:10.759282Z","shell.execute_reply.started":"2023-05-10T23:06:09.026887Z","shell.execute_reply":"2023-05-10T23:06:10.751301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Config.train_dir","metadata":{"execution":{"iopub.status.busy":"2023-05-10T23:06:10.760608Z","iopub.execute_input":"2023-05-10T23:06:10.761180Z","iopub.status.idle":"2023-05-10T23:06:10.768298Z","shell.execute_reply.started":"2023-05-10T23:06:10.761127Z","shell.execute_reply":"2023-05-10T23:06:10.767259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Training size | Size: \", len(train_df))\nfilepaths = train_df[\"filepath\"]\nlabels = train_df[\"label\"]\nframe2records(filepaths, labels, path_records=Config.train_dir, record_size=Config.records_size, name=\"train\")","metadata":{"execution":{"iopub.status.busy":"2023-05-10T23:06:10.769729Z","iopub.execute_input":"2023-05-10T23:06:10.770177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Config.valid_dir","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Validation size | Size: \", len(valid_df))\nfilepaths = valid_df[\"filepath\"]\nlabels = valid_df[\"label\"]\nframe2records(filepaths, labels, path_records=Config.valid_dir, record_size=Config.records_size, name=\"validation\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_metadata = pd.read_csv(os.path.join(\"/kaggle/working/birdclef-2023\", \"train.csv\"))\ndf_train_metadata","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_metadata = pd.read_csv(os.path.join(\"/kaggle/working/birdclef-2023\", \"validation.csv\"))\ndf_train_metadata","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Nice Notebook !!! -> End\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}