{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q tensorflow-addons==0.18.0\n!pip install -q tensorflow-probability==0.19.0\n\n# other utilies\n!pip install -q scikit-learn\n# !pip install -q tensorflow==2.12.0\n!pip install -q tensorflow-io==0.32.0","metadata":{"execution":{"iopub.status.busy":"2023-05-05T01:09:32.054113Z","iopub.execute_input":"2023-05-05T01:09:32.054784Z","iopub.status.idle":"2023-05-05T01:09:55.199584Z","shell.execute_reply.started":"2023-05-05T01:09:32.054752Z","shell.execute_reply":"2023-05-05T01:09:55.198533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport tensorflow as tf\nimport tensorflow_io as tfio\nimport tensorflow_probability as tfp\nimport random\nfrom kaggle_datasets import KaggleDatasets\nimport math\nfrom sklearn.model_selection import train_test_split\n# gpus = tf.config.list_logical_devices('GPU')\n# strategy = tf.distribute.MirroredStrategy(gpus)\ntpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect(tpu='local')\n# print('Device:', tpu.master())\n# instantiate a distribution strategy\ntpu_strategy = tf.distribute.TPUStrategy(tpu)\nBASE_PATH=\"/kaggle/input/birdclef-2023\"\n# GCS_PATH =BASE_PATH\n# from kaggle_secrets import UserSecretsClient\nGCS_PATH = KaggleDatasets().get_gcs_path(BASE_PATH.split('/')[-1])\n# print(BASE_PATH,GCS_PATH)\ndf = pd.read_csv(f'{BASE_PATH}/train_metadata.csv')\ncounts = df.primary_label.value_counts()\nclass_dist = df['primary_label'].value_counts()\n# identify the classes that have less than the threshold number of samples\ndown_classes = class_dist[class_dist < 20 ].index.tolist()\n# create an empty list to store the upsampled dataframes\nup_dfs = []\n# loop through the undersampled classes and upsample them\nfor c in down_classes:\n    # get the dataframe for the current class\n    class_df = df.query(\"primary_label==@c\")\n    # find number of samples to add\n    num_up = 25- class_df.shape[0]\n    # upsample the dataframe\n    class_df = class_df.sample(n=num_up, replace=True, random_state=2)\n    # append the upsampled dataframe to the list\n    up_dfs.append(class_df)\n# concatenate the upsampled dataframes and the original dataframe\ndf = pd.concat([df] + up_dfs, axis=0, ignore_index=True)\n# identify the classes that have less than the threshold number of samples\n# concatenate the upsampled dataframes and the original dataframe\ndf['filepath'] = GCS_PATH + '/train_audio/' + df.filename\n# print(pd.factorize(df[\"primary_label\"]).head())\ns={}\nt=0\nfor i in df[\"primary_label\"]:\n    if i not in s.keys():\n        s[i]=t\n        t+=1\nh=[]\ndf[\"target\"] = pd.factorize(df[\"primary_label\"])[0].astype(int)\nfor i in range(len(df[\"secondary_labels\"])):\n    # h.append(tf.cast(tf.one_hot([df.target[i]], 264), tf.float32))\n    # print(df.target[i])\n    if df.secondary_labels[i]!='[]':\n        g=[0]*264\n        f=df.secondary_labels[i][1:-1].split(',')\n        # print(f)\n        for j in f:\n            g[df.target[i]]=0.7\n            j=j.split(\"'\")\n            # print(s[j[1]])\n            g[s[j[1]]]=0.3/len(f)\n        g=tf.reshape(g, [1,264])\n        g=tf.cast(g, tf.float32)\n        h.append(g)\n    else:\n        h.append(tf.cast(tf.one_hot([df.target[i]], 264), tf.float32))\nh=np.array(h)\nindex1 = np.arange(len(df))\ntrain_df,valid_df = train_test_split(index1, test_size=0.25, random_state=4)\ntrain_paths = df.filepath[train_df].values; train_labels2=h[train_df]\nvalid_paths =df.filepath[valid_df].values; valid_labels= h[valid_df]\nindex = np.arange(len(train_paths))\nnp.random.shuffle(index)\ntrain_paths = train_paths[index]\ntrain_labels = train_labels2[index]\ndef audio_augmenter(dim=320000):\n    def augment(audio, dim=dim):\n        if random.randint(1,2) == 1:\n            audio = AudioAug(audio)\n        audio = tf.reshape(audio, [dim])\n        return audio\n    def augment_with_labels(audio, label):\n        return augment(audio), label\n    return augment_with_labels\ndef AudioAug(audio):\n    # Apply time shift and Gaussian noise to the audio signal\n    if  random.randint(1,2) == 1:\n        # Calculate random shift value\n        shift = random_int(shape=[], minval=0, maxval=320000)\n        # Randomly set the shift to be negative with 50% probability\n        if random.randint(1,2) == 1:\n            shift = -shift\n        # Roll the audio signal by the shift value\n        audio = tf.roll(audio, shift, axis=0)\n    # print(audio)\n    p=tf.argmax(audio,axis=0)\n    p=float(p)\n    std = random.uniform(0.01*p, 0.025*p)\n    # Randomly apply Gaussian noise with probability `prob`\n    if random.randint(1,2) == 1:\n        # Add random Gaussian noise to the audio signal\n        GN = tf.keras.layers.GaussianNoise(stddev=std)\n        audio = GN(audio, training=True)  # training=False don't apply noise to data\n    return audio\ndef SpecAug(spec):\n    # Convert the spectrogram to a 2D matrix and transpose it to get the shape [time, mel]\n    spec = tf.transpose(Img2Spec(spec), perm=[1, 0])\n    # Apply time and frequency masking to the spectrogram\n    spec = TimeFreqMask(spec, time_mask=30, freq_mask=20, prob=0.5)\n    # Transpose the spectrogram back to the original shape [mel, time] and convert it to an image\n    spec = tf.transpose(spec, perm=[1, 0])\n    spec = Spec2Img(spec)\n    return spec\n@tf.function\ndef Img2Spec(img):\n    # Extract the first channel of the image\n    return img[..., 0]\n@tf.function\ndef TimeFreqMask(spec, time_mask, freq_mask, prob=0.5):\n    if random.randint(1,2) == 1:\n        # Apply frequency masking to the spectrogram\n        spec = tfio.audio.freq_mask(spec, param=freq_mask)\n        # Apply time masking to the spectrogram\n        spec = tfio.audio.time_mask(spec, param=time_mask)\n    return spec\ndef spec_augmenter(dim=[128, 384]):\n    def augment(spec, dim=dim):\n        if random.randint(1,2) == 1:\n            spec = SpecAug(spec)\n        spec = tf.reshape(spec, [*dim, 3])\n        return spec\n    def augment_with_labels(spec, label):\n        return augment(spec), label\n    return augment_with_labels\ndef build_dataset(paths, labels=None, batch_size=128, target_size=[128, 256],\n                  audio_decode_fn=None,\n                  cache=True, cache_dir=\"\", drop_remainder=False,\n                  repeat=True, shuffle=1024,mixup=True):\n    # Create cache directory if cache is enabled\n#     if cache_dir != \"\" and cache is True:\n#         os.makedirs(cache_dir, exist_ok=True)\n    # Set default audio decode function if not provided\n    audio_decode_fn = audio_decoder( dim=320000)\n    spec_decode_fn = spec_decoder(dim=[128, 384])\n    audio_augment_fn = audio_augmenter(dim=320000)\n    spec_augment_fn=spec_augmenter(dim=[128, 384])\n    AUTO =tf.data.experimental.AUTOTUNE#自动生成线程读取数据\n    slices = paths if labels is None else (paths, labels)\n    # Create TensorFlow dataset from slices\n    ds = tf.data.Dataset.from_tensor_slices(slices)\n    # Map audio decode function to dataset\n    ds = ds.map(audio_decode_fn, num_parallel_calls=AUTO)#ds是数据集，map内是其对应函数\n#     Create TensorFlow dataset options\n    if repeat:\n        ds = ds.repeat(2)\n    opt = tf.data.Options()\n    # Shuffle dataset if shuffle is enabled\n    if shuffle:\n        ds = ds.shuffle(shuffle, seed=8)\n        opt.experimental_deterministic = False\n    ds = ds.with_options(opt)\n    opt.experimental_distribute.auto_shard_policy = tf.data.experimental.AutoShardPolicy.OFF\n    ds = ds.map(audio_augment_fn, num_parallel_calls=AUTO)\n    ds = ds.map(spec_decode_fn, num_parallel_calls=AUTO)\n    ds = ds.map(spec_augment_fn, num_parallel_calls=AUTO)\n    ds = ds.batch(batch_size, drop_remainder=drop_remainder)\n    if mixup:\n        ds = ds.map(MixUp(),num_parallel_calls=AUTO)\n    ds = ds.prefetch(AUTO)#预放入数据集到内存，使gpu，cpu一同运作\n    return ds\ndef MixUp(alpha=0.3):\n    \"\"\"Apply Spectrogram-MixUp augmentaiton. Apply Mixup to one batch and its shifted version\"\"\"\n    @tf.function\n    def apply(specs, labels, alpha=alpha):\n        if random.randint(0, 1) > 0:\n            return specs, labels\n        spec_shape = tf.shape(specs)\n        label_shape = tf.shape(labels)\n        # Select lambda from beta distribution\n        beta = tfp.distributions.Beta(alpha, alpha)\n        lam = beta.sample(1)[0]\n        # It's faster to roll the batch by one instead of shuffling it to create image pairs\n        specs = lam * specs + (1 - lam) * tf.roll(specs, shift=1, axis=0)  # mixup = [1, 2, 3]*lam + [3, 1, 2]*(1 - lam)\n        labels = lam * labels + (1 - lam) * tf.roll(labels, shift=1, axis=0)\n        specs = tf.reshape(specs, spec_shape)\n        labels = tf.reshape(labels, label_shape)\n        return specs, labels\n    return apply\ndef audio_decoder(dim=320000):\n    def get_audio(filepath):\n        file_bytes = tf.io.read_file(filepath)\n        audio = tfio.audio.decode_vorbis(file_bytes)  # decode .ogg file for .wave replace `decode_wav`\n        audio = tf.cast(audio, tf.float32)\n        audio = tf.squeeze(audio, axis=-1)\n        # if CFG.normalize:\n        #     audio = Normalize(audio)\n        return audio\n    def get_target(target):\n        target = tf.reshape(target, [264])\n        return target\n    def decode(path):\n        audio = get_audio(path)\n        audio = CropOrPad(audio, dim)  # crop or pad audio to keep a fixed length\n        audio = tf.reshape(audio, [dim])\n        return audio\n    def decode_with_labels(path, label):\n        label = get_target(label)\n        return decode(path), label\n    return decode_with_labels\ndef spec_decoder( dim=320000):\n    def decode(audio):\n        # Compute Spectrogram\n        spec = Audio2Spec(audio, spec_shape=dim, sr= 32000,\n                          nfft=2028, window=2048, fmin=20, fmax=16000)\n        # Spectrogram (H, W) to Image (H, W, C)\n        spec = Spec2Img(spec, num_channels=3)\n        spec = tf.reshape(spec, [*dim, 3])\n        return spec\n    def decode_with_labels(path, label):\n        return decode(path), label\n    return decode_with_labels\n@tf.function\ndef Spec2Img(spec, num_channels=3):\n    # If the original image has 1 channel, convert it to a 3 channel image by repeating the same image across channel axis\n    if num_channels > 1:\n        img = tf.tile(spec[..., tf.newaxis], [1, 1, num_channels])\n    else:\n        img = spec[..., tf.newaxis]\n    return img\ndef Audio2Spec(audio, spec_shape=[128, 384], sr=32000, nfft=2048, window=2048, fmin=500, fmax=14000):\n    spec_height = spec_shape[0]\n    spec_width = spec_shape[1]\n    audio_len = tf.shape(audio)[0]\n    hop_length = tf.cast((audio_len // (spec_width - 1)), tf.int32) # sample rate * duration / spec width - 1 == 627\n    spec = tfio.audio.spectrogram(audio, nfft=nfft, window=window, stride=hop_length)\n    mel_spec = tfio.audio.melscale(spec, rate=sr, mels=spec_height, fmin=fmin, fmax=fmax)\n#     db_mel_spec = tfio.audio.dbscale(mel_spec, top_db=80)\n    db_mel_spec = tf.transpose(mel_spec, perm=[1, 0])\n    if tf.shape(db_mel_spec)[1] > spec_width:\n        db_mel_spec = db_mel_spec[:, :spec_width]\n    db_mel_spec = tf.reshape(db_mel_spec, spec_shape)\n    return db_mel_spec\ndef random_int(shape=[], minval=0, maxval=1):\n    return tf.random.uniform(shape=shape, minval=minval, maxval=maxval, dtype=tf.int32)\ndef CropOrPad(audio, target_len, pad_mode='constant'):\n    # Get the length of the input audio\n    audio_len = tf.shape(audio)[0]\n    # If the length of the input audio is smaller than the target length, randomly pad the audio\n    if audio_len < target_len:\n        # Calculate the offset between the input audio and the target length\n        diff_len = (target_len - audio_len)\n        # Select a random location for padding\n        pad1 = random_int([], minval=0, maxval=diff_len)\n        # Calculate the second padding value\n        pad2 = diff_len - pad1\n        pad_len = [pad1, pad2]\n        # Apply padding to the audio data\n        audio = tf.pad(audio, paddings=[pad_len], mode=pad_mode)\n    # If the length of the input audio is larger than the target length, crop the audio\n    elif audio_len > target_len:\n        # Calculate the difference in length between the input audio and the target length\n        diff_len = (audio_len - target_len)\n#         Select a random location for cropping\n        idx = tf.random.uniform([], 0, diff_len, dtype=tf.int32)\n        # Crop the audio data\n        audio = audio[idx: (idx + target_len)]\n    # Reshape the audio data to the target length\n    audio = tf.reshape(audio, [target_len])\n    # Return the cropped or padded audio data\n    return audio\ntrain_ds = build_dataset(train_paths, train_labels,\n                         batch_size=128, cache=True, shuffle=True,\n                         drop_remainder=True,mixup=True)\nvalid_ds = build_dataset(valid_paths, valid_labels,\n                         batch_size=128, cache=True, shuffle=False,\n                          repeat=False, drop_remainder=True,mixup=False)\ndef get_lr_callback(batch_size=128, epochs=10):\n    # Returns a learning rate scheduler callback for a given batch size, mode, and number of epochs.\n    lr_start = 0.00005\n    lr_max = 0.00001 * batch_size\n    lr_min = 0.00001\n    lr_ramp_ep = 4\n    lr_sus_ep = 0\n    lr_decay = 0.8\n    # Function to update the lr\n    def lrfn(epoch):\n        if epoch < lr_ramp_ep:\n            lr = (lr_max - lr_start) / lr_ramp_ep * epoch + lr_start\n        elif epoch < lr_ramp_ep + lr_sus_ep:\n            lr = lr_max\n        else:\n            decay_total_epochs = epochs - lr_ramp_ep - lr_sus_ep + 3\n            decay_epoch_index = epoch - lr_ramp_ep - lr_sus_ep\n            phase = math.pi * decay_epoch_index / decay_total_epochs\n            cosine_decay = 0.5 * (1 + math.cos(phase))\n            lr = (lr_max - lr_min) * cosine_decay + lr_min\n        return lr\n    lr_callback = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose=False)\n    return lr_callback\ncallback=tf.keras.callbacks.ModelCheckpoint('best_model2.h5',\n                                           monitor='val_loss',\n                                           verbose=1,\n                                           save_best_only=True,\n                                           save_weights_only=False)\ndef creat_model():\n    model1=tf.keras.applications.efficientnet_v2.EfficientNetV2B3(include_top=False,weights='imagenet')\n    x = model1.output\n    x =tf.keras.layers.GlobalAveragePooling2D()(x)\n    x=tf.keras.layers.Dropout(0.2)(x)\n    predictions = tf.keras.layers.Dense(264, activation='softmax')(x)\n    # this is the model we will train\n    model =tf.keras.Model(inputs=model1.input, outputs=predictions)\n    return model\nwith tpu_strategy.scope():\n    model=creat_model()\n    model.load_weights('/kaggle/input/notebook5216810d7d/best_model2.h5')\n    model.summary()\n    #model.load_weights(\"/kaggle/input/notebook5216810d7d/best_model2.h5\")\n    # initial_learning_rate = 0.002\n    # lr_schedule = tf.keras.optimizers.schedules.ExponentialDecay(\n    #     initial_learning_rate,  # 设置初始学习率\n    #     decay_steps=64,  # 每隔多少个step衰减一次\n    #     decay_rate=0.96,  # 衰减系数\n    #     staircase=True)\n    # # # 将指数衰减学习率送入优化器\n    optimizer = tf.keras.optimizers.Adam(learning_rate=0.0003)\n    auc = tf.keras.metrics.AUC(curve='PR', name='auc', multi_label=False)  # auc on prcision-recall curve\n    acc = tf.keras.metrics.CategoricalAccuracy(name='acc')\n    metrics=[auc,acc]\n    model.compile(optimizer=optimizer,\n                  loss=tf.keras.losses.CategoricalCrossentropy(reduction='sum_over_batch_size'),\n                  metrics= metrics\n                  )\nepochs = 8\nhistory = model.fit(train_ds,\n                    validation_data=valid_ds,\n                    epochs=epochs,\n                    callbacks=callback)\n#                     callbacks=[callback,get_lr_callback(128,epochs=epochs)])\n\nmodel.save('keras_model_hdf5_version.h5')        ","metadata":{"_kg_hide-output":false,"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-05-05T01:10:47.732176Z","iopub.execute_input":"2023-05-05T01:10:47.732859Z"},"trusted":true},"execution_count":null,"outputs":[]}]}