{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport tensorflow_io as tfio\nfrom IPython.display import Audio\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nimport sklearn.metrics\nimport json\nimport tensorflow as tf\nimport os\nimport glob","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-12T08:22:27.046273Z","iopub.execute_input":"2023-05-12T08:22:27.046686Z","iopub.status.idle":"2023-05-12T08:22:27.052381Z","shell.execute_reply.started":"2023-05-12T08:22:27.046657Z","shell.execute_reply":"2023-05-12T08:22:27.051277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    image_size = [256, 256]\n    is_training = False\n    epochs = 10","metadata":{"execution":{"iopub.status.busy":"2023-05-12T08:22:27.054770Z","iopub.execute_input":"2023-05-12T08:22:27.055161Z","iopub.status.idle":"2023-05-12T08:22:27.068504Z","shell.execute_reply.started":"2023-05-12T08:22:27.055134Z","shell.execute_reply":"2023-05-12T08:22:27.067656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def padded_cmap(solution, submission, padding_factor=5):\n    solution = solution.drop(['row_id'], axis=1, errors='ignore')\n    submission = submission.drop(['row_id'], axis=1, errors='ignore')\n    new_rows = []\n    for i in range(padding_factor):\n        new_rows.append([1 for i in range(len(solution.columns))])\n    new_rows = pd.DataFrame(new_rows)\n    new_rows.columns = solution.columns\n    padded_solution = pd.concat([solution, new_rows]).reset_index(drop=True).copy()\n    padded_submission = pd.concat([submission, new_rows]).reset_index(drop=True).copy()\n    score = sklearn.metrics.average_precision_score(\n        padded_solution.values,\n        padded_submission.values,\n        average='macro',\n    )\n    return score\n\n\ndef preprocess(audio_url, label):\n    audio_string = tf.io.read_file(audio_url)\n    audio = tfio.audio.decode_vorbis(audio_string)\n    audio_tensor = tf.squeeze(audio, axis=[-1])\n    diff = tf.cast(tf.shape(audio_tensor)[0] - 5 * 32000, tf.float32)\n    begin = tf.cast(tf.random.uniform(shape=()) * diff, tf.int32)\n    start_position = tf.where(diff > 0, begin, 0)\n    end_position = tf.where(diff > 0, start_position + 5 * 32000, tf.shape(audio_tensor)[0])\n    audio_tensor = audio_tensor[start_position:end_position]\n    tensor = tf.cast(audio_tensor, tf.float32) / 32768.0\n    spectrogram = tfio.audio.spectrogram(tensor, nfft=512, window=512, stride=256)\n    spectrogram = tfio.audio.dbscale(spectrogram, top_db=80)\n    spectrogram = tf.expand_dims(spectrogram, axis=-1)\n    spectrogram = tf.image.resize(spectrogram, CFG.image_size)\n    spectrogram = (spectrogram - tf.reduce_min(spectrogram)) / (tf.reduce_max(spectrogram) - tf.reduce_min(spectrogram)) * 255.0\n    return spectrogram, label\n\ndef preprocess_test(audio_tensor):\n    tensor = tf.cast(audio_tensor, tf.float32) / 32768.0\n    spectrogram = tfio.audio.spectrogram(tensor, nfft=512, window=512, stride=256)\n    spectrogram = tfio.audio.dbscale(spectrogram, top_db=80)\n    spectrogram = tf.expand_dims(spectrogram, axis=-1)\n    spectrogram = tf.image.resize(spectrogram, CFG.image_size)\n    spectrogram = (spectrogram - tf.reduce_min(spectrogram)) / (tf.reduce_max(spectrogram) - tf.reduce_min(spectrogram)) * 255.0\n    return tf.expand_dims(spectrogram, axis=0)\n\ndef make_dataset(df, batch_size=128, shuffle=True):\n    ds = tf.data.Dataset.from_tensor_slices((df[\"file_path\"], df[\"label\"]))\n    ds = ds.map(preprocess)\n    if shuffle:\n        ds = ds.shuffle(batch_size * 4)\n    ds = ds.batch(batch_size).prefetch(tf.data.AUTOTUNE)\n    return ds\n\ndef make_inference(tensor):\n    image = preprocess_test(tensor)\n    return model.predict(image)\n\ndef frame_audio(\n      audio_array: np.ndarray,\n      window_size_s: float = 5.0,\n      hop_size_s: float = 5.0,\n      sample_rate = 32000,\n      ) -> np.ndarray:\n    \n    \"\"\"Helper function for framing audio for inference.\"\"\"\n    \"\"\" using tf.signal \"\"\"\n    if window_size_s is None or window_size_s < 0:\n        return audio_array[np.newaxis, :]\n    frame_length = int(window_size_s * sample_rate)\n    hop_length = int(hop_size_s * sample_rate)\n    framed_audio = tf.signal.frame(audio_array, frame_length, hop_length, pad_end=True)\n    return framed_audio\n\ndef ensure_sample_rate(waveform, original_sample_rate,\n                       desired_sample_rate=32000):\n    \"\"\"Resample waveform if required.\"\"\"\n    if original_sample_rate != desired_sample_rate:\n        waveform = tfio.audio.resample(waveform, original_sample_rate, desired_sample_rate)\n    return desired_sample_rate, waveform\n\ndef preprocess_test(audio_tensor):\n    tensor = tf.cast(audio_tensor, tf.float32) / 32768.0\n    spectrogram = tfio.audio.spectrogram(tensor, nfft=512, window=512, stride=256)\n    spectrogram = tfio.audio.dbscale(spectrogram, top_db=80)\n    spectrogram = tf.expand_dims(spectrogram, axis=-1)\n    spectrogram = tf.image.resize(spectrogram, (256, 256))\n    spectrogram = (spectrogram - tf.reduce_min(spectrogram)) / (tf.reduce_max(spectrogram) - tf.reduce_min(spectrogram)) * 255.0\n    return spectrogram\n\ndef predict_for_sample(filename, sample_submission, frame_limit_secs=None):\n    file_id = filename.split(\".ogg\")[0].split(\"/\")[-1]\n    audio = tfio.audio.AudioIOTensor(filename)\n    sample_rate = audio.rate.numpy()\n    audio_tensor = tf.squeeze(audio[0:], axis=[-1])\n    sample_rate, wav_data = ensure_sample_rate(audio_tensor, sample_rate)\n    fixed_tm = frame_audio(wav_data)\n    frame = 5\n    all_logits = make_inference(fixed_tm[:1])\n    for window in fixed_tm[1:]:\n        if frame_limit_secs and frame > frame_limit_secs:\n            continue\n        logits = make_inference(window[np.newaxis, :])\n        all_logits = np.concatenate([all_logits, logits], axis=0)\n        frame += 5\n    frame = 5\n    all_probabilities = []\n    for frame_logits in all_logits:\n        probabilities = tf.nn.softmax(frame_logits).numpy()\n        ## set the appropriate row in the sample submission\n        sample_submission.loc[sample_submission.row_id == file_id + \"_\" + str(frame), labels] = probabilities\n        frame += 5","metadata":{"execution":{"iopub.status.busy":"2023-05-12T08:22:27.070538Z","iopub.execute_input":"2023-05-12T08:22:27.071139Z","iopub.status.idle":"2023-05-12T08:22:27.099998Z","shell.execute_reply.started":"2023-05-12T08:22:27.071098Z","shell.execute_reply":"2023-05-12T08:22:27.098602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/birdclef-2023/train_metadata.csv\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-12T08:22:27.101672Z","iopub.execute_input":"2023-05-12T08:22:27.102058Z","iopub.status.idle":"2023-05-12T08:22:27.206669Z","shell.execute_reply.started":"2023-05-12T08:22:27.102026Z","shell.execute_reply":"2023-05-12T08:22:27.205424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv(\"../input/birdclef-2023/sample_submission.csv\")\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-12T08:22:27.209803Z","iopub.execute_input":"2023-05-12T08:22:27.210281Z","iopub.status.idle":"2023-05-12T08:22:27.239354Z","shell.execute_reply.started":"2023-05-12T08:22:27.210241Z","shell.execute_reply":"2023-05-12T08:22:27.237950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = list(submission.columns)\nlabels.remove(\"row_id\")\nprint(labels)","metadata":{"execution":{"iopub.status.busy":"2023-05-12T08:22:27.240905Z","iopub.execute_input":"2023-05-12T08:22:27.241314Z","iopub.status.idle":"2023-05-12T08:22:27.247140Z","shell.execute_reply.started":"2023-05-12T08:22:27.241282Z","shell.execute_reply":"2023-05-12T08:22:27.246016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.primary_label.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-05-12T08:22:27.248850Z","iopub.execute_input":"2023-05-12T08:22:27.249314Z","iopub.status.idle":"2023-05-12T08:22:27.269096Z","shell.execute_reply.started":"2023-05-12T08:22:27.249281Z","shell.execute_reply":"2023-05-12T08:22:27.267723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.secondary_labels.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-05-12T08:22:27.272081Z","iopub.execute_input":"2023-05-12T08:22:27.272721Z","iopub.status.idle":"2023-05-12T08:22:27.289029Z","shell.execute_reply.started":"2023-05-12T08:22:27.272691Z","shell.execute_reply":"2023-05-12T08:22:27.287800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"label\"] = train[\"primary_label\"].map(lambda primary_label: labels.index(primary_label))\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-12T08:22:27.293029Z","iopub.execute_input":"2023-05-12T08:22:27.293681Z","iopub.status.idle":"2023-05-12T08:22:27.372643Z","shell.execute_reply.started":"2023-05-12T08:22:27.293641Z","shell.execute_reply":"2023-05-12T08:22:27.371474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"file_path\"] = train[\"filename\"].apply(lambda filename: os.path.join(f\"/kaggle/input/birdclef-2023/train_audio/{filename}\"))\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-12T08:22:27.376171Z","iopub.execute_input":"2023-05-12T08:22:27.376782Z","iopub.status.idle":"2023-05-12T08:22:27.417566Z","shell.execute_reply.started":"2023-05-12T08:22:27.376740Z","shell.execute_reply":"2023-05-12T08:22:27.416422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train)","metadata":{"execution":{"iopub.status.busy":"2023-05-12T08:22:27.422167Z","iopub.execute_input":"2023-05-12T08:22:27.422533Z","iopub.status.idle":"2023-05-12T08:22:27.432205Z","shell.execute_reply.started":"2023-05-12T08:22:27.422494Z","shell.execute_reply":"2023-05-12T08:22:27.431072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio = tfio.audio.AudioIOTensor(\"/kaggle/input/birdclef-2023/train_audio/blakit1/XC115289.ogg\")\naudio_tensor = tf.squeeze(audio[0:], axis=[-1])\nAudio(audio_tensor.numpy(), rate=audio.rate.numpy())","metadata":{"execution":{"iopub.status.busy":"2023-05-12T08:22:27.433497Z","iopub.execute_input":"2023-05-12T08:22:27.434186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tensor = tf.cast(audio_tensor, tf.float32) / 32768.0\nplt.figure()\nplt.plot(tensor.numpy())","metadata":{"execution":{"iopub.status.busy":"2023-05-12T08:22:27.559727Z","iopub.execute_input":"2023-05-12T08:22:27.560088Z","iopub.status.idle":"2023-05-12T08:22:28.016036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert to spectrogram\ntensor = tf.cast(audio_tensor, tf.float32) \nspectrogram = tfio.audio.spectrogram(tensor, nfft=512, window=512, stride=256)\nspectrogram = tf.math.log(spectrogram)\nplt.imshow(spectrogram)","metadata":{"execution":{"iopub.status.busy":"2023-05-12T08:22:28.017442Z","iopub.execute_input":"2023-05-12T08:22:28.017854Z","iopub.status.idle":"2023-05-12T08:22:28.403352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert to spectrogram\nspectrogram = tfio.audio.spectrogram(tensor[0:audio.rate * 5], nfft=512, window=512, stride=256)\nspectrogram = tfio.audio.dbscale(spectrogram, top_db=80)\n\nspectrogram = (spectrogram - tf.reduce_min(spectrogram)) / (tf.reduce_max(spectrogram) - tf.reduce_min(spectrogram)) * 255.0\nplt.figure()\nplt.imshow(spectrogram.numpy())","metadata":{"execution":{"iopub.status.busy":"2023-05-12T08:22:28.404391Z","iopub.execute_input":"2023-05-12T08:22:28.404849Z","iopub.status.idle":"2023-05-12T08:22:28.655389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df, valid_df = train_test_split(train, test_size=0.2, shuffle=True, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-05-12T08:22:28.656683Z","iopub.execute_input":"2023-05-12T08:22:28.657339Z","iopub.status.idle":"2023-05-12T08:22:28.673868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-12T08:22:28.675457Z","iopub.execute_input":"2023-05-12T08:22:28.675887Z","iopub.status.idle":"2023-05-12T08:22:28.695524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-12T08:22:28.696972Z","iopub.execute_input":"2023-05-12T08:22:28.697377Z","iopub.status.idle":"2023-05-12T08:22:28.716089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_ds = make_dataset(valid_df, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-12T08:22:28.717446Z","iopub.execute_input":"2023-05-12T08:22:28.717874Z","iopub.status.idle":"2023-05-12T08:22:28.881787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for X, y in valid_ds.take(1):\n    print(X.shape, y.shape)","metadata":{"execution":{"iopub.status.busy":"2023-05-12T08:22:28.884225Z","iopub.execute_input":"2023-05-12T08:22:28.884636Z","iopub.status.idle":"2023-05-12T08:22:49.448385Z","shell.execute_reply.started":"2023-05-12T08:22:28.884606Z","shell.execute_reply":"2023-05-12T08:22:49.447281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if CFG.is_training:\n    train_ds = make_dataset(train_df)\n    def get_model():\n        inputs = tf.keras.Input(shape=(CFG.image_size[0], CFG.image_size[1], 1))\n        image_inputs = tf.concat([\n            inputs,\n            inputs,\n            inputs\n        ], axis=-1)\n        vector = efficent_net(image_inputs)\n        output = tf.keras.layers.Dense(264, activation=\"softmax\")(vector)\n        model = tf.keras.Model(inputs=inputs, outputs=output)\n        model.compile(loss=tf.keras.losses.SparseCategoricalCrossentropy(), optimizer=tf.keras.optimizers.Adam(1e-3), metrics=[\"accuracy\"])\n        return model\n    efficent_net = tf.keras.applications.EfficientNetV2S(include_top=False, pooling=\"max\")\n    efficent_net.trainable = False\n    efficent_net.summary()\n    model = get_model()\n    callbacks = [\n        tf.keras.callbacks.ModelCheckpoint(\n            \"model.h5\", \n            save_best_only=True\n        ),\n        tf.keras.callbacks.EarlyStopping(\n            min_delta=1e-4, \n            patience=10\n        ),\n        tf.keras.callbacks.ReduceLROnPlateau(\n            factor=0.3,\n            patience=2, \n            min_lr=1e-7\n        ),\n        tf.keras.callbacks.TerminateOnNaN()\n    ]\n    model.fit(train_ds, epochs=CFG.epochs, validation_data=valid_ds, callbacks=callbacks)\nelse:\n    model = tf.keras.models.load_model(\"/kaggle/input/bird-clef/model.h5\")\nmodel.summary()\ntf.keras.utils.plot_model(model, show_shapes=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-12T08:22:49.450595Z","iopub.execute_input":"2023-05-12T08:22:49.450921Z","iopub.status.idle":"2023-05-12T08:22:57.316156Z","shell.execute_reply.started":"2023-05-12T08:22:49.450894Z","shell.execute_reply":"2023-05-12T08:22:57.315000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_preds = model.predict(valid_ds)\ny_pred_labels = np.argmax(y_preds, axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-05-12T08:22:57.317456Z","iopub.execute_input":"2023-05-12T08:22:57.317796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = pd.DataFrame({\"row_id\": valid_df.index}).copy()\nfor i, column in enumerate(labels):\n    submission_df[column] = y_preds[:, i]\ntrue_labels = list(valid_df[\"label\"])\nsolution_df = pd.DataFrame({\"row_id\": valid_df.index}).copy()\nfor column in labels:\n    solution_df[column] = 0\nfor i in range(len(valid_df)):\n    secondary_labels = valid_df.iloc[i][\"secondary_labels\"]\n    secondary_labels = secondary_labels.replace(\"\\'\", \"\\\"\")\n    arr = json.loads(secondary_labels)\n    solution_df.loc[i, labels[true_labels[i]]] = 1\n    if len(arr) > 0:\n        for secondary_label in arr:\n            idx = labels.index(secondary_label)\n            if idx >= 0 and idx < len(labels):\n                solution_df.loc[i, labels[true_labels[idx]]] = 1\nscore = padded_cmap(solution_df, submission_df)\nprint(f\"CV:{score}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_samples = list(glob.glob(\"/kaggle/input/birdclef-2023/test_soundscapes/*.ogg\"))\nsubmission = pd.read_csv(\"../input/birdclef-2023/sample_submission.csv\")\nsubmission[labels] = submission[labels].astype(np.float32)\nfor filename in test_samples:\n    predict_for_sample(filename, submission, frame_limit_secs=15)\nsubmission.to_csv(\"submission.csv\", index=False)\nsubmission.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}