{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# BirdCLEF 2023 - Baseline Inference\n\nThis is the inference kernel of the baseline training kernel [BC23 - Baseline Training](https://www.kaggle.com/code/morodertobias/bc23-baseline-training). \n\nIt is the last step of the baseline pipeline to get some first scores.\n\n**All comments welcome!**\n\n## Imporant note!\nVersions prior to 13 include a critical error resulting in mismatching images and seconds of the submission!\n\n## Table of Contents\n- [Config](#Config)\n- [Preparation](#Preparation)\n- [Neural network](#Neural-network)\n- [Predict](#Predict)","metadata":{"papermill":{"duration":0.007486,"end_time":"2023-03-30T14:06:35.928733","exception":false,"start_time":"2023-03-30T14:06:35.921247","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import os, pathlib\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3' \nimport json\nimport collections\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom pydantic import BaseModel as ConfigBaseModel\nimport tensorflow as tf\nprint(\"tensorflow:\", tf.__version__)\nimport librosa\nprint(\"librosa:\", librosa.__version__)","metadata":{"papermill":{"duration":9.039381,"end_time":"2023-03-30T14:06:44.974606","exception":false,"start_time":"2023-03-30T14:06:35.935225","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-28T18:31:03.376439Z","iopub.execute_input":"2023-04-28T18:31:03.377545Z","iopub.status.idle":"2023-04-28T18:31:03.385301Z","shell.execute_reply.started":"2023-04-28T18:31:03.377504Z","shell.execute_reply":"2023-04-28T18:31:03.384145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"strategy = tf.distribute.MirroredStrategy()\nprint(\"Strategy:\", strategy)\nprint(\"Number of replicas:\", strategy.num_replicas_in_sync)","metadata":{"papermill":{"duration":0.119661,"end_time":"2023-03-30T14:06:45.100645","exception":false,"start_time":"2023-03-30T14:06:44.980984","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-28T18:31:17.728504Z","iopub.execute_input":"2023-04-28T18:31:17.729030Z","iopub.status.idle":"2023-04-28T18:31:17.848350Z","shell.execute_reply.started":"2023-04-28T18:31:17.728982Z","shell.execute_reply":"2023-04-28T18:31:17.847271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sorted(tf.config.list_logical_devices())","metadata":{"papermill":{"duration":0.018079,"end_time":"2023-03-30T14:06:45.125151","exception":false,"start_time":"2023-03-30T14:06:45.107072","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-28T18:31:18.660445Z","iopub.execute_input":"2023-04-28T18:31:18.660921Z","iopub.status.idle":"2023-04-28T18:31:18.669744Z","shell.execute_reply.started":"2023-04-28T18:31:18.660877Z","shell.execute_reply":"2023-04-28T18:31:18.668443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Config","metadata":{"papermill":{"duration":0.006244,"end_time":"2023-03-30T14:06:45.137891","exception":false,"start_time":"2023-03-30T14:06:45.131647","status":"completed"},"tags":[]}},{"cell_type":"code","source":"class Config(ConfigBaseModel):\n    ## general\n    testing = False\n    fit_verbose = 1 if (os.environ.get('KAGGLE_KERNEL_RUN_TYPE') == \"Interactive\") else 2\n    ## model\n    train_cfg = \"/kaggle/input/bc23-baseline-training/cfg.json\"\n    path_weights = \"/kaggle/input/bc23-baseline-training/weights_dev-b0-v8.h5\"\n    base_model_weights: None = None\n    dropout = 0.20\n    ## data\n    path_submission = \"/kaggle/input/birdclef-2023/sample_submission.csv\"\n    soundscape_dir = \"/kaggle/input/birdclef-2023/test_soundscapes/\"\n    sample_rate = 32_000    \n    label = \"label\"\n    n_label = 264\n    ## spec\n    img_size = (128, 256)\n    channels = 1\n    img_shape = (*img_size, channels)\n    seconds = 5\n    n_fft = 2048\n    n_mels = img_size[0]\n    hop_length = (seconds * sample_rate - n_fft) // (img_size[1] - 1) \n    center = False\n    fmin = 500\n    fmax = 12_500\n    top_db = 80    \n    ## predict\n    batch_size = 32\n    \n    \ncfg = Config()\nwith open(\"cfg.json\", \"w\") as f:\n    f.write(cfg.json(indent=2))\ncfg.dict()    ","metadata":{"papermill":{"duration":0.032329,"end_time":"2023-03-30T14:06:45.176664","exception":false,"start_time":"2023-03-30T14:06:45.144335","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-28T18:31:19.844816Z","iopub.execute_input":"2023-04-28T18:31:19.845915Z","iopub.status.idle":"2023-04-28T18:31:19.871715Z","shell.execute_reply.started":"2023-04-28T18:31:19.845860Z","shell.execute_reply":"2023-04-28T18:31:19.870306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preparation","metadata":{"papermill":{"duration":0.006281,"end_time":"2023-03-30T14:06:45.189777","exception":false,"start_time":"2023-03-30T14:06:45.183496","status":"completed"},"tags":[]}},{"cell_type":"code","source":"with open(cfg.train_cfg, \"r\") as f:\n    train_cfg = json.load(f)\ntrain_cfg","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-04-28T18:31:21.595223Z","iopub.execute_input":"2023-04-28T18:31:21.596152Z","iopub.status.idle":"2023-04-28T18:31:21.616068Z","shell.execute_reply.started":"2023-04-28T18:31:21.596089Z","shell.execute_reply":"2023-04-28T18:31:21.614927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load data","metadata":{"papermill":{"duration":0.006609,"end_time":"2023-03-30T14:06:45.203446","exception":false,"start_time":"2023-03-30T14:06:45.196837","status":"completed"},"tags":[]}},{"cell_type":"code","source":"submission = pd.read_csv(cfg.path_submission)\nsubmission","metadata":{"papermill":{"duration":0.066945,"end_time":"2023-03-30T14:06:45.277263","exception":false,"start_time":"2023-03-30T14:06:45.210318","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-28T18:31:23.251618Z","iopub.execute_input":"2023-04-28T18:31:23.252070Z","iopub.status.idle":"2023-04-28T18:31:23.309174Z","shell.execute_reply.started":"2023-04-28T18:31:23.252027Z","shell.execute_reply":"2023-04-28T18:31:23.307963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = submission.columns[1:].to_list()\nlen(labels), repr(labels)[:100]","metadata":{"papermill":{"duration":0.019254,"end_time":"2023-03-30T14:06:45.303643","exception":false,"start_time":"2023-03-30T14:06:45.284389","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-28T18:31:24.220078Z","iopub.execute_input":"2023-04-28T18:31:24.220510Z","iopub.status.idle":"2023-04-28T18:31:24.228709Z","shell.execute_reply.started":"2023-04-28T18:31:24.220470Z","shell.execute_reply":"2023-04-28T18:31:24.227323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if cfg.testing:\n    ser_row_id = pd.Series([f\"soundscape_29201_{x}\" for x in range(0, 600, 5)], name=\"row_id\")\n    submission = pd.concat([\n        ser_row_id.to_frame(),\n        pd.DataFrame(0, columns=labels, index=ser_row_id.index)\n    ], axis=1)\nsubmission","metadata":{"execution":{"iopub.status.busy":"2023-04-28T18:31:25.164066Z","iopub.execute_input":"2023-04-28T18:31:25.164523Z","iopub.status.idle":"2023-04-28T18:31:25.185828Z","shell.execute_reply.started":"2023-04-28T18:31:25.164487Z","shell.execute_reply":"2023-04-28T18:31:25.184433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dataset","metadata":{"papermill":{"duration":0.006693,"end_time":"2023-03-30T14:06:45.317305","exception":false,"start_time":"2023-03-30T14:06:45.310612","status":"completed"},"tags":[]}},{"cell_type":"code","source":"df = submission[[\"row_id\"]].copy()\ndf[[\"fname\", \"offset_sec_end\"]] = df[\"row_id\"].str.rsplit(\"_\", n=1, expand=True)\ndf[\"path_ogg\"] = cfg.soundscape_dir + df[\"fname\"] + \".ogg\"\ndf[\"offset_sec_end\"] = df[\"offset_sec_end\"].astype(\"int\")\ndf","metadata":{"papermill":{"duration":0.033768,"end_time":"2023-03-30T14:06:45.360381","exception":false,"start_time":"2023-03-30T14:06:45.326613","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-28T18:31:40.831642Z","iopub.execute_input":"2023-04-28T18:31:40.832054Z","iopub.status.idle":"2023-04-28T18:31:40.855131Z","shell.execute_reply.started":"2023-04-28T18:31:40.832018Z","shell.execute_reply":"2023-04-28T18:31:40.853756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.utils import Sequence\n\n\ndef show_img_stats(img):\n    print((img.shape, img.dtype, img.min(), img.max()))\n\n\ndef _get_duration(path_ogg):\n    \"\"\"Get duration in seconds\"\"\"\n    return librosa.get_duration(path=path_ogg)\n\n\ndef _get_mel_spec_db(path_ogg, offset_sec_end):\n    \"\"\"Get dB scaled mel power spectrum\"\"\"\n    required_len = cfg.seconds * cfg.sample_rate\n    sig, dr = librosa.load(\n        path=path_ogg,\n        sr=cfg.sample_rate,\n        offset=offset_sec_end - cfg.seconds,\n        duration=cfg.seconds,\n    )\n    sig = np.concatenate([sig, np.zeros((required_len - len(sig)), dtype=sig.dtype)])\n    mel_spec = librosa.feature.melspectrogram(\n        y=sig,\n        hop_length=cfg.hop_length,\n        sr=cfg.sample_rate,\n        n_fft=cfg.n_fft,\n        n_mels=cfg.n_mels,\n        center=cfg.center,\n        fmin=cfg.fmin,\n        fmax=cfg.fmax,\n    )\n    mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max, top_db=cfg.top_db)\n    return mel_spec_db\n\n\ndef _normalize_img(img):\n    \"\"\"Normalize to uint8 image range\"\"\"\n    assert img.ndim == 2, \"unexpected dimension\"\n    v_min, v_max = np.min(img), np.max(img)\n    return ((img - v_min) / (v_max - v_min) * 255).astype(\"uint8\")\n\n\nclass SpectrumReader(Sequence):\n    def __init__(self, data, batch_size=cfg.batch_size, **kwargs):\n        super().__init__(**kwargs)\n        assert {\"path_ogg\", \"offset_sec_end\"}.issubset(data.columns), \"missing columns\"\n        self.rows = data.to_dict(orient=\"records\")\n        self.batch_size = batch_size\n\n    def __len__(self):\n        return np.ceil(len(self.rows) / self.batch_size).astype(\"int\")\n\n    def __getitem__(self, idx):\n        start = idx * self.batch_size\n        stop = start + self.batch_size\n        imgs = np.asarray(\n            [\n                _normalize_img(\n                    _get_mel_spec_db(path_ogg=row[\"path_ogg\"], offset_sec_end=row[\"offset_sec_end\"])\n                )\n                for row in self.rows[start:stop]\n            ]\n        )\n        return imgs[..., np.newaxis]\n\n    def __repr__(self):\n        return f\"<SpectrumReader(df=[{len(self.rows)} row(s), batch_size={self.batch_size}])>\"","metadata":{"papermill":{"duration":0.029297,"end_time":"2023-03-30T14:06:45.396923","exception":false,"start_time":"2023-03-30T14:06:45.367626","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-28T18:31:54.188103Z","iopub.execute_input":"2023-04-28T18:31:54.188676Z","iopub.status.idle":"2023-04-28T18:31:54.211955Z","shell.execute_reply.started":"2023-04-28T18:31:54.188625Z","shell.execute_reply":"2023-04-28T18:31:54.210604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds = SpectrumReader(df, batch_size=cfg.batch_size)\nds_index = df.index\nds","metadata":{"papermill":{"duration":0.019205,"end_time":"2023-03-30T14:06:45.423631","exception":false,"start_time":"2023-03-30T14:06:45.404426","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-28T18:31:54.867909Z","iopub.execute_input":"2023-04-28T18:31:54.868335Z","iopub.status.idle":"2023-04-28T18:31:54.876510Z","shell.execute_reply.started":"2023-04-28T18:31:54.868294Z","shell.execute_reply":"2023-04-28T18:31:54.875434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs = ds[0]\nshow_img_stats(imgs)","metadata":{"papermill":{"duration":14.840484,"end_time":"2023-03-30T14:07:00.271234","exception":false,"start_time":"2023-03-30T14:06:45.430750","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-28T18:31:55.795189Z","iopub.execute_input":"2023-04-28T18:31:55.795631Z","iopub.status.idle":"2023-04-28T18:32:09.850265Z","shell.execute_reply.started":"2023-04-28T18:31:55.795584Z","shell.execute_reply":"2023-04-28T18:32:09.848471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(1, min(len(imgs), 5), sharex='all', sharey='all', figsize=(14, 5))\nfor i, ax in enumerate(axs.flat):\n    img = imgs[i]\n    show_img_stats(img)\n    ax.imshow(img, cmap=\"coolwarm\")\nplt.tight_layout()\nplt.show()","metadata":{"papermill":{"duration":0.790932,"end_time":"2023-03-30T14:07:01.082845","exception":false,"start_time":"2023-03-30T14:07:00.291913","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-28T18:32:22.370598Z","iopub.execute_input":"2023-04-28T18:32:22.371792Z","iopub.status.idle":"2023-04-28T18:32:23.027218Z","shell.execute_reply.started":"2023-04-28T18:32:22.371745Z","shell.execute_reply":"2023-04-28T18:32:23.025898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Neural network\nSeems that the latest TF version has some issue in saving the model; thus we have only stored the weights.","metadata":{"papermill":{"duration":0.013621,"end_time":"2023-03-30T14:07:01.111231","exception":false,"start_time":"2023-03-30T14:07:01.097610","status":"completed"},"tags":[]}},{"cell_type":"code","source":"from tensorflow.keras.applications.efficientnet import EfficientNetB0 as BaseModel\nfrom tensorflow.keras.applications.efficientnet import preprocess_input\nfrom tensorflow.keras import layers, losses, metrics, callbacks","metadata":{"papermill":{"duration":0.025627,"end_time":"2023-03-30T14:07:01.150955","exception":false,"start_time":"2023-03-30T14:07:01.125328","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-28T18:32:43.491728Z","iopub.execute_input":"2023-04-28T18:32:43.492913Z","iopub.status.idle":"2023-04-28T18:32:43.499473Z","shell.execute_reply.started":"2023-04-28T18:32:43.492859Z","shell.execute_reply":"2023-04-28T18:32:43.498640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model():\n    inputs = layers.Input(shape=cfg.img_shape, dtype=tf.float32)\n    x = tf.image.grayscale_to_rgb(inputs)\n    x = layers.Lambda(preprocess_input, name=\"preprocess_input\")(x)\n    base_model = BaseModel(include_top=False, weights=cfg.base_model_weights, pooling=\"avg\")\n    x = base_model(x, training=False)\n    x = layers.Dropout(cfg.dropout, name=\"top_dropout\")(x)\n    outputs = layers.Dense(cfg.n_label, name=\"logits\")(x)\n    model = tf.keras.Model(inputs=inputs, outputs=outputs)\n#     model.compile(\n#         optimizer=tf.keras.optimizers.Adam(learning_rate=lr),\n#         loss=tf.keras.losses.SparseCategoricalCrossentropy(from_logits=True),\n#         metrics=['acc']\n#     )\n    return model","metadata":{"papermill":{"duration":0.027081,"end_time":"2023-03-30T14:07:01.192454","exception":false,"start_time":"2023-03-30T14:07:01.165373","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-28T18:32:44.115473Z","iopub.execute_input":"2023-04-28T18:32:44.116329Z","iopub.status.idle":"2023-04-28T18:32:44.123416Z","shell.execute_reply.started":"2023-04-28T18:32:44.116287Z","shell.execute_reply":"2023-04-28T18:32:44.122124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.backend.clear_session()\nwith strategy.scope():\n    model = create_model()\n    model.load_weights(cfg.path_weights)\nmodel.summary(line_length=120)","metadata":{"papermill":{"duration":5.538452,"end_time":"2023-03-30T14:07:06.745641","exception":false,"start_time":"2023-03-30T14:07:01.207189","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-28T18:32:45.043115Z","iopub.execute_input":"2023-04-28T18:32:45.044282Z","iopub.status.idle":"2023-04-28T18:32:50.314209Z","shell.execute_reply.started":"2023-04-28T18:32:45.044236Z","shell.execute_reply":"2023-04-28T18:32:50.313069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict","metadata":{"papermill":{"duration":0.015369,"end_time":"2023-03-30T14:07:06.776570","exception":false,"start_time":"2023-03-30T14:07:06.761201","status":"completed"},"tags":[]}},{"cell_type":"code","source":"pred = model.predict(ds, verbose=cfg.fit_verbose, workers=os.cpu_count(), use_multiprocessing=True)\npred = pd.DataFrame(tf.nn.sigmoid(pred).numpy(), columns=labels, index=ds_index)\npred","metadata":{"papermill":{"duration":4.823291,"end_time":"2023-03-30T14:07:11.615691","exception":false,"start_time":"2023-03-30T14:07:06.792400","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-28T18:32:51.754448Z","iopub.execute_input":"2023-04-28T18:32:51.755365Z","iopub.status.idle":"2023-04-28T18:33:01.343333Z","shell.execute_reply.started":"2023-04-28T18:32:51.755312Z","shell.execute_reply":"2023-04-28T18:33:01.341763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.concat([submission[\"row_id\"], pred], axis=1)\nsubmission","metadata":{"papermill":{"duration":0.047607,"end_time":"2023-03-30T14:07:11.679267","exception":false,"start_time":"2023-03-30T14:07:11.631660","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-28T18:33:14.934822Z","iopub.execute_input":"2023-04-28T18:33:14.935289Z","iopub.status.idle":"2023-04-28T18:33:14.966845Z","shell.execute_reply.started":"2023-04-28T18:33:14.935245Z","shell.execute_reply":"2023-04-28T18:33:14.965425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not cfg.testing:\n    submission.to_csv(\"submission.csv\", index=False)","metadata":{"papermill":{"duration":0.033883,"end_time":"2023-03-30T14:07:11.729362","exception":false,"start_time":"2023-03-30T14:07:11.695479","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-28T18:33:15.394092Z","iopub.execute_input":"2023-04-28T18:33:15.394491Z","iopub.status.idle":"2023-04-28T18:33:15.408619Z","shell.execute_reply.started":"2023-04-28T18:33:15.394457Z","shell.execute_reply":"2023-04-28T18:33:15.407445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head submission.csv","metadata":{"papermill":{"duration":1.135814,"end_time":"2023-03-30T14:07:12.881333","exception":false,"start_time":"2023-03-30T14:07:11.745519","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-28T18:33:15.859171Z","iopub.execute_input":"2023-04-28T18:33:15.859949Z","iopub.status.idle":"2023-04-28T18:33:16.962089Z","shell.execute_reply.started":"2023-04-28T18:33:15.859903Z","shell.execute_reply":"2023-04-28T18:33:16.960462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.016408,"end_time":"2023-03-30T14:07:12.914371","exception":false,"start_time":"2023-03-30T14:07:12.897963","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]}]}