{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What impacts your CPU inference time?\n\nIn this competition, we only have **120 minutes of CPU inference time**.\n\nLet's explore what impacts the CPU inference time in this notebook:\n* Batch size\n* Choice of backbone\n* Spectrogram size\n\n\n**References:**\n* This notebook builds on the inference baselines from [[Pytorch] BirdCLEF23 Starter](https://www.kaggle.com/code/debarshichanda/pytorch-birdclef23-starter)\n* The function to estimate inference time on the test set is copied from [BirdCLEF23: Pretraining is All you Need [Infer]](https://www.kaggle.com/code/awsaf49/birdclef23-pretraining-is-all-you-need-infer/notebook#Submission-Time-%E2%8F%B0)","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nimport cv2\nimport math\nimport copy\nimport time\nimport random\n\n# For data manipulation\nimport numpy as np\nimport pandas as pd\n\n# Deep Learning framework\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\nfrom torch.optim import lr_scheduler\nfrom torch.utils.data import Dataset, DataLoader\n\n# Audio processing\nimport torchaudio\nimport torchaudio.transforms as T\nimport librosa\n\n# Pre-trained image models\nimport timm\n\n# Utils\nimport joblib\nfrom tqdm import tqdm\nfrom collections import defaultdict\nfrom pathlib import Path\n\nimport gc\n\nfrom sklearn.metrics import average_precision_score\n\nimport matplotlib.pyplot as plt\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n# For descriptive error messages\nos.environ['CUDA_LAUNCH_BLOCKING'] = \"1\"","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-15T13:00:56.513925Z","iopub.execute_input":"2023-04-15T13:00:56.514387Z","iopub.status.idle":"2023-04-15T13:00:56.524448Z","shell.execute_reply.started":"2023-04-15T13:00:56.514351Z","shell.execute_reply":"2023-04-15T13:00:56.522957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Batch size\nThe bigger the batch, the fast the throughput. \n\nTo keep it short, we will not run any experiments here.\n\nInstead, increase your batch size in powers of two until you get an error. \nThen use the biggest batch size your training and inference pipeline can handle. \n\n**For this pipeline, the biggest possible batch size turned out to be 128**","metadata":{}},{"cell_type":"code","source":"CONFIG = {\"seed\": 27032023,\n          \"epochs\": 5, # 15-30\n          \"model_name\": 'eca_nfnet_l0', #\"tf_efficientnet_b3_ns\",\n          \"embedding_size\": 768,\n          \"num_classes\": 264,\n          \"batch_size\": 128,\n          \"learning_rate\": 1e-3,\n          \"min_lr\": 1e-6,\n          \"T_max\": 500,\n          \"weight_decay\": 1e-6,\n          \"n_fold\": 5,\n          \"device\": torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\"),\n          \"competition\": \"BirdCLEF23\",\n          \"_wandb_kernel\": \"lemon\",\n          # Audio Specific\n          \"sample_rate\": 32000,\n          \"audio_length\": 5,\n          \"n_mels\": 128,\n          \"n_fft\": 1024, # 2048\n          \"hop_length\" : 512 # 512\n          }","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-04-15T13:00:56.535908Z","iopub.execute_input":"2023-04-15T13:00:56.537023Z","iopub.status.idle":"2023-04-15T13:00:56.543978Z","shell.execute_reply.started":"2023-04-15T13:00:56.536974Z","shell.execute_reply":"2023-04-15T13:00:56.542936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seed(seed=42):\n    '''Sets the seed of the entire notebook so results are the same every time we run.\n    This is for REPRODUCIBILITY.'''\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    # When running on the CuDNN backend, two further options must be set\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n    # Set a fixed value for the hash seed\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    \nset_seed(CONFIG['seed'])","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-15T13:00:56.549697Z","iopub.execute_input":"2023-04-15T13:00:56.550198Z","iopub.status.idle":"2023-04-15T13:00:56.564111Z","shell.execute_reply.started":"2023-04-15T13:00:56.550130Z","shell.execute_reply":"2023-04-15T13:00:56.562825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT_DIR = '../input/birdclef-2023'\nTRAIN_DIR = '../input/birdclef-2023/train_audio'\nTEST_DIR = '../input/birdclef-2023/test_soundscapes'","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-15T13:00:56.766354Z","iopub.execute_input":"2023-04-15T13:00:56.767133Z","iopub.status.idle":"2023-04-15T13:00:56.772316Z","shell.execute_reply.started":"2023-04-15T13:00:56.767095Z","shell.execute_reply":"2023-04-15T13:00:56.770846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_train_file_path(filename):\n    return f\"{TRAIN_DIR}/{filename}\"\n\ndf = pd.read_csv(f\"{ROOT_DIR}/train_metadata.csv\")\ndf['file_path'] = df['filename'].apply(get_train_file_path)\ndf.head()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-15T13:00:56.775294Z","iopub.execute_input":"2023-04-15T13:00:56.775648Z","iopub.status.idle":"2023-04-15T13:00:56.876465Z","shell.execute_reply.started":"2023-04-15T13:00:56.775615Z","shell.execute_reply":"2023-04-15T13:00:56.875508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.DataFrame(\n     [(path.stem, *path.stem.split(\"_\"), path) for path in Path('/kaggle/input/birdclef-2023/test_soundscapes/').glob(\"*.ogg\")],\n    columns = [\"filename\", \"name\" ,\"id\", \"path\"]\n)\nprint(df_test.shape)\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-15T13:00:56.877756Z","iopub.execute_input":"2023-04-15T13:00:56.878343Z","iopub.status.idle":"2023-04-15T13:00:56.895753Z","shell.execute_reply.started":"2023-04-15T13:00:56.878305Z","shell.execute_reply":"2023-04-15T13:00:56.894519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class AudioDataset(Dataset):\n    def __init__(self, \n                 df, \n                 target_sample_rate, \n                 audio_length, \n                   n_mels = CONFIG['n_mels'],\n                 hop_length = CONFIG['hop_length'],\n                step=None,):\n        self.file_paths = df['path'].values\n        self.target_sample_rate = target_sample_rate\n        self.num_samples = target_sample_rate * audio_length\n        self.step = step or self.num_samples\n        self.n_mels = n_mels\n        self.hop_length = hop_length\n        \n    def __len__(self):\n        return len(self.file_paths)\n    \n    def audio_to_image(self, audio):\n        mel_spectogram = T.MelSpectrogram(sample_rate=self.target_sample_rate, \n                                        n_mels=self.n_mels, \n                                        n_fft=CONFIG['n_fft'],\n                                       hop_length = self.hop_length\n                                       )\n        mel = mel_spectogram(audio)\n\n        # Convert to Image\n        image = torch.stack([mel])\n        \n        # Normalize Image\n        max_val = torch.abs(image).max()\n        image = image / max_val\n        return image\n\n    \n    def __getitem__(self, index):\n    \n        audio, sample_rate = torchaudio.load(self.file_paths[index])\n        audio = self.to_mono(audio)\n        \n        if sample_rate != self.target_sample_rate:\n            resample = T.Resample(sample_rate, self.target_sample_rate)\n            audio = resample(audio)\n        \n        audios = []\n        for i in range(self.num_samples, len(audio) + self.step, self.step):\n            start = max(0, i - self.num_samples)\n            end = start + self.num_samples\n            audios.append(audio[start:end])\n            \n        if len(audios[-1]) < self.num_samples:\n            audios = audios[:-1]\n            \n        images = [self.audio_to_image(audio) for audio in audios]\n        images = np.stack(images)\n        \n        return images\n            \n        \n    def to_mono(self, audio):\n        return torch.mean(audio, axis=0)","metadata":{"execution":{"iopub.status.busy":"2023-04-15T13:00:56.897497Z","iopub.execute_input":"2023-04-15T13:00:56.898057Z","iopub.status.idle":"2023-04-15T13:00:56.913487Z","shell.execute_reply.started":"2023-04-15T13:00:56.898022Z","shell.execute_reply":"2023-04-15T13:00:56.911877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class AudioModel(nn.Module):\n    def __init__(self,\n                 name,\n                 model_name = 'tf_efficientnet_b3_ns',\n                pretrained = True, \n                num_classes=CONFIG['num_classes']):\n        super(AudioModel, self).__init__()\n        self.model = timm.create_model(name, pretrained=pretrained, in_chans=1)\n        self.model.reset_classifier(num_classes=0) \n        in_features = self.model.num_features\n        self.fc = nn.Linear(in_features, CONFIG['num_classes'])\n\n    def forward(self, x):\n        x = self.model(x)\n        x = self.fc(x)\n        return x\n","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-04-15T13:00:56.917079Z","iopub.execute_input":"2023-04-15T13:00:56.918487Z","iopub.status.idle":"2023-04-15T13:00:56.930250Z","shell.execute_reply.started":"2023-04-15T13:00:56.918443Z","shell.execute_reply":"2023-04-15T13:00:56.928996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def criterion(outputs, labels):\n    return nn.CrossEntropyLoss()(outputs, labels)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-15T13:00:56.931819Z","iopub.execute_input":"2023-04-15T13:00:56.932740Z","iopub.status.idle":"2023-04-15T13:00:56.943078Z","shell.execute_reply.started":"2023-04-15T13:00:56.932700Z","shell.execute_reply":"2023-04-15T13:00:56.941947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def padded_cmap(solution, submission, padding_factor=5):\n    new_rows = []\n    for i in range(padding_factor):\n        new_rows.append([1 for i in range(len(solution.columns))])\n    new_rows = pd.DataFrame(new_rows)\n    new_rows.columns = solution.columns\n    padded_solution = pd.concat([solution, new_rows]).reset_index(drop=True).copy()\n    padded_submission = pd.concat([submission, new_rows]).reset_index(drop=True).copy()\n    score = average_precision_score(\n        padded_solution.values,\n        padded_submission.values,\n        average='macro',\n    )\n    return score","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-15T13:00:56.944816Z","iopub.execute_input":"2023-04-15T13:00:56.945319Z","iopub.status.idle":"2023-04-15T13:00:56.955341Z","shell.execute_reply.started":"2023-04-15T13:00:56.945268Z","shell.execute_reply":"2023-04-15T13:00:56.954131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict(data_loader, model):\n        \n    model.to('cpu')\n    model.eval()    \n    predictions = []\n    for en in range(len(data_loader)):\n        images = torch.from_numpy(data_loader[en])\n        with torch.no_grad():\n            outputs = model(images).sigmoid().detach().cpu().numpy()\n        predictions.append(outputs)\n    \n    return predictions","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-15T13:00:56.957070Z","iopub.execute_input":"2023-04-15T13:00:56.958401Z","iopub.status.idle":"2023-04-15T13:00:56.968911Z","shell.execute_reply.started":"2023-04-15T13:00:56.958348Z","shell.execute_reply":"2023-04-15T13:00:56.967593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def estimate_time(model_name, n_mels = CONFIG['n_mels'], hop_length = CONFIG['hop_length']):\n    # Start stopwatch\n    tick = time.time()\n\n    ds_test = AudioDataset(\n        df_test, \n        target_sample_rate = CONFIG['sample_rate'],\n        audio_length = CONFIG['audio_length'],\n        n_mels = n_mels,\n        hop_length = hop_length\n    )\n\n    model = AudioModel(model_name)\n\n\n    preds = predict(ds_test, model)   \n\n    gc.collect()\n    torch.cuda.empty_cache()\n    tock = time.time()\n\n\n    sub_time = (tock-tick)*200 # ~200 recording on the test data\n    sub_time = time.gmtime(sub_time)\n    sub_time = time.strftime(\"%H hr: %M min : %S sec\", sub_time)\n    print(f\">> Estimated Time for submission: ~ {sub_time} for {model_name}\")\n    return (tock-tick)*200/60\n\n","metadata":{"execution":{"iopub.status.busy":"2023-04-15T13:00:56.970710Z","iopub.execute_input":"2023-04-15T13:00:56.972268Z","iopub.status.idle":"2023-04-15T13:00:56.981824Z","shell.execute_reply.started":"2023-04-15T13:00:56.972197Z","shell.execute_reply":"2023-04-15T13:00:56.980864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Choice of backbone\n\nDue to the limitations of the inference time, you might not be able to ensemble multiple models if your choice of backbone is too time-consuming.\n\nLet's explore a few popular backbones from past BirdCLEF competitions:\n* 'tf_efficientnet_b3_ns', # https://www.kaggle.com/competitions/birdclef-2022/discussion/327047\n* 'tf_efficientnet_b0_ns', # https://www.kaggle.com/competitions/birdclef-2022/discussion/327193\n* 'tf_efficientnetv2_s_in21k', # https://www.kaggle.com/competitions/birdclef-2022/discussion/327193\n* 'tf_efficientnetv2_m_in21k', # https://www.kaggle.com/competitions/birdclef-2021/discussion/243463\n* 'resnet34', # https://www.kaggle.com/competitions/birdclef-2022/discussion/327193\n* 'resnet50', # https://www.kaggle.com/competitions/birdclef-2021/discussion/243351\n* 'convnext_tiny', # https://www.kaggle.com/competitions/birdclef-2022/discussion/327044\n* 'seresnext26t_32x4d', # https://www.kaggle.com/competitions/birdclef-2021/discussion/243463\n* 'eca_nfnet_l0', # https://www.kaggle.com/competitions/birdclef-2022/discussion/327047","metadata":{}},{"cell_type":"code","source":"#timm.list_models()\n\nmodel_zoo = ['tf_efficientnet_b3_ns', # https://www.kaggle.com/competitions/birdclef-2022/discussion/327047\n     'tf_efficientnet_b0_ns', # https://www.kaggle.com/competitions/birdclef-2022/discussion/327193\n     'tf_efficientnetv2_s_in21k', # https://www.kaggle.com/competitions/birdclef-2022/discussion/327193\n     'tf_efficientnetv2_m_in21k', # https://www.kaggle.com/competitions/birdclef-2021/discussion/243463\n     'resnet34', # https://www.kaggle.com/competitions/birdclef-2022/discussion/327193\n    'resnet50', # https://www.kaggle.com/competitions/birdclef-2021/discussion/243351\n    'convnext_tiny', # https://www.kaggle.com/competitions/birdclef-2022/discussion/327044\n     'seresnext26t_32x4d', # https://www.kaggle.com/competitions/birdclef-2021/discussion/243463\n     'eca_nfnet_l0', # https://www.kaggle.com/competitions/birdclef-2022/discussion/327047\n]\n\nt = []\n\nfor m in model_zoo:\n    estimated_time = estimate_time(m)\n    t.append(estimated_time)\n    \nfig = plt.figure(figsize = (5, 10))\nplt.barh(model_zoo, t)\n \nplt.ylabel(\"Backbone\")\nplt.xlabel(\"Minutes of estimated inference time\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-15T13:00:56.982922Z","iopub.execute_input":"2023-04-15T13:00:56.984158Z","iopub.status.idle":"2023-04-15T13:04:50.328337Z","shell.execute_reply.started":"2023-04-15T13:00:56.984118Z","shell.execute_reply":"2023-04-15T13:04:50.327066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Fro the above plot, we can see that 'tf_efficientnetv2_s_in21k' and 'tf_efficientnetv2_m_in21k' won't be suitable because the estimated inference time is longer than 120 mintes.","metadata":{}},{"cell_type":"markdown","source":"# Spectrogram Size\n\nThe spectrogram size is impacted by `n_mels` (= height of Spectrogram) and the `hop_length` inversly proportional to the width ((sample_rate * audio_length)/hop_length + 1)\n* the bigger n_mels, the bigger the spectrogram height.\n* the smaller hop_length, the the bigger the spectrogram width.","metadata":{}},{"cell_type":"code","source":"n_mels = CONFIG['n_mels'],\nhop_length = CONFIG['hop_length'],\n\nn_mels_list = [128, 256, 512]\nhop_length_list = [128, 256, 512 ]\n\nt = []\nfor hop_length in hop_length_list:\n    estimated_time = estimate_time('resnet34', CONFIG['n_mels'], hop_length)\n    t.append(estimated_time)\n    \nfig, ax = plt.subplots(1, 2, figsize = (10, 5))\nax[0].barh(hop_length_list,t)\nax[0].set_ylabel(\"hop_length\")\nax[0].set_xlabel(\"Minutes of estimated inference time\")\n\n\nt = []\nfor n_mels in n_mels_list:\n    estimated_time = estimate_time('resnet34', n_mels, CONFIG['hop_length'])\n    t.append(estimated_time)\n    \nax[1].barh(n_mels_list,t)\nax[1].set_ylabel(\"n_mels\")\nax[1].set_xlabel(\"Minutes of estimated inference time\")\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-15T13:08:03.686193Z","iopub.execute_input":"2023-04-15T13:08:03.686587Z","iopub.status.idle":"2023-04-15T13:10:14.526610Z","shell.execute_reply.started":"2023-04-15T13:08:03.686553Z","shell.execute_reply":"2023-04-15T13:10:14.525049Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Summary\n* **Batch size:** Should be as large as possible in your inference pipeline (For the code in this Notebook, the biggest batch size in powers of two is 128)\n* **Choice of backbone:** Find a trade-off between backbone performance and inference time to ensemble multiple models\n* **Spectrogram size:** find a trade-off between performance (bigger spectrogram, bigger n_mels, smaller hop_length) and inference time  (smaller spectrogram, smaller n_mels, bigger hop_length)\n    * the bigger n_mels, the bigger the spectrogram height.\n    * the smaller hop_length, the the bigger the spectrogram width.","metadata":{}},{"cell_type":"code","source":"\"\"\"#!pip install onnxruntime\n\nimport onnx\nimport onnxruntime\n\n# https://pytorch.org/tutorials/advanced/super_resolution_with_onnxruntime.html\nmodel = AudioModel(CONFIG['model_name'])\ndummy_input = torch.randn(16, 1, 3,3, requires_grad=False,  device=CONFIG['device'])\ntorch_out = model(dummy_input)\n\n# Providing input and output names sets the display names for values\n# within the model's graph. Setting these does not change the semantics\n# of the graph; it is only for readability.\n#\n# The inputs to the network consist of the flat list of inputs (i.e.\n# the values you would pass to the forward() method) followed by the\n# flat list of parameters. You can partially specify names, i.e. provide\n# a list here shorter than the number of inputs to the model, and we will\n# only set that subset of names, starting from the beginning.\n#input_names = [ \"actual_input_1\" ] + [ \"learned_%d\" % i for i in range(16) ]\n#output_names = [ \"output1\" ]\n\n# Export model to ONNX format\ntorch.onnx.export(model, \n                  dummy_input, \n                  f\"{CONFIG['model_name']}.onnx\", \n                  verbose=True, \n                  #input_names=input_names, \n                  #output_names=output_names\n                 )\n\n# https://pytorch.org/tutorials/advanced/super_resolution_with_onnxruntime.html\n\n\n# Load the ONNX model\nmodel = onnx.load(f\"/kaggle/working/{CONFIG['model_name']}.onnx\")\n\n# Check that the model is well formed\nonnx.checker.check_model(model)\n\n# Print a human readable representation of the graph\nprint(onnx.helper.printable_graph(model.graph))\n\n# https://pytorch.org/tutorials/advanced/super_resolution_with_onnxruntime.html\n\nort_session = onnxruntime.InferenceSession(f\"{CONFIG['model_name']}.onnx\")\n\ndef to_numpy(tensor):\n    return tensor.detach().cpu().numpy() if tensor.requires_grad else tensor.cpu().numpy()\n\n# compute ONNX Runtime output prediction\nort_inputs = {ort_session.get_inputs()[0].name: to_numpy(dummy_input)}\nort_outs = ort_session.run(None, ort_inputs)\nimg_out_y = ort_outs[0]\n\n# compare ONNX Runtime and PyTorch results\nnp.testing.assert_allclose(to_numpy(torch_out), ort_outs[0], rtol=1e-03, atol=1e-05)\n\nprint(\"Exported model has been tested with ONNXRuntime, and the result looks good!\")\n\n\ndef predict_with_onnx(data_loader, model):\n        \n    model.to('cpu')\n    model.eval()    \n    predictions = []\n    for en in range(len(data_loader)):\n        images = torch.from_numpy(data_loader[en])\n        with torch.no_grad():\n            outputs = model(images).sigmoid().detach().cpu().numpy()\n        predictions.append(outputs)\n    \n    return predictions\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-04-15T13:05:54.149865Z","iopub.status.idle":"2023-04-15T13:05:54.150823Z","shell.execute_reply.started":"2023-04-15T13:05:54.150553Z","shell.execute_reply":"2023-04-15T13:05:54.150584Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]}]}