{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":171041473,"sourceType":"kernelVersion"}],"dockerImageVersionId":30673,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Hello fellow Kagglers,\n\nThis notebook demonstrates the inference process and is a work in progress.\n\nUpdates will follow soon.\n\n[training notebook](https://www.kaggle.com/code/markwijkhuizen/birdclef-2024-efficientvit-training)\n\n[preprocessing notebook](https://www.kaggle.com/code/markwijkhuizen/birdclef-2024-eda-preprocessed-dataset)","metadata":{}},{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nfrom torch import nn\nfrom tqdm.notebook import tqdm\nimport IPython.display as ipd\n\nimport torch\nimport glob\nimport librosa\nimport re","metadata":{"execution":{"iopub.status.busy":"2024-04-08T19:27:03.948149Z","iopub.execute_input":"2024-04-08T19:27:03.949027Z","iopub.status.idle":"2024-04-08T19:27:09.821876Z","shell.execute_reply.started":"2024-04-08T19:27:03.948981Z","shell.execute_reply":"2024-04-08T19:27:09.820238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configuration","metadata":{}},{"cell_type":"code","source":"class Config():\n    # Horizontal melspectrogram resolution\n    MELSPEC_H = 128\n    # Competition Root Folder\n    ROOT_FOLDER = '/kaggle/input/birdclef-2024'\n    # Maximum decibel to clip audio to\n    TOP_DB = 100\n    # Minimum rating\n    MIN_RATING = 3.0\n    # Sample rate as provided in competition description\n    SR = 32000\n    N_FFT = 2000\n    HOP_LENGTH = 500\n    # Model input\n    HEIGHT = 128\n    WIDTH = 320\n    # Duration\n    DURATION_S = 240\n    WINDOW_S = 5\n    N_TEST_CHUNKS = DURATION_S // WINDOW_S\n    \nCONFIG = Config()","metadata":{"execution":{"iopub.status.busy":"2024-04-08T19:27:09.824479Z","iopub.execute_input":"2024-04-08T19:27:09.825053Z","iopub.status.idle":"2024-04-08T19:27:09.832188Z","shell.execute_reply.started":"2024-04-08T19:27:09.825016Z","shell.execute_reply":"2024-04-08T19:27:09.830847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Sample submission","metadata":{}},{"cell_type":"code","source":"sample_submission = pd.read_csv('/kaggle/input/birdclef-2024/sample_submission.csv')\n\n# Set labels\nCONFIG.LABELS = sample_submission.columns[1:]\nCONFIG.N_CLASSES = len(CONFIG.LABELS)\nprint(f'# classes: {CONFIG.N_CLASSES}')\n\ndisplay(sample_submission.head())","metadata":{"execution":{"iopub.status.busy":"2024-04-08T19:27:09.833949Z","iopub.execute_input":"2024-04-08T19:27:09.835256Z","iopub.status.idle":"2024-04-08T19:27:09.908326Z","shell.execute_reply.started":"2024-04-08T19:27:09.835212Z","shell.execute_reply":"2024-04-08T19:27:09.907024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# OGG → Melspectrogram Conversion","metadata":{}},{"cell_type":"code","source":"# Convert OOG audio files to melspectrogram encoded as PNG bytes\ndef ogg2melspectrogram(file_path):\n    # Load the audio file\n    y, _ = librosa.load(file_path, sr=CONFIG.SR)\n    # Normalize audio\n    y = librosa.util.normalize(y)\n    # Convert to mel spectrogram\n    spec = librosa.feature.melspectrogram(\n        y=y,\n        sr=CONFIG.SR, # sample rate\n        n_fft=CONFIG.N_FFT, # number of samples in window \n        hop_length=CONFIG.HOP_LENGTH, # step size of window\n        n_mels=CONFIG.MELSPEC_H, # horizontal resolution from fmin→fmax in log scale\n        fmin=40, # minimum frequency\n        fmax=15000, # maximum frequency\n        power=2.0, # intensity^power for log scale\n    )\n    # Convert to Db\n    spec = librosa.power_to_db(spec, ref=CONFIG.TOP_DB)\n    # Normalize 0-min\n    spec = spec - spec.min()\n    # Normalize 0-255\n    spec = (spec / spec.max() * 255).astype(np.uint8)\n    \n    return spec","metadata":{"execution":{"iopub.status.busy":"2024-04-08T19:27:09.910265Z","iopub.execute_input":"2024-04-08T19:27:09.911054Z","iopub.status.idle":"2024-04-08T19:27:09.921801Z","shell.execute_reply.started":"2024-04-08T19:27:09.911005Z","shell.execute_reply":"2024-04-08T19:27:09.920758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"# Count model parameters\ndef count_parameters(model):\n    return sum([p.numel() for p in model.parameters()])","metadata":{"execution":{"iopub.status.busy":"2024-04-08T19:27:09.924580Z","iopub.execute_input":"2024-04-08T19:27:09.925556Z","iopub.status.idle":"2024-04-08T19:27:09.936811Z","shell.execute_reply.started":"2024-04-08T19:27:09.925511Z","shell.execute_reply":"2024-04-08T19:27:09.935224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model from training notebook\nclass Model(nn.Module):\n    def __init__(self):\n        super().__init__()\n        # ImageNet Normalize Input\n        self.normalize = transforms.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225])\n        \n        # Backbone\n        self.backbone = timm.create_model(\n                CONFIG.BACKBONE,\n                pretrained=True,\n                num_classes=CONFIG.N_CLASSES,\n            )\n        \n    def forward(self, inputs):\n        # Go From HxW → 3xHxW\n        inputs = inputs.unsqueeze(1).expand(-1, 3, -1, -1)\n        # Normalize [0-255] → [0-1]\n        inputs = inputs.float() / 255\n        # Normalize\n        inputs = self.normalize(inputs)\n        \n        return self.backbone(inputs)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T19:27:09.938485Z","iopub.execute_input":"2024-04-08T19:27:09.938897Z","iopub.status.idle":"2024-04-08T19:27:09.958571Z","shell.execute_reply.started":"2024-04-08T19:27:09.938854Z","shell.execute_reply":"2024-04-08T19:27:09.957260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the saved model\nmodel = torch.load('/kaggle/input/birdclef-2024-efficientvit-training/model.pth', map_location=torch.device('cpu'))\nmodel.eval()\n\n# Number of parameters\nprint(f'# Model Parameters: {count_parameters(model):,}')","metadata":{"execution":{"iopub.status.busy":"2024-04-08T19:27:09.960376Z","iopub.execute_input":"2024-04-08T19:27:09.961022Z","iopub.status.idle":"2024-04-08T19:27:16.415825Z","shell.execute_reply.started":"2024-04-08T19:27:09.960971Z","shell.execute_reply":"2024-04-08T19:27:16.414350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference Loop","metadata":{}},{"cell_type":"code","source":"# List to save inference rows in\nINFERENCE_ROWS = []\n\n# Hidden test files\nif len(glob.glob(f'{CONFIG.ROOT_FOLDER}/test_soundscapes/*.ogg')) > 0:\n    ogg_file_paths = glob.glob(f'{CONFIG.ROOT_FOLDER}/test_soundscapes/*.ogg')\nelse:\n    ogg_file_paths = sorted(glob.glob(f'{CONFIG.ROOT_FOLDER}/unlabeled_soundscapes/*.ogg'))[:10]\n\n# Iterate over OGG files\nfor i, file_path in enumerate(tqdm(ogg_file_paths)):\n    # Extract filename\n    row_id = re.search(r'/([^/]+)\\.ogg$', file_path).group(1)\n    # Read OGG file and convert to melspectrogram\n    spec = ogg2melspectrogram(file_path)\n    # Pad spectogram to multiple of WIDTH\n    pad = CONFIG.WIDTH - (spec.shape[1] % CONFIG.WIDTH)\n    if pad > 0:\n        spec = np.pad(spec, ((0,0), (0,pad)))\n    # Reshape to BxHxW\n    spec = spec.reshape(CONFIG.HEIGHT,-1,CONFIG.WIDTH).transpose([1,0,2])\n    # Convert spec from Numpy array on CPU to Torch Tensor on GPU\n    spec = torch.Tensor(spec)\n    # Predict\n    with torch.no_grad():\n        outputs = model(spec).softmax(dim=1).numpy()\n    # Add to inference rows and limit to 4 minutes\n    for t, o in zip(range(CONFIG.N_TEST_CHUNKS), outputs):\n        # Predictions for each bird\n        predictions = dict([ (l,p) for l, p in zip(CONFIG.LABELS, o) ])\n        # Append to inference rows\n        INFERENCE_ROWS.append(\n            { 'row_id': f'{row_id}_{(t+1)*5}' } | predictions\n        )","metadata":{"execution":{"iopub.status.busy":"2024-04-08T19:28:08.107137Z","iopub.status.idle":"2024-04-08T19:28:08.108031Z","shell.execute_reply.started":"2024-04-08T19:28:08.107806Z","shell.execute_reply":"2024-04-08T19:28:08.107827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"# Create pandas DataFrame from inference rows\nsubmission_df = pd.DataFrame(INFERENCE_ROWS)\n\n# Display submission DataFrame\ndisplay(submission_df.head(30))","metadata":{"execution":{"iopub.status.busy":"2024-04-08T19:28:09.343186Z","iopub.execute_input":"2024-04-08T19:28:09.343688Z","iopub.status.idle":"2024-04-08T19:28:09.479014Z","shell.execute_reply.started":"2024-04-08T19:28:09.343651Z","shell.execute_reply":"2024-04-08T19:28:09.477590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Write CSV\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T19:28:10.504568Z","iopub.execute_input":"2024-04-08T19:28:10.505121Z","iopub.status.idle":"2024-04-08T19:28:10.694704Z","shell.execute_reply.started":"2024-04-08T19:28:10.505067Z","shell.execute_reply":"2024-04-08T19:28:10.693237Z"},"trusted":true},"execution_count":null,"outputs":[]}]}