{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":184024212,"sourceType":"kernelVersion"}],"dockerImageVersionId":30732,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nfrom tqdm.notebook import tqdm\nimport IPython.display as ipd\n\nimport torch\nimport glob\nimport librosa\nimport re","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:52:16.832468Z","iopub.execute_input":"2024-06-18T07:52:16.832855Z","iopub.status.idle":"2024-06-18T07:52:22.280553Z","shell.execute_reply.started":"2024-06-18T07:52:16.832820Z","shell.execute_reply":"2024-06-18T07:52:22.279229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config():\n    # Horizontal melspectrogram resolution\n    MELSPEC_H = 128\n    # Competition Root Folder\n    ROOT_FOLDER = '/kaggle/input/birdclef-2024'\n    MODEL_FOLDER = '/kaggle/input/birdclef-24-vgg19-train'\n    \n    # Maximum decibel to clip audio to\n    TOP_DB = 100\n    # Minimum rating\n    MIN_RATING = 3.0\n    # Sample rate as provided in competition description\n    SR = 32000\n    N_FFT = 1000\n    HOP_LENGTH = 500\n    # Model input\n    HEIGHT = 128\n    WIDTH = 320\n    # Duration\n    DURATION_S = 240\n    WINDOW_S = 5\n    CUSTOM_TEST = False\n    N_TEST_CHUNKS = DURATION_S // WINDOW_S\n    \nCONFIG = Config()","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:52:22.282695Z","iopub.execute_input":"2024-06-18T07:52:22.283350Z","iopub.status.idle":"2024-06-18T07:52:22.291624Z","shell.execute_reply.started":"2024-06-18T07:52:22.283307Z","shell.execute_reply":"2024-06-18T07:52:22.290223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv(f'{CONFIG.ROOT_FOLDER}/sample_submission.csv')\n\n# Set labels\nCONFIG.LABELS = sample_submission.columns[1:]\nCONFIG.N_CLASSES = len(CONFIG.LABELS)\nprint(f'# classes: {CONFIG.N_CLASSES}')","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:52:22.293203Z","iopub.execute_input":"2024-06-18T07:52:22.293630Z","iopub.status.idle":"2024-06-18T07:52:22.325082Z","shell.execute_reply.started":"2024-06-18T07:52:22.293591Z","shell.execute_reply":"2024-06-18T07:52:22.323737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert OOG audio files to melspectrogram encoded as PNG bytes\ndef ogg2melspectrogram(file_path):\n    # Load the audio file\n    y, _ = librosa.load(file_path, sr=CONFIG.SR)\n    # Normalize audio\n    y = librosa.util.normalize(y)\n    # Convert to mel spectrogram\n    spec = librosa.feature.melspectrogram(\n        y=y,\n        sr=CONFIG.SR, # sample rate\n        n_fft=CONFIG.N_FFT, # number of samples in window \n        hop_length=CONFIG.HOP_LENGTH, # step size of window\n        n_mels=CONFIG.MELSPEC_H, # horizontal resolution from fmin→fmax in log scale\n        fmin=40, # minimum frequency\n        fmax=15000, # maximum frequency\n        power=2.0, # intensity^power for log scale\n    )\n    # Convert to Db\n    spec = librosa.power_to_db(spec, ref=CONFIG.TOP_DB)\n    # Normalize 0-min\n    spec = spec - spec.min()\n    # Normalize 0-255\n    spec = (spec / spec.max() * 255).astype(np.uint8)\n    \n    return spec","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:52:22.328326Z","iopub.execute_input":"2024-06-18T07:52:22.328810Z","iopub.status.idle":"2024-06-18T07:52:22.338771Z","shell.execute_reply.started":"2024-06-18T07:52:22.328767Z","shell.execute_reply":"2024-06-18T07:52:22.337392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Count model parameters\ndef count_parameters(model):\n    return sum([p.numel() for p in model.parameters()])","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:52:22.340027Z","iopub.execute_input":"2024-06-18T07:52:22.340507Z","iopub.status.idle":"2024-06-18T07:52:22.350170Z","shell.execute_reply.started":"2024-06-18T07:52:22.340466Z","shell.execute_reply":"2024-06-18T07:52:22.348881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model from training notebook\nclass Model(torch.nn.Module):\n    def __init__(self):\n        super().__init__()\n        # ImageNet Normalize Input\n        self.normalize = transforms.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225])\n        \n        # Backbone\n        self.backbone = timm.create_model(\n                CONFIG.BACKBONE,\n                pretrained=True,\n                num_classes=CONFIG.N_CLASSES,\n            )\n        \n    def forward(self, inputs):\n        # Go From HxW → 3xHxW\n        inputs = inputs.unsqueeze(1).expand(-1, 3, -1, -1)\n        # Normalize [0-255] → [0-1]\n        inputs = inputs.float() / 255\n        # Normalize\n        inputs = self.normalize(inputs)\n        \n        return self.backbone(inputs)","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:52:22.351715Z","iopub.execute_input":"2024-06-18T07:52:22.352099Z","iopub.status.idle":"2024-06-18T07:52:22.368397Z","shell.execute_reply.started":"2024-06-18T07:52:22.352069Z","shell.execute_reply":"2024-06-18T07:52:22.367027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the saved model\nmodel = torch.load(f'{CONFIG.MODEL_FOLDER}/model_split_data.pth', map_location=torch.device('cpu'))\n\nmodel.eval()\n\n\n# Number of parameters\nprint(f'# Model Parameters: {count_parameters(model):,}')","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:52:22.369887Z","iopub.execute_input":"2024-06-18T07:52:22.370274Z","iopub.status.idle":"2024-06-18T07:52:26.614931Z","shell.execute_reply.started":"2024-06-18T07:52:22.370242Z","shell.execute_reply":"2024-06-18T07:52:26.613608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# List to save inference rows in\nINFERENCE_ROWS = []\n\n# Hidden test files\nif len(glob.glob(f'{CONFIG.ROOT_FOLDER}/test_soundscapes/*.ogg')) > 0:\n    ogg_file_paths = glob.glob(f'{CONFIG.ROOT_FOLDER}/test_soundscapes/*.ogg')\nelse:\n    ogg_file_paths = sorted(glob.glob(f'{CONFIG.ROOT_FOLDER}/unlabeled_soundscapes/*.ogg'))[:10]\n\n# Iterate over OGG files\nfor i, file_path in enumerate(tqdm(ogg_file_paths)):\n    # Extract filename\n    row_id = re.search(r'/([^/]+)\\.ogg$', file_path).group(1)\n    # Read OGG file and convert to melspectrogram\n    spec = ogg2melspectrogram(file_path)\n    # Pad spectogram to multiple of WIDTH\n    pad = CONFIG.WIDTH - (spec.shape[1] % CONFIG.WIDTH)\n    if pad > 0:\n        spec = np.pad(spec, ((0,0), (0,pad)))\n    # Reshape to BxHxW\n    spec = spec.reshape(CONFIG.HEIGHT,-1,CONFIG.WIDTH).transpose([1,0,2])\n    # Convert spec from Numpy array on CPU to Torch Tensor on GPU\n    spec = torch.Tensor(spec)\n    # Predict\n    with torch.no_grad():\n        outputs = model(spec).softmax(dim=1).numpy()\n    # Add to inference rows and limit to 4 minutes\n    for t, o in zip(range(CONFIG.N_TEST_CHUNKS), outputs):\n        # Predictions for each bird\n        predictions = dict([ (l,p) for l, p in zip(CONFIG.LABELS, o) ])\n        # Append to inference rows\n        INFERENCE_ROWS.append(\n            { 'row_id': f'{row_id}_{(t+1)*5}' } | predictions\n        )","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:52:26.616749Z","iopub.execute_input":"2024-06-18T07:52:26.617252Z","iopub.status.idle":"2024-06-18T07:53:07.309407Z","shell.execute_reply.started":"2024-06-18T07:52:26.617208Z","shell.execute_reply":"2024-06-18T07:53:07.308219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create pandas DataFrame from inference rows\nsubmission_df = pd.DataFrame(INFERENCE_ROWS)\n\n# Display submission DataFrame\ndisplay(submission_df)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:53:07.310657Z","iopub.execute_input":"2024-06-18T07:53:07.311280Z","iopub.status.idle":"2024-06-18T07:53:07.401333Z","shell.execute_reply.started":"2024-06-18T07:53:07.311247Z","shell.execute_reply":"2024-06-18T07:53:07.399975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:53:07.403015Z","iopub.execute_input":"2024-06-18T07:53:07.403484Z","iopub.status.idle":"2024-06-18T07:53:07.537888Z","shell.execute_reply.started":"2024-06-18T07:53:07.403443Z","shell.execute_reply":"2024-06-18T07:53:07.536500Z"},"trusted":true},"execution_count":null,"outputs":[]}]}