{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":8573823,"sourceType":"datasetVersion","datasetId":5126830},{"sourceId":57545,"sourceType":"modelInstanceVersion","modelInstanceId":48283}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nfrom torch import nn\nfrom tqdm.notebook import tqdm\nimport IPython.display as ipd\nimport imageio.v3 as imageio\nfrom torchvision import transforms\n\nimport albumentations as A\nfrom torch.utils.data import Dataset, DataLoader\n\nimport torch\nimport glob\nimport librosa\nimport re\n\nimport cv2\nimport pickle\nimport lzma\n\n\nimport torchmetrics\nimport timm\nimport pickle\nimport psutil\nimport time\nimport os","metadata":{"execution":{"iopub.status.busy":"2024-06-04T04:35:38.202342Z","iopub.execute_input":"2024-06-04T04:35:38.203568Z","iopub.status.idle":"2024-06-04T04:35:38.213583Z","shell.execute_reply.started":"2024-06-04T04:35:38.203459Z","shell.execute_reply":"2024-06-04T04:35:38.212033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    # Horizontal melspectrogram resolution\n    #MELSPEC_H = 128\n    MELSPEC_H = 224\n    # Competition Root Folder\n    ROOT_FOLDER = '/kaggle/input/birdclef-2024'\n    # Maximum decibel to clip audio to\n    TOP_DB = 80\n    # Minimum rating\n    MIN_RATING = 3.0\n    # Sample rate as provided in competition description\n    SR = 32000\n    N_FFT = 2000\n    HOP_LENGTH = 512\n    # Dataset\n    #HEIGHT = 128\n    #WIDTH = 320\n    HEIGHT = 224\n    WIDTH = 224\n    ROOT_FOLDER = '/kaggle/input/birdclef-2024'\n    WORK_FOLDER = '/kaggle/working'\n    # Training\n    BATCH_SIZE = 16\n    N_EPOCHS = 20\n    # Model\n    BACKBONE = 'efficientvit_m5.r224_in1k'\n    # Learning Rate Scheduler\n    LR_MAX = 3e-4\n    WEIGHT_DECAY = 0.00\n    # Others\n    SEED = 42\n    IS_INTERACTIVE = os.environ['KAGGLE_KERNEL_RUN_TYPE'] == 'Interactive'\n    # Duration\n    DURATION_S = 240\n    WINDOW_S = 10\n    N_TEST_CHUNKS = DURATION_S // WINDOW_S\n    \nCONFIG = Config()","metadata":{"execution":{"iopub.status.busy":"2024-06-04T04:35:38.216196Z","iopub.execute_input":"2024-06-04T04:35:38.216702Z","iopub.status.idle":"2024-06-04T04:35:38.228793Z","shell.execute_reply.started":"2024-06-04T04:35:38.216666Z","shell.execute_reply":"2024-06-04T04:35:38.227226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Count model parameters\ndef count_parameters(model):\n    return sum([p.numel() for p in model.parameters()])","metadata":{"execution":{"iopub.status.busy":"2024-06-04T04:35:38.230675Z","iopub.execute_input":"2024-06-04T04:35:38.231811Z","iopub.status.idle":"2024-06-04T04:35:38.243524Z","shell.execute_reply.started":"2024-06-04T04:35:38.231771Z","shell.execute_reply":"2024-06-04T04:35:38.241976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Model(nn.Module):\n    def __init__(self):\n        super().__init__()\n        # ImageNet Normalize Input\n        self.normalize = transforms.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225])\n        \n        # Backbone\n        self.backbone = timm.create_model(\n                CONFIG.BACKBONE,\n                pretrained=True,\n                num_classes=CONFIG.N_CLASSES,\n            )\n        \n    def forward(self, inputs):\n        # Go From HxW → 3xHxW\n        inputs = inputs.unsqueeze(1).expand(-1, 3, -1, -1)\n        # Normalize [0-255] → [0-1]\n        inputs = inputs.float() / 255\n        # Normalize\n        inputs = self.normalize(inputs)\n        \n        return self.backbone(inputs)","metadata":{"execution":{"iopub.status.busy":"2024-06-04T04:35:38.246216Z","iopub.execute_input":"2024-06-04T04:35:38.246828Z","iopub.status.idle":"2024-06-04T04:35:38.261813Z","shell.execute_reply.started":"2024-06-04T04:35:38.246788Z","shell.execute_reply":"2024-06-04T04:35:38.260113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv('/kaggle/input/birdclef-2024/sample_submission.csv')\n\n# Set labels\nCONFIG.LABELS = sample_submission.columns[1:]\nCONFIG.N_CLASSES = len(CONFIG.LABELS)\nprint(f'# classes: {CONFIG.N_CLASSES}')\n\ndisplay(sample_submission.head())","metadata":{"execution":{"iopub.status.busy":"2024-06-04T04:35:38.396373Z","iopub.execute_input":"2024-06-04T04:35:38.397911Z","iopub.status.idle":"2024-06-04T04:35:38.456920Z","shell.execute_reply.started":"2024-06-04T04:35:38.397861Z","shell.execute_reply":"2024-06-04T04:35:38.455345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert OOG audio files to melspectrogram encoded as PNG bytes\ndef ogg2melspectrogramspec(file_path):\n    # Load the audio file\n    y, _ = librosa.load(file_path, sr=CONFIG.SR)\n    # Normalize audio\n    y = librosa.util.normalize(y)\n    # Convert to mel spectrogram\n    spec = librosa.feature.melspectrogram(\n        y=y,\n        sr=CONFIG.SR, # sample rate\n        n_fft=CONFIG.N_FFT, # number of samples in window \n        hop_length=CONFIG.HOP_LENGTH, # step size of window\n        n_mels=CONFIG.MELSPEC_H, # horizontal resolution from fmin→fmax in log scale\n        fmin=40, # minimum frequency\n        fmax=15000, # maximum frequency\n        power=2.0, # intensity^power for log scale\n    )\n    # Convert to Db\n    spec = librosa.power_to_db(spec, ref=CONFIG.TOP_DB)\n    # Normalize 0-min\n    spec = spec - spec.min()\n    # Normalize 0-255\n    spec = (spec / spec.max() * 255).astype(np.uint8)\n    \n    return spec","metadata":{"execution":{"iopub.status.busy":"2024-06-04T04:35:38.460605Z","iopub.execute_input":"2024-06-04T04:35:38.461722Z","iopub.status.idle":"2024-06-04T04:35:38.471533Z","shell.execute_reply.started":"2024-06-04T04:35:38.461669Z","shell.execute_reply":"2024-06-04T04:35:38.469824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the saved model\nmodel = torch.load('/kaggle/input/efficientvitv2/modelnew1.pth', map_location=torch.device('cpu'))\nmodel.eval()\n\n# Number of parameters\nprint(f'# Model Parameters: {count_parameters(model):,}')","metadata":{"execution":{"iopub.status.busy":"2024-06-04T04:35:38.473019Z","iopub.execute_input":"2024-06-04T04:35:38.473458Z","iopub.status.idle":"2024-06-04T04:35:38.636113Z","shell.execute_reply.started":"2024-06-04T04:35:38.473422Z","shell.execute_reply":"2024-06-04T04:35:38.634635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# List to save inference rows in\nINFERENCE_ROWS = []\n\n# Hidden test files\nif len(glob.glob(f'{CONFIG.ROOT_FOLDER}/test_soundscapes/*.ogg')) > 0:\n    ogg_file_paths = glob.glob(f'{CONFIG.ROOT_FOLDER}/test_soundscapes/*.ogg')\nelse:\n    ogg_file_paths = sorted(glob.glob(f'{CONFIG.ROOT_FOLDER}/unlabeled_soundscapes/*.ogg'))\n\n# Iterate over OGG files\nfor i, file_path in enumerate(tqdm(ogg_file_paths)):\n    # Extract filename\n    row_id = re.search(r'/([^/]+)\\.ogg$', file_path).group(1)\n    # Read OGG file and convert to melspectrogram\n    spec = ogg2melspectrogramspec(file_path)\n    # Pad spectogram to multiple of WIDTH\n    pad = CONFIG.WIDTH - (spec.shape[1] % CONFIG.WIDTH)\n    if pad > 0:\n        spec = np.pad(spec, ((0,0), (0,pad)))\n    # Reshape to BxHxW\n    spec = spec.reshape(CONFIG.HEIGHT,-1,CONFIG.WIDTH).transpose([1,0,2])\n    # Convert spec from Numpy array on CPU to Torch Tensor on GPU\n    spec = torch.Tensor(spec)\n    # Predict\n    with torch.no_grad():\n        outputs = model(spec).softmax(dim=1).numpy()\n    # Add to inference rows and limit to 4 minutes\n    for t, o in zip(range(CONFIG.N_TEST_CHUNKS), outputs):\n        # Predictions for each bird\n        predictions = dict([ (l,p) for l, p in zip(CONFIG.LABELS, o) ])\n        # Append to inference rows\n        INFERENCE_ROWS.append(\n            { 'row_id': f'{row_id}_{(t+1)*5}' } | predictions\n        )","metadata":{"execution":{"iopub.status.busy":"2024-06-04T04:37:10.400363Z","iopub.execute_input":"2024-06-04T04:37:10.400896Z","iopub.status.idle":"2024-06-04T04:37:36.226459Z","shell.execute_reply.started":"2024-06-04T04:37:10.400851Z","shell.execute_reply":"2024-06-04T04:37:36.225221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create pandas DataFrame from inference rows\nsubmission_df = pd.DataFrame(INFERENCE_ROWS)\n\n# Display submission DataFrame\ndisplay(submission_df.head(30))","metadata":{"execution":{"iopub.status.busy":"2024-06-04T04:36:22.607550Z","iopub.execute_input":"2024-06-04T04:36:22.608581Z","iopub.status.idle":"2024-06-04T04:36:22.711477Z","shell.execute_reply.started":"2024-06-04T04:36:22.608532Z","shell.execute_reply":"2024-06-04T04:36:22.709800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Write CSV\nsubmission_df.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-06-04T04:36:22.713141Z","iopub.execute_input":"2024-06-04T04:36:22.713560Z","iopub.status.idle":"2024-06-04T04:36:22.810234Z","shell.execute_reply.started":"2024-06-04T04:36:22.713526Z","shell.execute_reply":"2024-06-04T04:36:22.808849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"reference","metadata":{}},{"cell_type":"markdown","source":"https://www.kaggle.com/code/markwijkhuizen/birdclef-2024-efficientvit-training?scriptVersionId=171041473&cellId=3","metadata":{}}]}