{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":8167877,"sourceType":"datasetVersion","datasetId":4833461},{"sourceId":8748557,"sourceType":"datasetVersion","datasetId":5213112}],"dockerImageVersionId":30684,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install --no-index --find-links /kaggle/input/openvino-lib openvino -q","metadata":{"execution":{"iopub.status.busy":"2025-04-12T12:58:34.089687Z","iopub.execute_input":"2025-04-12T12:58:34.090063Z","iopub.status.idle":"2025-04-12T12:58:47.785851Z","shell.execute_reply.started":"2025-04-12T12:58:34.090012Z","shell.execute_reply":"2025-04-12T12:58:47.784599Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport joblib\nimport openvino as ov\nimport librosa\nimport torch\nimport timm\nimport torch.nn.functional as F\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2025-04-12T12:58:47.787521Z","iopub.execute_input":"2025-04-12T12:58:47.787936Z","iopub.status.idle":"2025-04-12T12:58:56.752312Z","shell.execute_reply.started":"2025-04-12T12:58:47.787898Z","shell.execute_reply":"2025-04-12T12:58:56.751167Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Utils","metadata":{}},{"cell_type":"code","source":"SHAPE = [48, 1, 128, 320*2]\n\ndef mel(arr, sr=32_000):\n    arr = arr * 1024\n    S = librosa.feature.melspectrogram(y=arr, sr=sr,\n                                       n_fft=1024, hop_length=500, n_mels=128, \n                                       fmin=40, fmax=15000, power=2.0)\n    return S\n\ndef mel_preproc(x):\n    x = librosa.power_to_db(x, ref=1, top_db=100.0)\n    x = x.astype('float32')\n    return x\n\ndef torch_to_ov(model, input_shape=[48, 1, 128, 0], name='model'):\n    core = ov.Core()\n    ov_model = ov.convert_model(model)\n#     ov_model.reshape(input_shape)\n    compiled_model = core.compile_model(ov_model)\n    return compiled_model","metadata":{"execution":{"iopub.status.busy":"2025-04-12T12:59:27.424974Z","iopub.execute_input":"2025-04-12T12:59:27.425982Z","iopub.status.idle":"2025-04-12T12:59:27.432793Z","shell.execute_reply.started":"2025-04-12T12:59:27.425947Z","shell.execute_reply":"2025-04-12T12:59:27.431730Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load models","metadata":{}},{"cell_type":"code","source":"names = [\n    # EfficientNet_b0\n    'effnet_seg20_80low.ckpt',\n    'effnet_seg60_80low.ckpt',\n    'effnet_seg60_fold0.ckpt',\n    # RegNetY\n    'regnety_seg60_fold0.ckpt',\n    'regnety_seg30_fold0.ckpt',\n    'regnety_seg30_80low.ckpt',\n    ]\n\n\nmodels = []\nfor i, name in enumerate(names):\n    state_dict = torch.load(f'/kaggle/input/birdclef24-final/{name}', \n                            map_location=torch.device('cpu'))\n\n    model_name = 'efficientnet_b0'\n    if 'regnety' in name:\n        model_name = 'regnety_008.pycls_in1k'\n        \n    model = timm.create_model(\n        model_name, pretrained=None,\n        num_classes=182, in_chans=SHAPE[1]\n    )\n    \n    new_state_dict = {}\n    for key, val in state_dict['state_dict'].items():\n        if key.startswith('model.'):\n            new_state_dict[key[6:]] = val\n    model.load_state_dict(new_state_dict)\n    model.eval()\n    model = torch_to_ov(model, input_shape=SHAPE)\n    models.append(model)\n    \nprint(f'{len(models)} models are ready') ","metadata":{"execution":{"iopub.status.busy":"2025-04-12T12:59:31.048148Z","iopub.execute_input":"2025-04-12T12:59:31.049180Z","iopub.status.idle":"2025-04-12T12:59:48.221600Z","shell.execute_reply.started":"2025-04-12T12:59:31.049147Z","shell.execute_reply":"2025-04-12T12:59:48.220532Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load data and inference","metadata":{}},{"cell_type":"code","source":"test_audio_dir = '/kaggle/input/birdclef-2024/test_soundscapes/'\ntest_paths = [test_audio_dir+f for f in sorted(os.listdir(test_audio_dir))]\nif len(test_paths)==1:\n    test_audio_dir = '/kaggle/input/birdclef-2024/unlabeled_soundscapes/'\n    test_paths = [test_audio_dir+f for f in sorted(os.listdir(test_audio_dir))][:3]\ntest_df = pd.DataFrame(test_paths, columns=['filepath'])\ntest_df['filename'] = test_df.filepath.map(lambda x: x.split('/')[-1].replace('.ogg',''))","metadata":{"execution":{"iopub.status.busy":"2025-04-12T12:59:57.611224Z","iopub.execute_input":"2025-04-12T12:59:57.611593Z","iopub.status.idle":"2025-04-12T12:59:57.994351Z","shell.execute_reply.started":"2025-04-12T12:59:57.611565Z","shell.execute_reply":"2025-04-12T12:59:57.993297Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process(idx):\n    row = test_df.iloc[idx]\n    audiopath = row['filepath']\n    audio, sr = librosa.load(audiopath, sr=None)\n    chunk_size = sr * 5\n    chunks = audio.reshape(-1, chunk_size)\n    chunks_mel = mel(chunks, sr=sr)[:, :, :320].astype(np.float32)\n    return row['filename'], chunks_mel\n    \nindexes = test_df.index\noutput = joblib.Parallel(n_jobs=-1, backend=\"loky\")(\n        joblib.delayed(process)(idx) for idx in indexes\n    )\n\nmels_dict = dict(output)","metadata":{"execution":{"iopub.status.busy":"2025-04-12T13:01:27.167602Z","iopub.execute_input":"2025-04-12T13:01:27.167928Z","iopub.status.idle":"2025-04-12T13:01:29.905989Z","shell.execute_reply.started":"2025-04-12T13:01:27.167902Z","shell.execute_reply":"2025-04-12T13:01:29.904910Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ids = []\npreds = [np.empty(shape=(0, 182), dtype='float32') for _ in range(len(models))]\n\nfor filename in test_df.filename.tolist():\n    chunks = mels_dict[filename]\n    chunks = chunks[:, np.newaxis, :, :]\n    chunks_1 = np.concatenate([chunks[:1], chunks[:-1]], axis=0)\n    chunks_2 = np.concatenate([chunks[1:], chunks[-1:]], axis=0)\n    chunks = np.concatenate([chunks_1, chunks, chunks_2], axis=-1)\n    chunks = chunks[...,160:-160]\n    chunks = mel_preproc(chunks)\n    for m_idx in range(len(models)):\n        rec_preds = models[m_idx](torch.from_numpy(chunks))[0]\n        preds[m_idx] = np.concatenate([preds[m_idx], rec_preds], axis=0)\n\n    # create ID for each chunk in the audio with the filename and frame number\n    rec_ids = [f'{filename}_{(frame_id+1)*5}' for frame_id in range(rec_preds.shape[0])]\n    ids += rec_ids\npreds = F.sigmoid(torch.Tensor(np.array(preds))).numpy()","metadata":{"execution":{"iopub.status.busy":"2025-04-12T13:01:34.274001Z","iopub.execute_input":"2025-04-12T13:01:34.274387Z","iopub.status.idle":"2025-04-12T13:01:55.360329Z","shell.execute_reply.started":"2025-04-12T13:01:34.274358Z","shell.execute_reply":"2025-04-12T13:01:55.359432Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Aggregation","metadata":{}},{"cell_type":"code","source":"s0 = preds.shape[0]\npreds = preds.reshape(s0, -1, 48, 182)\nsmooth_preds = preds.copy()\nfor i in range(48):\n    smooth_preds[:, :, i] = preds[:, :, max(0,i-2):i+3].mean(axis=-2)\npreds = smooth_preds.reshape(s0, -1, 182)\n\n# preds[:3] = preds[:3].min(axis=0, keepdims=True)\n# preds[3:] = preds[3:].min(axis=0, keepdims=True)\npreds = preds.mean(axis=0, keepdims=True)\npreds = preds.squeeze()","metadata":{"execution":{"iopub.status.busy":"2025-04-12T13:01:26.031805Z","iopub.execute_input":"2025-04-12T13:01:26.032122Z","iopub.status.idle":"2025-04-12T13:01:26.042537Z","shell.execute_reply.started":"2025-04-12T13:01:26.032097Z","shell.execute_reply":"2025-04-12T13:01:26.041499Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission file","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv('/kaggle/input/birdclef-2024/train_metadata.csv')\nLABELS = sorted(list(data['primary_label'].unique()))\n\npred_df = pd.DataFrame(ids, columns=['row_id'])\npred_df.loc[:, LABELS] = preds\npred_df.to_csv('submission.csv', index=False)\npred_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2025-04-12T13:01:26.043820Z","iopub.execute_input":"2025-04-12T13:01:26.044153Z","iopub.status.idle":"2025-04-12T13:01:26.359456Z","shell.execute_reply.started":"2025-04-12T13:01:26.044115Z","shell.execute_reply":"2025-04-12T13:01:26.358076Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ims = plt.imshow(chunks[24][0], cmap='jet')\nplt.colorbar(ims, orientation='horizontal')\nplt.show()\nplt.bar(range(182), pred_df.iloc[24][LABELS])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-04-12T13:02:35.390041Z","iopub.execute_input":"2025-04-12T13:02:35.390926Z","iopub.status.idle":"2025-04-12T13:02:36.391476Z","shell.execute_reply.started":"2025-04-12T13:02:35.390893Z","shell.execute_reply":"2025-04-12T13:02:36.390531Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{},"outputs":[],"execution_count":null}]}