{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":21669,"databundleVersionId":1692278,"sourceType":"competition"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Import","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport shutil\nimport csv\nimport librosa\nimport librosa.display\nimport random\nimport matplotlib.pyplot as plt\nimport IPython.display as ipd\nfrom PIL import Image\nimport soundfile as sf\nimport warnings\nimport tensorflow as tf\nimport keras\nfrom keras import layers\nfrom keras.preprocessing import image_dataset_from_directory\nfrom keras.models import Sequential\nfrom keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout, BatchNormalization, Input\nfrom keras.losses import BinaryCrossentropy\nfrom keras.optimizers import Adam\nfrom keras.callbacks import EarlyStopping, ReduceLROnPlateau\nfrom sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay, multilabel_confusion_matrix, classification_report\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import MultiLabelBinarizer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T00:26:04.255267Z","iopub.execute_input":"2025-08-05T00:26:04.256007Z","iopub.status.idle":"2025-08-05T00:26:04.262246Z","shell.execute_reply.started":"2025-08-05T00:26:04.255979Z","shell.execute_reply":"2025-08-05T00:26:04.261562Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Beállítások","metadata":{}},{"cell_type":"code","source":"seed=100\nlength=5\nsr=48000\nimage_height=128\nimage_width=400\nbatch_size=16\nepochs=20\n\nsave='/kaggle/working/spectrograms'\nos.makedirs(save, exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T00:03:22.367186Z","iopub.execute_input":"2025-08-05T00:03:22.367679Z","iopub.status.idle":"2025-08-05T00:03:22.372016Z","shell.execute_reply.started":"2025-08-05T00:03:22.367651Z","shell.execute_reply":"2025-08-05T00:03:22.371192Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"warnings.filterwarnings(\"ignore\", category=UserWarning)\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T00:03:22.375085Z","iopub.execute_input":"2025-08-05T00:03:22.375313Z","iopub.status.idle":"2025-08-05T00:03:22.420652Z","shell.execute_reply.started":"2025-08-05T00:03:22.375293Z","shell.execute_reply":"2025-08-05T00:03:22.419948Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Elérés","metadata":{}},{"cell_type":"code","source":"input='/kaggle/input/rfcx-species-audio-detection'\ntrain_path=os.path.join(input, 'train')\ntest_path=os.path.join(input, 'test')\nfp_csv=os.path.join(input, 'train_fp.csv')\ntp_csv=os.path.join(input, 'train_tp.csv')\ntp_df=pd.read_csv(tp_csv)\nfp_df=pd.read_csv(fp_csv)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T00:03:22.421431Z","iopub.execute_input":"2025-08-05T00:03:22.421669Z","iopub.status.idle":"2025-08-05T00:03:22.477829Z","shell.execute_reply.started":"2025-08-05T00:03:22.421647Z","shell.execute_reply":"2025-08-05T00:03:22.477002Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Augmentation","metadata":{}},{"cell_type":"code","source":"data_augmentation=keras.Sequential([\n    layers.RandomTranslation(\n        height_factor=0,\n        width_factor=0.1,\n        fill_mode='nearest'\n    ),\n    layers.RandomContrast(0.2),\n    layers.RandomZoom(\n        height_factor=0,\n        width_factor=0.1\n    )\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T00:03:22.484960Z","iopub.execute_input":"2025-08-05T00:03:22.486489Z","iopub.status.idle":"2025-08-05T00:03:24.442507Z","shell.execute_reply.started":"2025-08-05T00:03:22.486470Z","shell.execute_reply":"2025-08-05T00:03:24.441829Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Random audio\n### Lejátszás","metadata":{}},{"cell_type":"code","source":"# flac_files=[f for f in os.listdir(train_path) if f.endswith('.flac')]\n# random_file=random.choice(flac_files)\n# random_file_path=os.path.join(train_path, random_file)\n# print(\"flac_files hossza: \", len(flac_files))\n# print('\\nHang: ', random_file)\n# recording_id=random_file.replace('.flac', '')\n# print(\"Id:\", recording_id)\n# record=tp_df[tp_df['recording_id']==recording_id]\n# y, sr=sf.read(random_file_path) # sr - sampling rate of y\n\n# print(f\"Sample rate: {sr}\")\n# if len(record)==0:\n#     print(\"False positive.\")\n# else:\n#     for _, row in record.iterrows():\n#         print(f\"Faj: {row['species_id']}\")\n#         print(f\"Típus: {row['songtype_id']}\")\n#         print(f\"Időtartam: {row['t_min']} - {row['t_max']}\")\n#         print(f\"Frekvencia: {row['f_min']} - {row['f_max']}\\n\")\n    \n# ipd.Audio(y, rate=sr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T00:03:24.443216Z","iopub.execute_input":"2025-08-05T00:03:24.443447Z","iopub.status.idle":"2025-08-05T00:03:24.447073Z","shell.execute_reply.started":"2025-08-05T00:03:24.443423Z","shell.execute_reply":"2025-08-05T00:03:24.446408Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Spectrogram","metadata":{}},{"cell_type":"code","source":"# S=librosa.feature.melspectrogram(y=y, sr=sr)\n# fig, ax=plt.subplots()\n# S_db=librosa.power_to_db(S, ref=np.max)\n# image=librosa.display.specshow(S_db, sr=sr, x_axis='time', y_axis='mel', ax=ax)\n# fig.colorbar(image, ax=ax)\n# ax.set(title='Spectrogram')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T00:03:24.447731Z","iopub.execute_input":"2025-08-05T00:03:24.448010Z","iopub.status.idle":"2025-08-05T00:03:24.465337Z","shell.execute_reply.started":"2025-08-05T00:03:24.447987Z","shell.execute_reply":"2025-08-05T00:03:24.464846Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Spectrogram - training","metadata":{}},{"cell_type":"code","source":"def spectrogram_gen(\n    file_path,\n    save,\n    recording_id,\n    species_id,\n    time_min=None,\n    time_max=None,\n    sr=48000,\n    length=10,\n    image_height=128,\n    image_width=400,\n):\n    slice_length=length*sr\n    audio, _=librosa.load(file_path, sr=sr)\n\n    center=(time_min+time_max)/2*sr\n    start=max(center-slice_length//2, 0)\n    end=start+slice_length\n    if end>len(audio):\n        end=len(audio)\n        start=end-slice_length\n    sliced_audio=audio[int(start):int(end)]\n\n    S=librosa.feature.melspectrogram(y=sliced_audio, sr=sr)\n    S_db=librosa.power_to_db(S, ref=np.max)\n    S_norm=(S_db-S_db.min())/(S_db.max()-S_db.min())\n    S_norm=(S_norm*255).astype(np.uint8)\n    S_image=Image.fromarray(S_norm)\n    S_image=S_image.resize((image_width, image_height))\n\n    species_path=os.path.join(save, species_id)\n    os.makedirs(species_path, exist_ok=True)\n\n    filename=f'{species_id}_{recording_id}_{center}.png' # {center} kell, hátha ugyanolyan nevű file keletkezne\n    save_path=os.path.join(species_path, filename)\n    S_image.save(save_path)\n    \n    return save_path # későbbi visszanézésre","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T00:03:24.465996Z","iopub.execute_input":"2025-08-05T00:03:24.466172Z","iopub.status.idle":"2025-08-05T00:03:24.483772Z","shell.execute_reply.started":"2025-08-05T00:03:24.466158Z","shell.execute_reply":"2025-08-05T00:03:24.483266Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## True positive","metadata":{}},{"cell_type":"code","source":"#tp_df=pd.read_csv(tp_csv)\nufiles=tp_df['recording_id'].nunique()\nprint(f\"True Positive - fájlok száma (egyedi): {ufiles}\")\nprint(f\"True Positive - fájlok száma (összes): {len(tp_df)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T00:03:24.484484Z","iopub.execute_input":"2025-08-05T00:03:24.485106Z","iopub.status.idle":"2025-08-05T00:03:24.519323Z","shell.execute_reply.started":"2025-08-05T00:03:24.485081Z","shell.execute_reply":"2025-08-05T00:03:24.518697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with open(tp_csv) as f:\n    reader=csv.reader(f)\n    next(reader) # fejlécet átugorjuk\n    for i, row in enumerate(reader):\n        recording_id=row[0]\n        species_id=row[1]\n        time_min=float(row[3])\n        time_max=float(row[5])\n        file_path=os.path.join(train_path, recording_id + '.flac')\n        audio, _=librosa.load(file_path, sr=sr)\n\n        spectrogram_gen(\n            file_path=file_path,\n            save=save,\n            recording_id=recording_id,\n            species_id=species_id,\n            time_min=time_min,\n            time_max=time_max,\n            sr=sr,\n            length=length,\n            image_height=image_height,\n            image_width=image_width,\n        )\n\n        if i%100==0:\n            print(f'{i} file feldolgozva.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T00:04:23.874857Z","iopub.execute_input":"2025-08-05T00:04:23.875352Z","iopub.status.idle":"2025-08-05T00:08:13.728712Z","shell.execute_reply.started":"2025-08-05T00:04:23.875329Z","shell.execute_reply":"2025-08-05T00:08:13.727954Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"species=len([f for f in os.listdir(save) if os.path.isdir(os.path.join(save, f))])\nprint(f\"Fajok száma: {species}\")\n\nsum_files=0\nprint(\"Fájlok száma az egyes species mappákban:\")\nfor f in os.listdir(save):\n    path=os.path.join(save, f)\n    if os.path.isdir(path):\n        files=len([name for name in os.listdir(path) if os.path.isfile(os.path.join(path, name))])\n        sum_files+=files\n        print(f\"{f}:\\t{files}\")\nprint(f\"Összes file: {sum_files}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T00:09:15.754411Z","iopub.execute_input":"2025-08-05T00:09:15.755004Z","iopub.status.idle":"2025-08-05T00:09:15.769167Z","shell.execute_reply.started":"2025-08-05T00:09:15.754979Z","shell.execute_reply":"2025-08-05T00:09:15.768577Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Mappa törlése","metadata":{}},{"cell_type":"code","source":"# # Mappa törlése\n# if os.path.exists(save):\n#     shutil.rmtree(save)\n#     print(\"Mappa törölve.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T00:09:15.770538Z","iopub.execute_input":"2025-08-05T00:09:15.770782Z","iopub.status.idle":"2025-08-05T00:09:15.794107Z","shell.execute_reply.started":"2025-08-05T00:09:15.770767Z","shell.execute_reply":"2025-08-05T00:09:15.793674Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## False positive","metadata":{}},{"cell_type":"code","source":"#fp_df=pd.read_csv(fp_csv)\n# ufiles=fp_df['recording_id'].nunique()\n# print(f\"False Positive - fájlok száma (egyedi): {ufiles}\")\n# print(f\"False Positive - fájlok száma (összes): {len(fp_df)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T00:09:15.794917Z","iopub.execute_input":"2025-08-05T00:09:15.795154Z","iopub.status.idle":"2025-08-05T00:09:15.818370Z","shell.execute_reply.started":"2025-08-05T00:09:15.795132Z","shell.execute_reply":"2025-08-05T00:09:15.817849Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# with open(fp_csv) as f:\n#     reader=csv.reader(f)\n#     next(reader) # fejlécet átugorjuk\n#     for i, row in enumerate(reader):\n#         recording_id=row[0]\n#         species_id=row[1]\n#         time_min=float(row[3])\n#         time_max=float(row[5])\n#         file_path=os.path.join(train_path, recording_id + '.flac')\n#         audio, _=librosa.load(file_path, sr=sr)\n\n#         spectrogram_gen(\n#             file_path=file_path,\n#             save=save,\n#             recording_id=recording_id,\n#             species_id=species_id,\n#             time_min=time_min,\n#             time_max=time_max,\n#             sr=sr,\n#             length=length,\n#             image_height=image_height,\n#             image_width=image_width\n#         )\n\n#         if i%100==0:\n#             print(f'{i} file feldolgozva.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T00:09:15.818968Z","iopub.execute_input":"2025-08-05T00:09:15.819125Z","iopub.status.idle":"2025-08-05T00:09:15.835385Z","shell.execute_reply.started":"2025-08-05T00:09:15.819113Z","shell.execute_reply":"2025-08-05T00:09:15.834907Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# species=len([f for f in os.listdir(save) if os.path.isdir(os.path.join(save, f))])\n# print(f\"Fajok száma: {species}\")\n\n# print(\"Fájlok száma az egyes species mappákban:\")\n# for f in os.listdir(save):\n#     path=os.path.join(save, f)\n#     if os.path.isdir(path):\n#         files=len([name for name in os.listdir(path) if os.path.isfile(os.path.join(path, name))])\n#         print(f\"{f}:\\t{files}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T00:09:15.836751Z","iopub.execute_input":"2025-08-05T00:09:15.837299Z","iopub.status.idle":"2025-08-05T00:09:15.852177Z","shell.execute_reply.started":"2025-08-05T00:09:15.837283Z","shell.execute_reply":"2025-08-05T00:09:15.851631Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Dataset gen","metadata":{}},{"cell_type":"code","source":"train_ds=tf.keras.utils.image_dataset_from_directory(\n    directory=save,\n    labels=\"inferred\",\n    label_mode=\"categorical\",\n    color_mode='grayscale',\n    batch_size=batch_size,\n    image_size=(image_height, image_width),\n    seed=seed,\n    validation_split=0.1,\n    subset=\"training\"\n)\n\nval_ds=tf.keras.utils.image_dataset_from_directory(\n    directory=save,\n    labels=\"inferred\",\n    label_mode=\"categorical\",\n    color_mode='grayscale',\n    batch_size=batch_size,\n    image_size=(image_height, image_width),\n    seed=seed,\n    validation_split=0.1,\n    subset=\"validation\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T00:09:15.852860Z","iopub.execute_input":"2025-08-05T00:09:15.853208Z","iopub.status.idle":"2025-08-05T00:09:18.600924Z","shell.execute_reply.started":"2025-08-05T00:09:15.853182Z","shell.execute_reply":"2025-08-05T00:09:18.600362Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class_labels=[]\nfor _, labels in train_ds:\n    labels=tf.argmax(labels, axis=1)\n    class_labels.extend(labels.numpy().tolist())\nclass_labels=np.array(class_labels)\n\nclass_weights=compute_class_weight(\n    class_weight='balanced',\n    classes=np.arange(species),\n    y=class_labels\n)\n\nprint(class_weights)\n\nclass_weights_dict=dict(enumerate(class_weights))\nprint(f\"Class weights: {class_weights_dict}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T00:09:18.601578Z","iopub.execute_input":"2025-08-05T00:09:18.601747Z","iopub.status.idle":"2025-08-05T00:09:19.209294Z","shell.execute_reply.started":"2025-08-05T00:09:18.601733Z","shell.execute_reply":"2025-08-05T00:09:19.208661Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Képek","metadata":{}},{"cell_type":"code","source":"# for images, _ in train_ds.take(1):\n#     plt.figure(figsize=(15, 10))\n#     for i in range(len(images)):\n#         plt.subplot(4, 4, i+1)\n#         plt.imshow(images[i])\n#         plt.axis('off')\n#     plt.suptitle(\"Első batch képei\", fontsize=15)\n#     plt.tight_layout()\n#     plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T00:09:19.210125Z","iopub.execute_input":"2025-08-05T00:09:19.210358Z","iopub.status.idle":"2025-08-05T00:09:19.213791Z","shell.execute_reply.started":"2025-08-05T00:09:19.210332Z","shell.execute_reply":"2025-08-05T00:09:19.213178Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model","metadata":{}},{"cell_type":"code","source":"model=Sequential([\n    Input(shape=(image_height, image_width, 1)),\n    #data_augmentation,\n    layers.Rescaling(1./255),\n    Conv2D(16, (3, 3), activation='relu'),\n    #MaxPooling2D(pool_size=(2, 2)),\n    Conv2D(32, (3, 3), activation='relu'),\n    #MaxPooling2D(pool_size=(2, 2)),\n    #Conv2D(64, (3, 3), activation='relu'),\n\n    Flatten(),\n    #Dense(128, activation='relu'),\n    #Dropout(0.25, seed=seed),\n    Dense(128, activation='relu'),\n    #Dropout(0.25, seed=seed),\n    Dense(species, activation='sigmoid')\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T01:09:53.625299Z","iopub.execute_input":"2025-08-05T01:09:53.625528Z","iopub.status.idle":"2025-08-05T01:09:53.668867Z","shell.execute_reply.started":"2025-08-05T01:09:53.625512Z","shell.execute_reply":"2025-08-05T01:09:53.668359Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"optimizer=Adam(\n    #learning_rate=0.01\n)\n\nreduce_lr=ReduceLROnPlateau(\n    factor=0.5,\n    patience=5,\n    verbose=1\n)\n\nearly_stop=EarlyStopping(\n    patience=5,\n    verbose=1,\n    restore_best_weights=True\n)\n\nmodel.compile(\n    optimizer=optimizer,\n    loss=keras.losses.BinaryCrossentropy(),\n    metrics=['accuracy']\n)\n\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T01:10:05.234871Z","iopub.execute_input":"2025-08-05T01:10:05.235414Z","iopub.status.idle":"2025-08-05T01:10:05.256438Z","shell.execute_reply.started":"2025-08-05T01:10:05.235394Z","shell.execute_reply":"2025-08-05T01:10:05.255742Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history=model.fit(\n    train_ds,\n    epochs=epochs,\n    validation_data=val_ds,\n    class_weight=class_weights_dict,\n    callbacks=[early_stop, reduce_lr]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T01:10:11.755841Z","iopub.execute_input":"2025-08-05T01:10:11.756614Z","iopub.status.idle":"2025-08-05T01:10:35.580264Z","shell.execute_reply.started":"2025-08-05T01:10:11.756586Z","shell.execute_reply":"2025-08-05T01:10:35.579676Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plt.figure(figsize=(10, 5))\n# plt.subplot(1,2,1)\n# plt.plot(history.history['accuracy'], label='Training accuracy')\n# plt.plot(history.history['val_accuracy'], label='Validation accuracy')\n# plt.title('Accuracy')\n# plt.xlabel('Epoch')\n# plt.ylabel('Accuracy')\n# plt.legend(loc='lower center')\n\n# plt.subplot(1,2,2)\n# plt.plot(history.history['loss'], label='Train loss')\n# plt.plot(history.history['val_loss'], label='Validation loss')\n# plt.title('Loss')\n# plt.xlabel('Epoch')\n# plt.ylabel('Loss')\n# plt.legend()\n\n# plt.tight_layout()\n# plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T01:12:30.892875Z","iopub.execute_input":"2025-08-05T01:12:30.893152Z","iopub.status.idle":"2025-08-05T01:12:31.227467Z","shell.execute_reply.started":"2025-08-05T01:12:30.893133Z","shell.execute_reply":"2025-08-05T01:12:31.226717Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Prediction","metadata":{}},{"cell_type":"markdown","source":"### Sliced","metadata":{}},{"cell_type":"code","source":"def gen_test_slice(\n    file_path,\n    sr=48000,\n    length=10,\n    image_height=128,\n    image_width=400):\n    \n    spectrograms = []\n    audio, _ = librosa.load(file_path, sr=sr)\n    slice_length = sr * length\n    n = len(audio) // slice_length\n\n    for i in range(n):\n        start = i * slice_length\n        end = start + slice_length\n        if end > len(audio):\n            end = len(audio)\n        sliced_audio = audio[start:end]\n\n        S = librosa.feature.melspectrogram(y=sliced_audio, sr=sr)\n        S_db = librosa.power_to_db(S, ref=np.max)\n        S_norm = (S_db - S_db.min()) / (S_db.max() - S_db.min())\n        S_norm = (S_norm * 255).astype(np.uint8)\n        image = Image.fromarray(S_norm).resize((image_width, image_height))\n        array = np.array(image) / 255.0\n        spectrograms.append(array)\n\n    return spectrograms","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T01:12:51.572637Z","iopub.execute_input":"2025-08-05T01:12:51.572921Z","iopub.status.idle":"2025-08-05T01:12:51.578587Z","shell.execute_reply.started":"2025-08-05T01:12:51.572900Z","shell.execute_reply":"2025-08-05T01:12:51.577916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict_test_slice(\n    model,\n    spectrograms,\n    threshold=0.5\n):\n    inputs = []\n    for s in spectrograms:\n        tensor = np.expand_dims(s, axis=(0, -1))  # 1, height, width, 1\n        inputs.append(tensor)\n    inputs = np.concatenate(inputs, axis=0)\n\n    outputs = model.predict(inputs, verbose=0)\n    pred = np.max(outputs, axis=0)\n    binary_pred = (pred > threshold).astype(int)\n    return pred, binary_pred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T01:13:00.050396Z","iopub.execute_input":"2025-08-05T01:13:00.051212Z","iopub.status.idle":"2025-08-05T01:13:00.055914Z","shell.execute_reply.started":"2025-08-05T01:13:00.051186Z","shell.execute_reply":"2025-08-05T01:13:00.055146Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_csv_slice(model, test_path, csv_file=None):\n    rows = []\n\n    for i, file in enumerate(sorted(os.listdir(test_path))):\n        if file.endswith('.flac'):\n            file_path = os.path.join(test_path, file)\n            recording_id = file.replace('.flac', '')\n            spectrograms = gen_test_slice(file_path)\n            pred, _ = predict_test_slice(model, spectrograms)\n            rows.append([recording_id] + list(pred))\n        if i % 100 == 0:\n            print(f\"{i} file feldolgozva.\")\n\n    df = pd.DataFrame(rows, columns=['recording_id'] + [f\"s{i}\" for i in range(24)])\n    if csv_file:\n        df.to_csv(csv_file, index=False)\n    else:\n        print(df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T01:13:03.618062Z","iopub.execute_input":"2025-08-05T01:13:03.618325Z","iopub.status.idle":"2025-08-05T01:13:03.623827Z","shell.execute_reply.started":"2025-08-05T01:13:03.618306Z","shell.execute_reply":"2025-08-05T01:13:03.623135Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Not sliced","metadata":{}},{"cell_type":"code","source":"# def spectrogram_gen_test(\n#     file_path,\n#     image_size=(image_height, image_width),\n#     normalize=True,\n#     sr=48000\n# ):\n#     y, _=librosa.load(file_path, sr=sr)\n#     S=librosa.feature.melspectrogram(y=y, sr=sr, n_mels=image_size[0])\n#     S=librosa.power_to_db(S, ref=np.max)\n    \n#     if normalize:\n#         S=(S-S.min())/(S.max()-S.min())\n        \n#     S=tf.expand_dims(S, axis=-1)   \n#     S=tf.image.resize(S, (image_height, image_width))\n#     S=tf.expand_dims(S, axis=0) # batch\n#     #print(S.shape)\n#     return S","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T01:02:01.825253Z","iopub.status.idle":"2025-08-05T01:02:01.825536Z","shell.execute_reply.started":"2025-08-05T01:02:01.825374Z","shell.execute_reply":"2025-08-05T01:02:01.825386Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def pred_image(\n#     file_path,\n#     image_size=(image_height, image_width),\n#     normalize=True,\n#     sr=48000,\n#     threshold=0.5\n# ):\n#     input=spectrogram_gen_test(file_path, image_size, normalize, sr)\n#     pred=model.predict(input, verbose=False)[0]\n#     binary_pred=(pred>threshold).astype(int)\n    \n#     # print(\"Prediction:\")\n#     # print(pred)\n#     # print(\"Binary prediction:\")\n#     # print(binary_pred)\n#     # print(f\"Pred:\\t{pred.shape}\")\n#     # print(f\"Binary_pred:\\t{binary_pred.shape}\")\n#     # for i, prob in enumerate(pred):\n#         # print(f\"{i}\\t{prob}\")\n\n#     # for i, prob in enumerate(pred):\n#     #     print(f\"{i}.\\t{prob}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T01:02:01.826487Z","iopub.status.idle":"2025-08-05T01:02:01.826738Z","shell.execute_reply.started":"2025-08-05T01:02:01.826633Z","shell.execute_reply":"2025-08-05T01:02:01.826646Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def create_csv(model, test_path, csv_file=None, \n#                                image_size=(image_height, image_width),\n#                                normalize=True, sr=48000, threshold=0.5):\n#     rows = []\n\n#     for i, file in enumerate(sorted(os.listdir(test_path))):\n#         if file.endswith('.flac'):\n#             file_path = os.path.join(test_path, file)\n#             recording_id = file.replace('.flac', '')\n#             input_tensor = spectrogram_gen_test(file_path, image_size, normalize, sr)\n#             pred = model.predict(input_tensor, verbose=0)[0]\n#             binary_pred = (pred > threshold).astype(int)\n#             rows.append([recording_id] + list(pred))\n\n#         if i % 100 == 0:\n#             print(f\"{i} fájl feldolgozva.\")\n\n#     df = pd.DataFrame(rows, columns=['recording_id'] + [f's{i}' for i in range(24)])\n#     if csv_file:\n#         df.to_csv(csv_file, index=False)\n#         print(f\"Mentve: {csv_file}\")\n#     else:\n#         print(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T01:02:01.829985Z","iopub.status.idle":"2025-08-05T01:02:01.830286Z","shell.execute_reply.started":"2025-08-05T01:02:01.830128Z","shell.execute_reply":"2025-08-05T01:02:01.830141Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Submmission file","metadata":{}},{"cell_type":"code","source":"# file_path='/kaggle/input/rfcx-species-audio-detection/train/c12e0a62b.flac'\n# gen_test_slice(file_path)\n# pred_image(file_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T01:02:01.828914Z","iopub.status.idle":"2025-08-05T01:02:01.829227Z","shell.execute_reply.started":"2025-08-05T01:02:01.829062Z","shell.execute_reply":"2025-08-05T01:02:01.829076Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_dir='/kaggle/working/csv'\nos.makedirs(submission_dir, exist_ok=True)\ncsv_file=os.path.join(submission_dir, 'submission.csv')\ncreate_csv_slice(model, test_path, csv_file=csv_file)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T01:13:07.258463Z","iopub.execute_input":"2025-08-05T01:13:07.258729Z","iopub.status.idle":"2025-08-05T01:23:47.975931Z","shell.execute_reply.started":"2025-08-05T01:13:07.258708Z","shell.execute_reply":"2025-08-05T01:23:47.975365Z"}},"outputs":[],"execution_count":null}]}