{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":21669,"databundleVersionId":1692278,"sourceType":"competition"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# General Settings","metadata":{}},{"cell_type":"markdown","source":"## Library Imports","metadata":{}},{"cell_type":"code","source":"#Data handling\nimport os\nimport pandas as pd\nfrom PIL import Image\n\n# Audio handling\n!pip install PySoundFile\nimport librosa\nfrom IPython.display import Audio\n\n# Visualization\nimport matplotlib.pyplot as plt\n\n# Feedback with progress bar\nfrom tqdm.notebook import tqdm\n\n# Math & Algorithms\nimport numpy as np","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-18T07:44:30.551695Z","iopub.execute_input":"2025-08-18T07:44:30.551993Z","iopub.status.idle":"2025-08-18T07:44:36.274351Z","shell.execute_reply.started":"2025-08-18T07:44:30.551967Z","shell.execute_reply":"2025-08-18T07:44:36.273668Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Settings","metadata":{}},{"cell_type":"code","source":"# Initialise random number generation\nrandom_seed = 42\nrng = np.random.default_rng()\n\n# NN training parameters\nslice_length = 3\nsr = None # Use recording native sr\nbatch_size = 16\nlr = 1e-3\nepochs = 30\npatience = 10\n\n# Folder for storing generated spectrograms\nspect_save_path='/kaggle/working/spectrograms'\nos.makedirs(spect_save_path, exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T07:44:36.275745Z","iopub.execute_input":"2025-08-18T07:44:36.276172Z","iopub.status.idle":"2025-08-18T07:44:36.281482Z","shell.execute_reply.started":"2025-08-18T07:44:36.276150Z","shell.execute_reply":"2025-08-18T07:44:36.280782Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### File path variables","metadata":{}},{"cell_type":"code","source":"# Root data path for RainForest Species\ninput_path='/kaggle/input/rfcx-species-audio-detection'\n\n# Train and Test audio recordings data\ntrain_path=os.path.join(input_path, 'train')\ntest_path=os.path.join(input_path, 'test')\n\n# Labels\ntp_label_csv_path=os.path.join(input_path, 'train_tp.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T07:44:36.282230Z","iopub.execute_input":"2025-08-18T07:44:36.282417Z","iopub.status.idle":"2025-08-18T07:44:36.300290Z","shell.execute_reply.started":"2025-08-18T07:44:36.282403Z","shell.execute_reply":"2025-08-18T07:44:36.299816Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training data preparation","metadata":{}},{"cell_type":"markdown","source":"# Data Import","metadata":{}},{"cell_type":"markdown","source":"### Import labels","metadata":{}},{"cell_type":"code","source":"df_labels=pd.read_csv(tp_label_csv_path)\ndf_labels['t_length'] = df_labels['t_max'] - df_labels['t_min']\nlengths = df_labels['t_length']\nprint(f\"Length of labeled audio segments: {lengths.mean():.2f}±{lengths.std():.2f} ({lengths.min():.2f}-{lengths.max():.2f}) [s]\")\ndf_labels","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T07:44:36.302176Z","iopub.execute_input":"2025-08-18T07:44:36.302466Z","iopub.status.idle":"2025-08-18T07:44:36.376583Z","shell.execute_reply.started":"2025-08-18T07:44:36.302426Z","shell.execute_reply":"2025-08-18T07:44:36.376024Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Exploration","metadata":{}},{"cell_type":"markdown","source":"### Play a selected or random audio recording","metadata":{}},{"cell_type":"code","source":"recording_id = None\n#recording_id = '5b5218aba'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T07:44:36.377312Z","iopub.execute_input":"2025-08-18T07:44:36.377609Z","iopub.status.idle":"2025-08-18T07:44:36.380947Z","shell.execute_reply.started":"2025-08-18T07:44:36.377584Z","shell.execute_reply":"2025-08-18T07:44:36.380357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if recording_id == None:\n    # Find all .flac files\n    flac_files=[f for f in os.listdir(train_path) if f.endswith('.flac')]\n    print(f'No. of .flac samples: {len(flac_files)}')\n\n    # Select and access a random .flac from the list\n    file=rng.choice(flac_files)\n    file_path=os.path.join(train_path, file)\n    recording_id=file.replace('.flac', '')\nelse:\n    file = recording_id + '.flac'\n    file_path=os.path.join(train_path, file)\n\nprint(f'Selected sample: {file}')\n\n# Read labels and audio data\nrecord = df_labels.loc[df_labels['recording_id'] == recording_id]\naudio, rec_sr = librosa.core.load(file_path, sr = sr, mono=False)\nprint(f\"Sample rate: {rec_sr}\")\n\n# Print label info\nprint(f\"\\nLabels:\")\nprint(f\"-------------\")\nif len(record)==0:\n    print(\"No associated label\")\nelse:\n    for _, row in record.iterrows():\n        print(f\"Faj: {row['species_id']}\")\n        print(f\"Típus: {row['songtype_id']}\")\n        print(f\"Időtartam: {row['t_min']} - {row['t_max']}\")\n        print(f\"Frekvencia: {row['f_min']} - {row['f_max']}\\n\")\n\n# Display player\nAudio(audio, rate=rec_sr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T07:44:36.381763Z","iopub.execute_input":"2025-08-18T07:44:36.381938Z","iopub.status.idle":"2025-08-18T07:44:50.028490Z","shell.execute_reply.started":"2025-08-18T07:44:36.381924Z","shell.execute_reply":"2025-08-18T07:44:50.027108Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Show the corresponding mel spectrogram","metadata":{}},{"cell_type":"code","source":"# Generate the Spectrogram\nS = librosa.feature.melspectrogram(y=audio, sr=rec_sr)\nS_db = librosa.power_to_db(S, ref=np.max)\n\n#Display\nfig, ax = plt.subplots(figsize=(16, 4))\nimage = librosa.display.specshow(S_db, sr=rec_sr, x_axis='time', y_axis='mel', ax=ax)\nfig.colorbar(image, ax=ax)\nax.set(title='Mel-Spectrogram')\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T07:44:50.029319Z","iopub.execute_input":"2025-08-18T07:44:50.029677Z","iopub.status.idle":"2025-08-18T07:44:52.795147Z","shell.execute_reply.started":"2025-08-18T07:44:50.029657Z","shell.execute_reply":"2025-08-18T07:44:52.794384Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Test mel spectrogram generation, saving and loading","metadata":{}},{"cell_type":"code","source":"# Generate spectrogram\nS = librosa.feature.melspectrogram(y=audio[0:slice_length*rec_sr], sr=rec_sr)\nS_db = librosa.power_to_db(S, ref=np.max)\n\n# Convert to image apropriate format\nS_norm = (S_db-S_db.min())/(S_db.max()-S_db.min())\nS_norm = (S_norm*255).astype(np.uint8)\n\n# Convert to PIL image\nimg = Image.fromarray(S_norm)\n\n# Save the image with PIL\nimg = Image.fromarray(S_norm)\nimg.save('/kaggle/working/test.png')\n\n# Load the image\nimg_loaded = Image.open('/kaggle/working/test.png')\nS_norm_loaded = np.array(img_loaded)\n\n# Assert equality of saved and loaded array\ntry:\n    np.testing.assert_array_equal(S_norm, S_norm_loaded)\nexcept AssertionError:\n    print(f\"The saved and loaded arrays are NOT identical:\")\nelse:\n    print(f\"The saved and loaded arrays are IDENTICAL:\")\nfinally:\n    print(f\"\\tSaved array: {S_norm.shape}; saved values: {S_norm.min()}-{S_norm.max()}; format: {type(S_norm)}\")\n    display(img)\n    print(f\"\\tLoaded array: {S_norm_loaded.shape}; saved values: {S_norm_loaded.min()}-{S_norm_loaded.max()}; format: {type(S_norm_loaded)}\")\n    display(img_loaded)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data generation","metadata":{}},{"cell_type":"markdown","source":"### Generate labeled spectrograms","metadata":{}},{"cell_type":"code","source":"def spectrogram_gen(audio_folder_path, recording_id, species_id,\n                    time_min, time_max, sr, slice_length,\n                    spect_save_path):\n    \"\"\"Generates spectrograms of from a given audio file\n\n    Generates spectrograms of 'slice_length' between 'time_min' and 'time_max'\n    of the given recording.\n\n    Parameters\n    ----------\n    audio_folder_path : path str\n        The folder in which the audio files are stored\n    recording_id : str\n        The identifier of the audio recording\n    species_id : int\n        The identifier of the species in the record\n    time_min: float\n        Start time of the vocalization of the species within the recording\n    time_max: float\n        End time of the vocalization of the species within the recording\n    sr : int\n        Forced sample_rate for laoding the audio (None uses the recording native sr)\n    slice_length: int\n        Length of recording slice to turn into spectrograms in seconds\n    spect_save_path: path str\n        Folder path to save generated spectrograms into\n\n    Returns\n    -------\n    save_path: path str\n        Location at which the generated spectrogram is saved at\n    \"\"\"\n\n    # Load the audio\n    file_path = os.path.join(audio_folder_path, recording_id + '.flac')\n    audio, rec_sr = librosa.core.load(file_path, sr=sr, mono=True)\n\n    # Generate audio slice(s)\n    slice_time = time_max - time_min\n    noSlices = max(int(np.round(slice_time/slice_length)), 1) # How many slices can fit into the given intervall, rounded to nearest int\n    \n    for i in range(noSlices): \n        # Find center time of the given slice\n        center = (time_min + i * (slice_time / (noSlices+1)))\n\n        # Find start and end sample of the given slice\n        start = int(max(center - slice_length/2, 0) * rec_sr)\n        end = start + int(slice_length * rec_sr)\n        if end > len(audio):\n            end = len(audio)\n            start = end - int(slice_length * rec_sr)\n\n        # Get the sliced audio\n        sliced_audio=audio[start:end]\n\n        # Generate Spectrogram\n        S = librosa.feature.melspectrogram(y = sliced_audio, sr=rec_sr)\n        S_db=librosa.power_to_db(S, ref=np.max)\n        \n        S_norm=(S_db-S_db.min())/(S_db.max()-S_db.min())\n        S_norm = (S_norm*255).astype(np.uint8)\n        spect_size = S_norm.shape\n\n        # Save the array as an image\n        species_path=os.path.join(spect_save_path, str(species_id))\n        os.makedirs(species_path, exist_ok=True)\n\n        filename = f'{species_id}_{recording_id}_{center:.2f}.png' # {center} kell, hátha ugyanolyan nevű file keletkezne\n        save_path = os.path.join(species_path, filename)\n\n        S_image = Image.fromarray(S_norm)\n        S_image.save(save_path)\n    \n    return spect_size, save_path # későbbi visszanézésre\n\n# Spectrogram generation progress (with TQDM progress bar)\ninput_size = None\nfor i in tqdm(range(len(df_labels))):\n    row = df_labels.iloc[i]\n    \n    recording_id=row['recording_id']\n    species_id=row['species_id']\n    time_min=float(row['t_min'])\n    time_max=float(row['t_max'])\n\n    # Generate spectrogram\n    spect_size, save_path = spectrogram_gen(audio_folder_path = train_path, recording_id = recording_id, species_id = species_id,\n                                            time_min = time_min, time_max = time_max, sr = sr, slice_length = slice_length, \n                                            spect_save_path = spect_save_path)\n    if input_size == None:\n        input_size = spect_size\n    else:\n        if (input_size != spect_size):\n            print(f\"WARNING: spectrogram size for label {i} ({spect_size}) does not match the spectrogram size for the first label ({input_size})\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T07:44:52.795977Z","iopub.execute_input":"2025-08-18T07:44:52.796169Z","iopub.status.idle":"2025-08-18T07:46:46.400759Z","shell.execute_reply.started":"2025-08-18T07:44:52.796153Z","shell.execute_reply":"2025-08-18T07:46:46.399784Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Summarise generated training data","metadata":{}},{"cell_type":"code","source":"species = [int(f) for f in os.listdir(spect_save_path) if os.path.isdir(os.path.join(spect_save_path, f))]\nspecies.sort()\nnumSpecies = len(species)\nprint(f\"Fajok száma: {numSpecies}\")\n\nsum_files=0\nprint(\"Fájlok száma az egyes species mappákban:\")\nfor f in species:\n    path = os.path.join(spect_save_path, str(f))\n    numFiles = len([name for name in os.listdir(path) if os.path.isfile(os.path.join(path, name))])\n    sum_files += numFiles\n    print(f\"{f}:\\t{numFiles}\")\nprint(f\"Összes spectrogram: {sum_files}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T07:46:46.401607Z","iopub.execute_input":"2025-08-18T07:46:46.401885Z","iopub.status.idle":"2025-08-18T07:46:46.417009Z","shell.execute_reply.started":"2025-08-18T07:46:46.401862Z","shell.execute_reply":"2025-08-18T07:46:46.416328Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Keras Implementation","metadata":{}},{"cell_type":"markdown","source":"## Library imports","metadata":{}},{"cell_type":"code","source":"# Imports\nimport keras\nfrom keras import layers\nfrom keras.metrics import BinaryAccuracy\nimport tensorflow as tf\nimport random\nfrom glob import glob\nfrom sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T07:46:46.419417Z","iopub.execute_input":"2025-08-18T07:46:46.419693Z","iopub.status.idle":"2025-08-18T07:47:04.144867Z","shell.execute_reply.started":"2025-08-18T07:46:46.419669Z","shell.execute_reply":"2025-08-18T07:47:04.144222Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Seeds and warnings","metadata":{}},{"cell_type":"code","source":"# Seeds\nos.environ['PYTHONHASHSEED'] = str(random_seed)\nnp.random.seed(random_seed)\ntf.random.set_seed(random_seed)\nrandom.seed(random_seed)\n\n# Hiding warnings\nos.environ['TF_DETERMINISTIC_OPS'] = '1'\nos.environ['TF_CUDNN_DETERMINISTIC'] = '1'\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '2'  # 2 -> warnings","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T07:47:04.145517Z","iopub.execute_input":"2025-08-18T07:47:04.145948Z","iopub.status.idle":"2025-08-18T07:47:04.150468Z","shell.execute_reply.started":"2025-08-18T07:47:04.145931Z","shell.execute_reply":"2025-08-18T07:47:04.149703Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_files = glob(\"/kaggle/working/spectrograms/*/*.png\")\nprint(f\"Total files: {len(all_files)}\")\ntrain_files, val_files = train_test_split(all_files, test_size=0.1, random_state=random_seed)\n\n# Input size\nimage_size = np.array(Image.open(all_files[0]))\ninput_shape = (image_size.shape[0], image_size.shape[1], 1) # adding grayscale dimension\nprint(f\"Image size: {image_size}\\nInput shape: {input_shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T07:47:04.151184Z","iopub.execute_input":"2025-08-18T07:47:04.151363Z","iopub.status.idle":"2025-08-18T07:47:04.185688Z","shell.execute_reply.started":"2025-08-18T07:47:04.151349Z","shell.execute_reply":"2025-08-18T07:47:04.184922Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Augmentation","metadata":{}},{"cell_type":"code","source":"# Data augmentation\ndata_augmentation=keras.Sequential([\n    layers.RandomTranslation(\n        height_factor=0,\n        width_factor=0.1,\n        fill_mode='nearest'\n    ),\n    layers.RandomContrast(0.2),\n    layers.RandomZoom(\n        height_factor=0,\n        width_factor=0.1\n    )\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T07:47:04.186489Z","iopub.execute_input":"2025-08-18T07:47:04.186744Z","iopub.status.idle":"2025-08-18T07:47:04.678531Z","shell.execute_reply.started":"2025-08-18T07:47:04.186722Z","shell.execute_reply":"2025-08-18T07:47:04.677728Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Dataset loader class","metadata":{}},{"cell_type":"code","source":"class AudioDataset(keras.utils.Sequence):\n    \n    \n    # Initialization\n    def __init__(self, files, num_species, batch_size=16, shuffle=False, seed=None, **kwargs):\n        \"\"\" Spectrogram dataset loader - reproducible\n            \n            Arguments:\n            files: list of paths to images\n            numSpecies: number of labels\n            batch_size: number of files in a batch\n            shuffle: whether to shuffle images\n            seed: seed for reproducibility\n        \"\"\"\n        super().__init__()      \n\n        self.files=files.copy()\n        self.num_species=num_species\n        self.batch_size=batch_size\n        self.shuffle=shuffle\n        self.seed=seed\n        self.rng=np.random.RandomState(seed) if seed is not None else np.random\n        \n        # setting seeds\n        if seed is not None:\n            np.random.seed(seed)\n            tf.random.set_seed(seed)\n            random.seed(seed)\n\n        # detecting image shape\n        with Image.open(self.files[0]) as image:\n            self.input_shape=(image.size[1], image.size[0], 1)\n\n        self.end_of_epoch()\n\n\n    # Number of batches\n    def __len__(self):\n        return int(np.ceil(len(self.files)/self.batch_size))\n\n\n    # Creates a single batch\n    def __getitem__(self, index):\n        batch_files=self.files[index*self.batch_size:(index+1)*self.batch_size]\n\n        batch_images=[]\n        batch_labels=[]\n        \n        for f in batch_files:\n            with Image.open(f) as image:\n                image=image.resize(self.input_shape[:2][::-1])\n                image=np.array(image, dtype=np.float32)/255.0\n                if image.ndim==2:\n                    image=image[...,np.newaxis]\n            batch_images.append(image)\n\n            label=int(os.path.basename(os.path.dirname(f)))\n            one_hot=np.zeros(self.num_species, dtype=np.float32)\n            one_hot[label]=1.0\n            batch_labels.append(one_hot)\n\n        return np.stack(batch_images), np.stack(batch_labels)\n\n    \n    # Shuffles files\n    def end_of_epoch(self):\n        if self.shuffle:\n            if self.seed is not None:\n                permutation=self.rng.permutation(len(self.files))\n                self.files=[self.files[i] for i in permutation]\n            else:\n                np.random.shuffle(self.files)\n        \n\n    # Shape of an image\n    def image_shape(self):\n        return self.input_shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T07:47:04.679302Z","iopub.execute_input":"2025-08-18T07:47:04.679821Z","iopub.status.idle":"2025-08-18T07:47:04.691484Z","shell.execute_reply.started":"2025-08-18T07:47:04.679795Z","shell.execute_reply":"2025-08-18T07:47:04.690199Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Dataset loading\ntrain_dataset=AudioDataset(files=train_files, num_species=numSpecies, batch_size=batch_size, shuffle=True, seed=random_seed)\nval_dataset=AudioDataset(files=val_files, num_species=numSpecies, batch_size=batch_size, shuffle=False, seed=random_seed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T07:47:04.719771Z","iopub.execute_input":"2025-08-18T07:47:04.719979Z","iopub.status.idle":"2025-08-18T07:47:04.740497Z","shell.execute_reply.started":"2025-08-18T07:47:04.719964Z","shell.execute_reply":"2025-08-18T07:47:04.739618Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Define model","metadata":{}},{"cell_type":"code","source":"# Model\nmodel_keras=keras.models.Sequential([\n    \n    layers.Input(shape=train_dataset.image_shape()),\n    data_augmentation,\n    \n    layers.Conv2D(16, (3, 3), activation='relu'),\n    layers.Conv2D(32, (3, 3), activation='relu'),\n    layers.Conv2D(32, (3, 3), activation='relu'),\n\n    layers.Flatten(),\n    layers.Dense(128, activation='relu'),\n    layers.Dense(numSpecies, activation='sigmoid')\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T07:47:04.741240Z","iopub.execute_input":"2025-08-18T07:47:04.741430Z","iopub.status.idle":"2025-08-18T07:47:06.374179Z","shell.execute_reply.started":"2025-08-18T07:47:04.741408Z","shell.execute_reply":"2025-08-18T07:47:06.373622Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Training logic","metadata":{}},{"cell_type":"code","source":"# Optimizer\noptimizer=keras.optimizers.Adam(learning_rate = lr)\n\n# Callbacks\nreduce_lr = keras.callbacks.ReduceLROnPlateau(factor = 0.5, patience = patience / 2, verbose=1)\nearly_stop = keras.callbacks.EarlyStopping(patience = patience, verbose = 1, restore_best_weights = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T07:47:06.374846Z","iopub.execute_input":"2025-08-18T07:47:06.375025Z","iopub.status.idle":"2025-08-18T07:47:06.384015Z","shell.execute_reply.started":"2025-08-18T07:47:06.375011Z","shell.execute_reply":"2025-08-18T07:47:06.383492Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Compile model","metadata":{}},{"cell_type":"code","source":"# Compile model\nmodel_keras.compile(\n    optimizer=optimizer,\n    loss=\"binary_crossentropy\",\n    metrics=[BinaryAccuracy()]\n)\nmodel_keras.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T07:47:06.384788Z","iopub.execute_input":"2025-08-18T07:47:06.384989Z","iopub.status.idle":"2025-08-18T07:47:06.415832Z","shell.execute_reply.started":"2025-08-18T07:47:06.384974Z","shell.execute_reply":"2025-08-18T07:47:06.415180Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train the model","metadata":{}},{"cell_type":"code","source":"history_keras = model_keras.fit(train_dataset, validation_data=val_dataset, epochs=epochs, callbacks=[early_stop, reduce_lr])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T07:47:06.416578Z","iopub.execute_input":"2025-08-18T07:47:06.416763Z","iopub.status.idle":"2025-08-18T07:50:03.698153Z","shell.execute_reply.started":"2025-08-18T07:47:06.416749Z","shell.execute_reply":"2025-08-18T07:50:03.697581Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Evaluate","metadata":{}},{"cell_type":"markdown","source":"### Generate spectrograms","metadata":{}},{"cell_type":"code","source":"# Generate spectrograms from a given file\ndef gen_test_spectrograms(file_path, sr, length, target_shape=(input_shape[0], input_shape[1])):\n    spectrograms=[]\n    audio, rec_sr = librosa.core.load(file_path, sr=sr, mono=True)\n    slice_length = rec_sr * length\n    n = len(audio) // slice_length\n\n    for i in range(n):\n        start = i * slice_length\n        end = start + slice_length\n        if end > len(audio):\n            end = len(audio)\n        sliced_audio = audio[start:end]\n\n        S = librosa.feature.melspectrogram(y=sliced_audio, sr=rec_sr, n_mels=target_shape[0])\n        S_db = librosa.power_to_db(S, ref=np.max)      \n\n        if S_db.shape[1]>target_shape[1]:\n            S_db=S_db[:, :target_shape[1]]\n        elif S_db.shape[1]<target_shape[1]:\n            pad_width=[(0, 0), (0, target_shape[1]-S_db.shape[1])]\n            S_db=np.pad(S_db, pad_width=pad_width, mode='constant')\n\n        S_norm = (S_db - S_db.min()) / (S_db.max() - S_db.min())\n        spectrograms.append(S_norm[..., None])  # Channel dimension\n\n    return spectrograms\n        ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T07:50:03.699148Z","iopub.execute_input":"2025-08-18T07:50:03.699673Z","iopub.status.idle":"2025-08-18T07:50:03.706428Z","shell.execute_reply.started":"2025-08-18T07:50:03.699648Z","shell.execute_reply":"2025-08-18T07:50:03.705711Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Predict on file","metadata":{}},{"cell_type":"code","source":"# Prediction on test files\ndef predict_test(model, spectrograms, threshold=0.5):\n    inputs=np.array(spectrograms)\n    outputs=model.predict(inputs, verbose=0)\n    pred=np.max(outputs, axis=0)\n    binary_pred=(pred>threshold).astype(int)\n    return pred, binary_pred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T07:50:03.707190Z","iopub.execute_input":"2025-08-18T07:50:03.707397Z","iopub.status.idle":"2025-08-18T07:50:03.732178Z","shell.execute_reply.started":"2025-08-18T07:50:03.707381Z","shell.execute_reply":"2025-08-18T07:50:03.731490Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Create submission file","metadata":{}},{"cell_type":"code","source":"# Creating .csv file for submission\ndef create_csv(model, test_path, csv_file=None):\n    rows = []\n    test_paths = os.listdir(test_path)\n\n    for i in tqdm(range(len(test_paths))):\n        file = test_paths[i]\n        \n        if file.endswith('.flac'):\n            file_path = os.path.join(test_path, file)\n            recording_id = file.replace('.flac', '')\n            spectrograms = gen_test_spectrograms(file_path, sr = sr, length = slice_length)\n            pred, _ = predict_test(model_keras, spectrograms)\n            rows.append([recording_id] + list(pred))\n            # _, binary_pred=predict_test(model_keras, spectrograms)\n            # rows.append([recording_id]+list(binary_pred))\n\n    df = pd.DataFrame(rows, columns=['recording_id']+[f\"s{i}\" for i in range(numSpecies)])\n    if csv_file:\n        df.to_csv(csv_file, float_format='%.5f', index=False)\n    else:\n        print(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T07:50:03.732965Z","iopub.execute_input":"2025-08-18T07:50:03.733263Z","iopub.status.idle":"2025-08-18T07:50:03.748807Z","shell.execute_reply.started":"2025-08-18T07:50:03.733243Z","shell.execute_reply":"2025-08-18T07:50:03.748290Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Saving submission file\nsubmission_dir = '/kaggle/working/csv'\nos.makedirs(submission_dir, exist_ok=True)\ncsv_file = os.path.join(submission_dir, 'rainForest_submission_keras.csv')\ncreate_csv(model_keras, test_path, csv_file=csv_file)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T07:50:03.777524Z","iopub.execute_input":"2025-08-18T07:50:03.777708Z","iopub.status.idle":"2025-08-18T08:03:04.197014Z","shell.execute_reply.started":"2025-08-18T07:50:03.777695Z","shell.execute_reply":"2025-08-18T08:03:04.196230Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Learning with unlabeled data","metadata":{}},{"cell_type":"markdown","source":"## Settings","metadata":{}},{"cell_type":"code","source":"num_files=500\n#patience=5","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T08:48:35.345342Z","iopub.execute_input":"2025-08-18T08:48:35.346035Z","iopub.status.idle":"2025-08-18T08:48:35.350295Z","shell.execute_reply.started":"2025-08-18T08:48:35.346000Z","shell.execute_reply":"2025-08-18T08:48:35.349624Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data loading","metadata":{}},{"cell_type":"code","source":"def load_audio_files(folder, rng, num_files=None, seed=42):\n    all_files=glob(os.path.join(folder, \"*.flac\"))\n    if num_files:\n        all_files=rng.choice(all_files, size=num_files, replace=False, shuffle=False).tolist()\n    return all_files","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T08:08:17.683586Z","iopub.execute_input":"2025-08-18T08:08:17.684392Z","iopub.status.idle":"2025-08-18T08:08:17.688573Z","shell.execute_reply.started":"2025-08-18T08:08:17.684367Z","shell.execute_reply":"2025-08-18T08:08:17.687762Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_audio(file_path, sr=22050):\n    audio, _=librosa.load(file_path, sr=sr)\n    return audio","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T08:08:17.689957Z","iopub.execute_input":"2025-08-18T08:08:17.690355Z","iopub.status.idle":"2025-08-18T08:08:17.710609Z","shell.execute_reply.started":"2025-08-18T08:08:17.690338Z","shell.execute_reply":"2025-08-18T08:08:17.709846Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Spectrogram generating and labeling","metadata":{}},{"cell_type":"code","source":"def gen_slices(\n    audio,\n    sr,\n    slice_length=3,\n):\n    num_samples=int(slice_length*sr)\n    num_slices=len(audio)//num_samples\n    specs=[]\n    for i in range(num_slices):\n        start=i*num_samples\n        end=start+num_samples\n        s=audio[start:end]\n        S=librosa.feature.melspectrogram(y=s, sr=sr, n_mels=128)\n        S_db=librosa.power_to_db(S, ref=np.max)\n        S_norm = (S_db - S_db.min()) / (S_db.max() - S_db.min())\n        specs.append(S_norm.astype(\"float32\"))\n\n    return specs","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T08:08:17.711521Z","iopub.execute_input":"2025-08-18T08:08:17.711777Z","iopub.status.idle":"2025-08-18T08:08:17.727348Z","shell.execute_reply.started":"2025-08-18T08:08:17.711753Z","shell.execute_reply":"2025-08-18T08:08:17.726855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def gen_spectrogram_files(\n    model,\n    train_path,\n    rng,\n    num_files=None,\n    slice_length=3,\n    sr=22050,\n    threshold=0.5,\n    seed=42\n):\n    files=load_audio_files(train_path, rng, num_files, seed)\n    all_spectrograms=[]\n    all_labels=[]\n    \n    for f in tqdm(files):\n        audio=load_audio(f, sr)\n        slices=gen_slices(audio, sr, slice_length)\n        if not slices:\n            continue\n        inputs=np.array(slices)[..., np.newaxis] # channel dimension\n        outputs_prob=model.predict(inputs, verbose=0)\n        outputs_pred=(outputs_prob>threshold).astype(np.float32)\n        \n        all_spectrograms.extend(inputs)\n        all_labels.extend(outputs_pred)\n\n    specs=np.array(all_spectrograms, dtype=np.float32)\n    labels=np.array(all_labels, dtype=np.float32)\n\n    return specs, labels","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T08:08:17.729039Z","iopub.execute_input":"2025-08-18T08:08:17.729274Z","iopub.status.idle":"2025-08-18T08:08:17.751106Z","shell.execute_reply.started":"2025-08-18T08:08:17.729260Z","shell.execute_reply":"2025-08-18T08:08:17.750578Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inputs, outputs=gen_spectrogram_files(model_keras, train_path, rng, num_files=num_files, slice_length=slice_length, sr=rec_sr, seed=random_seed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T08:09:05.800045Z","iopub.execute_input":"2025-08-18T08:09:05.800653Z","iopub.status.idle":"2025-08-18T08:12:35.314589Z","shell.execute_reply.started":"2025-08-18T08:09:05.800629Z","shell.execute_reply":"2025-08-18T08:12:35.313911Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Compile model","metadata":{}},{"cell_type":"code","source":"model_keras.compile(\n    optimizer=optimizer,\n    loss=\"binary_crossentropy\",\n    metrics=[BinaryAccuracy()]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T08:12:35.325402Z","iopub.execute_input":"2025-08-18T08:12:35.325859Z","iopub.status.idle":"2025-08-18T08:12:35.347381Z","shell.execute_reply.started":"2025-08-18T08:12:35.325836Z","shell.execute_reply":"2025-08-18T08:12:35.346798Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train the model","metadata":{}},{"cell_type":"code","source":"history_new=model_keras.fit(\n    inputs,\n    outputs,\n    batch_size=16,\n    validation_split=0.1,\n    epochs=epochs,\n    callbacks=[early_stop, reduce_lr]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T08:14:36.399616Z","iopub.execute_input":"2025-08-18T08:14:36.400426Z","iopub.status.idle":"2025-08-18T08:34:04.149130Z","shell.execute_reply.started":"2025-08-18T08:14:36.400402Z","shell.execute_reply":"2025-08-18T08:34:04.148349Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Create submission file","metadata":{}},{"cell_type":"code","source":"def gen_submission(\n    submission_dir,\n    model,\n    test_path,\n    filename\n):\n    os.makedirs(submission_dir, exist_ok=True)\n    csv_file=os.path.join(submission_dir, filename)\n    create_csv(model, test_path, csv_file=csv_file)\n    return csv_file","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T08:34:04.150825Z","iopub.execute_input":"2025-08-18T08:34:04.151062Z","iopub.status.idle":"2025-08-18T08:34:04.155128Z","shell.execute_reply.started":"2025-08-18T08:34:04.151038Z","shell.execute_reply":"2025-08-18T08:34:04.154410Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_dir = '/kaggle/working/csv'\nfilename='rainForest_submission_keras_new.csv'\ngen_submission(submission_dir, model_keras, test_path, filename)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T08:34:04.155933Z","iopub.execute_input":"2025-08-18T08:34:04.156148Z","iopub.status.idle":"2025-08-18T08:46:16.249080Z","shell.execute_reply.started":"2025-08-18T08:34:04.156134Z","shell.execute_reply":"2025-08-18T08:46:16.248496Z"}},"outputs":[],"execution_count":null}]}