{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":21669,"databundleVersionId":1692278,"sourceType":"competition"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# General Settings","metadata":{}},{"cell_type":"markdown","source":"## Library Imports","metadata":{}},{"cell_type":"code","source":"#Data handling\nimport os\nimport pandas as pd\nfrom PIL import Image\n\n# Audio handling\n!pip install PySoundFile\nimport librosa\nfrom IPython.display import Audio\n\n# Visualization\nimport matplotlib.pyplot as plt\n\n# Feedback with progress bar\nfrom tqdm.notebook import tqdm\n\n# Math & Algorithms\nimport numpy as np","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:18:42.968977Z","iopub.execute_input":"2025-12-11T22:18:42.969254Z","iopub.status.idle":"2025-12-11T22:18:47.817718Z","shell.execute_reply.started":"2025-12-11T22:18:42.969233Z","shell.execute_reply":"2025-12-11T22:18:47.817062Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Settings","metadata":{}},{"cell_type":"code","source":"# Initialise random number generation\nrandom_seed = 42\nrng = np.random.default_rng()\n\n# NN training parameters\nslice_length = 5\nsr = None # Use recording native sr\nbatch_size = 16\nlr = 1e-3\nepochs = 30\npatience = 10\n\n# Folder for storing generated spectrograms\nspect_save_path='/kaggle/working/spectrograms'\nos.makedirs(spect_save_path, exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:18:47.818875Z","iopub.execute_input":"2025-12-11T22:18:47.819337Z","iopub.status.idle":"2025-12-11T22:18:47.824377Z","shell.execute_reply.started":"2025-12-11T22:18:47.819311Z","shell.execute_reply":"2025-12-11T22:18:47.823618Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### File path variables","metadata":{}},{"cell_type":"code","source":"# Root data path for RainForest Species\ninput_path='/kaggle/input/rfcx-species-audio-detection'\n\n# Train and Test audio recordings data\ntrain_path=os.path.join(input_path, 'train')\ntest_path=os.path.join(input_path, 'test')\n\n# Labels\ntp_label_csv_path=os.path.join(input_path, 'train_tp.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:18:47.825136Z","iopub.execute_input":"2025-12-11T22:18:47.825343Z","iopub.status.idle":"2025-12-11T22:18:47.839640Z","shell.execute_reply.started":"2025-12-11T22:18:47.825320Z","shell.execute_reply":"2025-12-11T22:18:47.838877Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training data preparation","metadata":{}},{"cell_type":"markdown","source":"# Data Import","metadata":{}},{"cell_type":"markdown","source":"### Import labels","metadata":{}},{"cell_type":"code","source":"df_labels=pd.read_csv(tp_label_csv_path)\ndf_labels['t_length'] = df_labels['t_max'] - df_labels['t_min']\nlengths = df_labels['t_length']\nprint(f\"Length of labeled audio segments: {lengths.mean():.2f}±{lengths.std():.2f} ({lengths.min():.2f}-{lengths.max():.2f}) [s]\")\ndf_labels","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:18:47.841273Z","iopub.execute_input":"2025-12-11T22:18:47.841526Z","iopub.status.idle":"2025-12-11T22:18:47.902400Z","shell.execute_reply.started":"2025-12-11T22:18:47.841510Z","shell.execute_reply":"2025-12-11T22:18:47.901833Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Exploration","metadata":{}},{"cell_type":"markdown","source":"### Play a selected or random audio recording","metadata":{}},{"cell_type":"code","source":"recording_id = None\n#recording_id = '5b5218aba'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:18:47.903202Z","iopub.execute_input":"2025-12-11T22:18:47.903402Z","iopub.status.idle":"2025-12-11T22:18:47.906807Z","shell.execute_reply.started":"2025-12-11T22:18:47.903385Z","shell.execute_reply":"2025-12-11T22:18:47.906122Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if recording_id == None:\n    # Find all .flac files\n    flac_files=[f for f in os.listdir(train_path) if f.endswith('.flac')]\n    print(f'No. of .flac samples: {len(flac_files)}')\n\n    # Select and access a random .flac from the list\n    file=rng.choice(flac_files)\n    file_path=os.path.join(train_path, file)\n    recording_id=file.replace('.flac', '')\nelse:\n    file = recording_id + '.flac'\n    file_path=os.path.join(train_path, file)\n\nprint(f'Selected sample: {file}')\n\n# Read labels and audio data\nrecord = df_labels.loc[df_labels['recording_id'] == recording_id]\naudio, rec_sr = librosa.core.load(file_path, sr = sr, mono=False)\nprint(f\"Sample rate: {rec_sr}\")\n\n# Print label info\nprint(f\"\\nLabels:\")\nprint(f\"-------------\")\nif len(record)==0:\n    print(\"No associated label\")\nelse:\n    for _, row in record.iterrows():\n        print(f\"Faj: {row['species_id']}\")\n        print(f\"Típus: {row['songtype_id']}\")\n        print(f\"Időtartam: {row['t_min']} - {row['t_max']}\")\n        print(f\"Frekvencia: {row['f_min']} - {row['f_max']}\\n\")\n\n# Display player\nAudio(audio, rate=rec_sr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:18:47.907750Z","iopub.execute_input":"2025-12-11T22:18:47.907948Z","iopub.status.idle":"2025-12-11T22:18:53.192092Z","shell.execute_reply.started":"2025-12-11T22:18:47.907932Z","shell.execute_reply":"2025-12-11T22:18:53.190820Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Show the corresponding mel spectrogram","metadata":{}},{"cell_type":"code","source":"# Generate the Spectrogram\nS = librosa.feature.melspectrogram(y=audio, sr=rec_sr)\nS_db = librosa.power_to_db(S, ref=np.max)\n\n#Display\nfig, ax = plt.subplots(figsize=(16, 4))\nimage = librosa.display.specshow(S_db, sr=rec_sr, x_axis='time', y_axis='mel', ax=ax)\nfig.colorbar(image, ax=ax)\nax.set(title='Mel-Spectrogram')\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:18:53.193190Z","iopub.execute_input":"2025-12-11T22:18:53.193515Z","iopub.status.idle":"2025-12-11T22:19:03.604788Z","shell.execute_reply.started":"2025-12-11T22:18:53.193497Z","shell.execute_reply":"2025-12-11T22:19:03.604083Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Test mel spectrogram generation, saving and loading","metadata":{}},{"cell_type":"code","source":"# Generate spectrogram\nS = librosa.feature.melspectrogram(y=audio[0:int(slice_length*rec_sr)], sr=rec_sr)\nS_db = librosa.power_to_db(S, ref=np.max)\n\n# Convert to image apropriate format\nS_norm = (S_db-S_db.min())/(S_db.max()-S_db.min())\nS_norm = (S_norm*255).astype(np.uint8)\n\n# Convert to PIL image\nimg = Image.fromarray(S_norm)\n\n# Save the image with PIL\nimg = Image.fromarray(S_norm)\nimg.save('/kaggle/working/test.png')\n\n# Load the image\nimg_loaded = Image.open('/kaggle/working/test.png')\nS_norm_loaded = np.array(img_loaded)\n\n# Assert equality of saved and loaded array\ntry:\n    np.testing.assert_array_equal(S_norm, S_norm_loaded)\nexcept AssertionError:\n    print(f\"The saved and loaded arrays are NOT identical:\")\nelse:\n    print(f\"The saved and loaded arrays are IDENTICAL:\")\nfinally:\n    print(f\"\\tSaved array: {S_norm.shape}; saved values: {S_norm.min()}-{S_norm.max()}; format: {type(S_norm)}\")\n    display(img)\n    print(f\"\\tLoaded array: {S_norm_loaded.shape}; saved values: {S_norm_loaded.min()}-{S_norm_loaded.max()}; format: {type(S_norm_loaded)}\")\n    display(img_loaded)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:19:03.605541Z","iopub.execute_input":"2025-12-11T22:19:03.605800Z","iopub.status.idle":"2025-12-11T22:19:03.687701Z","shell.execute_reply.started":"2025-12-11T22:19:03.605769Z","shell.execute_reply":"2025-12-11T22:19:03.686916Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data generation","metadata":{}},{"cell_type":"markdown","source":"### Generate labeled spectrograms","metadata":{}},{"cell_type":"code","source":"def spectrogram_gen(audio_folder_path, recording_id, species_id,\n                    time_min, time_max, sr, slice_length,\n                    spect_save_path):\n    \"\"\"Generates spectrograms of from a given audio file\n\n    Generates spectrograms of 'slice_length' between 'time_min' and 'time_max'\n    of the given recording.\n\n    Parameters\n    ----------\n    audio_folder_path : path str\n        The folder in which the audio files are stored\n    recording_id : str\n        The identifier of the audio recording\n    species_id : int\n        The identifier of the species in the record\n    time_min: float\n        Start time of the vocalization of the species within the recording\n    time_max: float\n        End time of the vocalization of the species within the recording\n    sr : int\n        Forced sample_rate for laoding the audio (None uses the recording native sr)\n    slice_length: int\n        Length of recording slice to turn into spectrograms in seconds\n    spect_save_path: path str\n        Folder path to save generated spectrograms into\n\n    Returns\n    -------\n    save_path: path str\n        Location at which the generated spectrogram is saved at\n    \"\"\"\n\n    # Load the audio\n    file_path = os.path.join(audio_folder_path, recording_id + '.flac')\n    audio, rec_sr = librosa.core.load(file_path, sr=sr, mono=True)\n\n    # Generate audio slice(s)\n    slice_time = time_max - time_min\n    noSlices = max(int(np.round(slice_time/slice_length)), 1) # How many slices can fit into the given intervall, rounded to nearest int\n    \n    for i in range(noSlices): \n        # Find center time of the given slice\n        center = (time_min + i * (slice_time / (noSlices+1)))\n\n        # Find start and end sample of the given slice\n        start = int(max(center - slice_length/2, 0) * rec_sr)\n        end = start + int(slice_length * rec_sr)\n        if end > len(audio):\n            end = len(audio)\n            start = end - int(slice_length * rec_sr)\n\n        # Get the sliced audio\n        sliced_audio=audio[start:end]\n\n        # Generate Spectrogram\n        S = librosa.feature.melspectrogram(y = sliced_audio, sr=rec_sr)\n        S_db=librosa.power_to_db(S, ref=np.max)\n        \n        S_norm=(S_db-S_db.min())/(S_db.max()-S_db.min())\n        S_norm = (S_norm*255).astype(np.uint8)\n        spect_size = S_norm.shape\n\n        # Save the array as an image\n        species_path=os.path.join(spect_save_path, str(species_id))\n        os.makedirs(species_path, exist_ok=True)\n\n        filename = f'{species_id}_{recording_id}_{center:.2f}.png' # {center} kell, hátha ugyanolyan nevű file keletkezne\n        save_path = os.path.join(species_path, filename)\n\n        S_image = Image.fromarray(S_norm)\n        S_image.save(save_path)\n    \n    return spect_size, save_path # későbbi visszanézésre\n\n# Spectrogram generation progress (with TQDM progress bar)\ninput_size = None\nfor i in tqdm(range(len(df_labels))):\n    row = df_labels.iloc[i]\n    \n    recording_id=row['recording_id']\n    species_id=row['species_id']\n    time_min=float(row['t_min'])\n    time_max=float(row['t_max'])\n\n    # Generate spectrogram\n    spect_size, save_path = spectrogram_gen(audio_folder_path = train_path, recording_id = recording_id, species_id = species_id,\n                                            time_min = time_min, time_max = time_max, sr = sr, slice_length = slice_length, \n                                            spect_save_path = spect_save_path)\n    if input_size == None:\n        input_size = spect_size\n    else:\n        if (input_size != spect_size):\n            print(f\"WARNING: spectrogram size for label {i} ({spect_size}) does not match the spectrogram size for the first label ({input_size})\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:19:03.688605Z","iopub.execute_input":"2025-12-11T22:19:03.688822Z","iopub.status.idle":"2025-12-11T22:21:09.301732Z","shell.execute_reply.started":"2025-12-11T22:19:03.688807Z","shell.execute_reply":"2025-12-11T22:21:09.300865Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Summarise generated training data","metadata":{}},{"cell_type":"code","source":"species = [int(f) for f in os.listdir(spect_save_path) if os.path.isdir(os.path.join(spect_save_path, f))]\nspecies.sort()\nnumSpecies = len(species)\nprint(f\"Fajok száma: {numSpecies}\")\n\nsum_files=0\nprint(\"Fájlok száma az egyes species mappákban:\")\nfor f in species:\n    path = os.path.join(spect_save_path, str(f))\n    numFiles = len([name for name in os.listdir(path) if os.path.isfile(os.path.join(path, name))])\n    sum_files += numFiles\n    print(f\"{f}:\\t{numFiles}\")\nprint(f\"Összes spectrogram: {sum_files}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:21:09.304309Z","iopub.execute_input":"2025-12-11T22:21:09.304743Z","iopub.status.idle":"2025-12-11T22:21:09.319000Z","shell.execute_reply.started":"2025-12-11T22:21:09.304724Z","shell.execute_reply":"2025-12-11T22:21:09.318279Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Keras Implementation","metadata":{}},{"cell_type":"markdown","source":"## Library imports","metadata":{}},{"cell_type":"code","source":"# Imports\nimport keras\nfrom keras import layers\nfrom keras.metrics import BinaryAccuracy\nimport tensorflow as tf\nimport random\nfrom glob import glob\nfrom sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:21:09.319806Z","iopub.execute_input":"2025-12-11T22:21:09.320069Z","iopub.status.idle":"2025-12-11T22:21:23.362269Z","shell.execute_reply.started":"2025-12-11T22:21:09.320041Z","shell.execute_reply":"2025-12-11T22:21:23.361647Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Seeds and warnings","metadata":{}},{"cell_type":"code","source":"# Seeds\nos.environ['PYTHONHASHSEED'] = str(random_seed)\nnp.random.seed(random_seed)\ntf.random.set_seed(random_seed)\nrandom.seed(random_seed)\n\n# Hiding warnings\nos.environ['TF_DETERMINISTIC_OPS'] = '1'\nos.environ['TF_CUDNN_DETERMINISTIC'] = '1'\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '2'  # 2 -> warnings","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:21:23.362907Z","iopub.execute_input":"2025-12-11T22:21:23.363369Z","iopub.status.idle":"2025-12-11T22:21:23.368254Z","shell.execute_reply.started":"2025-12-11T22:21:23.363350Z","shell.execute_reply":"2025-12-11T22:21:23.367454Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_files = glob(\"/kaggle/working/spectrograms/*/*.png\")\nprint(f\"Total files: {len(all_files)}\")\ntrain_files, val_files = train_test_split(all_files, test_size=0.1, random_state=random_seed)\n\n# Input size\nimage_size = np.array(Image.open(all_files[0]))\ninput_shape = (image_size.shape[0], image_size.shape[1], 1) # adding grayscale dimension\nprint(f\"Image size: {image_size}\\nInput shape: {input_shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:21:23.369203Z","iopub.execute_input":"2025-12-11T22:21:23.369470Z","iopub.status.idle":"2025-12-11T22:21:23.399780Z","shell.execute_reply.started":"2025-12-11T22:21:23.369445Z","shell.execute_reply":"2025-12-11T22:21:23.399179Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Augmentation","metadata":{}},{"cell_type":"code","source":"# Data augmentation\ndata_augmentation=keras.Sequential([\n    layers.RandomTranslation(\n        height_factor=0,\n        width_factor=0.1,\n        fill_mode='nearest'\n    ),\n    layers.RandomContrast(0.2),\n    layers.RandomZoom(\n        height_factor=0,\n        width_factor=0.1\n    )\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:21:23.400487Z","iopub.execute_input":"2025-12-11T22:21:23.400668Z","iopub.status.idle":"2025-12-11T22:21:24.442750Z","shell.execute_reply.started":"2025-12-11T22:21:23.400654Z","shell.execute_reply":"2025-12-11T22:21:24.442104Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Dataset loader class","metadata":{}},{"cell_type":"code","source":"class AudioDataset(keras.utils.Sequence):\n    \n    \n    # Initialization\n    def __init__(self, files, num_species, batch_size=16, shuffle=False, seed=None, **kwargs):\n        \"\"\" Spectrogram dataset loader - reproducible\n            \n            Arguments:\n            files: list of paths to images\n            numSpecies: number of labels\n            batch_size: number of files in a batch\n            shuffle: whether to shuffle images\n            seed: seed for reproducibility\n        \"\"\"\n        super().__init__()      \n\n        self.files=files.copy()\n        self.num_species=num_species\n        self.batch_size=batch_size\n        self.shuffle=shuffle\n        self.seed=seed\n        self.rng=np.random.RandomState(seed) if seed is not None else np.random\n        \n        # setting seeds\n        if seed is not None:\n            np.random.seed(seed)\n            tf.random.set_seed(seed)\n            random.seed(seed)\n\n        # detecting image shape\n        with Image.open(self.files[0]) as image:\n            self.input_shape=(image.size[1], image.size[0], 1)\n\n        self.end_of_epoch()\n\n\n    # Number of batches\n    def __len__(self):\n        return int(np.ceil(len(self.files)/self.batch_size))\n\n\n    # Creates a single batch\n    def __getitem__(self, index):\n        batch_files=self.files[index*self.batch_size:(index+1)*self.batch_size]\n\n        batch_images=[]\n        batch_labels=[]\n        \n        for f in batch_files:\n            with Image.open(f) as image:\n                image=image.resize(self.input_shape[:2][::-1])\n                image=np.array(image, dtype=np.float32)/255.0\n                if image.ndim==2:\n                    image=image[...,np.newaxis]\n            batch_images.append(image)\n\n            label=int(os.path.basename(os.path.dirname(f)))\n            one_hot=np.zeros(self.num_species, dtype=np.float32)\n            one_hot[label]=1.0\n            batch_labels.append(one_hot)\n\n        return np.stack(batch_images), np.stack(batch_labels)\n\n    \n    # Shuffles files\n    def end_of_epoch(self):\n        if self.shuffle:\n            if self.seed is not None:\n                permutation=self.rng.permutation(len(self.files))\n                self.files=[self.files[i] for i in permutation]\n            else:\n                np.random.shuffle(self.files)\n        \n\n    # Shape of an image\n    def image_shape(self):\n        return self.input_shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:21:24.443400Z","iopub.execute_input":"2025-12-11T22:21:24.443590Z","iopub.status.idle":"2025-12-11T22:21:24.454167Z","shell.execute_reply.started":"2025-12-11T22:21:24.443575Z","shell.execute_reply":"2025-12-11T22:21:24.453351Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Dataset loading\ntrain_dataset=AudioDataset(files=train_files, num_species=numSpecies, batch_size=batch_size, shuffle=True, seed=random_seed)\nval_dataset=AudioDataset(files=val_files, num_species=numSpecies, batch_size=batch_size, shuffle=False, seed=random_seed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:21:24.454914Z","iopub.execute_input":"2025-12-11T22:21:24.455182Z","iopub.status.idle":"2025-12-11T22:21:24.470228Z","shell.execute_reply.started":"2025-12-11T22:21:24.455159Z","shell.execute_reply":"2025-12-11T22:21:24.469703Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Define model","metadata":{}},{"cell_type":"code","source":"# Model\nmodel_keras=keras.models.Sequential([\n    \n    layers.Input(shape=train_dataset.image_shape()),\n    data_augmentation,\n    \n    layers.Conv2D(16, (3, 3), activation='relu'),\n    layers.Conv2D(32, (3, 3), activation='relu'),\n    layers.Conv2D(32, (3, 3), activation='relu'),\n\n    layers.Flatten(),\n    layers.Dense(128, activation='relu'),\n    layers.Dense(numSpecies, activation='sigmoid')\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:21:24.471023Z","iopub.execute_input":"2025-12-11T22:21:24.471288Z","iopub.status.idle":"2025-12-11T22:21:25.878617Z","shell.execute_reply.started":"2025-12-11T22:21:24.471266Z","shell.execute_reply":"2025-12-11T22:21:25.878074Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Training logic","metadata":{}},{"cell_type":"code","source":"# Optimizer\noptimizer=keras.optimizers.Adam(learning_rate = lr)\n\n# Callbacks\nreduce_lr = keras.callbacks.ReduceLROnPlateau(factor = 0.5, patience = patience / 2, verbose=1)\nearly_stop = keras.callbacks.EarlyStopping(patience = patience, verbose = 1, restore_best_weights = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:21:25.879343Z","iopub.execute_input":"2025-12-11T22:21:25.879555Z","iopub.status.idle":"2025-12-11T22:21:25.888920Z","shell.execute_reply.started":"2025-12-11T22:21:25.879538Z","shell.execute_reply":"2025-12-11T22:21:25.888136Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Compile model","metadata":{}},{"cell_type":"code","source":"# Compile model\nmodel_keras.compile(\n    optimizer=optimizer,\n    loss=\"binary_crossentropy\",\n    metrics=[BinaryAccuracy()]\n)\nmodel_keras.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:21:25.889759Z","iopub.execute_input":"2025-12-11T22:21:25.890217Z","iopub.status.idle":"2025-12-11T22:21:25.913464Z","shell.execute_reply.started":"2025-12-11T22:21:25.890194Z","shell.execute_reply":"2025-12-11T22:21:25.912766Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train the model","metadata":{}},{"cell_type":"code","source":"history_keras = model_keras.fit(train_dataset, validation_data=val_dataset, epochs=epochs, callbacks=[early_stop, reduce_lr])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:21:25.914216Z","iopub.execute_input":"2025-12-11T22:21:25.914426Z","iopub.status.idle":"2025-12-11T22:25:58.628411Z","shell.execute_reply.started":"2025-12-11T22:21:25.914411Z","shell.execute_reply":"2025-12-11T22:25:58.627802Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Evaluate","metadata":{}},{"cell_type":"markdown","source":"### Generate spectrograms","metadata":{}},{"cell_type":"code","source":"# Generate spectrograms from a given file\ndef gen_test_spectrograms(file_path, sr, length, target_shape=(input_shape[0], input_shape[1])):\n    spectrograms=[]\n    audio, rec_sr = librosa.core.load(file_path, sr=sr, mono=True)\n    slice_length = int(rec_sr * length)\n    n = len(audio) // slice_length\n\n    for i in range(n):\n        start = i * slice_length\n        end = start + slice_length\n        if end > len(audio):\n            end = len(audio)\n        sliced_audio = audio[start:end]\n\n        S = librosa.feature.melspectrogram(y=sliced_audio, sr=rec_sr, n_mels=target_shape[0])\n        S_db = librosa.power_to_db(S, ref=np.max)      \n\n        if S_db.shape[1]>target_shape[1]:\n            S_db=S_db[:, :target_shape[1]]\n        elif S_db.shape[1]<target_shape[1]:\n            pad_width=[(0, 0), (0, target_shape[1]-S_db.shape[1])]\n            S_db=np.pad(S_db, pad_width=pad_width, mode='constant')\n\n        S_norm = (S_db - S_db.min()) / (S_db.max() - S_db.min())\n        spectrograms.append(S_norm[..., None])  # Channel dimension\n\n    return spectrograms\n        ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:25:58.629338Z","iopub.execute_input":"2025-12-11T22:25:58.629592Z","iopub.status.idle":"2025-12-11T22:25:58.636697Z","shell.execute_reply.started":"2025-12-11T22:25:58.629566Z","shell.execute_reply":"2025-12-11T22:25:58.635898Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Predict on file","metadata":{}},{"cell_type":"code","source":"# Prediction on test files\ndef predict_test(model, spectrograms, threshold=0.5):\n    inputs=np.array(spectrograms)\n    outputs=model.predict(inputs, verbose=0)\n    pred=np.max(outputs, axis=0)\n    binary_pred=(pred>threshold).astype(int)\n    return pred, binary_pred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:25:58.637384Z","iopub.execute_input":"2025-12-11T22:25:58.637571Z","iopub.status.idle":"2025-12-11T22:25:58.650639Z","shell.execute_reply.started":"2025-12-11T22:25:58.637556Z","shell.execute_reply":"2025-12-11T22:25:58.649961Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Create submission file","metadata":{}},{"cell_type":"code","source":"# Creating .csv file for submission\ndef create_csv(model, test_path, csv_file=None):\n    rows = []\n    test_paths = os.listdir(test_path)\n\n    for i in tqdm(range(len(test_paths))):\n        file = test_paths[i]\n        \n        if file.endswith('.flac'):\n            file_path = os.path.join(test_path, file)\n            recording_id = file.replace('.flac', '')\n            spectrograms = gen_test_spectrograms(file_path, sr = sr, length = slice_length)\n            pred, _ = predict_test(model_keras, spectrograms)\n            rows.append([recording_id] + list(pred))\n            # _, binary_pred=predict_test(model_keras, spectrograms)\n            # rows.append([recording_id]+list(binary_pred))\n\n    df = pd.DataFrame(rows, columns=['recording_id']+[f\"s{i}\" for i in range(numSpecies)])\n    if csv_file:\n        df.to_csv(csv_file, float_format='%.5f', index=False)\n    else:\n        print(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:25:58.651442Z","iopub.execute_input":"2025-12-11T22:25:58.651697Z","iopub.status.idle":"2025-12-11T22:25:58.662341Z","shell.execute_reply.started":"2025-12-11T22:25:58.651675Z","shell.execute_reply":"2025-12-11T22:25:58.661526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Saving submission file\nsubmission_dir = '/kaggle/working/csv'\nos.makedirs(submission_dir, exist_ok=True)\ncsv_file = os.path.join(submission_dir, 'rainForest_submission_keras.csv')\ncreate_csv(model_keras, test_path, csv_file=csv_file)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:25:58.663206Z","iopub.execute_input":"2025-12-11T22:25:58.663469Z","iopub.status.idle":"2025-12-11T22:38:13.945430Z","shell.execute_reply.started":"2025-12-11T22:25:58.663445Z","shell.execute_reply":"2025-12-11T22:38:13.944697Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Learning with unlabeled data","metadata":{}},{"cell_type":"markdown","source":"## Settings","metadata":{}},{"cell_type":"code","source":"# num_files=500\n# #patience=5","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:38:13.946327Z","iopub.execute_input":"2025-12-11T22:38:13.946515Z","iopub.status.idle":"2025-12-11T22:38:13.949522Z","shell.execute_reply.started":"2025-12-11T22:38:13.946500Z","shell.execute_reply":"2025-12-11T22:38:13.948918Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data loading","metadata":{}},{"cell_type":"code","source":"# def load_audio_files(folder, rng, num_files=None, seed=42):\n#     all_files=glob(os.path.join(folder, \"*.flac\"))\n#     if num_files:\n#         all_files=rng.choice(all_files, size=num_files, replace=False, shuffle=False).tolist()\n#     return all_files","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:38:13.950920Z","iopub.execute_input":"2025-12-11T22:38:13.951652Z","iopub.status.idle":"2025-12-11T22:38:13.965851Z","shell.execute_reply.started":"2025-12-11T22:38:13.951627Z","shell.execute_reply":"2025-12-11T22:38:13.965152Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def load_audio(file_path, sr=22050):\n#     audio, _=librosa.load(file_path, sr=sr)\n#     return audio","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:38:13.966490Z","iopub.execute_input":"2025-12-11T22:38:13.966690Z","iopub.status.idle":"2025-12-11T22:38:13.976536Z","shell.execute_reply.started":"2025-12-11T22:38:13.966675Z","shell.execute_reply":"2025-12-11T22:38:13.975845Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Spectrogram generating and labeling","metadata":{}},{"cell_type":"code","source":"# def gen_slices(\n#     audio,\n#     sr,\n#     slice_length=3,\n# ):\n#     num_samples=int(slice_length*sr)\n#     num_slices=len(audio)//num_samples\n#     specs=[]\n#     for i in range(num_slices):\n#         start=i*num_samples\n#         end=start+num_samples\n#         s=audio[start:end]\n#         S=librosa.feature.melspectrogram(y=s, sr=sr, n_mels=128)\n#         S_db=librosa.power_to_db(S, ref=np.max)\n#         S_norm = (S_db - S_db.min()) / (S_db.max() - S_db.min())\n#         specs.append(S_norm.astype(\"float32\"))\n\n#     return specs","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:38:13.979658Z","iopub.execute_input":"2025-12-11T22:38:13.979856Z","iopub.status.idle":"2025-12-11T22:38:13.989873Z","shell.execute_reply.started":"2025-12-11T22:38:13.979841Z","shell.execute_reply":"2025-12-11T22:38:13.989235Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def gen_spectrogram_files(\n#     model,\n#     train_path,\n#     rng,\n#     num_files=None,\n#     slice_length=3,\n#     sr=22050,\n#     threshold=0.5,\n#     seed=42\n# ):\n#     files=load_audio_files(train_path, rng, num_files, seed)\n#     all_spectrograms=[]\n#     all_labels=[]\n    \n#     for f in tqdm(files):\n#         audio=load_audio(f, sr)\n#         slices=gen_slices(audio, sr, slice_length)\n#         if not slices:\n#             continue\n#         inputs=np.array(slices)[..., np.newaxis] # channel dimension\n#         outputs_prob=model.predict(inputs, verbose=0)\n#         outputs_pred=(outputs_prob>threshold).astype(np.float32)\n        \n#         all_spectrograms.extend(inputs)\n#         all_labels.extend(outputs_pred)\n\n#     specs=np.array(all_spectrograms, dtype=np.float32)\n#     labels=np.array(all_labels, dtype=np.float32)\n\n#     return specs, labels","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:38:13.990478Z","iopub.execute_input":"2025-12-11T22:38:13.990636Z","iopub.status.idle":"2025-12-11T22:38:14.002907Z","shell.execute_reply.started":"2025-12-11T22:38:13.990622Z","shell.execute_reply":"2025-12-11T22:38:14.002224Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# inputs, outputs=gen_spectrogram_files(model_keras, train_path, rng, num_files=num_files, slice_length=slice_length, sr=rec_sr, seed=random_seed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:38:14.003564Z","iopub.execute_input":"2025-12-11T22:38:14.003784Z","iopub.status.idle":"2025-12-11T22:38:14.013888Z","shell.execute_reply.started":"2025-12-11T22:38:14.003769Z","shell.execute_reply":"2025-12-11T22:38:14.013236Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Compile model","metadata":{}},{"cell_type":"code","source":"# model_keras.compile(\n#     optimizer=optimizer,\n#     loss=\"binary_crossentropy\",\n#     metrics=[BinaryAccuracy()]\n# )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:38:14.014594Z","iopub.execute_input":"2025-12-11T22:38:14.015052Z","iopub.status.idle":"2025-12-11T22:38:14.024616Z","shell.execute_reply.started":"2025-12-11T22:38:14.015028Z","shell.execute_reply":"2025-12-11T22:38:14.024022Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train the model","metadata":{}},{"cell_type":"code","source":"# history_new=model_keras.fit(\n#     inputs,\n#     outputs,\n#     batch_size=16,\n#     validation_split=0.1,\n#     epochs=epochs,\n#     callbacks=[early_stop, reduce_lr]\n# )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:38:14.025353Z","iopub.execute_input":"2025-12-11T22:38:14.026103Z","iopub.status.idle":"2025-12-11T22:38:14.035586Z","shell.execute_reply.started":"2025-12-11T22:38:14.026077Z","shell.execute_reply":"2025-12-11T22:38:14.034993Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Create submission file","metadata":{}},{"cell_type":"code","source":"# def gen_submission(\n#     submission_dir,\n#     model,\n#     test_path,\n#     filename\n# ):\n#     os.makedirs(submission_dir, exist_ok=True)\n#     csv_file=os.path.join(submission_dir, filename)\n#     create_csv(model, test_path, csv_file=csv_file)\n#     return csv_file","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:38:14.036113Z","iopub.execute_input":"2025-12-11T22:38:14.036271Z","iopub.status.idle":"2025-12-11T22:38:14.045364Z","shell.execute_reply.started":"2025-12-11T22:38:14.036258Z","shell.execute_reply":"2025-12-11T22:38:14.044775Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission_dir = '/kaggle/working/csv'\n# filename='rainForest_submission_keras_new.csv'\n# gen_submission(submission_dir, model_keras, test_path, filename)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-11T22:38:14.045900Z","iopub.execute_input":"2025-12-11T22:38:14.046112Z","iopub.status.idle":"2025-12-11T22:38:14.057598Z","shell.execute_reply.started":"2025-12-11T22:38:14.046089Z","shell.execute_reply":"2025-12-11T22:38:14.056979Z"}},"outputs":[],"execution_count":null}]}