{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":21669,"databundleVersionId":1692278,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Global settings","metadata":{}},{"cell_type":"markdown","source":"## Import","metadata":{}},{"cell_type":"code","source":"#Data handling\nimport os\nimport pandas as pd\n# import itertools\nfrom PIL import Image\n\n# Randomization\nimport random\n\n# Audio handling\n# !pip install PySoundFile\nimport librosa\nfrom IPython.display import Audio\n\n# Visualization\nimport matplotlib.pyplot as plt\nfrom sklearn.manifold import TSNE\n\n# Feedback with progress bar\nfrom tqdm.notebook import tqdm\n\n# Math & Algorithms\nimport numpy as np\n\n# Model\nimport keras\nfrom keras import layers\nimport tensorflow as tf\nfrom tensorflow.keras import models\nfrom tensorflow.keras.layers import Resizing\n\n# Clustering\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.cluster import KMeans","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:17:20.658098Z","iopub.execute_input":"2025-11-25T13:17:20.658311Z","iopub.status.idle":"2025-11-25T13:17:38.733599Z","shell.execute_reply.started":"2025-11-25T13:17:20.658294Z","shell.execute_reply":"2025-11-25T13:17:38.733048Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Variable setting","metadata":{}},{"cell_type":"code","source":"# NN training parameters\nsegment_length = 1\nlatent_dim = 256\nbatch_size = 32\nlr = 1e-3\npatience = 10\nepochs = 100\nnum_files = 1000\nnum_clusters = 24\n\n# Initialize random number generation\nrandom_seed = 42\nrandom.seed(random_seed)\nrng = np.random.default_rng()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:18:30.188246Z","iopub.execute_input":"2025-11-25T13:18:30.188498Z","iopub.status.idle":"2025-11-25T13:18:30.193108Z","shell.execute_reply.started":"2025-11-25T13:18:30.188479Z","shell.execute_reply":"2025-11-25T13:18:30.192385Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## File path","metadata":{}},{"cell_type":"code","source":"# Folder for storing generated spectrograms\nsave_path='/kaggle/working/spectrograms'\nos.makedirs(save_path, exist_ok=True)\n\n# Root data path for RainForest Species\ninput_path='/kaggle/input/rfcx-species-audio-detection'\n\n# Train and Test audio recordings data\ntrain_path=os.path.join(input_path, 'train')\ntest_path=os.path.join(input_path, 'test')\n\n# Labels\ntp_label_csv_path=os.path.join(input_path, 'train_tp.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:18:33.429772Z","iopub.execute_input":"2025-11-25T13:18:33.430443Z","iopub.status.idle":"2025-11-25T13:18:33.435066Z","shell.execute_reply.started":"2025-11-25T13:18:33.430414Z","shell.execute_reply":"2025-11-25T13:18:33.434130Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Number of files\nnum_train_files=len([f for f in os.listdir(train_path) if os.path.isfile(os.path.join(train_path, f))])\nnum_test_files=len([f for f in os.listdir(test_path) if os.path.isfile(os.path.join(test_path, f))])\nnum_tp_rows=len(pd.read_csv(tp_label_csv_path))\nprint(f\"Number of training files: {num_train_files}\")\nprint(f\"Number of test files: {num_test_files}\")\nprint(f\"Number of labeled entries (true positive): {num_tp_rows}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:18:36.388653Z","iopub.execute_input":"2025-11-25T13:18:36.388978Z","iopub.status.idle":"2025-11-25T13:18:57.507022Z","shell.execute_reply.started":"2025-11-25T13:18:36.388949Z","shell.execute_reply":"2025-11-25T13:18:57.506186Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data generation","metadata":{}},{"cell_type":"markdown","source":"## Spectrogram generation - functions","metadata":{}},{"cell_type":"code","source":"# Chooses files randomly from the given folder\ndef random_files(source_path, num_files=1):\n\n    all_files=os.listdir(source_path)\n    chosen_files=random.sample(all_files, num_files)\n    return [os.path.join(source_path, f) for f in chosen_files]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:18:57.508727Z","iopub.execute_input":"2025-11-25T13:18:57.509503Z","iopub.status.idle":"2025-11-25T13:18:57.514910Z","shell.execute_reply.started":"2025-11-25T13:18:57.509475Z","shell.execute_reply":"2025-11-25T13:18:57.513708Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Returns random audio segment\ndef random_audio_segment(file_path, segment_length=3.0, sr = None):\n    \n    y, sr = librosa.load(file_path, sr = sr)\n    total_length = librosa.get_duration(y=y, sr = sr)\n\n    # Random start point\n    start_time = random.uniform(0, total_length-segment_length)\n    segment_samples = int(segment_length * sr)\n    start_sample = int(start_time * sr)\n    end_sample = start_sample+segment_samples\n    return y[start_sample:end_sample], sr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:18:57.515840Z","iopub.execute_input":"2025-11-25T13:18:57.516109Z","iopub.status.idle":"2025-11-25T13:18:57.528211Z","shell.execute_reply.started":"2025-11-25T13:18:57.516080Z","shell.execute_reply":"2025-11-25T13:18:57.527256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generates spectrogram from the given file\ndef spec_gen(file_path, sr = None):\n    \"\"\"\n    Generates a spectogram from a given audio file, and returns the first\n    target_shape[1] pixel columns from it.\n    \"\"\"\n    \n    audio, sr = librosa.core.load(file_path)\n    S = librosa.feature.melspectrogram(y = audio, sr = sr, n_mels = 128)\n    S_db = librosa.power_to_db(S, ref=np.max)\n\n    if S_db.shape[1]>target_shape[1]:\n        S_db=S_db[:, :target_shape[1]]\n    elif S_db.shape[1]<target_shape[1]:\n        pad_width=[(0, 0), (0, target_shape[1]-S_db.shape[1])]\n        S_db=np.pad(S_db, pad_width=pad_width, mode='constant')\n    \n    S_norm = (S_db - S_db.min()) / (S_db.max() - S_db.min())\n    \n    return S_norm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:18:57.530417Z","iopub.execute_input":"2025-11-25T13:18:57.531607Z","iopub.status.idle":"2025-11-25T13:18:57.542257Z","shell.execute_reply.started":"2025-11-25T13:18:57.531588Z","shell.execute_reply":"2025-11-25T13:18:57.541567Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def spec_gen_short(file_path, sr = None, n_mels = 128, segment_length = 3):\n    \n    audio, sr = random_audio_segment(file_path, segment_length, sr)\n    S = librosa.feature.melspectrogram(y = audio, sr = sr, n_mels = n_mels)\n    S_db = librosa.power_to_db(S, ref=np.max)\n    S_norm = (S_db - S_db.min()) / (S_db.max() - S_db.min())\n    \n    return S_norm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:18:57.542701Z","iopub.execute_input":"2025-11-25T13:18:57.542864Z","iopub.status.idle":"2025-11-25T13:18:57.553299Z","shell.execute_reply.started":"2025-11-25T13:18:57.542851Z","shell.execute_reply":"2025-11-25T13:18:57.552443Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def spec_save_short(source_path, save_path, num_files = 100, sr = None, segment_length = 3, n_mels = 128):\n    \"\"\"\n    Saves a random segment from the first n audio files from the\n    source path as a .png image to save_path.\n    \"\"\"\n\n    files = [f for f in os.scandir(source_path) if f.is_file()]\n    files = files[:num_files]\n    for f in tqdm(files):\n        output = os.path.join(save_path, os.path.splitext(f.name)[0]+\".png\")\n        spec = spec_gen_short(f.path, sr = sr, segment_length=segment_length)\n        spec = (spec*255).astype(np.uint8)\n        # spec = Image.fromarray(spec, mode = 'L')\n        spec = Image.fromarray(spec)\n        spec.save(output)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:18:57.554433Z","iopub.execute_input":"2025-11-25T13:18:57.555779Z","iopub.status.idle":"2025-11-25T13:18:57.568404Z","shell.execute_reply.started":"2025-11-25T13:18:57.555754Z","shell.execute_reply":"2025-11-25T13:18:57.567788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_images(source_path):\n\n    images=[]\n    for f in os.listdir(source_path):\n        if f.endswith('.png'):\n            image_path = os.path.join(source_path, f)\n            image = Image.open(image_path).convert('L')\n            image_array = np.array(image)/255.0\n            images.append(image_array)\n    return np.array(images)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:18:57.569116Z","iopub.execute_input":"2025-11-25T13:18:57.569320Z","iopub.status.idle":"2025-11-25T13:18:57.576565Z","shell.execute_reply.started":"2025-11-25T13:18:57.569304Z","shell.execute_reply":"2025-11-25T13:18:57.575868Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data loader class","metadata":{}},{"cell_type":"code","source":"class AudioDataset(keras.utils.Sequence):\n\n    # Initialization\n    def __init__(self, images, batch_size, shuffle=False, seed=None, **kwargs):\n\n        super().__init__()\n\n        self.images=images\n        self.batch_size=batch_size\n        self.shuffle=shuffle\n        self.seed=seed\n        self.rng=np.random.RandomState(seed) if seed is not None else np.random\n\n        # setting seeds\n        if seed is not None:\n            np.random.seed(seed)\n            tf.random.set_seed(seed)\n            random.seed(seed)\n\n        # detecting image shape\n        self.input_shape=self.images[0].shape\n        self.indices=np.arange(len(self.images))\n\n        self.end_of_epoch()\n\n\n    # Number of batches\n    def __len__(self):\n        return int(np.ceil(len(self.images)/self.batch_size))\n\n\n    # Creates a single batch\n    def __getitem__(self, index):\n        batch_idx=self.indices[index*self.batch_size:(index+1)*self.batch_size]\n        batch_images=[self.images[i] for i in batch_idx]\n        batch_images=np.stack(batch_images).astype(\"float32\")\n\n        if batch_images.ndim==3:\n            batch_images=np.expand_dims(batch_images, -1)\n\n        if batch_images.max()>1.0:\n            batch_images=batch_images/255.0\n\n        return batch_images, batch_images\n\n\n    # Shuffles files\n    def end_of_epoch(self):\n        if self.shuffle:\n            if self.seed is not None:\n                self.rng.shuffle(self.indices)\n            else:\n                np.random.shuffle(self.indices)\n\n\n    # Shape of an image\n    def image_shape(self):\n        return self.input_shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:19:31.677565Z","iopub.execute_input":"2025-11-25T13:19:31.678395Z","iopub.status.idle":"2025-11-25T13:19:31.686796Z","shell.execute_reply.started":"2025-11-25T13:19:31.678363Z","shell.execute_reply":"2025-11-25T13:19:31.686036Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Spectrogram generation","metadata":{}},{"cell_type":"code","source":"# Getting sampling rate\nrandom_file, sr = librosa.core.load(random_files(train_path)[0], sr = None)\nprint(sr)\n\n# Saving a short spec segment from each file\nspec_save_short(train_path, save_path, num_files, segment_length = segment_length, n_mels=128)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:19:40.650683Z","iopub.execute_input":"2025-11-25T13:19:40.651356Z","iopub.status.idle":"2025-11-25T13:21:42.013274Z","shell.execute_reply.started":"2025-11-25T13:19:40.651333Z","shell.execute_reply":"2025-11-25T13:21:42.012424Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Training dataset","metadata":{}},{"cell_type":"code","source":"# Calculating input shape\nexample_path=os.path.join(save_path, os.listdir(save_path)[0])\nexample_image=Image.open(example_path).convert('L')\nexample_array=np.array(example_image)\ntarget_shape=example_array.shape\nprint(f\"Example image path: {example_path}\")\nprint(f\"Target shape: {target_shape}\")\n\n# Showing example image\nplt.imshow(example_image, cmap='viridis')\n\n# Loading and reshaping images\nimages=load_images(save_path).reshape(-1, target_shape[0], target_shape[1], 1)\nprint(images.shape)\nplt.imshow(images[0], cmap='viridis')\n\n# Building the training dataset for the autoencoder\ntrain_dataset = AudioDataset(\n    images=images,\n    batch_size=batch_size,\n    shuffle=True,\n    seed=42\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:21:42.014411Z","iopub.execute_input":"2025-11-25T13:21:42.014998Z","iopub.status.idle":"2025-11-25T13:21:42.717287Z","shell.execute_reply.started":"2025-11-25T13:21:42.014977Z","shell.execute_reply":"2025-11-25T13:21:42.716657Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model training","metadata":{}},{"cell_type":"markdown","source":"## Autoencoder","metadata":{}},{"cell_type":"code","source":"def conv_autoencoder(input_shape, latent_dim=64):\n\n    # Encoder\n    encoder_input=layers.Input(shape=input_shape)\n    x=layers.Conv2D(128, (3, 3), activation='relu', padding='same')(encoder_input)\n    x=layers.BatchNormalization()(x)\n    x=layers.MaxPooling2D((2, 2), padding='same')(x)\n    x=layers.Dropout(0.2)(x)\n    \n    x=layers.Conv2D(64, (3, 3), activation='relu', padding='same')(x)\n    x=layers.BatchNormalization()(x)\n    x=layers.MaxPooling2D((2, 2), padding='same')(x)\n    x=layers.Dropout(0.2)(x)\n    \n    x=layers.Conv2D(32, (3, 3), activation='relu', padding='same')(x)\n    x=layers.BatchNormalization()(x)\n    \n    x_shape=x.shape[1:]\n    x_prod=np.prod(x_shape)\n\n    # Bottleneck\n    x=layers.Flatten()(x)\n    latent=layers.Dense(latent_dim, activation='sigmoid')(x)\n    encoder=models.Model(encoder_input, latent)\n\n    # Decoder\n    decoder_input=layers.Input(shape=(latent_dim,))\n    x=layers.Dense(x_prod, activation='relu')(decoder_input)\n    x=layers.Reshape((x_shape))(x)\n\n    x=layers.Conv2DTranspose(16, (3,3), strides=2, activation='relu', padding='same')(x)\n    x = layers.BatchNormalization()(x)\n    x=layers.Conv2DTranspose(32, (3,3), strides=2, activation='relu', padding='same')(x)\n    x = layers.BatchNormalization()(x)\n    x=layers.Conv2DTranspose(64, (3,3), strides=2, activation='relu', padding='same')(x)\n    x = layers.BatchNormalization()(x)\n    decoder_output=layers.Conv2D(1, (3, 3), activation='sigmoid', padding='same')(x)\n    decoder_output=Resizing(input_shape[0], input_shape[1])(decoder_output)\n    decoder=models.Model(decoder_input, decoder_output)\n\n    # Autoencoder\n    autoencoder_output=decoder(encoder(encoder_input))\n    autoencoder=models.Model(encoder_input, autoencoder_output)\n\n    return autoencoder, encoder, decoder","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:21:42.718051Z","iopub.execute_input":"2025-11-25T13:21:42.718254Z","iopub.status.idle":"2025-11-25T13:21:42.726648Z","shell.execute_reply.started":"2025-11-25T13:21:42.718238Z","shell.execute_reply":"2025-11-25T13:21:42.725817Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Training logic","metadata":{}},{"cell_type":"code","source":"# Optimizer\noptimizer=keras.optimizers.Adam(learning_rate = lr)\n\n# Callbacks\nreduce_lr = keras.callbacks.ReduceLROnPlateau(factor = 0.5, patience = patience / 2, verbose=1)\nearly_stop = keras.callbacks.EarlyStopping(patience = patience, verbose = 1, restore_best_weights = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:21:42.728351Z","iopub.execute_input":"2025-11-25T13:21:42.728652Z","iopub.status.idle":"2025-11-25T13:21:43.598083Z","shell.execute_reply.started":"2025-11-25T13:21:42.728611Z","shell.execute_reply":"2025-11-25T13:21:43.597265Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Compile model","metadata":{}},{"cell_type":"code","source":"autoencoder, encoder, decoder=conv_autoencoder(input_shape=(target_shape[0], target_shape[1], 1), latent_dim=latent_dim)\nautoencoder.compile(optimizer=optimizer, loss='mse')\nautoencoder.summary()\nencoder.summary()\ndecoder.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:21:43.599002Z","iopub.execute_input":"2025-11-25T13:21:43.599329Z","iopub.status.idle":"2025-11-25T13:21:44.956451Z","shell.execute_reply.started":"2025-11-25T13:21:43.599303Z","shell.execute_reply":"2025-11-25T13:21:44.955756Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Training","metadata":{}},{"cell_type":"code","source":"history = autoencoder.fit(\n    train_dataset,\n    # images,\n    # images,\n    epochs=epochs,\n    # batch_size = batch_size,\n    # validation_split = 0.2,\n    callbacks = [early_stop, reduce_lr],\n    verbose=1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:21:44.957186Z","iopub.execute_input":"2025-11-25T13:21:44.957378Z","iopub.status.idle":"2025-11-25T13:37:37.460385Z","shell.execute_reply.started":"2025-11-25T13:21:44.957361Z","shell.execute_reply":"2025-11-25T13:37:37.459594Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualization","metadata":{}},{"cell_type":"code","source":"def visualize_latent_features(decoder, latent_dim, n_cols=8):\n\n    # n_rows=int(np.ceil(latent_dim/n_cols))\n    # plt.figure(figsize=(n_cols*2, n_rows*2))\n    # for i in range(latent_dim):\n    #     latent_vector=np.zeros((1, latent_dim))\n    #     latent_vector[0, i] = 10\n    #     latent_img = decoder.predict(latent_vector)\n    #     latent_img = latent_img.squeeze()\n    #     plt.subplot(n_rows, n_cols, i+1)\n    #     plt.imshow(latent_img, cmap ='viridis')\n    #     plt.title(f\"Feature {i+1}\")\n    #     plt.axis('off')\n    # plt.tight_layout()\n    # plt.show()\n\n    n_rows=int(np.ceil(latent_dim/n_cols))\n    latent_matrix = np.zeros((latent_dim, latent_dim), dtype=np.float32)\n    for i in range(latent_dim):\n        latent_matrix[i, i] = 10\n    decoded = decoder.predict(latent_matrix)\n    plt.figure(figsize=(n_cols * 2, n_rows * 2))\n    for i in range(latent_dim):\n        img = decoded[i].squeeze()\n        plt.subplot(n_rows, n_cols, i + 1)\n        plt.imshow(img, cmap='viridis')\n        plt.title(f\"Feature {i+1}\")\n        plt.axis('off')\n    plt.tight_layout()\n    plt.show()\n    print(decoded.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:37:37.461505Z","iopub.execute_input":"2025-11-25T13:37:37.461742Z","iopub.status.idle":"2025-11-25T13:37:37.467703Z","shell.execute_reply.started":"2025-11-25T13:37:37.461719Z","shell.execute_reply":"2025-11-25T13:37:37.466838Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualizing latent space\nvisualize_latent_features(decoder, latent_dim, 8)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:37:37.468406Z","iopub.execute_input":"2025-11-25T13:37:37.468608Z","iopub.status.idle":"2025-11-25T13:37:53.394717Z","shell.execute_reply.started":"2025-11-25T13:37:37.468593Z","shell.execute_reply":"2025-11-25T13:37:53.393919Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Testing all-zero image\ndark = np.zeros([1, target_shape[0], target_shape[1], 1])\nlatent_dark = encoder.predict(dark)\nrec_dark = decoder.predict(latent_dark)\nzero_in = decoder.predict(np.zeros([1, latent_dim]))\nprint(latent_dark)\n# print(zero_in)\n\nplt.figure(figsize=(12, 5))\nplt.subplot(1, 3, 1)\nplt.imshow(dark[0, :, :, 0], cmap='viridis')\nplt.title('Original')\nplt.subplot(1, 3, 2)\nplt.imshow(rec_dark[0, :, :, 0], cmap='viridis')\nplt.title('Reconstructed')\nplt.subplot(1, 3, 3)\nplt.imshow(zero_in[0, :, :, 0], cmap='viridis')\nplt.title('Zero latent vector')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:48:33.744852Z","iopub.execute_input":"2025-11-25T13:48:33.745606Z","iopub.status.idle":"2025-11-25T13:48:34.431144Z","shell.execute_reply.started":"2025-11-25T13:48:33.745581Z","shell.execute_reply":"2025-11-25T13:48:34.430523Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Clustering","metadata":{}},{"cell_type":"markdown","source":"## Feature extraction","metadata":{}},{"cell_type":"code","source":"def feature_extraction(spectrograms, encoder, n_clusters=24):\n    \n    X_features = spectrograms.reshape(-1, target_shape[0], target_shape[1], 1)\n    latent_features = encoder.predict(X_features, verbose=0)\n    print(f\"Shape of latent_features: {latent_features.shape}\")\n    normalized_latent_features = StandardScaler().fit_transform(latent_features)\n    kmeans=KMeans(\n        n_clusters=n_clusters,\n        n_init=10,\n        random_state=random_seed,\n        verbose=1\n    )\n    cluster_labels=kmeans.fit_predict(normalized_latent_features)\n    return cluster_labels, latent_features, kmeans","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:37:54.369645Z","iopub.execute_input":"2025-11-25T13:37:54.369944Z","iopub.status.idle":"2025-11-25T13:37:54.374585Z","shell.execute_reply.started":"2025-11-25T13:37:54.369926Z","shell.execute_reply":"2025-11-25T13:37:54.373850Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# K-means clustering\ncluster_labels, latent_features, kmeans=feature_extraction(images, encoder, num_clusters)\nprint(f\"Found {len(np.unique(cluster_labels))} clusters.\")\n\n# Printing cluster numbers\ncluster_counts=np.bincount(cluster_labels)\nprint(\"Number of files in each cluster:\")\nfor i, count in enumerate(cluster_counts):\n    print(f\"Cluster {i}:\\t{count}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:37:54.375324Z","iopub.execute_input":"2025-11-25T13:37:54.375554Z","iopub.status.idle":"2025-11-25T13:37:55.524144Z","shell.execute_reply.started":"2025-11-25T13:37:54.375538Z","shell.execute_reply":"2025-11-25T13:37:55.521675Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualization","metadata":{}},{"cell_type":"code","source":"def find_process_files(consistent_df, train_path):\n    \n    spectrograms = []\n    valid_recording_ids = []\n    valid_species_labels = []\n\n    # Get list of .flac files\n    flac_files = [f for f in os.listdir(train_path) if f.endswith('.flac')]\n    print(f\"Number of .flac files: {len(flac_files)}\")\n\n    # Mapping\n    flac_mapping = {}\n    for flac_file in flac_files:\n        recording_id = os.path.splitext(flac_file)[0]\n        flac_mapping[recording_id] = os.path.join(train_path, flac_file)\n\n    # Processing files\n    for idx, row in tqdm(consistent_df.iterrows(), total=len(consistent_df), desc=\"Processing files\"):\n        recording_id = row['recording_id']\n        species_id = row['species_id']\n        \n        if recording_id in flac_mapping:\n            flac_path = flac_mapping[recording_id]\n            spectrogram = spec_gen(flac_path)\n            spectrograms.append(spectrogram)\n            valid_recording_ids.append(recording_id)\n            valid_species_labels.append(species_id)\n        else:\n            print(f\".flac file not found for: {recording_id}\")\n\n    spectrograms_array = np.array(spectrograms)\n    print(f\"Generated spectrograms: {len(spectrograms_array)}\")\n    print(f\"Spectrogram shape: {spectrograms_array.shape}\")\n    \n    return spectrograms_array, valid_recording_ids, valid_species_labels","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:49:22.256927Z","iopub.execute_input":"2025-11-25T13:49:22.257210Z","iopub.status.idle":"2025-11-25T13:49:22.264306Z","shell.execute_reply.started":"2025-11-25T13:49:22.257190Z","shell.execute_reply":"2025-11-25T13:49:22.263343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def true_label_vis(recording_id, cluster_labels, true_labels, latent_features):\n\n    tsne = TSNE(n_components=2, random_state=random_seed, perplexity=30)\n    X_tsne = tsne.fit_transform(latent_features)\n\n    # Create plot\n    colors=plt.cm.tab20.colors+plt.cm.tab10.colors[:4]\n    cmap=plt.matplotlib.colors.ListedColormap(colors)\n\n    plt.figure(figsize=(14, 10))\n    scatter = plt.scatter(X_tsne[:, 0], X_tsne[:, 1], \n                        c=cluster_labels, \n                        cmap=cmap,\n                        alpha=0.7,\n                        linewidth=0.5)\n    for i, (x, y) in enumerate(X_tsne):\n        plt.annotate(str(species_labels[i]), \n                    (x, y), \n                    fontsize=8,\n                    fontweight='bold',\n                    alpha=0.8)    \n    plt.colorbar(scatter, label='Clusters')\n    plt.title('t-SNE visualization')\n    plt.xlabel('t-SNE component 1')\n    plt.ylabel('t-SNE component 2')\n\n    plt.tight_layout()\n    plt.show()\n    \n    return X_tsne","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:49:24.693801Z","iopub.execute_input":"2025-11-25T13:49:24.694278Z","iopub.status.idle":"2025-11-25T13:49:24.700110Z","shell.execute_reply.started":"2025-11-25T13:49:24.694254Z","shell.execute_reply":"2025-11-25T13:49:24.699282Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Getting true positive recording ids\ndf=pd.read_csv(tp_label_csv_path)\nmulti_species=df.groupby('recording_id')['species_id'].nunique()\nconflicting_ids = multi_species[multi_species > 1].index\nsame_species_ids = multi_species[multi_species == 1].index\n\nprint(f\"\\nRecording_ids with conflicting species: {len(conflicting_ids)}\")\nprint(f\"Recording_ids with consistent species: {len(same_species_ids)}\")\n\n# Consistent labeling only\nconsistent_df = df[df['recording_id'].isin(same_species_ids)].drop_duplicates(subset=['recording_id'])\nprint(f\"Consistent recording_ids after processing: {len(consistent_df)}\")\nprint(consistent_df)\n\nspectrograms, recording_ids, species_labels = find_process_files(consistent_df, train_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:49:27.569575Z","iopub.execute_input":"2025-11-25T13:49:27.570113Z","iopub.status.idle":"2025-11-25T13:49:31.403963Z","shell.execute_reply.started":"2025-11-25T13:49:27.570088Z","shell.execute_reply":"2025-11-25T13:49:31.402852Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cluster_labels, latent_features, kmeans = feature_extraction(spectrograms, encoder, num_clusters)\nprint(f\"Found {len(np.unique(cluster_labels))} clusters.\")\n\nX_tsne = true_label_vis(recording_ids, cluster_labels, species_labels, latent_features)\ncluster_counts = np.bincount(cluster_labels)\nprint(\"\\nNumber of files in each cluster:\")\nfor i, count in enumerate(cluster_counts):\n    print(f\"Cluster {i}:\\t{count} files\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:49:50.490139Z","iopub.execute_input":"2025-11-25T13:49:50.490666Z","iopub.status.idle":"2025-11-25T13:49:58.711924Z","shell.execute_reply.started":"2025-11-25T13:49:50.490641Z","shell.execute_reply":"2025-11-25T13:49:58.711131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def visualization(images, cluster_labels, latent_features):\n\n    # t-SNE visualization\n    tsne=TSNE(n_components=2, perplexity=30, random_state=random_seed, verbose=1)\n    features_tsne=tsne.fit_transform(latent_features)\n    plt.figure(figsize=(15, 5))\n    plt.subplot(1, 1, 1)\n    colors=plt.cm.tab20.colors+plt.cm.tab10.colors[:4]\n    cmap=plt.matplotlib.colors.ListedColormap(colors)\n    scatter=plt.scatter(features_tsne[:, 0], features_tsne[:, 1], c=cluster_labels, cmap=cmap, alpha=0.6)\n    plt.colorbar(scatter, ticks=range(num_clusters))\n    plt.title('t-SNE visualization of the clusters')\n    plt.xlabel('t-SNE 1')\n    plt.ylabel('t-SNE 2')\n    plt.show()\n\n    # Example spectrogram from each cluster\n    unique_clusters = np.unique(cluster_labels)\n    for i, c in enumerate(unique_clusters):\n        plt.subplot(4, 6, i+1)\n        cluster_indices = np.where(cluster_labels==c)[0]\n        if len(cluster_indices)>0:\n            plt.imshow(images[cluster_indices[0]])\n            plt.title(f'Cluster {c}')\n            plt.axis('off')\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:44:09.535637Z","iopub.execute_input":"2025-11-25T13:44:09.535900Z","iopub.status.idle":"2025-11-25T13:44:09.542120Z","shell.execute_reply.started":"2025-11-25T13:44:09.535877Z","shell.execute_reply":"2025-11-25T13:44:09.541561Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualizing cluster labeling\nvisualization(images, cluster_labels, latent_features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:44:09.543814Z","iopub.execute_input":"2025-11-25T13:44:09.544100Z","iopub.status.idle":"2025-11-25T13:44:14.788303Z","shell.execute_reply.started":"2025-11-25T13:44:09.544084Z","shell.execute_reply":"2025-11-25T13:44:14.787394Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Reconstruction","metadata":{}},{"cell_type":"code","source":"def segment_audio(file_path, segment_duration=3.0, sr = None):\n\n    audio, sr = librosa.load(file_path, sr = sr)\n    segment_length = int(segment_duration * sr)\n    segments = []\n    \n    for start in range(0, len(audio), segment_length):\n        end = start + segment_length\n        segment = audio[start:end]\n        if len(segment) < segment_length:\n            segment = np.pad(segment, (0, segment_length - len(segment)))\n        segments.append(segment)\n    \n    return segments, audio, sr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:44:14.789297Z","iopub.execute_input":"2025-11-25T13:44:14.789593Z","iopub.status.idle":"2025-11-25T13:44:14.795520Z","shell.execute_reply.started":"2025-11-25T13:44:14.789567Z","shell.execute_reply":"2025-11-25T13:44:14.794901Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def segments_to_specs(segments, sr):\n    \n    specs = []\n    for seg in segments:\n        S = librosa.feature.melspectrogram(y=seg, sr = sr, n_mels=128)\n        S_db = librosa.power_to_db(S, ref=np.max)\n        S_norm = (S_db - S_db.min()) / (S_db.max() - S_db.min())\n        specs.append(S_norm)\n    return np.array(specs)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:44:14.796130Z","iopub.execute_input":"2025-11-25T13:44:14.796390Z","iopub.status.idle":"2025-11-25T13:44:14.807609Z","shell.execute_reply.started":"2025-11-25T13:44:14.796363Z","shell.execute_reply":"2025-11-25T13:44:14.807037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reconstruct_segments(specs, autoencoder):\n    \n    reconstructed = []\n    for S in specs:\n        h, w = S.shape\n        input_image = S.reshape(1, h, w, 1)\n        rec = autoencoder.predict(input_image, verbose=0)[0, :, :, 0]\n        reconstructed.append(rec)\n    rec_h, rec_w=reconstructed[0].shape\n    return np.array(reconstructed), rec_h, rec_w","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:44:14.808315Z","iopub.execute_input":"2025-11-25T13:44:14.808538Z","iopub.status.idle":"2025-11-25T13:44:14.821336Z","shell.execute_reply.started":"2025-11-25T13:44:14.808522Z","shell.execute_reply":"2025-11-25T13:44:14.820543Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def combine_specs(specs):\n    return np.concatenate(specs, axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:44:14.822105Z","iopub.execute_input":"2025-11-25T13:44:14.822483Z","iopub.status.idle":"2025-11-25T13:44:14.835724Z","shell.execute_reply.started":"2025-11-25T13:44:14.822457Z","shell.execute_reply":"2025-11-25T13:44:14.835019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reconstruct_full_spectrogram(file_path, autoencoder, segment_length=3.0):\n    \n    segments, audio, sr = segment_audio(file_path, segment_length)\n    specs = segments_to_specs(segments, sr)\n    rec_specs, rec_h, rec_w = reconstruct_segments(specs, autoencoder)\n    full_rec_spec = combine_specs(rec_specs)\n\n    S = librosa.feature.melspectrogram(y=audio, sr = sr, n_mels=128)\n    S_db = librosa.power_to_db(S, ref=np.max)\n    original = (S_db - S_db.min()) / (S_db.max() - S_db.min())\n    \n    plt.figure(figsize=(12, 5))\n    plt.subplot(1, 2, 1)\n    plt.imshow(original, cmap='viridis', aspect='auto') # aspect nem kell\n    plt.title(\"Original\")\n    plt.subplot(1, 2, 2)\n    plt.imshow(full_rec_spec, cmap='viridis', aspect='auto')\n    plt.title(\"Reconstructed\")\n    plt.tight_layout()\n    plt.show()\n\n    print(original.shape)\n    print(full_rec_spec.shape)\n    \n    return full_rec_spec, audio, sr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:44:14.836384Z","iopub.execute_input":"2025-11-25T13:44:14.836584Z","iopub.status.idle":"2025-11-25T13:44:14.843385Z","shell.execute_reply.started":"2025-11-25T13:44:14.836567Z","shell.execute_reply":"2025-11-25T13:44:14.842807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Reconstructing a random segment\ndef reconstruct_slice(file_path, segment_length, autoencoder, sr=None):\n    random_slice, sr=random_audio_segment(file_path, segment_length, sr)\n    random_slice_spec=segments_to_specs([random_slice], sr)\n    random_slice_rec, rec_h, rec_w=reconstruct_segments(random_slice_spec, autoencoder)\n    \n    S = librosa.feature.melspectrogram(y=random_slice, sr = sr, n_mels=128)\n    S_db = librosa.power_to_db(S, ref=np.max)\n    original = (S_db - S_db.min()) / (S_db.max() - S_db.min())\n    \n    plt.figure(figsize=(12, 5))\n    plt.subplot(1, 2, 1)\n    plt.imshow(original, cmap='viridis', aspect='auto')\n    plt.title(\"Original\")\n    plt.subplot(1, 2, 2)\n    plt.imshow(random_slice_rec[0], cmap='viridis', aspect='auto')\n    plt.title(\"Reconstructed\")\n    plt.tight_layout()\n    plt.show()\n\n    print(original.shape)\n    print(random_slice_rec.shape)\n    \n    return random_slice_rec, random_slice, sr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:44:14.845004Z","iopub.execute_input":"2025-11-25T13:44:14.845363Z","iopub.status.idle":"2025-11-25T13:44:14.858025Z","shell.execute_reply.started":"2025-11-25T13:44:14.845346Z","shell.execute_reply":"2025-11-25T13:44:14.857327Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Reconstructing example slice\nexample_file=random_files(train_path)[0]\nprint(example_file)\nrec_spec, audio, sr = reconstruct_slice(example_file, segment_length, autoencoder)\nAudio(audio, rate = sr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:51:01.774048Z","iopub.execute_input":"2025-11-25T13:51:01.774339Z","iopub.status.idle":"2025-11-25T13:51:02.289798Z","shell.execute_reply.started":"2025-11-25T13:51:01.774316Z","shell.execute_reply":"2025-11-25T13:51:02.289256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Reconstructing example file\nexample_file=random_files(train_path)[0]\nprint(example_file)\nrec_spec, audio, sr = reconstruct_full_spectrogram(example_file, autoencoder, segment_length)\nAudio(audio, rate = sr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T13:51:12.432297Z","iopub.execute_input":"2025-11-25T13:51:12.433126Z","iopub.status.idle":"2025-11-25T13:51:18.861945Z","shell.execute_reply.started":"2025-11-25T13:51:12.433096Z","shell.execute_reply":"2025-11-25T13:51:18.859701Z"}},"outputs":[],"execution_count":null}]}