{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":21669,"databundleVersionId":1692278,"sourceType":"competition"}],"dockerImageVersionId":31154,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Import","metadata":{}},{"cell_type":"code","source":"#Data handling\nimport os\nimport pandas as pd\nimport itertools\nfrom PIL import Image\n\n# Randomization\nimport random\n\n# Audio handling\n!pip install PySoundFile\nimport librosa\nfrom IPython.display import Audio\n\n# Visualization\nimport matplotlib.pyplot as plt\nfrom sklearn.manifold import TSNE\n\n# Feedback with progress bar\nfrom tqdm.notebook import tqdm\n\n# Math & Algorithms\nimport numpy as np\n\n# Model\nimport keras\nfrom keras import layers\nimport tensorflow as tf\nfrom tensorflow.keras import models, layers\nfrom tensorflow.keras.layers import Resizing\n\n# Clustering\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.cluster import KMeans","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:18:40.796302Z","iopub.execute_input":"2025-11-18T14:18:40.796491Z","iopub.status.idle":"2025-11-18T14:19:02.289722Z","shell.execute_reply.started":"2025-11-18T14:18:40.796474Z","shell.execute_reply":"2025-11-18T14:19:02.289041Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Variable setting","metadata":{}},{"cell_type":"code","source":"# Initialize random number generation\nrandom_seed = 42\nrandom.seed(random_seed)\nrng = np.random.default_rng()\n\n# NN training parameters\n# target_shape=(256, 512)\nsegment_length = 1.5\noverlap=0.5\nlatent_dim = 256\nbatch_size = 32\nlr = 1e-3\npatience = 10\nepochs = 50\nnum_files = 1000 # number of files used for training\nnum_clusters = 24\n\n# Folder for storing generated spectrograms\nsave_path='/kaggle/working/spectrograms'\nos.makedirs(save_path, exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:43:19.058693Z","iopub.execute_input":"2025-11-18T14:43:19.058978Z","iopub.status.idle":"2025-11-18T14:43:19.064248Z","shell.execute_reply.started":"2025-11-18T14:43:19.058958Z","shell.execute_reply":"2025-11-18T14:43:19.063391Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## File path","metadata":{}},{"cell_type":"code","source":"# Root data path for RainForest Species\ninput_path='/kaggle/input/rfcx-species-audio-detection'\n\n# Train and Test audio recordings data\ntrain_path=os.path.join(input_path, 'train')\ntest_path=os.path.join(input_path, 'test')\n\n# Labels\ntp_label_csv_path=os.path.join(input_path, 'train_tp.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:19:02.299530Z","iopub.execute_input":"2025-11-18T14:19:02.299824Z","iopub.status.idle":"2025-11-18T14:19:02.313280Z","shell.execute_reply.started":"2025-11-18T14:19:02.299801Z","shell.execute_reply":"2025-11-18T14:19:02.312667Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Number of files\nnum_train_files=len([f for f in os.listdir(train_path) if os.path.isfile(os.path.join(train_path, f))])\nnum_test_files=len([f for f in os.listdir(test_path) if os.path.isfile(os.path.join(test_path, f))])\nnum_tp_rows=len(pd.read_csv(tp_label_csv_path))\nprint(f\"Number of training files: {num_train_files}\")\nprint(f\"Number of test files: {num_test_files}\")\nprint(f\"Number of labeled entries (true positive): {num_tp_rows}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:19:02.314085Z","iopub.execute_input":"2025-11-18T14:19:02.314340Z","iopub.status.idle":"2025-11-18T14:19:26.194607Z","shell.execute_reply.started":"2025-11-18T14:19:02.314299Z","shell.execute_reply":"2025-11-18T14:19:26.193847Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Chooses files randomly from the given folder\ndef random_files(source_path, num_files=1):\n\n    all_files=os.listdir(source_path)\n    chosen_files=random.sample(all_files, num_files)\n    return [os.path.join(source_path, f) for f in chosen_files]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:19:26.196378Z","iopub.execute_input":"2025-11-18T14:19:26.196705Z","iopub.status.idle":"2025-11-18T14:19:26.201930Z","shell.execute_reply.started":"2025-11-18T14:19:26.196679Z","shell.execute_reply":"2025-11-18T14:19:26.201037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"random_file, sr = librosa.core.load(random_files(train_path)[0], sr = None)\nprint(sr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:19:26.202802Z","iopub.execute_input":"2025-11-18T14:19:26.203082Z","iopub.status.idle":"2025-11-18T14:19:38.760436Z","shell.execute_reply.started":"2025-11-18T14:19:26.203058Z","shell.execute_reply":"2025-11-18T14:19:38.759713Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data generation","metadata":{}},{"cell_type":"markdown","source":"### Short segment","metadata":{}},{"cell_type":"code","source":"# Returns random audio segment\ndef random_audio_segment(file_path, segment_length=3.0, sr = None):\n    \n    y, sr = librosa.load(file_path, sr = sr)\n    total_length = librosa.get_duration(y=y, sr = sr)\n\n    # Random start point\n    start_time = random.uniform(0, total_length-segment_length)\n    segment_samples = int(segment_length * sr)\n    start_sample = int(start_time * sr)\n    end_sample = start_sample+segment_samples\n    return y[start_sample:end_sample], sr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:19:38.762031Z","iopub.execute_input":"2025-11-18T14:19:38.762735Z","iopub.status.idle":"2025-11-18T14:19:38.767191Z","shell.execute_reply.started":"2025-11-18T14:19:38.762713Z","shell.execute_reply":"2025-11-18T14:19:38.766515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def spec_gen_short(file_path, sr = None, n_mels = 128, segment_length = 3):\n    \n    audio, sr = random_audio_segment(file_path, segment_length, sr)\n    S = librosa.feature.melspectrogram(y = audio, sr = sr, n_mels = n_mels)\n    S_db = librosa.power_to_db(S, ref=np.max)\n    S_norm = (S_db - S_db.min()) / (S_db.max() - S_db.min())\n    \n    return S_norm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:19:38.767937Z","iopub.execute_input":"2025-11-18T14:19:38.768130Z","iopub.status.idle":"2025-11-18T14:19:38.781427Z","shell.execute_reply.started":"2025-11-18T14:19:38.768114Z","shell.execute_reply":"2025-11-18T14:19:38.780436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def spec_save_short(source_path, save_path, num_files = 100, sr = None, segment_length = 3, n_mels = 128):\n    \"\"\"\n    Saves a random segment from the first n audio files from the\n    source path as a .png image to save_path.\n    \"\"\"\n\n    files = [f for f in os.scandir(source_path) if f.is_file()]\n    files = files[:num_files]\n    for f in tqdm(files):\n        output = os.path.join(save_path, os.path.splitext(f.name)[0]+\".png\")\n        spec = spec_gen_short(f.path, sr = sr, segment_length=segment_length)\n        spec = (spec*255).astype(np.uint8)\n        spec = Image.fromarray(spec, mode = 'L')\n        spec.save(output)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:19:38.782340Z","iopub.execute_input":"2025-11-18T14:19:38.782642Z","iopub.status.idle":"2025-11-18T14:19:38.794885Z","shell.execute_reply.started":"2025-11-18T14:19:38.782620Z","shell.execute_reply":"2025-11-18T14:19:38.794145Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Original","metadata":{}},{"cell_type":"code","source":"# Generates spectrogram from the given file\ndef spec_gen(file_path, target_shape, sr = None):\n    \"\"\"\n    Generates a spectogram from a given audio file, and reutrns the first\n    target_shape[1] pixel columns from it.\n    \"\"\"\n    \n    audio, sr = librosa.core.load(file_path)\n    S = librosa.feature.melspectrogram(y = audio, sr = sr, n_mels = target_shape[0]) # f_max, f_min \n    S_db = librosa.power_to_db(S, ref=np.max)\n\n    if S_db.shape[1]>target_shape[1]:\n        S_db=S_db[:, :target_shape[1]]\n    elif S_db.shape[1]<target_shape[1]:\n        pad_width=[(0, 0), (0, target_shape[1]-S_db.shape[1])]\n        S_db=np.pad(S_db, pad_width=pad_width, mode='constant')\n    \n    S_norm = (S_db - S_db.min()) / (S_db.max() - S_db.min())\n    \n    return S_norm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:19:38.795717Z","iopub.execute_input":"2025-11-18T14:19:38.795979Z","iopub.status.idle":"2025-11-18T14:19:38.807046Z","shell.execute_reply.started":"2025-11-18T14:19:38.795962Z","shell.execute_reply":"2025-11-18T14:19:38.806200Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generates and saves spectrograms from a given number of files\ndef spec_save(source_path, save_path, target_shape, num_files=100, sr = None):\n    \n    files=[f for f in os.scandir(source_path) if f.is_file()]\n    files=files[:num_files]\n    for f in tqdm(files):\n        output=os.path.join(save_path, os.path.splitext(f.name)[0]+\".png\")\n        spec=spec_gen(f.path, target_shape, sr)\n        spec=(spec*255).astype(np.uint8)\n        spec=Image.fromarray(spec, mode='L')\n        spec.save(output)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:19:38.808667Z","iopub.execute_input":"2025-11-18T14:19:38.808926Z","iopub.status.idle":"2025-11-18T14:19:38.822769Z","shell.execute_reply.started":"2025-11-18T14:19:38.808909Z","shell.execute_reply":"2025-11-18T14:19:38.822065Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# spec_save(train_path, save_path, target_shape, num_files=num_files, sr=sr)\nspec_save_short(train_path, save_path, num_files, segment_length = segment_length, n_mels=128)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:19:38.823452Z","iopub.execute_input":"2025-11-18T14:19:38.823695Z","iopub.status.idle":"2025-11-18T14:21:20.841829Z","shell.execute_reply.started":"2025-11-18T14:19:38.823673Z","shell.execute_reply":"2025-11-18T14:21:20.840967Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Autoencoder","metadata":{}},{"cell_type":"code","source":"def conv_autoencoder(input_shape, latent_dim=64):\n\n    # Encoder\n    encoder_input=layers.Input(shape=input_shape)\n    x=layers.Conv2D(128, (3, 3), activation='relu', padding='same')(encoder_input)\n    x=layers.BatchNormalization()(x)\n    x=layers.MaxPooling2D((2, 2), padding='same')(x)\n    x=layers.Dropout(0.2)(x)\n    \n    x=layers.Conv2D(64, (3, 3), activation='relu', padding='same')(x)\n    x=layers.BatchNormalization()(x)\n    x=layers.MaxPooling2D((2, 2), padding='same')(x)\n    x=layers.Dropout(0.2)(x)\n    \n    x=layers.Conv2D(32, (3, 3), activation='relu', padding='same')(x)\n    x=layers.BatchNormalization()(x)\n    \n    x_shape=x.shape[1:]\n    x_prod=np.prod(x_shape)\n\n    # Bottleneck\n    x=layers.Flatten()(x)\n    latent=layers.Dense(latent_dim, activation='relu')(x)\n    encoder=models.Model(encoder_input, latent)\n\n    # Decoder\n    decoder_input=layers.Input(shape=(latent_dim,))\n    x=layers.Dense(x_prod, activation='relu')(decoder_input)\n    x=layers.Reshape((x_shape))(x)\n\n    # x = layers.Conv2DTranspose(128, (3,3), strides=(2,2), activation='relu', padding='same')(x)\n    x=layers.Conv2DTranspose(16, (3,3), strides=2, activation='relu', padding='same')(x)\n    x = layers.BatchNormalization()(x)\n    x=layers.Conv2DTranspose(32, (3,3), strides=2, activation='relu', padding='same')(x)\n    x = layers.BatchNormalization()(x)\n    x=layers.Conv2DTranspose(64, (3,3), strides=2, activation='relu', padding='same')(x)\n    x = layers.BatchNormalization()(x)\n    # x=layers.Conv2D(16, (3,3), activation='relu', padding='same')(x)\n    # x=layers.UpSampling2D((2, 2))(x)\n    # x=layers.Conv2D(32, (3,3), activation='relu', padding='same')(x)\n    # x=layers.UpSampling2D((2, 2))(x)\n    # x=layers.Conv2D(64, (3,3), activation='relu', padding='same')(x)\n    # x=layers.UpSampling2D((2, 2))(x)\n    # x=layers.Conv2D(128, (3,3), activation='relu', padding='same')(x)\n    # x=layers.UpSampling2D((2, 2))(x)\n    decoder_output=layers.Conv2D(1, (3, 3), activation='sigmoid', padding='same')(x)\n    decoder_output=Resizing(input_shape[0], input_shape[1])(decoder_output)\n    decoder=models.Model(decoder_input, decoder_output)\n\n    # Autoencoder\n    autoencoder_output=decoder(encoder(encoder_input))\n    autoencoder=models.Model(encoder_input, autoencoder_output)\n\n    return autoencoder, encoder, decoder","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:35:00.581576Z","iopub.execute_input":"2025-11-18T14:35:00.581872Z","iopub.status.idle":"2025-11-18T14:35:00.592044Z","shell.execute_reply.started":"2025-11-18T14:35:00.581850Z","shell.execute_reply":"2025-11-18T14:35:00.591145Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Training logic","metadata":{}},{"cell_type":"code","source":"# Optimizer\noptimizer=keras.optimizers.Adam(learning_rate = lr)\n\n# Callbacks\nreduce_lr = keras.callbacks.ReduceLROnPlateau(factor = 0.5, patience = patience / 2, verbose=1)\nearly_stop = keras.callbacks.EarlyStopping(patience = patience, verbose = 1, restore_best_weights = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:35:03.328304Z","iopub.execute_input":"2025-11-18T14:35:03.328962Z","iopub.status.idle":"2025-11-18T14:35:03.337667Z","shell.execute_reply.started":"2025-11-18T14:35:03.328938Z","shell.execute_reply":"2025-11-18T14:35:03.336769Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Compile model","metadata":{}},{"cell_type":"code","source":"# Calculate input shape\nexample_path=os.path.join(save_path, os.listdir(save_path)[0])\nexample_image=Image.open(example_path).convert('L')\nplt.imshow(example_image, cmap='viridis')\n\nexample_array=np.array(example_image)\ntarget_shape=example_array.shape\n\nprint(f\"Example image path: {example_path}\")\nprint(f\"Target shape: {target_shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:35:05.024162Z","iopub.execute_input":"2025-11-18T14:35:05.024602Z","iopub.status.idle":"2025-11-18T14:35:05.310259Z","shell.execute_reply.started":"2025-11-18T14:35:05.024573Z","shell.execute_reply":"2025-11-18T14:35:05.309357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"autoencoder, encoder, decoder=conv_autoencoder(input_shape=(target_shape[0], target_shape[1], 1), latent_dim=latent_dim)\nautoencoder.compile(optimizer=optimizer, loss='mse')\nautoencoder.summary()\nencoder.summary()\ndecoder.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:35:07.929182Z","iopub.execute_input":"2025-11-18T14:35:07.929695Z","iopub.status.idle":"2025-11-18T14:35:08.089249Z","shell.execute_reply.started":"2025-11-18T14:35:07.929671Z","shell.execute_reply":"2025-11-18T14:35:08.088700Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Data Loader","metadata":{}},{"cell_type":"markdown","source":"## Data loading","metadata":{}},{"cell_type":"code","source":"def load_images(source_path):\n\n    images=[]\n    for f in os.listdir(source_path):\n        if f.endswith('.png'):\n            image_path = os.path.join(source_path, f)\n            image = Image.open(image_path).convert('L')\n            image_array = np.array(image)/255.0\n            images.append(image_array)\n    return np.array(images)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:36:04.567195Z","iopub.execute_input":"2025-11-18T14:36:04.567759Z","iopub.status.idle":"2025-11-18T14:36:04.571971Z","shell.execute_reply.started":"2025-11-18T14:36:04.567738Z","shell.execute_reply":"2025-11-18T14:36:04.571230Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# reshape -> adds the required channel dimension\nimages=load_images(save_path).reshape(-1, target_shape[0], target_shape[1], 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:35:17.349186Z","iopub.execute_input":"2025-11-18T14:35:17.349961Z","iopub.status.idle":"2025-11-18T14:35:17.995288Z","shell.execute_reply.started":"2025-11-18T14:35:17.349935Z","shell.execute_reply":"2025-11-18T14:35:17.994506Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(images.shape)\nplt.imshow(images[0], cmap='viridis')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:35:20.641190Z","iopub.execute_input":"2025-11-18T14:35:20.641480Z","iopub.status.idle":"2025-11-18T14:35:20.902203Z","shell.execute_reply.started":"2025-11-18T14:35:20.641460Z","shell.execute_reply":"2025-11-18T14:35:20.901361Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Training","metadata":{}},{"cell_type":"code","source":"class AudioDataset(keras.utils.Sequence):\n\n    # Initialization\n    def __init__(self, images, batch_size, shuffle=False, seed=None, **kwargs):\n\n        super().__init__()\n\n        self.images=images\n        self.batch_size=batch_size\n        self.shuffle=shuffle\n        self.seed=seed\n        self.rng=np.random.RandomState(seed) if seed is not None else np.random\n\n        # setting seeds\n        if seed is not None:\n            np.random.seed(seed)\n            tf.random.set_seed(seed)\n            random.seed(seed)\n\n        # detecting image shape\n        self.input_shape=self.images[0].shape\n        self.indices=np.arange(len(self.images))\n\n        self.end_of_epoch()\n\n\n    # Number of batches\n    def __len__(self):\n        return int(np.ceil(len(self.images)/self.batch_size))\n\n\n    # Creates a single batch\n    def __getitem__(self, index):\n        batch_idx=self.indices[index*self.batch_size:(index+1)*self.batch_size]\n        batch_images=[self.images[i] for i in batch_idx]\n        batch_images=np.stack(batch_images).astype(\"float32\")\n\n        if batch_images.ndim==3:\n            batch_images=np.expand_dims(batch_images, -1)\n\n        if batch_images.max()>1.0:\n            batch_images=batch_images/255.0\n\n        return batch_images, batch_images\n\n\n    # Shuffles files\n    def end_of_epoch(self):\n        if self.shuffle:\n            if self.seed is not None:\n                self.rng.shuffle(self.indices)\n            else:\n                np.random.shuffle(self.indices)\n\n\n    # Shape of an image\n    def image_shape(self):\n        return self.input_shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:36:08.334733Z","iopub.execute_input":"2025-11-18T14:36:08.335372Z","iopub.status.idle":"2025-11-18T14:36:08.342977Z","shell.execute_reply.started":"2025-11-18T14:36:08.335344Z","shell.execute_reply":"2025-11-18T14:36:08.342197Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset = AudioDataset(\n    images=images,\n    batch_size=batch_size,\n    shuffle=True,\n    seed=42\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:36:44.735260Z","iopub.execute_input":"2025-11-18T14:36:44.735908Z","iopub.status.idle":"2025-11-18T14:36:44.740107Z","shell.execute_reply.started":"2025-11-18T14:36:44.735884Z","shell.execute_reply":"2025-11-18T14:36:44.739297Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = autoencoder.fit(\n    # images,\n    # images,\n    train_dataset,\n    epochs = epochs,\n    # batch_size = batch_size,\n    # validation_split = 0.2,\n    callbacks = [early_stop, reduce_lr],\n    verbose=1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:37:28.092078Z","iopub.execute_input":"2025-11-18T14:37:28.092691Z","iopub.status.idle":"2025-11-18T14:42:08.879098Z","shell.execute_reply.started":"2025-11-18T14:37:28.092667Z","shell.execute_reply":"2025-11-18T14:42:08.878530Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature extraction","metadata":{}},{"cell_type":"code","source":"def feature_extraction(spectrograms, encoder, n_clusters=24):\n    \n    X_features = spectrograms.reshape(-1, target_shape[0], target_shape[1], 1)\n    latent_features = encoder.predict(X_features, verbose=0)\n    print(f\"Shape of latent_features: {latent_features.shape}\")\n    normalized_latent_features = StandardScaler().fit_transform(latent_features)\n    kmeans=KMeans(\n        n_clusters=n_clusters,\n        n_init=10,\n        random_state=random_seed,\n        verbose=1\n    )\n    cluster_labels=kmeans.fit_predict(normalized_latent_features)\n    return cluster_labels, latent_features, kmeans","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:29:20.835287Z","iopub.execute_input":"2025-11-18T14:29:20.835540Z","iopub.status.idle":"2025-11-18T14:29:20.840682Z","shell.execute_reply.started":"2025-11-18T14:29:20.835521Z","shell.execute_reply":"2025-11-18T14:29:20.839921Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cluster_labels, latent_features, kmeans=feature_extraction(images, encoder, num_clusters)\nprint(f\"Found {len(np.unique(cluster_labels))} clusters.\")\n\ncluster_counts=np.bincount(cluster_labels)\nprint(\"Number of files in each cluster:\")\nfor i, count in enumerate(cluster_counts):\n    print(f\"Cluster {i}:\\t{count}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:29:20.841430Z","iopub.execute_input":"2025-11-18T14:29:20.841596Z","iopub.status.idle":"2025-11-18T14:29:22.562938Z","shell.execute_reply.started":"2025-11-18T14:29:20.841584Z","shell.execute_reply":"2025-11-18T14:29:22.562202Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualization","metadata":{}},{"cell_type":"code","source":"def visualization(images, cluster_labels, latent_features):\n\n    # t-SNE visualization\n    tsne=TSNE(n_components=2, perplexity=30, random_state=random_seed, verbose=1)\n    features_tsne=tsne.fit_transform(latent_features)\n    plt.figure(figsize=(15, 5))\n    plt.subplot(1, 1, 1)\n    colors=plt.cm.tab20.colors+plt.cm.tab10.colors[:4]\n    cmap=plt.matplotlib.colors.ListedColormap(colors)\n    scatter=plt.scatter(features_tsne[:, 0], features_tsne[:, 1], c=cluster_labels, cmap=cmap, alpha=0.6)\n    plt.colorbar(scatter, ticks=range(num_clusters))\n    plt.title('t-SNE visualization of the clusters')\n    plt.xlabel('t-SNE 1')\n    plt.ylabel('t-SNE 2')\n    plt.show()\n\n    # Example spectrogram from each cluster\n    unique_clusters = np.unique(cluster_labels)\n    for i, c in enumerate(unique_clusters):\n        plt.subplot(4, 6, i+1)\n        cluster_indices = np.where(cluster_labels==c)[0]\n        if len(cluster_indices)>0:\n            plt.imshow(images[cluster_indices[0]])\n            plt.title(f'Cluster {c}')\n            plt.axis('off')\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:29:22.563732Z","iopub.execute_input":"2025-11-18T14:29:22.563944Z","iopub.status.idle":"2025-11-18T14:29:22.571528Z","shell.execute_reply.started":"2025-11-18T14:29:22.563928Z","shell.execute_reply":"2025-11-18T14:29:22.570603Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"visualization(images, cluster_labels, latent_features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:29:22.571980Z","iopub.execute_input":"2025-11-18T14:29:22.572688Z","iopub.status.idle":"2025-11-18T14:29:28.114814Z","shell.execute_reply.started":"2025-11-18T14:29:22.572668Z","shell.execute_reply":"2025-11-18T14:29:28.114031Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def visualize_latent_features(decoder, latent_dim, n_cols=8):\n\n    n_rows=int(np.ceil(latent_dim/n_cols))\n    plt.figure(figsize=(n_cols*2, n_rows*2))\n    for i in range(latent_dim):\n        latent_vector=np.zeros((1, latent_dim))\n        latent_vector[0, i] = 10\n        latent_img = decoder.predict(latent_vector)\n        latent_img = latent_img.squeeze()\n        plt.subplot(n_rows, n_cols, i+1)\n        plt.imshow(latent_img, cmap ='viridis')\n        plt.title(f\"Feature {i+1}\")\n        plt.axis('off')\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:29:28.115926Z","iopub.execute_input":"2025-11-18T14:29:28.116124Z","iopub.status.idle":"2025-11-18T14:29:28.121810Z","shell.execute_reply.started":"2025-11-18T14:29:28.116109Z","shell.execute_reply":"2025-11-18T14:29:28.120848Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dark = np.zeros([1, 128,141, 1])\nlatent_dark = encoder.predict(dark)\nrec_dark = decoder.predict(latent_dark)\nprint(latent_dark)\n\nplt.figure(figsize=(12, 5))\nplt.subplot(1, 2, 1)\nplt.imshow(dark[0, :, :, 0], cmap='viridis')\nplt.title('Original')\nplt.subplot(1, 2, 2)\nplt.imshow(rec_dark[0, :, :, 0], cmap='viridis')\nplt.title('Reconstructed')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:29:28.124485Z","iopub.execute_input":"2025-11-18T14:29:28.124736Z","iopub.status.idle":"2025-11-18T14:29:30.480555Z","shell.execute_reply.started":"2025-11-18T14:29:28.124707Z","shell.execute_reply":"2025-11-18T14:29:30.479855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"visualize_latent_features(decoder, latent_dim, 8)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:29:30.481463Z","iopub.execute_input":"2025-11-18T14:29:30.481748Z","iopub.status.idle":"2025-11-18T14:30:08.337484Z","shell.execute_reply.started":"2025-11-18T14:29:30.481724Z","shell.execute_reply":"2025-11-18T14:30:08.336209Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Reconstruction","metadata":{}},{"cell_type":"markdown","source":"### Short version","metadata":{}},{"cell_type":"code","source":"def segment_audio(file_path, segment_duration=3.0, overlap=0.5, sr = None):\n    \n    # audio, sr = librosa.load(file_path)\n    # segment_length = int(segment_duration * sr)\n    # hop_length = int(segment_length * (1 - overlap))\n    \n    # segments = []\n    # starts = []\n    # for start in range(0, len(audio) - segment_length + 1, hop_length):\n    #     end = start + segment_length\n    #     segment = audio[start:end]\n    #     segments.append(segment)\n    #     starts.append(start)\n    \n    # return np.array(segments), starts, sr   \n\n\n    audio, sr = librosa.load(file_path, sr = sr)\n    segment_length = int(segment_duration * sr)\n    segments = []\n    \n    for start in range(0, len(audio), segment_length):\n        end = start + segment_length\n        segment = audio[start:end]\n        if len(segment) < segment_length:\n            segment = np.pad(segment, (0, segment_length - len(segment)))\n        segments.append(segment)\n    \n    return segments, audio, sr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:30:08.338950Z","iopub.execute_input":"2025-11-18T14:30:08.339179Z","iopub.status.idle":"2025-11-18T14:30:08.347075Z","shell.execute_reply.started":"2025-11-18T14:30:08.339160Z","shell.execute_reply":"2025-11-18T14:30:08.346249Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def segments_to_specs(segments, sr):\n    \n    specs = []\n    for seg in segments:\n        S = librosa.feature.melspectrogram(y=seg, sr = sr, n_mels=128)\n        S_db = librosa.power_to_db(S, ref=np.max)\n        S_norm = (S_db - S_db.min()) / (S_db.max() - S_db.min())\n        specs.append(S_norm)\n    return np.array(specs)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:30:08.348168Z","iopub.execute_input":"2025-11-18T14:30:08.348617Z","iopub.status.idle":"2025-11-18T14:30:08.363115Z","shell.execute_reply.started":"2025-11-18T14:30:08.348583Z","shell.execute_reply":"2025-11-18T14:30:08.362258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reconstruct_segments(specs, autoencoder):\n    \n    reconstructed = []\n    for S in specs:\n        h, w = S.shape\n        input_image = S.reshape(1, h, w, 1)\n        rec = autoencoder.predict(input_image, verbose=0)[0, :, :, 0]\n        reconstructed.append(rec)\n    rec_h, rec_w=reconstructed[0].shape\n    return np.array(reconstructed), rec_h, rec_w","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:30:08.363989Z","iopub.execute_input":"2025-11-18T14:30:08.364247Z","iopub.status.idle":"2025-11-18T14:30:08.370877Z","shell.execute_reply.started":"2025-11-18T14:30:08.364223Z","shell.execute_reply":"2025-11-18T14:30:08.370073Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def combine_specs_with_overlap(rec_specs, overlap=0.5)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:30:08.371684Z","iopub.execute_input":"2025-11-18T14:30:08.371879Z","iopub.status.idle":"2025-11-18T14:30:08.382741Z","shell.execute_reply.started":"2025-11-18T14:30:08.371864Z","shell.execute_reply":"2025-11-18T14:30:08.381860Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def combine_specs(specs):\n    return np.concatenate(specs, axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:30:08.383558Z","iopub.execute_input":"2025-11-18T14:30:08.384179Z","iopub.status.idle":"2025-11-18T14:30:08.390465Z","shell.execute_reply.started":"2025-11-18T14:30:08.384151Z","shell.execute_reply":"2025-11-18T14:30:08.389758Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reconstruct_full_spectrogram(file_path, autoencoder, segment_length=3.0, overlap=0.5):\n    \n    segments, audio, sr = segment_audio(file_path, segment_length, overlap)\n    specs = segments_to_specs(segments, sr)\n    rec_specs, rec_h, rec_w = reconstruct_segments(specs, autoencoder)\n    # full_rec_spec=combine_specs_with_overlap(rec_specs, overlap)\n    full_rec_spec = combine_specs(rec_specs)\n\n    # original=spec_gen(file_path, (rec_h, rec_w), sr)\n    S = librosa.feature.melspectrogram(y=audio, sr = sr, n_mels=128)\n    S_db = librosa.power_to_db(S, ref=np.max)\n    original = (S_db - S_db.min()) / (S_db.max() - S_db.min())\n    \n    plt.figure(figsize=(12, 5))\n    plt.subplot(1, 2, 1)\n    plt.imshow(original, cmap='viridis', aspect='auto') # aspect nem kell\n    plt.title(\"Original\")\n    plt.subplot(1, 2, 2)\n    plt.imshow(full_rec_spec, cmap='viridis', aspect='auto')\n    plt.title(\"Reconstructed\")\n    plt.tight_layout()\n    plt.show()\n\n    print(original.shape)\n    print(full_rec_spec.shape)\n    \n    return full_rec_spec, audio, sr\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:30:08.391213Z","iopub.execute_input":"2025-11-18T14:30:08.391440Z","iopub.status.idle":"2025-11-18T14:30:08.399483Z","shell.execute_reply.started":"2025-11-18T14:30:08.391414Z","shell.execute_reply":"2025-11-18T14:30:08.398779Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Original version","metadata":{}},{"cell_type":"code","source":"def spec_reconstruct(file_path, autoencoder, segment_legth):\n\n    S_norm = spec_gen_short(file_path)\n    input_image = S_norm.reshape(1, S_norm.shape[0], S_norm.shape[1], 1)\n    reconstructed_image = autoencoder.predict(input_image, verbose=1)[0, :, :, 0]\n\n    plt.figure(figsize=(12, 5))\n    plt.subplot(1, 2, 1)\n    plt.imshow(S_norm, cmap='viridis')\n    plt.title('Original')\n    plt.subplot(1, 2, 2)\n    plt.imshow(reconstructed_image, cmap='viridis')\n    plt.title('Reconstructed')\n    plt.tight_layout()\n    plt.show()\n    print(target_shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:30:08.400410Z","iopub.execute_input":"2025-11-18T14:30:08.400962Z","iopub.status.idle":"2025-11-18T14:30:08.409763Z","shell.execute_reply.started":"2025-11-18T14:30:08.400945Z","shell.execute_reply":"2025-11-18T14:30:08.408843Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"example_file=random_files(train_path)[0]\n# spec_reconstruct(example_file, autoencoder, segment_length)\nrec_spec, audio, sr = reconstruct_full_spectrogram(example_file, autoencoder, segment_length, overlap)\n\nAudio(audio, rate = sr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-18T14:43:24.863584Z","iopub.execute_input":"2025-11-18T14:43:24.864147Z","iopub.status.idle":"2025-11-18T14:43:30.382877Z","shell.execute_reply.started":"2025-11-18T14:43:24.864124Z","shell.execute_reply":"2025-11-18T14:43:30.380680Z"}},"outputs":[],"execution_count":null}]}