{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":21669,"databundleVersionId":1692278,"sourceType":"competition"}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Global settings","metadata":{}},{"cell_type":"markdown","source":"## Import","metadata":{}},{"cell_type":"code","source":"#Data handling\nimport os\nimport pandas as pd\nfrom PIL import Image\nimport shutil\n\n# Randomization\nimport random\n\n# Audio handling\nimport librosa\nfrom IPython.display import Audio\nimport soundfile as sf\n\n# Visualization\nimport matplotlib.pyplot as plt\nfrom sklearn.manifold import TSNE\n\n# Feedback with progress bar\nfrom tqdm.notebook import tqdm\n\n# Math & Algorithms\nimport numpy as np\n\n# Model\nimport keras\nfrom keras import layers\nimport tensorflow as tf\nfrom tensorflow.keras import models\nfrom tensorflow.keras.layers import Resizing\n\n# Clustering\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.cluster import KMeans","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-08T15:21:59.094581Z","iopub.execute_input":"2025-12-08T15:21:59.095214Z","iopub.status.idle":"2025-12-08T15:22:17.899342Z","shell.execute_reply.started":"2025-12-08T15:21:59.095180Z","shell.execute_reply":"2025-12-08T15:22:17.898752Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Variable setting","metadata":{}},{"cell_type":"code","source":"# NN training parameters\nsegment_length = 0.8\nn_mels=128\nnum_batches=20\nnum_files = 5000\n\n# Initialize random number generation\nrandom_seed = 42\nrandom.seed(random_seed)\nrng = np.random.default_rng()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-08T15:22:17.905704Z","iopub.execute_input":"2025-12-08T15:22:17.906475Z","iopub.status.idle":"2025-12-08T15:22:18.169307Z","shell.execute_reply.started":"2025-12-08T15:22:17.906457Z","shell.execute_reply":"2025-12-08T15:22:18.168637Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Spectrogram generation","metadata":{}},{"cell_type":"code","source":"# Returns None if file cannot be read\ndef safe_load(path, sr):\n    try:\n        y, s = sf.read(path)\n        if sr is not None and s != sr:\n            y = librosa.resample(y, orig_sr=s, target_sr=sr)\n        return y\n    except Exception:\n        try:\n            y, _ = librosa.load(path, sr=sr)\n            return y\n        except Exception:\n            return None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-07T12:17:24.964573Z","iopub.execute_input":"2025-12-07T12:17:24.964888Z","iopub.status.idle":"2025-12-07T12:17:24.986770Z","shell.execute_reply.started":"2025-12-07T12:17:24.964867Z","shell.execute_reply":"2025-12-07T12:17:24.986092Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generates spectrogram from the audio - normalized\ndef create_spec(audio, n_mels):\n    S=librosa.power_to_db(librosa.feature.melspectrogram(y=audio, n_mels=n_mels), ref=np.max)\n    return (S-S.min())/(S.max()-S.min())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-08T15:22:18.170511Z","iopub.execute_input":"2025-12-08T15:22:18.170906Z","iopub.status.idle":"2025-12-08T15:22:18.179272Z","shell.execute_reply.started":"2025-12-08T15:22:18.170886Z","shell.execute_reply":"2025-12-08T15:22:18.178582Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Returns starting points of segments based on the given parameters\ndef create_slices(y_duration, segment_length, n_slices):\n    max_possible = int(y_duration // segment_length)\n    n_slices = min(n_slices, max_possible)\n    step=y_duration/n_slices\n    starts=[i*step for i in range(n_slices)]\n    return starts","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-07T12:17:25.002813Z","iopub.execute_input":"2025-12-07T12:17:25.003058Z","iopub.status.idle":"2025-12-07T12:17:25.014878Z","shell.execute_reply.started":"2025-12-07T12:17:25.003025Z","shell.execute_reply":"2025-12-07T12:17:25.014234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Creates spectrogram from a file\ndef gen_spec_from_file(file_path, n_slices, n_mels, sr, segment_length=None):\n    y=safe_load(file_path, sr)\n    if y is None:\n        return []\n    y_duration=librosa.get_duration(y=y, sr=sr)\n    specs=[]\n\n     # Whole file\n    if segment_length is None:\n        specs.append(create_spec(y, n_mels))\n        return specs\n\n    # Segmented\n    starts=create_slices(y_duration, segment_length, n_slices)\n    for s in starts:\n        start_sample=int(sr*s)\n        end_sample=start_sample+int(sr*segment_length)\n        segment=y[start_sample:end_sample]\n        specs.append(create_spec(segment, n_mels))\n    return specs","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-07T12:17:25.015602Z","iopub.execute_input":"2025-12-07T12:17:25.015832Z","iopub.status.idle":"2025-12-07T12:17:25.029417Z","shell.execute_reply.started":"2025-12-07T12:17:25.015809Z","shell.execute_reply":"2025-12-07T12:17:25.028814Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generates spectrograms from several files (and optionally saves them)\ndef gen_specs(source_path, n_files, n_slices, n_mels, sr, segment_length=None, save_path=None, files=None):\n    if files is None:\n        files = [f for f in os.scandir(source_path) if f.is_file()]\n        selected_files=files[:n_files]\n    else:\n        selected_files=files\n    specs=[]\n    \n    for f in tqdm(selected_files):\n        file_specs=gen_spec_from_file(f, n_slices, n_mels, sr, segment_length)\n        if len(file_specs)==0:\n            continue\n        # Saving spectrograms\n        if save_path is not None:\n            base=os.path.splitext(os.path.basename(f))[0]\n            for i, spec in enumerate(file_specs):\n                spec=(spec*255).astype(np.uint8)\n                spec=Image.fromarray(spec)\n                output=f\"{base}_{i}.png\"\n                spec.save(os.path.join(save_path, output))\n        else:\n            specs.extend(file_specs)\n    return specs","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-07T12:17:25.030008Z","iopub.execute_input":"2025-12-07T12:17:25.030686Z","iopub.status.idle":"2025-12-07T12:17:25.047292Z","shell.execute_reply.started":"2025-12-07T12:17:25.030668Z","shell.execute_reply":"2025-12-07T12:17:25.046623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Returns a random file from the given folder\ndef random_files(source_path, num_files=1):\n    all_files=os.listdir(source_path)\n    chosen_files=random.sample(all_files, num_files)\n    return [os.path.join(source_path, f) for f in chosen_files]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-07T12:17:25.047852Z","iopub.execute_input":"2025-12-07T12:17:25.048485Z","iopub.status.idle":"2025-12-07T12:17:25.061450Z","shell.execute_reply.started":"2025-12-07T12:17:25.048466Z","shell.execute_reply":"2025-12-07T12:17:25.060711Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Exploration","metadata":{}},{"cell_type":"markdown","source":"## File path","metadata":{}},{"cell_type":"code","source":"# Folder for storing generated spectrograms\nsave_path='/kaggle/working/spectrograms'\nlabeled_path='/kaggle/working/labeled'\npred_path='/kaggle/working/test'\nos.makedirs(save_path, exist_ok=True)\nos.makedirs(labeled_path, exist_ok=True)\nos.makedirs(pred_path, exist_ok=True)\n\n# Root data path for RainForest Species\ninput_path='/kaggle/input/rfcx-species-audio-detection'\n\n# Train and Test audio recordings data\ntrain_path=os.path.join(input_path, 'train')\ntest_path=os.path.join(input_path, 'test')\n\n# Labels\ntp_label_csv_path=os.path.join(input_path, 'train_tp.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-08T15:23:09.450311Z","iopub.execute_input":"2025-12-08T15:23:09.451009Z","iopub.status.idle":"2025-12-08T15:23:09.455650Z","shell.execute_reply.started":"2025-12-08T15:23:09.450972Z","shell.execute_reply":"2025-12-08T15:23:09.455085Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Number of files\ntraining_files=[f.path for f in os.scandir(train_path) if f.is_file()]\ntest_files=[f.path for f in os.scandir(test_path) if f.is_file()]\nnum_train_files=len(training_files)\nnum_test_files=len(test_files)\nnum_tp_rows=len(pd.read_csv(tp_label_csv_path))\nprint(f\"Number of training files: {num_train_files}\")\nprint(f\"Number of test files: {num_test_files}\")\nprint(f\"Number of labeled entries (true positive): {num_tp_rows}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-08T15:24:44.043084Z","iopub.execute_input":"2025-12-08T15:24:44.043408Z","iopub.status.idle":"2025-12-08T15:24:55.033148Z","shell.execute_reply.started":"2025-12-08T15:24:44.043386Z","shell.execute_reply":"2025-12-08T15:24:55.032351Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Shows an example file\nexample_file=random_files(train_path)[0]\nexample_audio, sr=librosa.core.load(example_file, sr=None)\nprint(f\"Sampling rate: {sr}\")\n\nop1=gen_spec_from_file(example_file, 1, n_mels, sr)\nop2=gen_spec_from_file(example_file, 1, n_mels, sr, 10)\nop3=gen_spec_from_file(example_file, 1, n_mels, sr, 3)\nop4=gen_spec_from_file(example_file, 1, n_mels, sr, 1)\n\nplt.figure(figsize=(15,10))\n\nplt.subplot(4, 1, 1)\nplt.imshow(op1[0], cmap='viridis', aspect='auto')\nplt.title('Whole file')\n\nplt.subplot(4, 1, 2)\nplt.imshow(op2[0], cmap='viridis', aspect='auto')\nplt.title('10 sec')\n\nplt.subplot(4, 1, 3)\nplt.imshow(op3[0], cmap='viridis', aspect='auto')\nplt.title('3 sec')\n\nplt.subplot(4, 1, 4)\nplt.imshow(op4[0], cmap='viridis', aspect='auto')\nplt.title('1 sec')\n\nplt.tight_layout()\nplt.savefig('/kaggle/working/spec_4.png', bbox_inches='tight')\nplt.show()\n\n\nAudio(example_file, rate=sr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-07T12:17:40.102018Z","iopub.execute_input":"2025-12-07T12:17:40.102338Z","iopub.status.idle":"2025-12-07T12:17:56.146453Z","shell.execute_reply.started":"2025-12-07T12:17:40.102320Z","shell.execute_reply":"2025-12-07T12:17:56.145374Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Creating dataset - unsupervised","metadata":{}},{"cell_type":"markdown","source":"## Batches","metadata":{}},{"cell_type":"code","source":"# Create batches from file paths\ndef make_batches(files, n_batches):\n    total=len(files)\n    batch_size=int(np.ceil(total/n_batches))\n\n    batches=[]\n    for i in range(0, total, batch_size):\n        batches.append(files[i:i+batch_size])\n    return batches","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T09:57:37.242286Z","iopub.execute_input":"2025-12-04T09:57:37.242766Z","iopub.status.idle":"2025-12-04T09:57:37.248400Z","shell.execute_reply.started":"2025-12-04T09:57:37.242747Z","shell.execute_reply":"2025-12-04T09:57:37.247700Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"batches=make_batches(training_files, num_batches)\nfor i, batch in enumerate(batches):\n    print(i, len(batch))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T09:57:37.249175Z","iopub.execute_input":"2025-12-04T09:57:37.249367Z","iopub.status.idle":"2025-12-04T09:57:37.294584Z","shell.execute_reply.started":"2025-12-04T09:57:37.249352Z","shell.execute_reply":"2025-12-04T09:57:37.293850Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Save spectrograms","metadata":{}},{"cell_type":"code","source":"for i in range(num_batches):\n    # Create images\n    gen_specs(train_path, num_files, 75, n_mels, sr, segment_length=segment_length, save_path=save_path, files=batches[1])\n    # Create zip\n    zip_name=f\"/kaggle/working/spectrograms_batch_{i}\"\n    shutil.make_archive(zip_name, 'zip', '/kaggle/working/spectrograms')\n    # Delete existing images\n    for f in os.listdir(save_path):\n        path = os.path.join(save_path, f)\n        if os.path.isfile(path):\n            os.remove(path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T10:52:18.938647Z","iopub.execute_input":"2025-12-04T10:52:18.938928Z","iopub.status.idle":"2025-12-04T10:54:45.264345Z","shell.execute_reply.started":"2025-12-04T10:52:18.938907Z","shell.execute_reply":"2025-12-04T10:54:45.263524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Deletes .zip file\n# zip_path = \"/kaggle/working/spectrograms_batch_0.zip\"\n# if os.path.exists(zip_path):\n#     os.remove(zip_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T09:57:37.328029Z","iopub.execute_input":"2025-12-04T09:57:37.328204Z","iopub.status.idle":"2025-12-04T09:57:37.341379Z","shell.execute_reply.started":"2025-12-04T09:57:37.328191Z","shell.execute_reply":"2025-12-04T09:57:37.340654Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Showing an example image\nexample_path=random_files(save_path, 1)[0]\nexample_image = Image.open(example_path)\nplt.figure(figsize=(8, 6))\nplt.imshow(example_image, cmap='viridis')\nplt.title(\"Example spectrogram for training\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T10:55:19.660960Z","iopub.execute_input":"2025-12-04T10:55:19.661550Z","iopub.status.idle":"2025-12-04T10:55:19.840727Z","shell.execute_reply.started":"2025-12-04T10:55:19.661528Z","shell.execute_reply":"2025-12-04T10:55:19.840142Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Creating dataset - labeled","metadata":{}},{"cell_type":"markdown","source":"## Reading csv","metadata":{}},{"cell_type":"code","source":"df=pd.read_csv(tp_label_csv_path)\ndf['t_length'] = df['t_max'] - df['t_min']\nlengths = df['t_length']\nprint(f\"Length of labeled audio segments: {lengths.mean():.2f}±{lengths.std():.2f} ({lengths.min():.2f}-{lengths.max():.2f}) [s]\")\ndf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-07T12:17:56.147274Z","iopub.execute_input":"2025-12-07T12:17:56.147689Z","iopub.status.idle":"2025-12-07T12:17:56.182725Z","shell.execute_reply.started":"2025-12-07T12:17:56.147671Z","shell.execute_reply":"2025-12-07T12:17:56.182089Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Generate labeled images","metadata":{}},{"cell_type":"code","source":"# Generates and saves spectrograms from a given audio file\ndef gen_and_save_spec_labeled(audio_folder_path, recording_id, species_id,\n                    time_min, time_max, segment_length, n_mels, save_path):\n\n    # Load the audio\n    file_path = os.path.join(audio_folder_path, recording_id + '.flac')\n    audio, sr = librosa.core.load(file_path, sr=None)\n\n    # Number of segments in time interval\n    slice_length=time_max-time_min\n    num_segments=max(int(slice_length // segment_length), 1)\n\n    for i in range(num_segments):\n        # Find center time of the given segment\n        center = (time_min + i * (slice_length / (num_segments+1)))\n\n        # Find start and end sample of the given slice\n        start = int(max(center - segment_length/2, 0) * sr)\n        end = start + int(segment_length * sr)\n        if end > len(audio):\n            end = len(audio)\n            start = end - int(segment_length * sr)\n    \n        # Get the sliced audio\n        audio_short=audio[start:end]\n\n        # Generate spectrogram\n        S = create_spec(audio_short, n_mels)\n        S = (S * 255).astype(np.uint8)\n\n        \n        # For saving .png\n        filename = f'{species_id}_{recording_id}_{center:.2f}.png'\n        species_path=os.path.join(save_path, str(species_id))\n        os.makedirs(species_path, exist_ok=True)\n        save_path_final = os.path.join(species_path, filename)\n\n        # Saving image\n        image=Image.fromarray(S)\n        image.save(save_path_final)\n\n    return save_path_final, S, S.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-07T12:19:31.515320Z","iopub.execute_input":"2025-12-07T12:19:31.515595Z","iopub.status.idle":"2025-12-07T12:19:31.522611Z","shell.execute_reply.started":"2025-12-07T12:19:31.515574Z","shell.execute_reply":"2025-12-07T12:19:31.522061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in tqdm(range(len(df))):\n    row = df.iloc[i]\n    \n    rec_id = row[\"recording_id\"]\n    species_id = row[\"species_id\"]\n    time_min = float(row[\"t_min\"])\n    time_max = float(row[\"t_max\"])\n\n    gen_and_save_spec_labeled(train_path, rec_id, species_id, time_min, time_max, segment_length=segment_length,n_mels=n_mels, save_path=labeled_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-07T12:19:33.696264Z","iopub.execute_input":"2025-12-07T12:19:33.696925Z","iopub.status.idle":"2025-12-07T12:23:10.897549Z","shell.execute_reply.started":"2025-12-07T12:19:33.696893Z","shell.execute_reply":"2025-12-07T12:23:10.896711Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create .zip for download\nzip_name=f\"/kaggle/working/spectrograms_labeled\"\nshutil.make_archive(zip_name, 'zip', '/kaggle/working/labeled')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-07T12:24:49.443264Z","iopub.execute_input":"2025-12-07T12:24:49.443561Z","iopub.status.idle":"2025-12-07T12:24:50.136700Z","shell.execute_reply.started":"2025-12-07T12:24:49.443538Z","shell.execute_reply":"2025-12-07T12:24:50.136153Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Summarise generated labeled training data","metadata":{}},{"cell_type":"code","source":"# Number of species\nspecies = [int(f) for f in os.listdir(labeled_path) if os.path.isdir(os.path.join(labeled_path, f))]\nspecies.sort()\nnum_species = len(species)\nprint(f\"Species: {num_species}\")\n\nsum_files=0\nprint(\"Number of files in each species folder:\")\nfor f in species:\n    path = os.path.join(labeled_path, str(f))\n    num_files = len([name for name in os.listdir(path) if os.path.isfile(os.path.join(path, name))])\n    sum_files += num_files\n    print(f\"{f}:\\t{num_files}\")\nprint(f\"Spectrograms: {sum_files}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-07T12:23:10.898707Z","iopub.execute_input":"2025-12-07T12:23:10.899007Z","iopub.status.idle":"2025-12-07T12:23:10.923629Z","shell.execute_reply.started":"2025-12-07T12:23:10.898988Z","shell.execute_reply":"2025-12-07T12:23:10.922919Z"}},"outputs":[],"execution_count":null}]}