{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":762,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":629,"modelId":52}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n! pip install pandarallel\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/yamnet/'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        pass\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"06865f8c-efd2-44a8-9292-7e9f22e0885a","_cell_guid":"c0492972-a264-4978-981e-127588cc3709","trusted":true,"execution":{"iopub.status.busy":"2025-05-04T18:33:35.148538Z","iopub.execute_input":"2025-05-04T18:33:35.148863Z","iopub.status.idle":"2025-05-04T18:33:42.700495Z","shell.execute_reply.started":"2025-05-04T18:33:35.148837Z","shell.execute_reply":"2025-05-04T18:33:42.699445Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels = {'brtpar1', '1139490', 'compau', 'chbant1', 'yehcar1', 'yecspi2', 'watjac1', 'grasal4', 'grbhaw1', \n          'yebfly1', 'neocor', '81930', 'spbwoo1', '64862', 'grepot1', 'ruther1', 'banana', 'whttro1', \n          '1462711', '42087', '66531', 'soulap1', 'amakin1', '41970', '65373', '714022', 'bafibi1', 'blcant4', \n          'rutjac1', 'plbwoo1', 'anhing', 'yehbla2', '21211', 'recwoo1', 'blbgra1', 'creoro1', 'shtfly1', \n          'amekes', 'blchaw1', '21116', '566513', 'bugtan', 'strcuc1', '1564122', '1462737', 'purgal2', \n          'socfly1', 'gohman1', 'gycwor1', 'bubwre1', 'blhpar1', '65336', 'solsan', '134933', '24292', \n          '42113', 'plukit1', 'savhaw1', 'sobtyr1', 'chfmac1', 'yebsee1', '66016', 'blbwre1', 'mastit1', \n          'smbani', 'whfant1', 'strfly1', 'roahaw', 'rumfly1', '476537', 'butsal1', 'bucmot3', 'colcha1', \n          'bobfly1', '67082', 'rebbla1', 'pavpig2', '1192948', 'whbman1', 'verfly', 'eardov1', 'norscr1', \n          'rinkin1', '67252', 'greibi1', 'greegr', 'cattyr', 'laufal1', 'trokin', 'grekis', 'crebob1', \n          'bubcur1', 'fotfly', 'palhor2', '476538', '24322', 'tropar', 'whwswa1', 'yercac1', '517119', \n          '24272', 'cocher1', 'labter1', 'bicwre1', 'compot1', 'olipic1', 'blcjay1', 'colara1', 'spepar1', \n          'cregua1', 'cargra1', '22976', 'plctan1', '715170', 'leagre', '22973', 'bkmtou1', 'yelori1', \n          'trsowl', 'strher', 'ragmac1', 'yeofly1', '548639', 'tbsfin1', '135045', '65344', 'bkcdon', \n          'stbwoo2', 'piepuf1', '868458', '963335', 'blctit1', 'saffin', 'rtlhum', 'royfly1', '66893', \n          'rutpuf1', 'linwoo1', 'wbwwre1', 'srwswa1', '126247', 'gretin1', 'grnkin', 'littin1', 'secfly1', \n          '41778', '528041', 'bbwduc', 'greani1', 'rubsee1', 'orcpar', 'rosspo1', 'yebela1', '47067', \n          'crcwoo1', '65349', 'snoegr', 'gybmar', 'thbeup1', '66578', 'turvul', 'rugdov', 'baymac', \n          'speowl1', 'cocwoo1', 'cotfly1', 'y00678', '65419', 'bobher1', '52884', '41663', '22333', \n          'piwtyr1', '21038', '787625', 'rufmot1', '65962', 'paltan1', '48124', '555142', '65547', \n          'crbtan1', '1194042', 'ywcpar', 'shghum1', 'cinbec1', 'thlsch3', '1346504', '555086', 'sahpar1', \n          'grysee1', 'blkvul', '523060', 'strowl1', 'whbant1', 'whmtyr1', '65448', 'ampkin1', 'whtdov', \n          'yectyr1', '42007', '46010', 'pirfly1', 'woosto', 'babwar', '50186'}\nprint(len(labels))","metadata":{"_uuid":"7d1582c7-b934-4308-bef1-2cb926448b35","_cell_guid":"d12fcec0-c322-427d-94c5-a35fb6545838","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-05-04T16:41:17.128358Z","iopub.execute_input":"2025-05-04T16:41:17.128724Z","iopub.status.idle":"2025-05-04T16:41:17.137781Z","shell.execute_reply.started":"2025-05-04T16:41:17.128690Z","shell.execute_reply":"2025-05-04T16:41:17.136773Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport os\nfrom IPython.display import Audio\nfrom scipy.io import wavfile\nimport soundfile as sf\nimport tensorflow as tf\nimport tensorflow_hub as hub\nimport warnings\nwarnings.filterwarnings('ignore')\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom sklearn.metrics import classification_report\nimport scipy.signal\nfrom tqdm import tqdm\nflag = 0","metadata":{"_uuid":"f2cccdd0-2b59-4596-8461-962c18bb52b7","_cell_guid":"7e2e0e2d-61f6-4f04-926a-6d3213cde37c","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-05-04T16:41:17.138773Z","iopub.execute_input":"2025-05-04T16:41:17.139077Z","iopub.status.idle":"2025-05-04T16:41:39.431775Z","shell.execute_reply.started":"2025-05-04T16:41:17.139048Z","shell.execute_reply":"2025-05-04T16:41:39.430681Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def ensure_sample_rate(original_sample_rate, waveform, desired_sample_rate=22000):\n    if original_sample_rate != desired_sample_rate:\n        desired_length = int(\n            round(float(len(waveform))/original_sample_rate * desired_sample_rate))\n        waveform = scipy.signal.resample(waveform, desired_length)\n    return desired_sample_rate, waveform","metadata":{"_uuid":"50f45f5a-29ad-4b38-842e-763412cd51ea","_cell_guid":"6bfaabde-4d94-47b0-b0d5-ca75ba8e961e","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-05-04T16:41:39.433041Z","iopub.execute_input":"2025-05-04T16:41:39.433860Z","iopub.status.idle":"2025-05-04T16:41:39.438983Z","shell.execute_reply.started":"2025-05-04T16:41:39.433817Z","shell.execute_reply":"2025-05-04T16:41:39.437785Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def read_audio(filename):\n    wav_data, sample_rate = sf.read(file=filename, dtype=np.int16)\n    if len(wav_data.shape) > 1:\n        wav_data = np.mean(wav_data, axis=1)\n    sample_rate, wav_data = ensure_sample_rate(sample_rate, wav_data)\n    return sample_rate, wav_data","metadata":{"_uuid":"815b5ee9-300a-46c7-a58f-b34ef3e1b410","_cell_guid":"864fe43d-6dc2-4b7f-9aaf-eae3ca7506c0","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-05-04T16:41:39.439921Z","iopub.execute_input":"2025-05-04T16:41:39.440248Z","iopub.status.idle":"2025-05-04T16:41:39.544591Z","shell.execute_reply.started":"2025-05-04T16:41:39.440214Z","shell.execute_reply":"2025-05-04T16:41:39.543725Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# Define the root directory containing the audio files\nroot_dir = '/kaggle/input/birdclef-2025/train_audio' # Adjust if your path is different\n\n# List to store the data for the DataFrame\naudio_data_list = []\n\n# Walk through the directory structure\nfor dirname, _, filenames in os.walk(root_dir):\n    # Skip the root directory itself if it doesn't contain class folders directly\n    # (Adjust this condition if your structure is different)\n    if dirname == root_dir:\n        continue\n\n    # Extract the class name (subdirectory name)\n    # os.path.basename gets the last part of the directory path\n    class_name = os.path.basename(dirname)\n\n    # Iterate through files in the current directory\n    for filename in filenames:\n        # Construct the full path to the audio file\n        full_path = os.path.join(dirname, filename)\n\n        # Append the file path and its class to the list\n        audio_data_list.append([full_path, class_name])\n        # You can remove the print statement if you don't need it anymore\n        # print(full_path) # Optional: print the path as it's processed\n\n# Create the Pandas DataFrame\naudio_dataframe = pd.DataFrame(audio_data_list, columns=[\"audio_path\", \"class\"])\n\n# Display the first few rows of the DataFrame (optional)\nprint(audio_dataframe.head())\n\n# Display the shape of the DataFrame (optional)\nprint(f\"\\nDataFrame shape: {audio_dataframe.shape}\")","metadata":{"_uuid":"4121086d-c342-4272-89c9-d3dc8774759d","_cell_guid":"1dca9713-a909-4e49-b6b0-82330ce3d220","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-05-04T16:41:39.546710Z","iopub.execute_input":"2025-05-04T16:41:39.546972Z","iopub.status.idle":"2025-05-04T16:42:26.904216Z","shell.execute_reply.started":"2025-05-04T16:41:39.546952Z","shell.execute_reply":"2025-05-04T16:42:26.903241Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = audio_dataframe.iloc[0]\n\nfor i in audio_dataframe['class'].unique():\n    df_sub = audio_dataframe[audio_dataframe['class'] == i]\n    num_rows = 3\n    # print(len(df_sub.index))\n    if len(df_sub.index) < num_rows:\n        num_rows = len(df_sub.index)\n    # print(num_rows)\n    df_sub = audio_dataframe[audio_dataframe['class'] == i].iloc[:num_rows]\n    # print(df_sub)\n    df = pd.concat([df,df_sub])\ndf['class']\n\n# flag = 0","metadata":{"_uuid":"e22d37fd-03e9-47dd-85d6-0df6a5124199","_cell_guid":"e34eed76-cb7b-440e-9833-13cee54e7eaa","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-05-04T16:42:26.905600Z","iopub.execute_input":"2025-05-04T16:42:26.905872Z","iopub.status.idle":"2025-05-04T16:42:27.989455Z","shell.execute_reply.started":"2025-05-04T16:42:26.905851Z","shell.execute_reply":"2025-05-04T16:42:27.988707Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport concurrent.futures\nfrom tqdm.auto import tqdm\nimport time # Just for mock function\nimport os # To get cpu count\nif flag == 0:\n    \n    # Get the list of paths to process\n    paths_to_process = df['audio_path'].tolist()\n\n    # Determine number of workers (e.g., number of CPU cores)\n    # Adjust based on memory constraints and task type (CPU vs I/O bound)\n    num_workers = os.cpu_count()\n    print(f\"Using {num_workers} workers.\")\n\n    results = [None] * len(paths_to_process) # Preallocate results list\n\n    print(\"Starting parallel processing with concurrent.futures...\")\n    # Use ProcessPoolExecutor for CPU-bound tasks\n    with concurrent.futures.ProcessPoolExecutor(max_workers=num_workers) as executor:\n        # Use executor.map to apply the function in parallel\n        # Wrap executor.map with tqdm for a progress bar\n        # executor.map preserves the order of the input iterable\n        future_to_path = {executor.submit(read_audio, path): i for i, path in enumerate(paths_to_process)}\n\n        for future in tqdm(concurrent.futures.as_completed(future_to_path), total=len(paths_to_process), desc=\"Reading audio files\"):\n            index = future_to_path[future]\n            try:\n                results[index] = future.result()\n            except Exception as exc:\n                print(f'Path at index {index} generated an exception: {exc}')\n                results[index] = None # Or some other error indicator\n    flag = 1\n    print(\"Finished parallel processing.\")\n\n    # Assign the results back to the DataFrame\n    df['audio_data'] = results\n    df.to_pickle('/kaggle/working/df.pk1')\n    print(df.head())","metadata":{"_uuid":"a8b30ee5-27f6-4929-8939-116228480d8e","_cell_guid":"977fe82e-9a6e-4262-858d-9bdf56e4bd1a","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-05-04T16:42:27.990304Z","iopub.execute_input":"2025-05-04T16:42:27.990546Z","iopub.status.idle":"2025-05-04T16:45:21.038110Z","shell.execute_reply.started":"2025-05-04T16:42:27.990516Z","shell.execute_reply":"2025-05-04T16:45:21.036854Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import re\n\ndef audio_data_unfucker(fucked_tuple: tuple) -> list:\n        assert(type(fucked_tuple[1]) != int)\n        return fucked_tuple[1]","metadata":{"_uuid":"c4191425-c246-40c9-9374-e5ba95ff31ef","_cell_guid":"a0870745-efe2-43a5-84ee-eb84e58b272d","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-05-04T16:45:21.039631Z","iopub.execute_input":"2025-05-04T16:45:21.040053Z","iopub.status.idle":"2025-05-04T16:45:21.046625Z","shell.execute_reply.started":"2025-05-04T16:45:21.040015Z","shell.execute_reply":"2025-05-04T16:45:21.045570Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if flag == 1:\n    pass\nelse:\n    flag = 1\ndf = pd.read_pickle('/kaggle/working/df.pk1')\ndf = df.iloc[2:]\nprint(df.columns)\ndf['audio_data'] = df['audio_data'].apply(audio_data_unfucker)","metadata":{"_uuid":"e5124c66-5846-45c9-b0de-c024b252b875","_cell_guid":"4d65d6a6-763d-400d-b9ed-ba0683348fda","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-05-04T16:45:21.047514Z","iopub.execute_input":"2025-05-04T16:45:21.047723Z","iopub.status.idle":"2025-05-04T16:45:23.917421Z","shell.execute_reply.started":"2025-05-04T16:45:21.047704Z","shell.execute_reply":"2025-05-04T16:45:23.916607Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['audio_data']","metadata":{"_uuid":"5858ca67-8a01-4f2a-b6ac-a69d9418105a","_cell_guid":"4596ff4f-2490-45db-ab98-d1d2d0bdbf5b","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-05-04T16:45:23.918236Z","iopub.execute_input":"2025-05-04T16:45:23.918534Z","iopub.status.idle":"2025-05-04T16:45:23.929013Z","shell.execute_reply.started":"2025-05-04T16:45:23.918502Z","shell.execute_reply":"2025-05-04T16:45:23.928250Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"audio_data = df['audio_data'].copy()","metadata":{"_uuid":"3a81fb7f-2eed-4a4d-a30d-989cb47f221d","_cell_guid":"a645c135-06d9-4af0-abf6-31e6fdf726f5","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-05-04T16:45:23.929719Z","iopub.execute_input":"2025-05-04T16:45:23.929971Z","iopub.status.idle":"2025-05-04T16:45:25.444367Z","shell.execute_reply.started":"2025-05-04T16:45:23.929952Z","shell.execute_reply":"2025-05-04T16:45:25.443278Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"max_len = 1000  # Your desired fixed length\nprocessed_audio = []\nprocessed_labels = []\n\nprint(f\"Processing audio data with max_len = {max_len}\")\nprint(f\"Original DataFrame has {len(df)} rows.\")\n\n# Iterate through the DataFrame rows using itertuples (generally faster)\n# 'Index' is the DataFrame index, 'audio_data' and 'label' match column names\nfor row in tqdm(df.itertuples(), total=len(df), desc=\"Processing Audio\"):\n    # --- Get data for the current row ---\n    # Use getattr for robustness if column names might vary slightly\n    # print(row)\n    original_seq = getattr(row, 'audio_data')\n    original_label = getattr(row, '_3') # Fetch the label\n\n    # Ensure the sequence is a numpy array with the desired dtype\n    # This handles cases where 'audio_data' might contain lists\n    seq = np.array(original_seq, dtype=np.float32)\n    current_len = len(seq)\n\n    # --- Case 1: Sequence is shorter than max_len ---\n    if current_len < max_len:\n        padding_needed = max_len - current_len\n        # Use np.pad which can be cleaner for padding\n        padded_seq = np.pad(seq, (0, padding_needed), mode='constant', constant_values=0)\n        processed_audio.append(padded_seq)\n        processed_labels.append(original_label) # Append the original label\n\n    # --- Case 2: Sequence is exactly max_len ---\n    elif current_len == max_len:\n        processed_audio.append(seq) # No padding or truncation needed\n        processed_labels.append(original_label) # Append the original label\n\n    # --- Case 3: Sequence is longer than max_len ---\n    else:\n        # Calculate how many full chunks we can get\n        num_full_chunks = current_len // max_len\n\n        # Iterate through the sequence, extracting non-overlapping chunks\n        for i in range(num_full_chunks):\n            start_index = i * max_len\n            end_index = start_index + max_len\n            chunk = seq[start_index:end_index]\n            processed_audio.append(chunk)\n            # Append the SAME original label for EACH chunk\n            processed_labels.append(original_label)\n\n        # --- Handle the remainder (the part left over after full chunks) ---\n        remainder_len = current_len % max_len\n        if remainder_len > 0:\n            # Extract the remainder\n            remainder_start_index = num_full_chunks * max_len\n            remainder_chunk = seq[remainder_start_index:]\n\n            # Pad the remainder to max_len\n            padding_needed = max_len - remainder_len\n            padded_remainder = np.pad(remainder_chunk, (0, padding_needed), mode='constant', constant_values=0)\n\n            processed_audio.append(padded_remainder)\n            # Append the SAME original label for the padded remainder chunk\n            processed_labels.append(original_label)\n\n\n# --- Final Output ---\n# Convert lists to numpy arrays (common practice for ML/DL)\nfinal_audio_array = np.array(processed_audio)\n# Labels can be kept as a list or converted to numpy array/pandas Series\nfinal_labels = processed_labels # Or np.array(processed_labels)\n\nprint(f\"\\nFinished processing.\")\nprint(f\"Resulting audio array shape: {final_audio_array.shape}\")\n# Based on example: (1+1+2+1+1) = 6 samples -> (6, 100)\nprint(f\"Number of resulting labels: {len(final_labels)}\")\n# Based on example: 6 labels -> ['Short', 'Exact', 'Long', 'Long', 'Long', 'Short_2']\n# print(\"Example processed labels:\", final_labels)","metadata":{"_uuid":"532a3805-4956-4a69-8092-9cc4ecb70f22","_cell_guid":"88fd1701-e577-4f4d-819b-40827b85c463","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-05-04T16:45:25.445432Z","iopub.execute_input":"2025-05-04T16:45:25.445684Z","iopub.status.idle":"2025-05-04T16:45:29.706451Z","shell.execute_reply.started":"2025-05-04T16:45:25.445663Z","shell.execute_reply":"2025-05-04T16:45:29.705707Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n\nprint(f\"Original audio array shape: {final_audio_array.shape}\")\nprint(f\"Original number of labels: {len(final_labels)}\")\n\n\n### VARIABLE TO CHANGE\nlimit_per_label = 2000\n\n# 1. Create a DataFrame with labels and original indices\n#    This avoids putting the large audio array into the DataFrame, saving memory.\nprint(\"\\nCreating temporary DataFrame with labels and indices...\")\nindices_df = pd.DataFrame({\n    'label': final_labels,\n    'original_index': np.arange(len(final_labels)) # Store 0, 1, 2, ... N-1\n})\n\n# 2. Group by label and select the first 'limit_per_label' indices for each group\nprint(f\"Grouping by label and selecting up to {limit_per_label} indices per group...\")\n# .head(n) conveniently takes min(n, group_size) automatically\nselected_indices_df = indices_df.groupby('label', observed=True).head(limit_per_label)\n# 'observed=True' can speed up grouping if labels are categorical\n\n# --- Alternative: Random Sampling (if you don't want the *first* 100) ---\n# def sample_or_head(group, n):\n#     group_size = len(group)\n#     # Sample if group is large enough, otherwise take all (head)\n#     return group.sample(n=min(group_size, n), random_state=42) # Use random_state for reproducibility\n#\n# selected_indices_df = indices_df.groupby('label', observed=True).apply(sample_or_head, n=limit_per_label).reset_index(drop=True)\n# print(f\"Grouping by label and randomly sampling up to {limit_per_label} indices per group...\")\n# --- End Alternative ---\n\n\n# 3. Get the selected original indices\nselected_indices = selected_indices_df['original_index'].values\n\n# Ensure the indices are sorted if you want the final array order to be somewhat grouped by label\n# This is optional, .head() preserves original relative order within groups.\n# selected_indices = np.sort(selected_indices)\n\nprint(f\"Total indices selected: {len(selected_indices)}\")\n\n# 4. Use the selected indices to filter your original arrays\nprint(\"Filtering original audio array and labels using selected indices...\")\nlimited_audio_array = final_audio_array[selected_indices]\n\n# Ensure final_labels is a numpy array if it isn't already for fancy indexing\nfinal_labels_array = np.array(final_labels)\nlimited_labels_array = final_labels_array[selected_indices]\n\n# --- Verification (Optional) ---\nprint(\"\\n--- Verification ---\")\nprint(f\"Limited audio array shape: {limited_audio_array.shape}\")\nprint(f\"Limited labels array length: {len(limited_labels_array)}\")\n\n# Check counts per label in the limited set\nunique_labels, counts = np.unique(limited_labels_array, return_counts=True)\nprint(\"Counts per label in the limited dataset:\")\nfor label, count in zip(unique_labels, counts):\n    print(f\"  Label '{label}': {count} samples\")\n    if count > limit_per_label:\n        print(f\"  WARNING: Label '{label}' has more than {limit_per_label} samples!\") # Should not happen with .head()\n\nprint(\"\\nDownsampling complete.\")\n\n# Now use 'limited_audio_array' and 'limited_labels_array' for further steps","metadata":{"_uuid":"992bd950-31c2-419f-b60e-7bedf893bf28","_cell_guid":"a2f4c61b-20fe-438a-9c86-e120ac966b44","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-05-04T16:45:29.707307Z","iopub.execute_input":"2025-05-04T16:45:29.707522Z","iopub.status.idle":"2025-05-04T16:45:30.339821Z","shell.execute_reply.started":"2025-05-04T16:45:29.707503Z","shell.execute_reply":"2025-05-04T16:45:30.339098Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow_hub as hub\nimport os\n\n# Define the path to the model directory provided by Kaggle\n# This path will be inside '/kaggle/input/'. The exact name depends\n# on how Kaggle named the model input folder (e.g., 'yamnet', 'yamnet-yamnet-v1', etc.)\n# You might need to check your /kaggle/input/ directory to confirm the exact name.\n\n# Example - Adjust 'yamnet-yamnet-v1' based on your actual input folder name\nmodel_dir_path = '/kaggle/input/yamnet/tensorflow2/yamnet/1/' #<-- CHECK AND CHANGE THIS if needed\n\n# Verify the path contains the saved_model.pb file\nif os.path.exists(os.path.join(model_dir_path, 'saved_model.pb')):\n    print(f\"Loading model from: {model_dir_path}\")\n    # Load the model from the local SavedModel directory\n    model_yamnet = hub.load(model_dir_path)\n    print(\"Model loaded successfully!\")\n\n    # Now you can use the model as before\n    # Example: Check the model's expected input signature\n    # print(model_yamnet.signatures['default'].pretty_print())\n\nelse:\n    print(f\"Error: 'saved_model.pb' not found in {model_dir_path}\")\n    print(\"Please check the directory path and the contents of /kaggle/input/\")\n    print(\"Contents of /kaggle/input/:\")\n    print(os.system('ls /kaggle/input/'))","metadata":{"_uuid":"47139e48-33e0-4a9d-897d-eec7987fad55","_cell_guid":"5d537e1d-6936-4c82-becb-f7edcbdfae58","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-05-04T18:34:28.522396Z","iopub.execute_input":"2025-05-04T18:34:28.522715Z","iopub.status.idle":"2025-05-04T18:34:32.283707Z","shell.execute_reply.started":"2025-05-04T18:34:28.522690Z","shell.execute_reply":"2025-05-04T18:34:32.282672Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"audio_embeddings = []\nfor i in tqdm(limited_audio_array, desc=\"Generating Audio Embeddings\"):\n    waveform = i / tf.int16.max\n    scores, embeddings, spectrogram = model_yamnet(waveform)\n    audio_embeddings.append(embeddings)","metadata":{"_uuid":"74922e51-60fc-4d4b-8d64-9282189624b9","_cell_guid":"072300bf-dab1-40f2-ba2d-90a0ef11906a","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-05-04T18:34:35.732032Z","iopub.execute_input":"2025-05-04T18:34:35.732360Z","iopub.status.idle":"2025-05-04T18:41:05.676148Z","shell.execute_reply.started":"2025-05-04T18:34:35.732335Z","shell.execute_reply":"2025-05-04T18:41:05.674672Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = np.array(audio_embeddings) # Shape will be (N, 1024)\nX_reshaped = np.squeeze(X, axis=1) # Specify axis=1 to remove the middle dimension\nprint(f\"Generated embeddings array X with shape: {X_reshaped.shape}\")\n\n# Ensure you have the corresponding labels (length N)\ny = limited_labels_array","metadata":{"_uuid":"6978e048-de39-46ec-9e1b-2b8d0a94c65e","_cell_guid":"072fdaf9-4b36-4b9a-ba89-3e5ed7d926d3","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-05-04T17:00:55.105150Z","iopub.execute_input":"2025-05-04T17:00:55.105417Z","iopub.status.idle":"2025-05-04T17:01:08.003523Z","shell.execute_reply.started":"2025-05-04T17:00:55.105396Z","shell.execute_reply":"2025-05-04T17:01:08.002775Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n# --- Label Encoding ---\nprint(\"\\nEncoding labels...\")\n# 1. Encode string/object labels to integers (0, 1, 2, ...)\nlabel_encoder = LabelEncoder()\ny_integer_encoded = label_encoder.fit_transform(y)\n\n# 2. Determine the number of unique classes\nnum_classes = len(label_encoder.classes_)\nprint(f\"Found {num_classes} unique classes: {label_encoder.classes_}\")\n\n# 3. Convert integer labels to one-hot encoding\ny_one_hot = tf.keras.utils.to_categorical(y_integer_encoded, num_classes=num_classes)\nprint(f\"One-hot encoded labels shape: {y_one_hot.shape}\") # Should be (N, num_classes)\n\n# --- Train/Test Split ---\nprint(\"\\nSplitting data into training and testing sets...\")\nxtrain, xtest, ytrain_one_hot, ytest_one_hot = train_test_split(\n    X_reshaped,                      # Your embeddings array (N, 1024)\n    y_one_hot,              # Your one-hot encoded labels (N, num_classes)\n    test_size=0.2,          # Fraction for the test set\n    random_state=42,        # For reproducibility\n    stratify=y_integer_encoded # IMPORTANT: Stratify based on original integer labels\n                               # to ensure class balance in train/test sets\n)\n\nprint(f\"xtrain shape: {xtrain.shape}\") # (N * 0.8, 1024)\nprint(f\"ytrain_one_hot shape: {ytrain_one_hot.shape}\") # (N * 0.8, num_classes)\nprint(f\"xtest shape: {xtest.shape}\") # (N * 0.2, 1024)\nprint(f\"ytest_one_hot shape: {ytest_one_hot.shape}\") # (N * 0.2, num_classes)","metadata":{"_uuid":"96703e48-f9a8-4045-8758-44dcf03a392f","_cell_guid":"da422fe0-2261-4a53-8774-3d2cd857e466","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-05-04T17:01:08.016976Z","iopub.execute_input":"2025-05-04T17:01:08.017213Z","iopub.status.idle":"2025-05-04T17:01:11.058923Z","shell.execute_reply.started":"2025-05-04T17:01:08.017193Z","shell.execute_reply":"2025-05-04T17:01:11.058145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\nDefining and compiling the model...\")\n# --- Define Model ---\nmodel = models.Sequential([\n    # Input shape is now (1024,) after averaging embeddings\n    layers.Input(shape=(1024,)),\n    # Flatten layer is not needed if input is 1D before Dense\n    # layers.Flatten(), # Remove this\n    layers.Dense(300, activation='relu'),\n    layers.Dropout(0.1),\n    layers.Dense(300, activation='relu'),\n    layers.Dropout(0.1),\n    layers.Dense(200, activation='relu'),\n    # Output layer must have 'num_classes' units\n    layers.Dense(num_classes, activation='softmax')\n])\n\nmodel.summary()\n\n# --- Compile Model ---\n# Using categorical_crossentropy because ytrain is one-hot encoded\nmodel.compile(optimizer='adam',\n              loss='categorical_crossentropy',\n              metrics=['accuracy'])\n\nprint(\"Model compiled successfully.\")","metadata":{"_uuid":"c5c31e39-febd-40c2-ae5e-f3263d92a9d0","_cell_guid":"c0d42a1c-377d-4870-a195-8ab8fb3b7380","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-05-04T17:01:11.059845Z","iopub.execute_input":"2025-05-04T17:01:11.060189Z","iopub.status.idle":"2025-05-04T17:01:12.408593Z","shell.execute_reply.started":"2025-05-04T17:01:11.060155Z","shell.execute_reply":"2025-05-04T17:01:12.407937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping\n\nearly_stopping = EarlyStopping(\n    monitor='val_loss',  # Metric to monitor (usually validation loss or accuracy)\n    patience=3,         # Number of epochs with no improvement after which training will be stopped\n    verbose=1,           # Set to 1 to print messages when stopping happens\n    mode='min',          # 'min' for loss/error metrics, 'max' for accuracy metrics\n    restore_best_weights=True  # Restore model weights from the epoch with the best value of the monitored quantity.\n)","metadata":{"_uuid":"c587b256-b5b9-4798-b931-57815075d174","_cell_guid":"626c0064-1aa5-4726-8e0a-2bef034c3163","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-05-04T17:01:12.411302Z","iopub.execute_input":"2025-05-04T17:01:12.411526Z","iopub.status.idle":"2025-05-04T17:01:12.428607Z","shell.execute_reply.started":"2025-05-04T17:01:12.411506Z","shell.execute_reply":"2025-05-04T17:01:12.428038Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check for GPU availability\ngpu_devices = tf.config.list_physical_devices('GPU')\nprint(\"Num GPUs Available: \", len(gpu_devices))\n\nif len(gpu_devices) > 0:\n    print(\"GPU(s) found:\")\n    for gpu in gpu_devices:\n        print(f\"- {gpu.name}\")\n        # Try to print details (might require specific CUDA versions)\n        try:\n            tf.config.experimental.get_device_details(gpu)\n            print(\"  Details:\", tf.config.experimental.get_device_details(gpu))\n        except: # Handle cases where details might not be fetchable\n             print(\"  Could not fetch device details.\")\n    # TensorFlow automatically places operations on the GPU if available\n    print(\"\\nTensorFlow should automatically use the GPU for training.\")\nelse:\n    print(\"\\nNo GPU found. TensorFlow will use the CPU.\")\n    print(\"Ensure NVIDIA drivers, CUDA Toolkit, and cuDNN are installed correctly and compatible.\")\n\n\n# --- Train Model ---\nprint(\"\\nStarting model training...\")\nepochs = 100\nhistory = model.fit(\n    xtrain,\n    ytrain_one_hot,\n    epochs=epochs,\n    validation_split=0.1, # Optional: use part of training data for validation during training\n    callbacks=[early_stopping]\n    # Or use validation_data=(xtest, ytest_one_hot) - be careful not to \"tune\" on test set\n)\n\nprint(\"Model training finished.\")\n\n# --- Evaluate Model (Optional) ---\nprint(\"\\nEvaluating model on the test set...\")\nloss, accuracy = model.evaluate(xtest, ytest_one_hot, verbose=0)\nprint(f\"Test Loss: {loss:.4f}\")\nprint(f\"Test Accuracy: {accuracy:.4f}\")","metadata":{"_uuid":"06142eb2-30f2-4552-9f83-bd1ba457ffe4","_cell_guid":"e17912b0-2a76-4801-ab44-24280d799315","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-05-04T17:04:57.438226Z","iopub.execute_input":"2025-05-04T17:04:57.438533Z","iopub.status.idle":"2025-05-04T17:07:54.923971Z","shell.execute_reply.started":"2025-05-04T17:04:57.438510Z","shell.execute_reply":"2025-05-04T17:07:54.922955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# Define the root directory containing the audio files\nroot_dir = '/kaggle/input/birdclef-2025/test_audio' # Adjust if your path is different\n\n# List to store the data for the DataFrame\naudio_data_list = []\n\n# Walk through the directory structure\nfor dirname, _, filenames in os.walk(root_dir):\n    # Skip the root directory itself if it doesn't contain class folders directly\n    # (Adjust this condition if your structure is different)\n    if dirname == root_dir:\n        continue\n\n    # Extract the class name (subdirectory name)\n    # os.path.basename gets the last part of the directory path\n    class_name = os.path.basename(dirname)\n\n    # Iterate through files in the current directory\n    for filename in filenames:\n        # Construct the full path to the audio file\n        full_path = os.path.join(dirname, filename)\n\n        # Append the file path and its class to the list\n        audio_data_list.append([full_path, class_name])\n        # You can remove the print statement if you don't need it anymore\n        # print(full_path) # Optional: print the path as it's processed\n\n# Create the Pandas DataFrame\naudio_dataframe = pd.DataFrame(audio_data_list, columns=[\"audio_path\", \"class\"])\n\n# Display the first few rows of the DataFrame (optional)\nprint(audio_dataframe.head())\n\n# Display the shape of the DataFrame (optional)\nprint(f\"\\nDataFrame shape: {audio_dataframe.shape}\")","metadata":{"_uuid":"de8f3436-a9a2-4c75-adaa-782966619c61","_cell_guid":"62e3b71b-89d1-4e54-86a9-550af561402b","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-05-04T17:09:07.738112Z","iopub.execute_input":"2025-05-04T17:09:07.738455Z","iopub.status.idle":"2025-05-04T17:09:07.753974Z","shell.execute_reply.started":"2025-05-04T17:09:07.738429Z","shell.execute_reply":"2025-05-04T17:09:07.753048Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import soundfile as sf\nimport numpy as np\n\ndef segment_audio(audio, sample_rate, segment_length=5):\n    \"\"\"Split audio into non-overlapping segments of segment_length seconds.\"\"\"\n    segment_samples = int(segment_length * sample_rate)\n    total_segments = len(audio) // segment_samples\n    segments = []\n    for i in range(total_segments):\n        start = i * segment_samples\n        end = start + segment_samples\n        segments.append(audio[start:end])\n    return segments\n\ndef get_yamnet_embedding(segment, model_yamnet):\n    # YAMNet expects float32 waveform in [-1.0, 1.0]\n    waveform = segment.astype(np.float32) / np.abs(segment).max()\n    scores, embeddings, spectrogram = model_yamnet(waveform)\n    # Average embeddings over time axis (axis=0)\n    return np.mean(embeddings.numpy(), axis=0)\n\nimport pandas as pd\nimport os\n\ntest_dir = '/kaggle/input/birdclef-2025/test_soundscapes'\nsubmission_rows = []\n\nfor filename in os.listdir(test_dir):\n    if not filename.endswith('.ogg'):\n        continue\n    filepath = os.path.join(test_dir, filename)\n    audio, sr = sf.read(filepath)\n    segments = segment_audio(audio, sr, segment_length=5)\n    for i, segment in enumerate(segments):\n        embedding = get_yamnet_embedding(segment, model_yamnet)\n        # Model expects shape (1, 1024)\n        pred = model.predict(embedding.reshape(1, -1))\n        # Format row_id as required: soundscape_xxxxxx_[end_time]\n        end_time = (i + 1) * 5\n        row_id = f\"{filename.replace('.ogg','')}_{end_time}\"\n        # pred is shape (1, 206)\n        submission_rows.append([row_id] + pred.flatten().tolist())\n\n# Get species columns from sample_submission.csv\nsample_sub = pd.read_csv('/kaggle/input/birdclef-2025/sample_submission.csv')\nspecies_cols = sample_sub.columns[1:]\n\n# Build DataFrame\nsubmission_df = pd.DataFrame(submission_rows, columns=['row_id'] + list(species_cols))\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T18:28:04.649520Z","iopub.execute_input":"2025-05-04T18:28:04.649906Z","iopub.status.idle":"2025-05-04T18:28:04.669641Z","shell.execute_reply.started":"2025-05-04T18:28:04.649873Z","shell.execute_reply":"2025-05-04T18:28:04.668727Z"}},"outputs":[],"execution_count":null}]}