{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nfrom collections import Counter, defaultdict\n\n# for audio files\n!pip install -q mutagen\nfrom mutagen.oggvorbis import OggVorbis\nfrom mutagen import File\n\n# play audio\nfrom IPython.display import Audio\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-24T00:26:03.759892Z","iopub.execute_input":"2025-03-24T00:26:03.760252Z","iopub.status.idle":"2025-03-24T00:26:05.014596Z","shell.execute_reply.started":"2025-03-24T00:26:03.760216Z","shell.execute_reply":"2025-03-24T00:26:05.013019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(os.getcwd())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T21:56:55.742634Z","iopub.execute_input":"2025-03-23T21:56:55.743201Z","iopub.status.idle":"2025-03-23T21:56:55.748907Z","shell.execute_reply.started":"2025-03-23T21:56:55.743165Z","shell.execute_reply":"2025-03-23T21:56:55.747887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# dataset name and paths\ndataset_name = 'birdclef-2025'\npath = f'/kaggle/input/{dataset_name}'\nprint('dataset_name: ',dataset_name)\nprint('path: ',path)\n\ntrain_audio_name = 'train_audio'\ntrain_soundscapes_name = 'train_soundscapes'\n\ntrain_audio_path = f'/kaggle/input/{dataset_name}/{train_audio_name}'\ntrain_soundscapes_path = f'/kaggle/input/{dataset_name}/{train_soundscapes_name}'\n# test_path = f'/kaggle/input/{dataset_name}/test'\n\nprint('\\ntrain_audio_path: ',train_audio_path)\nprint('train_soundscapes_path: ',train_soundscapes_path)\n\nsample_submission_name = 'sample_submission.csv'\ntrain_name = 'train.csv'\ntaxonomy_name = 'taxonomy.csv'\nrecording_location_name = 'recording_location.txt'\ntest_soundscapes_name = 'test_soundscapes/readme.txt'\n\nsample_submission_path = f'/kaggle/input/{dataset_name}/{sample_submission_name}'\ntrain_path = f'/kaggle/input/{dataset_name}/{train_name}'\ntaxonomy_path = f'/kaggle/input/{dataset_name}/{taxonomy_name}'\n\nrecording_location_path = f'/kaggle/input/{dataset_name}/recording_location.txt'\nreadme_path = f'/kaggle/input/{dataset_name}/{test_soundscapes_name}'\n\nprint('\\nsample_submission_path: ',sample_submission_path)\nprint('train_path: ',train_path)\nprint('taxonomy_path: ',taxonomy_path)\nprint('\\nrecording_location_path: ',recording_location_path)\nprint('readme_path: ',readme_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T00:43:46.731312Z","iopub.execute_input":"2025-03-24T00:43:46.731689Z","iopub.status.idle":"2025-03-24T00:43:46.742499Z","shell.execute_reply.started":"2025-03-24T00:43:46.731660Z","shell.execute_reply":"2025-03-24T00:43:46.741078Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_sample_submission = pd.read_csv(f'{path}/{sample_submission_name}')\nprint('df_sample_submission shape: ',df_sample_submission.shape)\ndf_sample_submission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T22:00:59.477085Z","iopub.execute_input":"2025-03-23T22:00:59.477419Z","iopub.status.idle":"2025-03-23T22:00:59.553606Z","shell.execute_reply.started":"2025-03-23T22:00:59.477395Z","shell.execute_reply":"2025-03-23T22:00:59.552567Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_csv(f'{path}/{train_name}')\nprint('df_train shape: ',df_train.shape)\ndf_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T22:01:10.852725Z","iopub.execute_input":"2025-03-23T22:01:10.853152Z","iopub.status.idle":"2025-03-23T22:01:11.079104Z","shell.execute_reply.started":"2025-03-23T22:01:10.853122Z","shell.execute_reply":"2025-03-23T22:01:11.077957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_taxonomy_name = pd.read_csv(f'{path}/{taxonomy_name}')\nprint('df_taxonomy_name shape: ',df_taxonomy_name.shape)\ndf_taxonomy_name.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T22:01:14.264869Z","iopub.execute_input":"2025-03-23T22:01:14.265204Z","iopub.status.idle":"2025-03-23T22:01:14.284533Z","shell.execute_reply.started":"2025-03-23T22:01:14.265180Z","shell.execute_reply":"2025-03-23T22:01:14.283651Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# test soundscapes\nwith open(readme_path, 'r') as f:\n    content = f.read()\n\n# Print the content\nprint(content)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T22:04:16.646730Z","iopub.execute_input":"2025-03-23T22:04:16.647107Z","iopub.status.idle":"2025-03-23T22:04:16.659155Z","shell.execute_reply.started":"2025-03-23T22:04:16.647078Z","shell.execute_reply":"2025-03-23T22:04:16.658182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train soundscapes\nwith open(recording_location_path, 'r') as f:\n    content = f.read()\n\n# Print the content\nprint(content)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-23T22:04:38.079445Z","iopub.execute_input":"2025-03-23T22:04:38.079793Z","iopub.status.idle":"2025-03-23T22:04:38.092342Z","shell.execute_reply.started":"2025-03-23T22:04:38.079740Z","shell.execute_reply":"2025-03-23T22:04:38.091026Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# folder summary function\n\ndef analyze_folder_detailed(folder_path):\n    total_file_count = 0\n    extension_counter_global = Counter()\n    subdir_stats = []\n\n    for root, dirs, files in os.walk(folder_path):\n        # Determine relative subdir name\n        subdir_name = os.path.relpath(root, folder_path)\n        if subdir_name == \".\":\n            subdir_name = \"[root]\"\n\n        file_count = len(files)\n        total_file_count += file_count\n\n        # Count extensions in this directory\n        extension_counter_local = Counter()\n        for file in files:\n            ext = os.path.splitext(file)[1].lower()\n            extension_counter_local[ext] += 1\n            extension_counter_global[ext] += 1\n\n        # Record stats for this folder only if it contains files\n        if file_count > 0:\n            row = {\n                \"subdirectory\": subdir_name,\n                \"total_files\": file_count,\n            }\n            row.update(extension_counter_local)\n            subdir_stats.append(row)\n\n    # Create DataFrame\n    df = pd.DataFrame(subdir_stats)\n    if not df.empty:\n        df.fillna(0, inplace=True)\n        df = df.astype({col: int for col in df.columns if col != 'subdirectory'})\n\n    # Summary Info\n    num_subdirs_with_files = df[df[\"subdirectory\"] != \"[root]\"].shape[0]\n    has_root_files = \"[root]\" in df[\"subdirectory\"].values\n\n    print(f\"Analyzing folder: {folder_path}\")\n    print(f\"Number of subdirectories with files: {num_subdirs_with_files}\")\n    if has_root_files:\n        print(\"Root folder also contains files.\")\n    print(f\"Total number of files: {total_file_count}\")\n    print(f\"Unique file extensions: {len(extension_counter_global)}\\n\")\n\n    print(\"File Extension Counts (Global):\")\n    for ext, count in extension_counter_global.items():\n        print(f\"  {ext or '[No Extension]'}: {count}\")\n\n    return df\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T01:13:30.092791Z","iopub.execute_input":"2025-03-24T01:13:30.093218Z","iopub.status.idle":"2025-03-24T01:13:30.102313Z","shell.execute_reply.started":"2025-03-24T01:13:30.093175Z","shell.execute_reply":"2025-03-24T01:13:30.100870Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" # folder summary- train_audio\ndf_train_metadata = analyze_folder_detailed(train_audio_path)\nprint('df_train_metadata: ',df_train_metadata.shape)\ncolumn_name = 'total_files'\nmax_value = df_train_metadata[column_name].max()\nmin_value = df_train_metadata[column_name].min()\n\nprint(f\"Maximum value in '{column_name}': {max_value}\")\nprint(f\"Minimum value in '{column_name}': {min_value}\")\ndf_train_metadata.sort_values(by = column_name,ascending=False).head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T01:13:33.845359Z","iopub.execute_input":"2025-03-24T01:13:33.845680Z","iopub.status.idle":"2025-03-24T01:13:46.567659Z","shell.execute_reply.started":"2025-03-24T01:13:33.845656Z","shell.execute_reply":"2025-03-24T01:13:46.566684Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" # folder summary- train_soundscapes\ndf_train_soundscapes_metadata = analyze_folder_detailed(train_soundscapes_path)\nprint('df_train_soundscapes_metadata: ',df_train_soundscapes_metadata.shape)\ncolumn_name = 'total_files'\nmax_value = df_train_soundscapes_metadata[column_name].max()\nmin_value = df_train_soundscapes_metadata[column_name].min()\n\nprint(f\"Maximum value in '{column_name}': {max_value}\")\nprint(f\"Minimum value in '{column_name}': {min_value}\")\ndf_train_soundscapes_metadata.sort_values(by = column_name,ascending=False).head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T01:14:04.533619Z","iopub.execute_input":"2025-03-24T01:14:04.533972Z","iopub.status.idle":"2025-03-24T01:14:07.394272Z","shell.execute_reply.started":"2025-03-24T01:14:04.533947Z","shell.execute_reply":"2025-03-24T01:14:07.393110Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load and print metadata of the audio file\ndef audio_metadata(audio_path):\n    audio_file = File(audio_path)\n    print(\"Metadata:\")\n    for key, value in audio_file.items():\n        print(f\"{key}: {value}\")\n        print(\"Duration (s):\", audio_file.info.length)\n        print(\"Bitrate (bps):\", audio_file.info.bitrate)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T01:22:50.598786Z","iopub.execute_input":"2025-03-24T01:22:50.599233Z","iopub.status.idle":"2025-03-24T01:22:50.604561Z","shell.execute_reply.started":"2025-03-24T01:22:50.599200Z","shell.execute_reply":"2025-03-24T01:22:50.603597Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# metadata of the audio file - train_audio\naudio_path1 = f'/kaggle/input/{dataset_name}/train_audio/1192948/CSA36388.ogg'\naudio_metadata(audio_path1)\n\n# Play audio\nAudio(filename=audio_path1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T01:28:59.601954Z","iopub.execute_input":"2025-03-24T01:28:59.602402Z","iopub.status.idle":"2025-03-24T01:28:59.649914Z","shell.execute_reply.started":"2025-03-24T01:28:59.602371Z","shell.execute_reply":"2025-03-24T01:28:59.648664Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# metadata of the audio file - train_soundscapes\naudio_path2 = f'/kaggle/input/{dataset_name}/train_soundscapes/H02_20230420_074000.ogg'\naudio_metadata(audio_path2)\n# Play audio\nAudio(filename=audio_path2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T01:26:53.691368Z","iopub.execute_input":"2025-03-24T01:26:53.691723Z","iopub.status.idle":"2025-03-24T01:26:53.716107Z","shell.execute_reply.started":"2025-03-24T01:26:53.691697Z","shell.execute_reply":"2025-03-24T01:26:53.715032Z"}},"outputs":[],"execution_count":null}]}