{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_io as tfio\nimport os\nimport numpy as np\nimport pandas as pd\n\nbase_path = '/kaggle/input/birdclef-2024/train_audio'\ndata = []  # This list will hold our data\nlabels = []  # This list will hold our labels\n\n# List all directories in the base path\nfor dir_name in os.listdir(base_path):\n    dir_path = os.path.join(base_path, dir_name)  # Full path of the directory\n    if os.path.isdir(dir_path):  # Check if it is indeed a directory\n        for filename in os.listdir(dir_path):\n            file_path = os.path.join(dir_path, filename)\n            print(f\"Processing file: {file_path}\")\n            \n            # Read the audio file\n            audio = tfio.audio.AudioIOTensor(file_path)\n            \n            # If the audio has more than one channel, take the first one (mono)\n            if audio.shape[1] > 1:\n                audio_tensor = audio.to_tensor()[:, 0]\n            else:\n                audio_tensor = audio.to_tensor()[:, 0]\n\n            # Convert to float32\n            audio_tensor = tf.cast(audio_tensor, tf.float32)\n            \n            # Normalization\n            max_val = tf.reduce_max(tf.abs(audio_tensor))\n            if max_val.numpy() != 0:\n                audio_tensor = audio_tensor / max_val\n\n            # Take the first 1000 frames, pad if necessary\n            if tf.shape(audio_tensor)[0] < 1000:\n                # Pad array if it has less than 1000 frames\n                padding = tf.constant([[0, 1000 - tf.shape(audio_tensor)[0]]], dtype=tf.int32)\n                audio_tensor = tf.pad(audio_tensor, padding, \"CONSTANT\")\n                audio_tensor = audio_tensor[:1000]  # Ensuring the tensor is exactly 1000 frames long\n            else:\n                audio_tensor = audio_tensor[:1000]  # Take only the first 1000 frames\n            \n            # Convert to numpy and store\n            audio_numpy = audio_tensor.numpy()\n            data.append(audio_numpy)\n            labels.append(dir_path.split('/')[-1])  # Use the filename as the label\n\n# Convert lists to a Pandas DataFrame\ndf = pd.DataFrame(data)\ndf['label'] = labels  # Adding a new column for labels\n\n# Save DataFrame to CSV (optional)\ndf.to_csv('audio_dataset.csv', index=False)\nprint(\"Dataset has been created and saved to audio_dataset.csv\")\n","metadata":{"execution":{"iopub.status.busy":"2024-04-11T10:21:45.227443Z","iopub.execute_input":"2024-04-11T10:21:45.227834Z","iopub.status.idle":"2024-04-11T10:21:56.772372Z","shell.execute_reply.started":"2024-04-11T10:21:45.227805Z","shell.execute_reply":"2024-04-11T10:21:56.770822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text = dir_path\ntext.split('/')[-1]\n","metadata":{"execution":{"iopub.status.busy":"2024-04-11T10:23:48.242007Z","iopub.execute_input":"2024-04-11T10:23:48.242391Z","iopub.status.idle":"2024-04-11T10:23:48.249716Z","shell.execute_reply.started":"2024-04-11T10:23:48.242363Z","shell.execute_reply":"2024-04-11T10:23:48.248497Z"},"trusted":true},"execution_count":null,"outputs":[]}]}