{"metadata":{"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":1664376,"sourceType":"datasetVersion","datasetId":985270},{"sourceId":5181249,"sourceType":"datasetVersion","datasetId":3012199},{"sourceId":5190993,"sourceType":"datasetVersion","datasetId":3018185},{"sourceId":5196408,"sourceType":"datasetVersion","datasetId":3018885},{"sourceId":8036535,"sourceType":"datasetVersion","datasetId":4737648}],"dockerImageVersionId":30684,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.13"},"papermill":{"default_parameters":{},"duration":85.328891,"end_time":"2024-04-06T18:24:33.296159","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-04-06T18:23:07.967268","version":"2.5.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"I'm looking at how to make use of pretrained models, such as https://www.kaggle.com/datasets/awsaf49/birdclef23-effnetb1-fsr-pretrain-10s-ds\n\nReference notebook: https://www.kaggle.com/code/shadesh/bird-species-classification-using-machine-learning","metadata":{}},{"cell_type":"code","source":"import sys, os  # Importing the sys and os modules for system-related operations\nsys.path.append('/kaggle/input/efficientnet-keras-dataset/efficientnet_kaggle')  # Appending a directory to the Python path where EfficientNet is located\n!pip install -q /kaggle/input/tensorflow-extra-lib-ds/tensorflow_extra-1.0.2-py3-none-any.whl --no-deps  # Installing a specific version of TensorFlow extras library from a local wheel file silently without dependencies","metadata":{"papermill":{"duration":23.25675,"end_time":"2024-04-06T18:23:34.791789","exception":false,"start_time":"2024-04-06T18:23:11.535039","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-04-11T16:05:50.455143Z","iopub.execute_input":"2024-04-11T16:05:50.455670Z","iopub.status.idle":"2024-04-11T16:06:12.984864Z","shell.execute_reply.started":"2024-04-11T16:05:50.455634Z","shell.execute_reply":"2024-04-11T16:06:12.983143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport tensorflow as tf  # Importing TensorFlow library for machine learning tasks\ntf.get_logger().setLevel('ERROR')  # Setting the logging level of TensorFlow to 'ERROR' to suppress unnecessary messages\ntf.autograph.set_verbosity(0)  # Setting the verbosity level of TensorFlow Autograph to 0 to suppress warnings\nimport os  # Importing os module for operating system related functionalities\nimport pandas as pd  # Importing pandas library for data manipulation and analysis\nimport numpy as np  # Importing NumPy library for numerical computing\nimport random  # Importing random module for generating random numbers\nfrom glob import glob  # Importing glob function from glob module for file path pattern matching\nfrom tqdm import tqdm  # Importing tqdm library for displaying progress bars during iterations\ntqdm.pandas()  # Enabling tqdm progress bars for pandas operations\nimport gc  # Importing gc module for garbage collection\nimport librosa  # Importing librosa library for audio signal processing\nimport time  # Importing time module for time-related functions","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:06:12.987722Z","iopub.execute_input":"2024-04-11T16:06:12.988156Z","iopub.status.idle":"2024-04-11T16:06:13.000412Z","shell.execute_reply.started":"2024-04-11T16:06:12.988120Z","shell.execute_reply":"2024-04-11T16:06:12.999022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib as mpl  # Importing matplotlib library for creating visualizations\nimport matplotlib.pyplot as plt  # Importing pyplot module from matplotlib for creating plots\nimport librosa.display as lid  # Importing display module from librosa for audio display\nimport IPython.display as ipd  # Importing display module from IPython for audio display","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:06:13.002195Z","iopub.execute_input":"2024-04-11T16:06:13.002916Z","iopub.status.idle":"2024-04-11T16:06:13.016353Z","shell.execute_reply.started":"2024-04-11T16:06:13.002852Z","shell.execute_reply":"2024-04-11T16:06:13.015028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf  # Importing TensorFlow library again (duplicate import)\ntf.config.optimizer.set_jit(True)  # Enabling XLA (Accelerated Linear Algebra) for speed up in TensorFlow optimization\nimport tensorflow_io as tfio  # Importing tensorflow_io library for TensorFlow IO functionalities\nimport tensorflow.keras.backend as K  # Importing Keras backend module from TensorFlow for backend operations","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:06:13.017780Z","iopub.execute_input":"2024-04-11T16:06:13.018143Z","iopub.status.idle":"2024-04-11T16:06:13.028870Z","shell.execute_reply.started":"2024-04-11T16:06:13.018112Z","shell.execute_reply":"2024-04-11T16:06:13.027590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import efficientnet.tfkeras as efn  # Importing EfficientNet model from TensorFlow.keras for image classification\nimport tensorflow_extra as tfe  # Importing additional TensorFlow utilities from tensorflow_extra module","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:06:13.032176Z","iopub.execute_input":"2024-04-11T16:06:13.032581Z","iopub.status.idle":"2024-04-11T16:06:13.040064Z","shell.execute_reply.started":"2024-04-11T16:06:13.032548Z","shell.execute_reply":"2024-04-11T16:06:13.039180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    debug = False  # Flag for debugging mode\n    verbose = 0  # Verbosity level for printing messages\n    \n    device = 'CPU'  # Device used for training (default: CPU)\n    seed = 42  # Seed value for random number generation for reproducibility\n    \n    # Input image size and batch size\n    img_size = [128, 384]  # Size of input images [height, width]\n    batch_size = 16  # Batch size for training\n    infer_bs = 2  # Batch size for inference\n    tta = 1  # Number of test time augmentations\n    drop_remainder = True  # Whether to drop the remaining samples in the last incomplete batch\n    \n    # STFT parameters\n    duration = 5  # Duration of audio clips for testing\n    train_duration = 10  # Duration of audio clips for training\n    sample_rate = 32000  # Sampling rate of audio clips\n    downsample = 1  # Downsample factor\n    audio_len = duration * sample_rate  # Length of audio clips\n    nfft = 2028  # Number of FFT points\n    window = 2048  # Size of STFT window\n    hop_length = train_duration * 32000 // (img_size[1] - 1)  # Hop length for STFT computation\n    fmin = 20  # Minimum frequency for STFT\n    fmax = 16000  # Maximum frequency for STFT\n    normalize = True  # Flag to normalize the input\n    \n    # Data Preprocessing Settings\n    class_names = sorted(os.listdir('/kaggle/input/birdclef-2024/train_audio/'))  # List of class names extracted from the directory of train_audio\n    num_classes = len(class_names)  # Number of classes\n    class_labels = list(range(num_classes))  # List of class labels\n    label2name = dict(zip(class_labels, class_names))  # Mapping of label to class name\n    name2label = {v: k for k, v in label2name.items()}  # Mapping of class name to label\n    \n    target_col = ['target']  # Target column name in the dataset\n    tab_cols = ['filename', 'common_name', 'rate']  # Columns to be used from the tabular data","metadata":{"_kg_hide-input":false,"papermill":{"duration":0.043831,"end_time":"2024-04-06T18:23:55.926667","exception":false,"start_time":"2024-04-06T18:23:55.882836","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-04-11T16:06:13.041658Z","iopub.execute_input":"2024-04-11T16:06:13.042048Z","iopub.status.idle":"2024-04-11T16:06:13.058378Z","shell.execute_reply.started":"2024-04-11T16:06:13.042014Z","shell.execute_reply":"2024-04-11T16:06:13.057240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.set_random_seed(CFG.seed)","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.021933,"end_time":"2024-04-06T18:23:55.984654","exception":false,"start_time":"2024-04-06T18:23:55.962721","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-04-11T16:06:13.059683Z","iopub.execute_input":"2024-04-11T16:06:13.060090Z","iopub.status.idle":"2024-04-11T16:06:13.071454Z","shell.execute_reply.started":"2024-04-11T16:06:13.060054Z","shell.execute_reply":"2024-04-11T16:06:13.070390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_device():\n    \"Detect and initializes GPU/TPU automatically\"\n    # Check TPU category\n    tpu = 'local' if CFG.device == 'TPU-VM' else None\n    try:\n        # Connect to TPU\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect(tpu=tpu) \n        # Set TPU strategy\n        strategy = tf.distribute.TPUStrategy(tpu)\n        print(f'> Running on {CFG.device} ', tpu.master(), end=' | ')\n        print('Num of TPUs: ', strategy.num_replicas_in_sync)\n        device = CFG.device\n    except:\n        # If TPU is not available, detect GPUs\n        gpus = tf.config.list_logical_devices('GPU')\n        ngpu = len(gpus)\n        # Check number of GPUs\n        if ngpu:\n            # Set GPU strategy\n            strategy = tf.distribute.MirroredStrategy(gpus)  # single-GPU or multi-GPU\n            # Print GPU details\n            print(\"> Running on GPU\", end=' | ')\n            print(\"Num of GPUs: \", ngpu)\n            device = 'GPU'\n        else:\n            # If no GPUs are available, use CPU\n            print(\"> Running on CPU\")\n            strategy = tf.distribute.get_strategy()\n            device = 'CPU'\n    return strategy, device, tpu","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.035385,"end_time":"2024-04-06T18:23:56.061052","exception":false,"start_time":"2024-04-06T18:23:56.025667","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-04-11T16:06:13.073015Z","iopub.execute_input":"2024-04-11T16:06:13.073533Z","iopub.status.idle":"2024-04-11T16:06:13.086370Z","shell.execute_reply.started":"2024-04-11T16:06:13.073490Z","shell.execute_reply":"2024-04-11T16:06:13.085002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize GPU/TPU/TPU-VM\nstrategy, CFG.device, tpu = get_device()\nCFG.replicas = strategy.num_replicas_in_sync","metadata":{"papermill":{"duration":0.031624,"end_time":"2024-04-06T18:23:56.105848","exception":false,"start_time":"2024-04-06T18:23:56.074224","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-04-11T16:06:13.088588Z","iopub.execute_input":"2024-04-11T16:06:13.089171Z","iopub.status.idle":"2024-04-11T16:06:13.100279Z","shell.execute_reply.started":"2024-04-11T16:06:13.089120Z","shell.execute_reply":"2024-04-11T16:06:13.099138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_PATH = '/kaggle/input/birdclef-2024'\nGCS_PATH = BASE_PATH","metadata":{"papermill":{"duration":0.022691,"end_time":"2024-04-06T18:23:56.167179","exception":false,"start_time":"2024-04-06T18:23:56.144488","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-04-11T16:06:13.101874Z","iopub.execute_input":"2024-04-11T16:06:13.102336Z","iopub.status.idle":"2024-04-11T16:06:13.110854Z","shell.execute_reply.started":"2024-04-11T16:06:13.102298Z","shell.execute_reply":"2024-04-11T16:06:13.109723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Directory containing the test audio files\ntest_audio_dir = '/kaggle/input/birdclef-2024/test_soundscapes/'\n\n# List of file paths for test audio files\ntest_paths = [test_audio_dir + f for f in sorted(os.listdir(test_audio_dir))]\n\n# Checking if there's only one file in the test directory\nif len(test_paths) == 1:\n    # If only one file is found, update the directory to unlabeled_soundscapes\n    test_audio_dir = '/kaggle/input/birdclef-2024/unlabeled_soundscapes/'\n    \n    # List the first two files in the unlabeled_soundscapes directory\n    test_paths = [test_audio_dir + f for f in sorted(os.listdir(test_audio_dir))][:2]\n\n# Creating a DataFrame to store the test file paths and corresponding filenames\ntest_df = pd.DataFrame(test_paths, columns=['filepath'])\n\n# Extracting filenames from file paths and storing them in a new column 'filename'\ntest_df['filename'] = test_df.filepath.map(lambda x: x.split('/')[-1].replace('.ogg', ''))\n\n# Displaying the first few rows of the DataFrame\ntest_df.head()","metadata":{"papermill":{"duration":0.345655,"end_time":"2024-04-06T18:23:56.551086","exception":false,"start_time":"2024-04-06T18:23:56.205431","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-04-11T16:06:13.112586Z","iopub.execute_input":"2024-04-11T16:06:13.113029Z","iopub.status.idle":"2024-04-11T16:06:13.147189Z","shell.execute_reply.started":"2024-04-11T16:06:13.112981Z","shell.execute_reply":"2024-04-11T16:06:13.146177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.io.gfile.exists(test_df.filepath.iloc[0])","metadata":{"papermill":{"duration":0.026736,"end_time":"2024-04-06T18:23:56.590731","exception":false,"start_time":"2024-04-06T18:23:56.563995","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-04-11T16:06:13.148730Z","iopub.execute_input":"2024-04-11T16:06:13.150192Z","iopub.status.idle":"2024-04-11T16:06:13.158680Z","shell.execute_reply.started":"2024-04-11T16:06:13.150150Z","shell.execute_reply":"2024-04-11T16:06:13.157121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_audio(filepath, sr=32000, normalize=True):\n    # Load audio file using librosa\n    audio, orig_sr = librosa.load(filepath, sr=None)\n    \n    # Resample audio if specified sample rate differs from original sample rate\n    if sr != orig_sr:\n        audio = librosa.resample(audio, orig_sr, sr)\n    \n    # Convert audio to float32 and flatten the array\n    audio = audio.astype('float32').ravel()\n    \n    # Convert audio to TensorFlow tensor\n    audio = tf.convert_to_tensor(audio)\n    \n    return audio\n\n@tf.function(jit_compile=True)\ndef MakeFrame(audio, duration=5, sr=32000):\n    # Calculate frame length and step size based on duration and sample rate\n    frame_length = int(duration * sr)\n    frame_step = int(duration * sr)\n    \n    # Split audio into frames using TensorFlow's frame function\n    chunks = tf.signal.frame(audio, frame_length, frame_step, pad_end=True)\n    \n    return chunks","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.026981,"end_time":"2024-04-06T18:23:56.656559","exception":false,"start_time":"2024-04-06T18:23:56.629578","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-04-11T16:06:13.160611Z","iopub.execute_input":"2024-04-11T16:06:13.161192Z","iopub.status.idle":"2024-04-11T16:06:13.177084Z","shell.execute_reply.started":"2024-04-11T16:06:13.161135Z","shell.execute_reply":"2024-04-11T16:06:13.175501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil  # Importing shutil module for file operations\n# Directory containing checkpoints\nCKPT_DIR = '/kaggle/input/birdclef24-pretraining-train-model'\n\n# Get file paths of all trained models in the directory\nCKPT_PATHS = sorted([x for x in glob(f'{CKPT_DIR}/fold-*keras')])\nprint(\"Checkpoints: \", CKPT_PATHS)\n\n# Define a writable directory\nWRITABLE_DIR = '/kaggle/working/models/'\n\n# Create the writable directory if it does not exist\nif not os.path.exists(WRITABLE_DIR):\n    os.makedirs(WRITABLE_DIR)\n\n# Copy the model files to the writable directory\nfor ckpt_path in CKPT_PATHS:\n    shutil.copy(ckpt_path, WRITABLE_DIR)\n\n# Update the checkpoint paths to the writable directory\nCKPT_PATHS = sorted([f'{WRITABLE_DIR}/{os.path.basename(x)}' for x in glob(f'{CKPT_DIR}/fold-*keras')])\n\n# Load all the models in memory to speed up\nCKPTS = [tf.keras.models.load_model(x, compile=False) for x in tqdm(CKPT_PATHS, desc=\"Loading ckpts \")]\n# Number of checkpoints to use\nNUM_CKPTS = 1","metadata":{"papermill":{"duration":8.586829,"end_time":"2024-04-06T18:24:20.941753","exception":false,"start_time":"2024-04-06T18:24:12.354924","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-04-11T16:06:13.182640Z","iopub.execute_input":"2024-04-11T16:06:13.183198Z","iopub.status.idle":"2024-04-11T16:06:22.543293Z","shell.execute_reply.started":"2024-04-11T16:06:13.183157Z","shell.execute_reply":"2024-04-11T16:06:22.541990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Start stopwatch\ntick = time.time()\n\n# Initialize empty list to store ids\nids = []\n# Initialize empty array to store predictions\npreds = np.empty(shape=(0, 182), dtype='float32')\n\n# Iterate over each audio file in the test dataset\nfor filepath in tqdm(test_df.filepath.tolist(), 'test '):\n    # Extract the filename without the extension\n    filename = filepath.split('/')[-1].replace('.ogg','')\n    \n    # Load audio from file and create audio frames, each recording will be a batch input\n    audio = load_audio(filepath)\n    chunks = MakeFrame(audio)\n    \n    # Predict bird species for all frames in a recording using all trained models\n    chunk_preds = np.zeros(shape=(len(chunks), 182), dtype=np.float32)\n    for model in CKPTS[:NUM_CKPTS]:\n        # Get the model's predictions for the current audio frames\n        rec_preds = model(chunks, training=False).numpy()\n        # Ensemble all predictions with average\n        chunk_preds += rec_preds / len(CKPTS)\n    \n    # Create an ID for each frame in a recording using the filename and frame number\n    rec_ids = [f'{filename}_{(frame_id+1)*5}' for frame_id in range(len(chunks))]\n    \n    # Concatenate the ids\n    ids += rec_ids\n    # Concatenate the predictions\n    preds = np.concatenate([preds, chunk_preds], axis=0)\n    \n# Stop stopwatch\ntock = time.time()","metadata":{"papermill":{"duration":8.25077,"end_time":"2024-04-06T18:24:29.254015","exception":false,"start_time":"2024-04-06T18:24:21.003245","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-04-11T16:06:22.545516Z","iopub.execute_input":"2024-04-11T16:06:22.546069Z","iopub.status.idle":"2024-04-11T16:06:31.315907Z","shell.execute_reply.started":"2024-04-11T16:06:22.546016Z","shell.execute_reply":"2024-04-11T16:06:31.314664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds.shape","metadata":{"papermill":{"duration":0.032619,"end_time":"2024-04-06T18:24:29.351816","exception":false,"start_time":"2024-04-06T18:24:29.319197","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-04-11T16:06:31.319141Z","iopub.execute_input":"2024-04-11T16:06:31.319567Z","iopub.status.idle":"2024-04-11T16:06:31.327918Z","shell.execute_reply.started":"2024-04-11T16:06:31.319530Z","shell.execute_reply":"2024-04-11T16:06:31.326446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df = pd.DataFrame(ids, columns=['row_id'])  # Create DataFrame with row_ids\npred_df.loc[:, CFG.class_names] = preds  # Add predicted probabilities for each class as columns\npred_df.to_csv('submission.csv', index=False)  # Save DataFrame to CSV file for submission\npred_df.sample(5)  # Display the DataFrame","metadata":{"papermill":{"duration":0.220742,"end_time":"2024-04-06T18:24:29.593948","exception":false,"start_time":"2024-04-06T18:24:29.373206","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-04-11T16:06:31.329851Z","iopub.execute_input":"2024-04-11T16:06:31.330349Z","iopub.status.idle":"2024-04-11T16:06:31.509196Z","shell.execute_reply.started":"2024-04-11T16:06:31.330305Z","shell.execute_reply":"2024-04-11T16:06:31.507963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_time = (tock - tick) * 550  # Calculate estimated submission time for ~1100 recordings\nsub_time = time.gmtime(sub_time)  # Convert seconds to a time tuple\nsub_time = time.strftime(\"%H hr: %M min : %S sec\", sub_time)  # Format time tuple as string\nprint(f\">> Time for submission: ~ {sub_time}\")  # Print estimated submission time","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.034257,"end_time":"2024-04-06T18:24:29.793347","exception":false,"start_time":"2024-04-06T18:24:29.75909","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-04-11T16:06:31.511003Z","iopub.execute_input":"2024-04-11T16:06:31.511496Z","iopub.status.idle":"2024-04-11T16:06:31.519219Z","shell.execute_reply.started":"2024-04-11T16:06:31.511450Z","shell.execute_reply":"2024-04-11T16:06:31.517760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}