{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":1664376,"sourceType":"datasetVersion","datasetId":985270},{"sourceId":5181249,"sourceType":"datasetVersion","datasetId":3012199},{"sourceId":5190993,"sourceType":"datasetVersion","datasetId":3018185},{"sourceId":5196408,"sourceType":"datasetVersion","datasetId":3018885},{"sourceId":8036535,"sourceType":"datasetVersion","datasetId":4737648},{"sourceId":8108072,"sourceType":"datasetVersion","datasetId":4789213},{"sourceId":8319412,"sourceType":"datasetVersion","datasetId":4941521}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This work is an ensemble of two high-scoring public notebooks:\n\nhttps://www.kaggle.com/code/tc0000/birdclef-starter-notebook <br>\nhttps://www.kaggle.com/code/aikhmelnytskyy/birdclef24-pretraining-is-all-you-need-infer <br>\n\nAll credit to the authors of those, amazing work!\n\nAdditionally, to improve generalization, since we knew the shakeup and potential of overfitting on public could be a problem, I averaged my predictions over each audiofile, and used this as 75% of the predictions for each row, with 25% the original predictions. ","metadata":{}},{"cell_type":"markdown","source":"# Install Libraries 🛠","metadata":{"papermill":{"duration":0.062037,"end_time":"2022-03-08T03:15:20.082763","exception":false,"start_time":"2022-03-08T03:15:20.020726","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import sys, os\nsys.path.append('/kaggle/input/efficientnet-keras-dataset/efficientnet_kaggle')\n!pip install -q /kaggle/input/tensorflow-extra-lib-ds/tensorflow_extra-1.0.2-py3-none-any.whl --no-deps","metadata":{"execution":{"iopub.status.busy":"2024-04-06T18:22:19.845061Z","iopub.execute_input":"2024-04-06T18:22:19.846166Z","iopub.status.idle":"2024-04-06T18:22:42.113014Z","shell.execute_reply.started":"2024-04-06T18:22:19.846127Z","shell.execute_reply":"2024-04-06T18:22:42.111493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import Libraries 📚","metadata":{"papermill":{"duration":0.065343,"end_time":"2022-03-08T03:18:11.885586","exception":false,"start_time":"2022-03-08T03:18:11.820243","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import tensorflow as tf\ntf.get_logger().setLevel('ERROR')\ntf.autograph.set_verbosity(0)\nimport os\nimport pandas as pd\nimport numpy as np\nimport random\nfrom glob import glob\nfrom tqdm import tqdm\ntqdm.pandas()\nimport gc\nimport librosa\nimport sklearn\nimport time\n\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\nimport librosa.display as lid\nimport IPython.display as ipd\n\nimport tensorflow as tf\ntf.config.optimizer.set_jit(True) # enable xla for speed up\nimport tensorflow_io as tfio\nimport tensorflow.keras.backend as K\n\nimport efficientnet.tfkeras as efn\nimport tensorflow_extra as tfe","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":2.632068,"end_time":"2022-03-08T03:18:14.585094","exception":false,"start_time":"2022-03-08T03:18:11.953026","status":"completed"},"tags":[],"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-04-06T18:22:42.118359Z","iopub.execute_input":"2024-04-06T18:22:42.1188Z","iopub.status.idle":"2024-04-06T18:22:42.130377Z","shell.execute_reply.started":"2024-04-06T18:22:42.118761Z","shell.execute_reply":"2024-04-06T18:22:42.129115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Library Version","metadata":{"papermill":{"duration":0.065649,"end_time":"2022-03-08T03:18:14.717311","exception":false,"start_time":"2022-03-08T03:18:14.651662","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print('np:', np.__version__)\nprint('pd:', pd.__version__)\nprint('sklearn:', sklearn.__version__)\nprint('librosa:', librosa.__version__)\nprint('tf:', tf.__version__)\nprint('tfio:', tfio.__version__)","metadata":{"papermill":{"duration":0.155095,"end_time":"2022-03-08T03:18:14.939054","exception":false,"start_time":"2022-03-08T03:18:14.783959","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-04-06T18:22:42.132205Z","iopub.execute_input":"2024-04-06T18:22:42.132725Z","iopub.status.idle":"2024-04-06T18:22:42.142514Z","shell.execute_reply.started":"2024-04-06T18:22:42.132683Z","shell.execute_reply":"2024-04-06T18:22:42.141555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configuration ⚙️","metadata":{"papermill":{"duration":0.066353,"end_time":"2022-03-08T03:18:18.099835","exception":false,"start_time":"2022-03-08T03:18:18.033482","status":"completed"},"tags":[]}},{"cell_type":"code","source":"class CFG:\n    debug = False\n    verbose = 0\n    \n    device = 'CPU'\n    seed = 42\n    \n    # Input image size and batch size\n    img_size = [128, 384]\n    batch_size = 16\n    infer_bs = 2\n    tta = 1\n    drop_remainder = True\n    \n    # STFT parameters\n    duration = 5 # duration for test\n    train_duration = 10\n    sample_rate = 32000\n    downsample = 1\n    audio_len = duration*sample_rate\n    nfft = 2028\n    window = 2048\n    hop_length = train_duration*32000 // (img_size[1] - 1)\n    fmin = 20\n    fmax = 16000\n    normalize = True\n\n    # Data Preprocessing Settings\n    class_names = sorted(os.listdir('/kaggle/input/birdclef-2024/train_audio/'))\n    num_classes = len(class_names)\n    class_labels = list(range(num_classes))\n    label2name = dict(zip(class_labels, class_names))\n    name2label = {v:k for k,v in label2name.items()}\n    \n    target_col = ['target']\n    tab_cols = ['filename','common_name','rate']","metadata":{"papermill":{"duration":0.156464,"end_time":"2022-03-08T03:18:18.322809","exception":false,"start_time":"2022-03-08T03:18:18.166345","status":"completed"},"tags":[],"_kg_hide-input":false,"execution":{"iopub.status.busy":"2024-04-06T18:22:42.145771Z","iopub.execute_input":"2024-04-06T18:22:42.146674Z","iopub.status.idle":"2024-04-06T18:22:42.158579Z","shell.execute_reply.started":"2024-04-06T18:22:42.146616Z","shell.execute_reply":"2024-04-06T18:22:42.157303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Reproducibility ♻️\nSets value for random seed to produce similar result in each run.","metadata":{"papermill":{"duration":0.070351,"end_time":"2022-03-08T03:18:18.46058","exception":false,"start_time":"2022-03-08T03:18:18.390229","status":"completed"},"tags":[]}},{"cell_type":"code","source":"tf.keras.utils.set_random_seed(CFG.seed)","metadata":{"papermill":{"duration":0.153451,"end_time":"2022-03-08T03:18:18.685056","exception":false,"start_time":"2022-03-08T03:18:18.531605","status":"completed"},"tags":[],"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-04-06T18:22:42.160328Z","iopub.execute_input":"2024-04-06T18:22:42.160673Z","iopub.status.idle":"2024-04-06T18:22:42.219982Z","shell.execute_reply.started":"2024-04-06T18:22:42.160625Z","shell.execute_reply":"2024-04-06T18:22:42.218425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Set Up Device  📱\nFollowing codes automatically detects hardware(tpu or tpu-vm or gpu). ","metadata":{"papermill":{"duration":0.065779,"end_time":"2022-03-08T03:18:18.817867","exception":false,"start_time":"2022-03-08T03:18:18.752088","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def get_device():\n    \"Detect and intializes GPU/TPU automatically\"\n    # Check TPU category\n    tpu = 'local' if CFG.device=='TPU-VM' else None\n    try:\n        # Connect to TPU\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect(tpu=tpu) \n        # Set TPU strategy\n        strategy = tf.distribute.TPUStrategy(tpu)\n        print(f'> Running on {CFG.device} ', tpu.master(), end=' | ')\n        print('Num of TPUs: ', strategy.num_replicas_in_sync)\n        device=CFG.device\n    except:\n        # If TPU is not available, detect GPUs\n        gpus = tf.config.list_logical_devices('GPU')\n        ngpu = len(gpus)\n         # Check number of GPUs\n        if ngpu:\n            # Set GPU strategy\n            strategy = tf.distribute.MirroredStrategy(gpus) # single-GPU or multi-GPU\n            # Print GPU details\n            print(\"> Running on GPU\", end=' | ')\n            print(\"Num of GPUs: \", ngpu)\n            device='GPU'\n        else:\n            # If no GPUs are available, use CPU\n            print(\"> Running on CPU\")\n            strategy = tf.distribute.get_strategy()\n            device='CPU'\n    return strategy, device, tpu","metadata":{"_kg_hide-input":true,"papermill":{"duration":7.941725,"end_time":"2022-03-08T03:18:26.826553","exception":false,"start_time":"2022-03-08T03:18:18.884828","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-04-06T18:22:42.225155Z","iopub.execute_input":"2024-04-06T18:22:42.225636Z","iopub.status.idle":"2024-04-06T18:22:42.239621Z","shell.execute_reply.started":"2024-04-06T18:22:42.225594Z","shell.execute_reply":"2024-04-06T18:22:42.238614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize GPU/TPU/TPU-VM\nstrategy, CFG.device, tpu = get_device()\nCFG.replicas = strategy.num_replicas_in_sync","metadata":{"execution":{"iopub.status.busy":"2024-04-06T18:22:42.240892Z","iopub.execute_input":"2024-04-06T18:22:42.241353Z","iopub.status.idle":"2024-04-06T18:22:42.257251Z","shell.execute_reply.started":"2024-04-06T18:22:42.241319Z","shell.execute_reply":"2024-04-06T18:22:42.255839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataset Path 📁","metadata":{}},{"cell_type":"code","source":"BASE_PATH = '/kaggle/input/birdclef-2024'\nGCS_PATH = BASE_PATH","metadata":{"execution":{"iopub.status.busy":"2024-04-06T18:22:42.258707Z","iopub.execute_input":"2024-04-06T18:22:42.259583Z","iopub.status.idle":"2024-04-06T18:22:42.266864Z","shell.execute_reply.started":"2024-04-06T18:22:42.259538Z","shell.execute_reply":"2024-04-06T18:22:42.265921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Meta Data 📖\n* **test_soundscapes/** - directory contains $~200$ recordings to be used for scoring when a notebook is submitted. Without submission only $1$ recording is accessible.  All recordings are $10$ minutes long and in `.ogg` audio format.\n* **sample_submission.csv** - is the valid sample submission.\n    * `row_id`: A slug of [soundscape_id]_[end_time] for the prediction.\n    * `[bird_id]`: There are $264$ bird ID columns. The probability of the presence of each bird for each row needs to be predicted.","metadata":{"papermill":{"duration":0.067107,"end_time":"2022-03-08T03:18:26.962626","exception":false,"start_time":"2022-03-08T03:18:26.895519","status":"completed"},"tags":[]}},{"cell_type":"code","source":"test_audio_dir = '/kaggle/input/birdclef-2024/test_soundscapes/'\n\ntest_paths = [test_audio_dir+f for f in sorted(os.listdir(test_audio_dir))]\nif len(test_paths)==1:\n    test_audio_dir = '/kaggle/input/birdclef-2024/unlabeled_soundscapes/'\n\n    test_paths = [test_audio_dir+f for f in sorted(os.listdir(test_audio_dir))][:2]\n    \ntest_df = pd.DataFrame(test_paths, columns=['filepath'])\ntest_df['filename'] = test_df.filepath.map(lambda x: x.split('/')[-1].replace('.ogg',''))\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-06T18:22:42.26821Z","iopub.execute_input":"2024-04-06T18:22:42.268552Z","iopub.status.idle":"2024-04-06T18:22:42.296988Z","shell.execute_reply.started":"2024-04-06T18:22:42.268517Z","shell.execute_reply":"2024-04-06T18:22:42.295772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.io.gfile.exists(test_df.filepath.iloc[0])","metadata":{"papermill":{"duration":0.244976,"end_time":"2022-03-08T03:18:33.994955","exception":false,"start_time":"2022-03-08T03:18:33.749979","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-04-06T18:22:42.301394Z","iopub.execute_input":"2024-04-06T18:22:42.301774Z","iopub.status.idle":"2024-04-06T18:22:42.309258Z","shell.execute_reply.started":"2024-04-06T18:22:42.301745Z","shell.execute_reply":"2024-04-06T18:22:42.308098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Loader 🍚","metadata":{"papermill":{"duration":0.077969,"end_time":"2022-03-08T03:18:36.503797","exception":false,"start_time":"2022-03-08T03:18:36.425828","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def load_audio(filepath, sr=32000, normalize=True):\n    audio, orig_sr = librosa.load(filepath, sr=None)\n    if sr!=orig_sr:\n        audio = librosa.resample(y, orig_sr, sr)\n    audio = audio.astype('float32').ravel()\n    audio = tf.convert_to_tensor(audio)\n    return audio\n\n@tf.function(jit_compile=True)\ndef MakeFrame(audio, duration=5, sr=32000):\n    frame_length = int(duration * sr)\n    frame_step = int(duration * sr)\n    chunks = tf.signal.frame(audio, frame_length, frame_step, pad_end=True)\n    return chunks","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.251393,"end_time":"2022-03-08T03:18:36.833376","exception":false,"start_time":"2022-03-08T03:18:36.581983","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-04-06T18:22:42.311128Z","iopub.execute_input":"2024-04-06T18:22:42.311555Z","iopub.status.idle":"2024-04-06T18:22:42.324468Z","shell.execute_reply.started":"2024-04-06T18:22:42.311517Z","shell.execute_reply":"2024-04-06T18:22:42.323226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA 🎨","metadata":{"papermill":{"duration":0.091062,"end_time":"2022-03-08T03:18:37.019504","exception":false,"start_time":"2022-03-08T03:18:36.928442","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## Utility","metadata":{}},{"cell_type":"code","source":"def display_audio(row):\n    # Caption for viz\n    caption = f'Id: {row.filename}'\n    # Read audio file\n    audio = load_audio(row.filepath)\n    # Keep fixed length audio\n    audio = audio[:CFG.audio_len]\n    # Display audio\n    print(\"# Audio:\")\n    display(ipd.Audio(audio.numpy(), rate=CFG.sample_rate))\n    print('# Visualization:')\n    plt.figure(figsize=(12, 3))\n    plt.title(caption)\n    # Waveplot\n    lid.waveshow(audio.numpy(),\n                 sr=CFG.sample_rate,)\n                 \n    plt.xlabel('');\n    plt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-04-06T18:22:42.327793Z","iopub.execute_input":"2024-04-06T18:22:42.328275Z","iopub.status.idle":"2024-04-06T18:22:42.339768Z","shell.execute_reply.started":"2024-04-06T18:22:42.328232Z","shell.execute_reply":"2024-04-06T18:22:42.338887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check","metadata":{}},{"cell_type":"code","source":"display_audio(test_df.iloc[0])","metadata":{"execution":{"iopub.status.busy":"2024-04-06T18:22:42.341201Z","iopub.execute_input":"2024-04-06T18:22:42.341611Z","iopub.status.idle":"2024-04-06T18:22:43.139992Z","shell.execute_reply.started":"2024-04-06T18:22:42.341574Z","shell.execute_reply":"2024-04-06T18:22:43.138729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference Configs 🔧","metadata":{}},{"cell_type":"code","source":"import shutil\n# Directory of checkpoint\nCKPT_DIR = '/kaggle/input/birdclef24-pretraining-train-model'\n\n# Get file paths of all trained models in the directory\nCKPT_PATHS = sorted([x for x in glob(f'{CKPT_DIR}/fold-*keras')])\nprint(\"Checkpoints: \", CKPT_PATHS)\n\n\n# Define a writable directory\nWRITABLE_DIR = '/kaggle/working/models/'\n\n# Create the writable directory if it does not exist\nif not os.path.exists(WRITABLE_DIR):\n    os.makedirs(WRITABLE_DIR)\n\n\n# Copy the model files to the writable directory\nfor ckpt_path in CKPT_PATHS:\n    shutil.copy(ckpt_path, WRITABLE_DIR)\n\n# Update the checkpoint paths to the writable directory\nCKPT_PATHS = sorted([f'{WRITABLE_DIR}/{os.path.basename(x)}' for x in glob(f'{CKPT_DIR}/fold-*keras')])\n\n# Load all the models in memory to speed up\nCKPTS = [tf.keras.models.load_model(x, compile=False) for x in tqdm(CKPT_PATHS, desc=\"Loading ckpts \")]\n# Num of ckpt to use\nNUM_CKPTS = 1\n\n# Submit or Interactive mode\n#SUBMIT = pd.read_csv('/kaggle/input/birdclef-2024/sample_submission.csv').shape[0] != 3","metadata":{"execution":{"iopub.status.busy":"2024-04-06T18:22:43.141382Z","iopub.execute_input":"2024-04-06T18:22:43.142437Z","iopub.status.idle":"2024-04-06T18:22:49.493602Z","shell.execute_reply.started":"2024-04-06T18:22:43.1424Z","shell.execute_reply":"2024-04-06T18:22:49.492405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference 🧪","metadata":{"papermill":{"duration":0.151237,"end_time":"2022-03-08T03:18:47.959873","exception":false,"start_time":"2022-03-08T03:18:47.808636","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Start stopwatch\ntick = time.time()\n\n# Initialize empty list to store ids\nids = []\n# Initialize empty array to store predictions\npreds = np.empty(shape=(0, 182), dtype='float32')\n\n# Iterate over each audio file in the test dataset\nfor filepath in tqdm(test_df.filepath.tolist(), 'test '):\n    # Extract the filename without the extension\n    filename = filepath.split('/')[-1].replace('.ogg','')\n    \n    # Load audio from file and create audio frames, each recording will be a batch input\n    audio = load_audio(filepath)\n    chunks = MakeFrame(audio)\n    \n    # Predict bird species for all frames in a recording using all trained models\n    chunk_preds = np.zeros(shape=(len(chunks), 182), dtype=np.float32)\n    for model in CKPTS[:NUM_CKPTS]:\n        # Get the model's predictions for the current audio frames\n        rec_preds = model(chunks, training=False).numpy()\n        # Ensemble all prediction with average\n        chunk_preds += rec_preds/len(CKPTS)\n    \n    # Create a ID for each frame in a recording using the filename and frame number\n    rec_ids = [f'{filename}_{(frame_id+1)*5}' for frame_id in range(len(chunks))]\n    \n    # Concatenate the ids\n    ids += rec_ids\n    # Concatenate the predictions\n    preds = np.concatenate([preds, chunk_preds], axis=0)\n    \n# Stop stopwatch\ntock = time.time()","metadata":{"execution":{"iopub.status.busy":"2024-04-06T18:22:49.495755Z","iopub.execute_input":"2024-04-06T18:22:49.496624Z","iopub.status.idle":"2024-04-06T18:22:54.337523Z","shell.execute_reply.started":"2024-04-06T18:22:49.496583Z","shell.execute_reply":"2024-04-06T18:22:54.336304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission 📮","metadata":{}},{"cell_type":"code","source":"preds.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-06T18:22:54.338923Z","iopub.execute_input":"2024-04-06T18:22:54.339274Z","iopub.status.idle":"2024-04-06T18:22:54.346122Z","shell.execute_reply.started":"2024-04-06T18:22:54.339244Z","shell.execute_reply":"2024-04-06T18:22:54.345048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Submit prediction\npred_df = pd.DataFrame(ids, columns=['row_id'])\npred_df.loc[:, CFG.class_names] = preds\npred_df.to_csv('submission1.csv',index=False)\npred_df","metadata":{"execution":{"iopub.status.busy":"2024-04-06T18:22:54.347504Z","iopub.execute_input":"2024-04-06T18:22:54.347836Z","iopub.status.idle":"2024-04-06T18:22:54.497737Z","shell.execute_reply.started":"2024-04-06T18:22:54.34781Z","shell.execute_reply":"2024-04-06T18:22:54.496577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check Submission","metadata":{}},{"cell_type":"code","source":"#if not SUBMIT:\n #   pred_labels = pred_df[pred_df.columns[1:]].values.argmax(axis=1)\n #   pred_classes = list(map(lambda x: CFG.label2name[x], pred_labels))\n #   print(pred_classes)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-04-06T18:22:54.499212Z","iopub.execute_input":"2024-04-06T18:22:54.499553Z","iopub.status.idle":"2024-04-06T18:22:54.504134Z","shell.execute_reply.started":"2024-04-06T18:22:54.499525Z","shell.execute_reply":"2024-04-06T18:22:54.502982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission Time ⏰\nEstimated time to complete the submission.\n> **Note**: There are nearly ~$200$ recordings on the test data.","metadata":{}},{"cell_type":"code","source":"sub_time = (tock-tick)*550 # ~1100 recording on the test data\nsub_time = time.gmtime(sub_time)\nsub_time = time.strftime(\"%H hr: %M min : %S sec\", sub_time)\nprint(f\">> Time for submission: ~ {sub_time}\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-04-06T18:22:54.505535Z","iopub.execute_input":"2024-04-06T18:22:54.505883Z","iopub.status.idle":"2024-04-06T18:22:54.517461Z","shell.execute_reply.started":"2024-04-06T18:22:54.505855Z","shell.execute_reply":"2024-04-06T18:22:54.516417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"KAGGLE = True\n\nisTrain = False\nisInference = True\n\nif isTrain == False and isInference == True:\n    newDir = False\nelse:\n    newDir = True\n    \nif KAGGLE == True:\n    !pip install /kaggle/input/onnxruntime/humanfriendly-10.0-py2.py3-none-any.whl --no-index --find-links /kaggle/input/onnxruntime\n    !pip install /kaggle/input/onnxruntime/coloredlogs-15.0.1-py2.py3-none-any.whl --no-index --find-links /kaggle/input/onnxruntime\n    !pip install /kaggle/input/onnxruntime/onnxruntime-1.17.3-cp310-cp310-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl --no-index --find-links /kaggle/input/onnxruntime","metadata":{"execution":{"iopub.status.busy":"2024-06-07T16:14:00.79887Z","iopub.execute_input":"2024-06-07T16:14:00.799245Z","iopub.status.idle":"2024-06-07T16:14:00.833128Z","shell.execute_reply.started":"2024-06-07T16:14:00.799215Z","shell.execute_reply":"2024-06-07T16:14:00.831771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nimport sys\nimport glob\nimport time\nimport shutil\nimport random\nimport warnings\nwarnings.simplefilter(\"ignore\")\n\nimport wandb\n\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import KFold, GroupKFold, StratifiedGroupKFold\n\nfrom tqdm.notebook import tqdm\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom torch.cuda import amp\nimport torch\nprint(f\"pytorch version is {torch.__version__}\")\nimport torch.nn as nn\nfrom torch.cuda import amp\n\nimport torchvision\nfrom torchvision.transforms import v2 as transforms\n\nimport librosa\nimport torchaudio\nimport torchaudio.transforms as audioT\n\nif KAGGLE == False:\n    import nnAudio\n    from nnAudio import features\n    import albumentations\n    from audiomentations import Compose, SpecCompose, OneOf, AddGaussianNoise, AddColorNoise\n    from audiomentations import TimeStretch, PitchShift, Shift, SpecFrequencyMask, TimeMask\n    from audiomentations import Gain, GainTransition\n    from torcheval.metrics.functional import multiclass_auroc, multiclass_f1_score, multiclass_precision, multiclass_recall, multilabel_accuracy\nif KAGGLE == False:\n    from adan_pytorch import Adan\nimport timm","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## decide name\nif KAGGLE == False:\n    if isTrain == True:\n        name = sorted(glob.glob(\"exp1*.ipynb\"))[-1][:-6]\n        print(f\"filename is {name}\")\n    else:\n        name = \"exp1057\"\n        print(f\"filename is {name}\")\nelse:\n    name = \"exp1057\"\n    name = f'bird2024{name}'\n    print(f\"filename is {name}\")\ntrial = \"trial1\"\np_name = f\"BirdCLEF_cv_ver2\"\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class config:\n    if KAGGLE:\n        dir = \"/kaggle/input/birdclef-2024/\"\n    else:\n        dir = \"/mnt/d/kaggle/birdclef-2024/\"\n\n    # wave_path = \"/mnt/d/kaggle/birdclef-2024/original_waves/second_30/\"\n    wave_path = \"original_waves/second_30/\"\n\n    # model_name = 'eca_nfnet_l0'\n    model_name = 'tf_efficientnet_b0'\n\n    pool_type = 'avg'\n\n    \n    train_duration = 30 ##学習に使うデータの秒数\n    slice_duration = 5 ##実際にSTFTをして食わせるデータのサイズ\n\n    test_duration = 5\n\n    train_drop_duration = 1\n    \n    ###spectrogram parameters\n    sr = 32000\n    fmin = 20\n    fmax = 15000\n\n    n_mels = 128\n    n_fft = n_mels*8\n    size_x = 512\n    \n    hop_length = int(sr*slice_duration / size_x)\n    test_hop_length = int(sr*test_duration / size_x)\n    \n    bins_per_octave = 12\n\n    nfolds = 5\n    inference_folds = [4]\n    \n    enable_amp = True\n    train_batchsize = 32\n    valid_batchsize = 1\n\n    # loss_type = \"BCEWithLogitsLoss\"\n    loss_type = \"BCEFocalLoss\"\n\n    lr = 1.0e-03 #for tf_efficientnet_b0\n    # lr = 1.0e-04 #for movilenet\n    # lr = 1.0e-05 #for movilenet\n\n    optimizer='adan'\n    # optimizer='adamW'\n    weight_decay = 1.0e-02\n    es_patience =  5\n    deterministic = True\n    enable_amp = True\n\n    max_epoch = 9\n    aug_epoch = 6\n    \n\n    useSecondary =True\n    secondary_label_value = 0.5\n    oversample =False\n    oversample_threthold = 60\n    \n    seed = 42\n\n    wandb = True\n\n    ###augmentation flags\n    aug_noise            = 0.\n    aug_gain             = 0.0\n    aug_wave_pitchshift  = 0.0#効果はあるので入れたいが、重いので実験中は使わない\n    aug_wave_shift       = 0.\n\n    aug_spec_xymasking   = 0.\n    aug_spec_coarsedrop  = 0.\n    aug_spec_hflip       = 0.\n\n    ##mixup param\n    aug_wave_mixup       = 1.0\n    aug_spec_mixup       = 0.0\n    aug_spec_mixup_prob  = 0.5 #specmixup++をさせる確率\n    alpha=0.95\n\n    smoothing_value      = 0.0\n    # spec_mix_mask_percent = 20\n    \ncfg = config()\n\ndevice = torch.device('cuda:0') if torch.cuda.is_available() else torch.device('cpu')\n\n\n# device = torch.device('cpu')\n\nprint(device)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if isTrain== True:\n\n    normal_augment = Compose([\n        OneOf([\n            Gain(min_gain_in_db=-15, max_gain_in_db=15, p=1.0),\n            GainTransition(min_gain_in_db=-24.0, max_gain_in_db=6.0,\n                           min_duration=0.2, max_duration=6.0,  p=1.0)\n        ], p=cfg.aug_gain),\n        \n        OneOf([\n            AddGaussianNoise(p=1),\n            AddColorNoise(p=1, min_snr_db=5, max_snr_db=20, min_f_decay=-3.01, max_f_decay=-3.01)\n        ],p=cfg.aug_noise),\n        # OneOf([\n        #     AddGaussianNoise(min_amplitude=0.001, max_amplitude=0.002, p=1),\n        #     AddColorNoise(p=1, max_f_decay=-3.01, min_snr_db=5, max_snr_db=20, n_fft=cfg.n_fft)\n        # ],p=cfg.aug_noise),\n    \n        PitchShift(min_semitones=-1, max_semitones=1, p=cfg.aug_wave_pitchshift),\n        Shift(p=cfg.aug_wave_shift)\n    ])\n    alb_transform = [\n        ### num_masks_x=1, num_masks_y=1 will change\n        albumentations.XYMasking(num_masks_x=2, num_masks_y=1, \n                                 mask_x_length=cfg.size_x//30, mask_y_length=cfg.n_mels//30,\n                                 fill_value=0, mask_fill_value=0, p=cfg.aug_spec_xymasking),\n        albumentations.CoarseDropout(fill_value=0, min_holes=20, max_holes=50, p=cfg.aug_spec_coarsedrop),\n        albumentations.HorizontalFlip(p=cfg.aug_spec_hflip)    \n    ]\n    albumentations_augment = albumentations.Compose(alb_transform)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mixup(data, targets, alpha, mode=\"same_wave\"):\n    \n    if mode == \"same_wave\":\n        data = torch.tensor(data)\n        indices = torch.randperm(data.size(0))\n        shuffled_data = data[indices]\n        # shuffled_targets = targets[indices]\n        #print(indices)\n        lam = np.random.beta(alpha, alpha)\n        #print(lam)\n        new_data = data * lam + shuffled_data * (1 - lam)\n        # new_targets = targets * lam + shuffled_targets * (1 - lam)\n        return new_data.numpy()\n        \n    elif mode == \"other_wave\":\n        indices = torch.randperm(data.size(0))\n        shuffled_data = data[indices]\n        shuffled_targets = targets[indices]\n    \n        lam = np.random.beta(alpha, alpha)\n        new_data = data * lam + shuffled_data * (1 - lam)\n        new_targets = targets * lam + shuffled_targets * (1 - lam)\n    \n        return new_data, new_targets","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if isTrain== True:\n    spec_xymasking = albumentations.XYMasking(num_masks_x=2, num_masks_y=1, \n                                              mask_x_length=cfg.size_x // 10, mask_y_length=cfg.n_mels // 10,\n                                              fill_value=0, mask_fill_value=0, p=1)\n\ndef spec_mixup(data, targets):\n    type = data.dtype\n\n    indices = torch.randperm(data.size(0))\n    # print(indices)\n    shuffled_data = data[indices]\n    shuffled_targets = targets[indices]\n\n    ##masking\n    data = np.array(data)\n    data_transposed = np.transpose(data, (2, 3, 1, 0))\n    data_transposed = spec_xymasking(image=data_transposed)[\"image\"]\n    data_transposed = np.transpose(data_transposed, (3, 2, 0, 1))  \n\n    ##masking した場所を取り出す\n    diff = data - data_transposed\n    mask = (diff != 0).astype(int)\n\n    ##mask箇所を、ほかのデータから取り出す\n    shuffled_data_masked = (shuffled_data * mask)\n\n    ##mixup\n    new_data = torch.tensor(data_transposed, dtype=type) + torch.tensor(shuffled_data_masked, dtype=type)\n\n    #lamはmask量で決める\n    # print(data.size(0))\n    lam = mask.sum() / len(data) / (cfg.n_mels*cfg.size_x)\n    # print(lam)\n    new_targets = targets * (1-lam) + shuffled_targets *lam\n\n    return new_data, new_targets","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spec_layer = torchaudio.transforms.MelSpectrogram(\n    sample_rate=cfg.sr, hop_length=cfg.hop_length, n_fft=cfg.n_fft,\n    n_mels=cfg.n_mels,f_min=cfg.fmin,f_max=cfg.fmax,mel_scale='slaney',center=True, pad_mode='reflect'\n    # normalized=True\n).to(device)\n\nvalid_spec_layer = torchaudio.transforms.MelSpectrogram(\n    sample_rate=cfg.sr, hop_length=cfg.test_hop_length, n_fft=cfg.n_fft,\n    n_mels=cfg.n_mels,f_min=cfg.fmin,f_max=cfg.fmax,mel_scale='slaney',center=True, pad_mode='reflect'\n    # normalized=True\n).to(device)\n\ntest_spec_layer = torchaudio.transforms.MelSpectrogram(\n    sample_rate=cfg.sr, hop_length=cfg.test_hop_length, n_fft=cfg.n_fft,\n    n_mels=cfg.n_mels,f_min=cfg.fmin,f_max=cfg.fmax,mel_scale='slaney',center=True, pad_mode='reflect'\n    # normalized=True\n).cpu()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if KAGGLE == False:\n    if cfg.wandb == True:\n        wandb.login(key=\"1def2aa62f171b4a95415cc2ad76b5eb1bb37b41\")\n\n    if newDir == True:\n        new_dir_path_recursive = f\"{name}/checkpoint\"\n    \n        os.makedirs(new_dir_path_recursive, exist_ok=True)\n        shutil.rmtree(new_dir_path_recursive)\n        os.makedirs(new_dir_path_recursive, exist_ok=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv(cfg.dir+\"sample_submission.csv\")\nLABELS = list(sample_submission.set_index(\"row_id\").columns)\nLABELS[:5]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if isTrain:\nif KAGGLE == False:\n    train_csv = pd.read_csv(cfg.dir+\"train_eda.csv\")\n    train_csv[\"fileID\"] = train_csv[\"filename\"].map(lambda x:x.split(\"/\")[1][:-4])\n    train_csv.head(2)\nelse:\n    train_csv = pd.read_csv(cfg.dir+\"train_metadata.csv\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import ast\ntrain_csv['new_target'] = train_csv['primary_label'] + ' ' + train_csv['secondary_labels'].map(lambda x: ' '.join(ast.literal_eval(x)))\ntrain_csv['len_new_target'] =train_csv['new_target'].map(lambda x: len(x.split()))\ntrain_csv[\"len_new_target\"].value_counts().plot(kind=\"bar\", figsize=(4,2))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv[\"filename_tmp\"] = train_csv[\"filename\"].map(lambda x:x.split(\"/\")[1][:-4])\nduplicated_filenames = train_csv[\"filename_tmp\"].value_counts()[train_csv[\"filename_tmp\"].value_counts() > 1].index\n\ntrain_csv = train_csv[~train_csv[\"filename_tmp\"].isin(duplicated_filenames)]\ntrain_csv = train_csv.reset_index(drop=True)\n\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BirdCLEF_Dataset(torch.utils.data.Dataset):\n    def __init__(self, df, augmentation=False, mode='train'):\n        if mode == 'train':\n            self.df = df.reset_index(drop=True)\n        elif mode == 'valid':\n            self.df = df.reset_index(drop=True)\n        else:\n            self.df = df\n        self.mode = mode\n        self.augmentation = augmentation\n    \n    def __len__(self):\n        return len(self.df)\n\n    def normalize(self, x):\n        valid_values = x[x != float('-inf')]\n        mean_value = np.mean(valid_values)\n        x[x == float('-inf')] = mean_value\n        # x[x == float('-inf')] = 0\n\n        x = x - x.min()\n        x = x / x.max()\n        return x\n\n    def wave_tile_and_cutoff(self, data):\n        ### ---PREPROCESS wave length to train duration & slice duration-----\n        drop_duration = cfg.sr*cfg.train_drop_duration\n        use_duration  = cfg.sr*cfg.train_duration\n        \n        if len(data[0]) > drop_duration: #最初の1秒を捨てる\n            data = data[:,drop_duration:]\n\n        if len(data[0]) < use_duration: #30秒に満たない場合はタイルする\n            iter = 1 + (use_duration) // len(data[0])\n            data = np.tile(data, (1, iter))\n\n        data = data[:,:use_duration]\n        return data\n\n    def label_smoothing(self, idx, target):\n    \n        secondary_target = target * cfg.secondary_label_value\n    \n        out_of_target_noise_intensity = cfg.smoothing_value/(len(LABELS)-1) #該当label以外に、smootiong_valueを分散させる\n        out_of_target_noise_array = torch.ones(target.shape) * out_of_target_noise_intensity\n        \n        secondary_target_with_noise = secondary_target + out_of_target_noise_array\n        secondary_target_with_noise = torch.clip(secondary_target_with_noise, min=0, max=cfg.secondary_label_value)\n    \n        primary_target = np.isin(LABELS, self.df.loc[idx, \"primary_label\"]).astype(int)\n        primary_target = torch.tensor(primary_target, dtype=torch.float32)\n\n        primary_and_secondary_target_with_noise = primary_target + secondary_target_with_noise\n        new_target = torch.clip(primary_and_secondary_target_with_noise, min=0, max=1)\n    \n        new_target = new_target - primary_target * cfg.smoothing_value\n    \n        return new_target\n\n    \n    def __getitem__(self, idx):\n\n        if self.mode == 'train':\n\n            ### ------------------READ DATA  ------------------------------------\n            if cfg.useSecondary == True:\n                target = np.isin(LABELS, self.df.loc[idx, \"new_target\"].split()).astype(int)\n            else:\n                target = np.isin(LABELS, self.df.loc[idx, \"primary_label\"].split()).astype(int)\n            target = torch.tensor(target, dtype=torch.float32)\n            ### ------------------label smoothing  --------------------------\n            target = self.label_smoothing(idx, target)\n            \n            fileID = self.df.loc[idx, 'fileID'] #filename : ****/****.ogg\n            \n            path = f\"{cfg.wave_path}{fileID}.npy\"\n            wave = np.load(path)\n            ### -----------------------------------------------------------------\n\n            # ---PREPROCESS wave length to train duration & slice duration-------\n            wave = self.wave_tile_and_cutoff(data=wave)\n\n            \n            input_duration = cfg.sr * cfg.slice_duration\n            # middle_shift = cfg.sr * cfg.train_duration // 2 \n            \n            if self.augmentation == True:\n                ### ------------------wave time mixup  --------------------------\n                if cfg.aug_wave_mixup > np.random.random(): #同じwave内でmixupをする\n                    #train_duration -> slice_duration\n                    wave_reshape = wave.reshape(-1, input_duration)\n                    wave = mixup(data=wave_reshape, targets=target, alpha=cfg.alpha, mode=\"same_wave\")\n                    wave = wave[:1,:]\n                else:\n                    wave = wave[:, :input_duration]\n                \n                ### ------------------wave augmentation  ------------------------\n                wave = normal_augment(samples=wave, sample_rate=cfg.sr)\n\n                ### ------------------MAKE SPECTROGRAM  -------------------------\n                wave = torch.tensor(wave).to(device)\n                mel_spec = spec_layer(wave)\n                mel_spec = np.array(mel_spec.cpu())\n\n                mel_spec = np.log(mel_spec)\n                for i in range(len(mel_spec)):\n                    mel_spec[i] = self.normalize(mel_spec[i])\n                mel_spec = torch.tensor(mel_spec)\n                mel_spec = mel_spec[:,:,:cfg.size_x]\n\n                ### ------------------spec augmentation  ------------------------\n                mel_spec = np.array(mel_spec.cpu())\n                mel_spec = np.transpose(mel_spec, (1, 2, 0))                \n                mel_spec = albumentations_augment(image=mel_spec)[\"image\"]\n                mel_spec = np.transpose(mel_spec, (2, 0, 1))\n\n                # ### ------------------label smoothing  --------------------------\n                # target = self.label_smoothing(idx, target)\n                \n            else:\n                wave = wave[:, :input_duration]\n                \n                ### ------------------MAKE SPECTROGRAM  -------------------------\n                wave = torch.tensor(wave).to(device)\n                mel_spec = spec_layer(wave)\n                mel_spec = np.array(mel_spec.cpu())\n\n                mel_spec = np.log(mel_spec)\n\n                for i in range(len(mel_spec)):\n                    mel_spec[i] = self.normalize(mel_spec[i])\n                    \n\n                mel_spec = torch.tensor(mel_spec)\n                mel_spec = mel_spec[:,:,:cfg.size_x]\n\n            \n            mel_spec = torch.tensor(mel_spec)\n\n            \n            return mel_spec, target\n\n        elif self.mode == 'valid':\n            \n            ### ------------------READ DATA  ------------------------------------\n            if cfg.useSecondary == True:\n                target = np.isin(LABELS, self.df.loc[idx, \"new_target\"].split()).astype(int)\n            else:\n                target = np.isin(LABELS, self.df.loc[idx, \"primary_target\"].split()).astype(int)\n            target = torch.tensor(target, dtype=torch.float32)\n            \n            fileID = self.df.loc[idx, 'fileID'] #filename : ****/****.ogg\n            \n            path = f\"{cfg.wave_path}{fileID}.npy\"\n            wave = np.load(path)\n            ### -----------------------------------------------------------------\n\n            # ---PREPROCESS wave length to train duration & slice duration-------\n            wave = self.wave_tile_and_cutoff(data=wave)\n\n            input_duration = cfg.sr*cfg.test_duration\n            wave_reshape = wave.reshape(-1, input_duration)\n\n            wave_reshape = torch.tensor(wave_reshape).to(device)\n            mel_specs = valid_spec_layer(wave_reshape)\n            mel_specs = mel_specs.cpu().numpy()\n\n            mel_specs = np.log(mel_specs)\n            for i in range(len(mel_specs)):\n                mel_specs[i] = self.normalize(mel_specs[i])\n            mel_specs = torch.tensor(mel_specs)\n            \n            mel_specs = mel_specs[:,:,:cfg.size_x]\n\n            targets = torch.tile(target, dims=(mel_specs.shape[0],1))\n            return mel_specs, targets\n\n        ###---------------------------------------------------------------------------------------------------------------------\n        elif self.mode == 'test':\n\n            filepath = self.df[idx]\n            wave, _  = torchaudio.load(filepath)\n            wave = wave[:,:60*4*32000]\n\n            wave_reshaped = wave.reshape(-1, 1, cfg.test_duration*cfg.sr)\n            \n            mel_spec = test_spec_layer(wave_reshaped)\n            mel_spec = np.log(mel_spec)\n\n            mel_spec = np.array(mel_spec)\n            for i in range(len(mel_spec)):\n                mel_spec[i] = self.normalize(mel_spec[i])\n            mel_spec = torch.tensor(mel_spec)\n\n            mel_spec = mel_spec[:,:,:cfg.size_x]\n            return mel_spec\n\n        elif self.mode == 'clean':\n\n            filepath = self.df[idx]\n            wave, _  = torchaudio.load(filepath)\n\n            wave = wave[:, :6*cfg.test_duration*cfg.sr]\n\n            chunk_length = len(wave[0]) // (cfg.test_duration*cfg.sr)\n            \n            wave = wave[:,:chunk_length*cfg.test_duration*cfg.sr]\n\n            wave_reshaped = wave.reshape(-1, 1, cfg.test_duration*cfg.sr)\n            \n            mel_spec = test_spec_layer(wave_reshaped)\n            mel_spec = np.log(mel_spec)\n\n            mel_spec = np.array(mel_spec)\n            for i in range(len(mel_spec)):\n                mel_spec[i] = self.normalize(mel_spec[i])\n            mel_spec = torch.tensor(mel_spec)\n\n            return mel_spec, filepath","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if isTrain:\n    print(\"train data\")\n    dataset = BirdCLEF_Dataset(df=train_csv, augmentation=True,  mode=\"train\")\n    data, target = dataset[270]\n    fig, ax = plt.subplots(figsize=(6,4))\n    plt.imshow(data[0], cmap=\"jet\", origin=\"lower\")\n    plt.show()\n    \n    print(\"validation data\")\n    dataset = BirdCLEF_Dataset(df=train_csv, augmentation=True,  mode=\"valid\")\n    data, target = dataset[270]\n    fig, axes = plt.subplots(figsize=(12,8), nrows=len(data), tight_layout=True)\n    for idx, ax in enumerate(axes.ravel()):\n        ax.imshow(data[idx], cmap=\"jet\", origin=\"lower\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BirdModel(torch.nn.Module):\n    def __init__(self, model_name, pretrained, in_channels, num_classes, pool=\"default\"):\n        super().__init__()\n\n        self.pool = pool\n        self.normalize = transforms.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225])\n        \n        if pool == \"default\":\n            self.backbone = timm.create_model(\n                model_name=model_name, pretrained=pretrained,\n                num_classes=0, in_chans=3)\n        else:\n            self.backbone = timm.create_model(\n                model_name=model_name, pretrained=pretrained,\n                num_classes=0, in_chans=3, global_pool=\"\")\n\n        in_features = self.backbone.num_features\n\n\n        # self.pooling = torch.nn.MaxPool2d()\n        # self.pooling = torch.nn.AvgPool2d()\n        self.max_pooling = torch.nn.Sequential(torch.nn.AdaptiveMaxPool2d(1),\n                                               torch.nn.Flatten(start_dim=1, end_dim=-1))\n        self.avg_pooling = torch.nn.Sequential(torch.nn.AdaptiveAvgPool2d(1),\n                                               torch.nn.Flatten(start_dim=1, end_dim=-1))\n        self.both_pooling_neck = torch.nn.Sequential(torch.nn.BatchNorm1d(2*in_features),\n                                                     torch.nn.Linear(in_features=2*in_features, out_features=in_features))\n        \n        self.head = torch.nn.Sequential(\n            torch.nn.BatchNorm1d(in_features),\n            torch.nn.Linear(in_features=in_features, out_features=256),\n            torch.nn.Hardswish(inplace=True),torch.nn.Dropout(0.1),\n            torch.nn.Linear(in_features=256, out_features=len(LABELS))  \n        )\n\n\n\n        self.active = torch.nn.Sigmoid()\n    def forward(self, x):\n        x = x.expand(-1, 3, -1, -1)\n        x = self.normalize(x)\n        x = self.backbone(x)\n\n        if self.pool == \"max\":\n            x = self.max_pooling(x)\n        elif self.pool == \"avg\":\n            x = self.avg_pooling(x)\n        elif self.pool == \"both\":\n            x_max = self.max_pooling(x)\n            x_avg = self.avg_pooling(x)\n            x = x_max + x_avg\n            # x = torch.cat([x_max, x_avg], dim=1)\n            # x = self.both_pooling_neck(x)\n            \n        x = self.head(x)\n        # x = self.active(x)\n        return x","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if isTrain:\nfrom sklearn.model_selection import KFold, StratifiedKFold, GroupKFold\nskf = StratifiedKFold(n_splits=cfg.nfolds, shuffle=True, random_state=cfg.seed)\nfor fold, (train_index, valid_index) in enumerate(skf.split(train_csv, train_csv['primary_label'])):\n    train_csv.loc[valid_index, 'fold'] = int(fold)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_random_seed(seed: int = 42, deterministic: bool = False):\n    \"\"\"Set seeds\"\"\"\n    random.seed(seed)\n    np.random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)  # type: ignore\n    torch.backends.cudnn.deterministic = deterministic  # type: ignore","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BCEFocalLoss(nn.Module):\n    def __init__(self, alpha=0.25, gamma=2.0):\n        super().__init__()\n        self.alpha = alpha\n        self.gamma = gamma\n\n    def forward(self, preds, targets):\n        bce_loss = nn.BCEWithLogitsLoss(reduction='none')(preds, targets)\n        probas = torch.sigmoid(preds)\n\n        \n\n        tmp = targets * self.alpha * (1. - probas)**self.gamma * bce_loss\n        smp = (1. - targets) * probas**self.gamma * bce_loss\n        \n        loss = tmp + smp\n        loss = loss.mean()\n        return loss","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def initialization():\n    model = BirdModel(model_name=cfg.model_name, pretrained=True, in_channels=3, num_classes=len(LABELS), pool=cfg.pool_type)\n    \n    if cfg.optimizer=='adan':\n        optimizer = Adan(model.parameters(), lr=cfg.lr, betas=(0.02, 0.08, 0.01), weight_decay=cfg.weight_decay)\n    else:\n        optimizer = torch.optim.AdamW(params=model.parameters(), lr=cfg.lr, weight_decay=cfg.weight_decay)\n    \n    scheduler = torch.optim.lr_scheduler.OneCycleLR(\n        optimizer=optimizer, epochs=cfg.max_epoch,\n        pct_start=0.0, steps_per_epoch=len(train_dataloader),\n        max_lr=cfg.lr, div_factor=25, final_div_factor=4.0e-01\n    )\n    \n    scaler = amp.GradScaler(enabled=cfg.enable_amp)\n    if cfg.loss_type == \"BCEWithLogitsLoss\":\n        loss_func = torch.nn.BCEWithLogitsLoss()\n    elif cfg.loss_type == \"BCEFocalLoss\":\n        loss_func = BCEFocalLoss(alpha=1)\n    \n    \n    \n    # loss_func = torch.nn.CrossEntropyLoss()\n    # loss_func = torch.nn.BCELoss()\n\n    return model.to(device), optimizer, scheduler, scaler, loss_func.to(device)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import metrics\nfrom sklearn.metrics import mean_squared_error, roc_auc_score\n\ndef get_oversampled_df(df):\n    \n    new_df = [df]\n\n    low_sample_birds = df[\"primary_label\"].value_counts()[df[\"primary_label\"].value_counts() < cfg.oversample_threthold].index\n    for bird in low_sample_birds:\n        tmp = df[df[\"primary_label\"] == bird]\n        data_num = len(tmp)\n    \n        tiles = 1 + cfg.oversample_threthold // data_num\n    \n        tile_df = []\n        for i in range(tiles):\n            tile_df.append(tmp)\n    \n        tiled_df = pd.concat(tile_df)\n        piece = tiled_df[data_num:cfg.oversample_threthold]\n        new_df.append(piece)\n    \n    return pd.concat(new_df)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = dict()\nmodels_names = dict()\n# for fold in range(cfg.nfolds):\nfor fold in cfg.inference_folds:\n    if KAGGLE == True:\n        bestmodel_path = sorted(glob.glob(f\"/kaggle/input/{name}/checkpoint/fold_{fold}*.pth\"))[-1]\n    else:\n        bestmodel_path = sorted(glob.glob(f\"{name}/checkpoint/fold_{fold}*.pth\"))[-1]\n    print(bestmodel_path)\n    model = BirdModel(model_name=cfg.model_name, pretrained=False, in_channels=1, num_classes=len(LABELS))\n    model.load_state_dict(torch.load(bestmodel_path, map_location=torch.device('cpu')))\n    model = model.eval()\n    models[fold] = model\n\n    models_names[fold] = bestmodel_path.split(\".\")[0]+\".onnx\"\n    print(models_names[fold])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if KAGGLE == True:\n    test_audio_dir = f\"{cfg.dir}test_soundscapes/\"\n    file_list = glob.glob(test_audio_dir+\"*.ogg\")\n    file_list = sorted(file_list)\n\nif KAGGLE == False:\n    test_audio_dir = f\"{cfg.dir}unlabeled_soundscapes/\"\n    file_list = glob.glob(test_audio_dir+\"*.ogg\")\n    file_list = sorted(file_list)[:3]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = BirdCLEF_Dataset(df=file_list, mode=\"test\")\ntest_dataloader = torch.utils.data.DataLoader(dataset=test_dataset, \n                                              batch_size=1, \n                                              shuffle=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_tensor = torch.randn((48, 1, cfg.n_mels, cfg.size_x+1))  # input shape\noutput_names=['output']\ninput_names=[\"x\"]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import onnx\nimport onnxruntime as ort","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# models_names = []\nmodels_names = dict()\n# for fold in range(cfg.nfolds):\nfor fold in cfg.inference_folds:\n    if KAGGLE == True:\n        onnxmodel_path = sorted(glob.glob(f\"/kaggle/input/{name}/checkpoint/fold_{fold}*.onnx\"))[-1]\n    else:\n        onnxmodel_path = sorted(glob.glob(f\"{name}/checkpoint/fold_{fold}*.onnx\"))[-1]\n    print(onnxmodel_path)\n#     models_names.append(onnxmodel_path)\n    models_names[fold] = onnxmodel_path","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"onnx_sessions = dict()\n# for fold in range(cfg.nfolds):\nfor fold in cfg.inference_folds:\n\n    onnx_model = onnx.load(models_names[fold])\n    onnx_model_graph = onnx_model.graph\n    onnx_session = ort.InferenceSession(onnx_model.SerializeToString())\n\n    onnx_sessions[fold] = onnx_session","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"start_time = time.time()\n\npredictions = []\nfor data in tqdm(test_dataloader):\n    \n    preds = []\n    \n#     for fold, session in enumerate(onnx_sessions):\n    for fold in cfg.inference_folds:\n        session = onnx_sessions[fold]\n        pred = session.run(output_names, {input_names[0]: data[0].numpy()})[0]\n        \n        pred = torch.sigmoid(torch.tensor(pred))\n        preds.append(pred)\n    preds_per_batch = torch.stack(preds, axis=0).mean(axis=0)\n    \n    predictions.extend(preds_per_batch)\n    \nif len(predictions)>0:\n    predictions = torch.stack(predictions)\nelse:\n    predictions = predictions\nend_time = time.time()\nuse_time = end_time - start_time","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"{cfg.inference_folds}fold +     3ogg is {round(use_time,1)}[s]\")\nprint(f\"{cfg.inference_folds}fold + 1,100ogg is {round(1100*use_time/3,1)}[s], {round(1100*use_time/3/60,1)}[m]\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bird_cols = sample_submission.columns[1:]\ndf = pd.DataFrame(columns=['row_id']+list(bird_cols))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"row_list = []\nfor file in file_list:\n    dataname = file.split(\"/\")[-1][:-4]\n    for i in range(int(4*60/5)):\n        row = f\"{dataname}_{(i+1)*5}\"\n        row_list.append(row)\n        \ndf['row_id'] = row_list\n\nif len(predictions) < 1:\n    pass\nelse:\n    df[bird_cols] = predictions\n    \ndf.to_csv('submission2.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1 = pd.read_csv('/kaggle/working/submission1.csv').sort_values(by=['row_id'])\ndf2 = pd.read_csv('/kaggle/working/submission2.csv').sort_values(by=['row_id'])\nrow_ids = df1['row_id']\n\ndf1.drop(columns=['row_id'], inplace=True)\ndf2.drop(columns=['row_id'], inplace=True)\n\ndf = 0.5 * df1 + 0.5 * df2\n\ndf['row_id'] = row_ids","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col_names = [col for col in df.columns if col not in ['row_id', 'audiofile']]\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SMOOTHING = True\n\nif SMOOTHING:\n    df['audiofile'] = [x.split('_')[0] for x in df['row_id']]\n\n    # Function to apply smoothing\n    def smooth_predictions(group):\n        # Calculate the mean of predictions for the group\n        mean_predictions = group[col_names].mean()\n        # Apply the smoothing formula\n        group[col_names] = 0.25 * group[col_names] + 0.75 * mean_predictions\n        return group\n\n    # Apply the smoothing to each group\n    df = df.groupby('audiofile').apply(smooth_predictions).reset_index(drop=True)\n\n    # Drop the extra audiofile column if not needed\n    df.drop(columns=['audiofile'], inplace=True)\n\ndf.to_csv('submission.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]}]}