{"metadata":{"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":1664376,"sourceType":"datasetVersion","datasetId":985270},{"sourceId":5181249,"sourceType":"datasetVersion","datasetId":3012199},{"sourceId":8108072,"sourceType":"datasetVersion","datasetId":4789213},{"sourceId":8319412,"sourceType":"datasetVersion","datasetId":4941521},{"sourceId":8478505,"sourceType":"datasetVersion","datasetId":5056677},{"sourceId":8605414,"sourceType":"datasetVersion","datasetId":5149186},{"sourceId":8036535,"sourceType":"datasetVersion","datasetId":4737648},{"sourceId":8679105,"sourceType":"datasetVersion","datasetId":5202665},{"sourceId":8618216,"sourceType":"datasetVersion","datasetId":5093037},{"sourceId":8593760,"sourceType":"datasetVersion","datasetId":4904171},{"sourceId":8627864,"sourceType":"datasetVersion","datasetId":4807916}],"dockerImageVersionId":30698,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.13"},"papermill":{"default_parameters":{},"duration":72.272366,"end_time":"2024-05-27T08:24:12.863444","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-05-27T08:23:00.591078","version":"2.5.0"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"087aee61943640ff866fad504d3bb8b7":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"DescriptionStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"2412ab76a24d443693ab5c99798255d6":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":"20px"}},"3051ba21b6bb4960912b45a267d7954e":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"5b524d1526854b07a9cbd17a164389ac":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_3051ba21b6bb4960912b45a267d7954e","placeholder":"​","style":"IPY_MODEL_087aee61943640ff866fad504d3bb8b7","value":""}},"603297b6cf5a4516b17db45b449d8bcd":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_aee52a129532428f97b4ffa466195640","placeholder":"​","style":"IPY_MODEL_98758643ec6f4277946a0b2523b33a62","value":" 0/0 [00:00&lt;?, ?it/s]"}},"7de4c22f5bde4876bd7045b85b99ed48":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_5b524d1526854b07a9cbd17a164389ac","IPY_MODEL_919bacdd323e48c4a80b2ef6b3bdc060","IPY_MODEL_603297b6cf5a4516b17db45b449d8bcd"],"layout":"IPY_MODEL_85db54c088da49578b1a07af47f9565e"}},"85db54c088da49578b1a07af47f9565e":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"919bacdd323e48c4a80b2ef6b3bdc060":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"ProgressView","bar_style":"success","description":"","description_tooltip":null,"layout":"IPY_MODEL_2412ab76a24d443693ab5c99798255d6","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_dd98675913fb42bca53f7133abd10dae","value":0}},"98758643ec6f4277946a0b2523b33a62":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"DescriptionStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"aee52a129532428f97b4ffa466195640":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"dd98675913fb42bca53f7133abd10dae":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","bar_color":null,"description_width":""}}},"version_major":2,"version_minor":0}}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\n    !pip install /kaggle/input/onnxruntime/humanfriendly-10.0-py2.py3-none-any.whl --no-index --find-links /kaggle/input/onnxruntime\n    !pip install /kaggle/input/onnxruntime/coloredlogs-15.0.1-py2.py3-none-any.whl --no-index --find-links /kaggle/input/onnxruntime\n    !pip install /kaggle/input/onnxruntime/onnxruntime-1.17.3-cp310-cp310-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl --no-index --find-links /kaggle/input/onnxruntime","metadata":{"papermill":{"duration":47.930075,"end_time":"2024-05-27T08:23:51.84393","exception":false,"start_time":"2024-05-27T08:23:03.913855","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-07-04T07:30:06.661474Z","iopub.execute_input":"2024-07-04T07:30:06.661918Z","iopub.status.idle":"2024-07-04T07:30:55.836331Z","shell.execute_reply.started":"2024-07-04T07:30:06.661887Z","shell.execute_reply":"2024-07-04T07:30:55.834786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nimport sys\nimport glob\nimport time\nimport shutil\nimport random\nimport ast\n\nimport warnings\nwarnings.simplefilter(\"ignore\")\nimport onnx\nimport onnxruntime as ort\nimport wandb\n\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import KFold, GroupKFold, StratifiedGroupKFold\nfrom sklearn.model_selection import KFold, StratifiedKFold, GroupKFold\nfrom sklearn import metrics\nfrom sklearn.metrics import mean_squared_error, roc_auc_score\nfrom tqdm.notebook import tqdm\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom torch.cuda import amp\nimport torch\nprint(f\"pytorch version is {torch.__version__}\")\nimport torch.nn as nn\nfrom torch.cuda import amp\n\nisTrain = False\nname = 'bird2024exp1057'\n\nimport torchvision\nfrom torchvision.transforms import v2 as transforms\n\nimport librosa\nimport torchaudio\nimport torchaudio.transforms as audioT\n\nimport timm","metadata":{"papermill":{"duration":13.435679,"end_time":"2024-05-27T08:24:05.355498","exception":false,"start_time":"2024-05-27T08:23:51.919819","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-07-04T07:30:55.839029Z","iopub.execute_input":"2024-07-04T07:30:55.839452Z","iopub.status.idle":"2024-07-04T07:31:07.363293Z","shell.execute_reply.started":"2024-07-04T07:30:55.839399Z","shell.execute_reply":"2024-07-04T07:31:07.362101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class config:\n    dir = \"/kaggle/input/birdclef-2024/\"\n\n\n    wave_path = \"original_waves/second_30/\"\n\n    model_name = 'tf_efficientnet_b0'\n\n    pool_type = 'avg'\n\n    \n    train_duration = 30 \n    slice_duration = 5 \n\n    test_duration = 5\n\n    train_drop_duration = 1\n    \n    # spectrogram parameters\n    sr = 32000\n    fmin = 20\n    fmax = 15000\n\n    n_mels = 128\n    n_fft = n_mels*8\n    size_x = 512\n    \n    hop_length = int(sr*slice_duration / size_x)\n    test_hop_length = int(sr*test_duration / size_x)\n    \n    bins_per_octave = 12\n\n    nfolds = 5\n    inference_folds = [4]\n    \n    enable_amp = True\n    train_batchsize = 32\n    valid_batchsize = 1\n\n    # loss_type = \"BCEWithLogitsLoss\"\n    loss_type = \"BCEFocalLoss\"\n\n    #调整学习率，变大，收敛快一点\n    lr = 2.0e-04 \n\n\n    optimizer='adan'\n    weight_decay = 1.0e-03  #更改过拟合2\n    es_patience =  5\n    deterministic = True\n    enable_amp = True\n\n    max_epoch = 9\n    aug_epoch = 7   #数据增强\n    \n\n    useSecondary =True\n    #置信度\n    secondary_label_value = 0.6\n    #决定是否对数据进行过采样。如果设置为 True，表明你打算增加少数类样本的数量，以此来减少类别不平衡的问题。\n    oversample =True\n    oversample_threthold = 5\n    \n    seed = 42\n\n    wandb = True\n\n    ###augmentation flags   音频增强参数设置\n    aug_noise            = 0.\n    aug_gain             = 0.0\n    aug_wave_pitchshift  = 0.0\n    aug_wave_shift       = 0.\n\n    aug_spec_xymasking   = 0.\n    aug_spec_coarsedrop  = 0.\n    aug_spec_hflip       = 0.\n\n    ##mixup param\n    aug_wave_mixup       = 1.0\n    aug_spec_mixup       = 0.1\n    aug_spec_mixup_prob  = 0.5 \n    alpha=0.96\n\n    smoothing_value      = 0.0\n    # spec_mix_mask_percent = 20\n    \ncfg = config()\n\ndevice = torch.device('cuda:0') if torch.cuda.is_available() else torch.device('cpu')\n\nprint(device)","metadata":{"papermill":{"duration":0.045294,"end_time":"2024-05-27T08:24:05.579365","exception":false,"start_time":"2024-05-27T08:24:05.534071","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-07-04T07:31:07.365071Z","iopub.execute_input":"2024-07-04T07:31:07.365845Z","iopub.status.idle":"2024-07-04T07:31:07.381129Z","shell.execute_reply.started":"2024-07-04T07:31:07.365799Z","shell.execute_reply":"2024-07-04T07:31:07.379714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if isTrain== True:\n#多个数据增强操作\n    normal_augment = Compose([\n        OneOf([\n            Gain(min_gain_in_db=-15, max_gain_in_db=15, p=1.0),\n            GainTransition(min_gain_in_db=-24.0, max_gain_in_db=6.0,\n                           min_duration=0.2, max_duration=6.0,  p=1.0)\n        ], p=cfg.aug_gain),\n        \n        OneOf([\n            AddGaussianNoise(p=1),\n            AddColorNoise(p=1, min_snr_db=5, max_snr_db=20, min_f_decay=-3.01, max_f_decay=-3.01)\n        ],p=cfg.aug_noise),\n\n    \n        PitchShift(min_semitones=-1, max_semitones=1, p=cfg.aug_wave_pitchshift),\n        Shift(p=cfg.aug_wave_shift)\n    ])\n    alb_transform = [\n        albumentations.XYMasking(num_masks_x=2, num_masks_y=1, \n                                 mask_x_length=cfg.size_x//30, mask_y_length=cfg.n_mels//30,\n                                 fill_value=0, mask_fill_value=0, p=cfg.aug_spec_xymasking),\n        albumentations.CoarseDropout(fill_value=0, min_holes=20, max_holes=50, p=cfg.aug_spec_coarsedrop),\n        albumentations.HorizontalFlip(p=cfg.aug_spec_hflip)    \n    ]\n    albumentations_augment = albumentations.Compose(alb_transform)","metadata":{"papermill":{"duration":0.041072,"end_time":"2024-05-27T08:24:05.746803","exception":false,"start_time":"2024-05-27T08:24:05.705731","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-07-04T07:31:07.384372Z","iopub.execute_input":"2024-07-04T07:31:07.384811Z","iopub.status.idle":"2024-07-04T07:31:07.409267Z","shell.execute_reply.started":"2024-07-04T07:31:07.384769Z","shell.execute_reply":"2024-07-04T07:31:07.407935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mixup(data, targets, alpha, mode=\"same_wave\"):\n    \n    if mode == \"same_wave\":\n        data = torch.tensor(data)\n        indices = torch.randperm(data.size(0))\n        shuffled_data = data[indices]\n\n        lam = np.random.beta(alpha, alpha)\n        new_data = data * lam + shuffled_data * (1 - lam)\n        return new_data.numpy()\n     #保证数据以及标签都有泛化   \n    elif mode == \"other_wave\":\n        indices = torch.randperm(data.size(0))\n        shuffled_data = data[indices]\n        shuffled_targets = targets[indices]\n    \n        lam = np.random.beta(alpha, alpha)\n        new_data = data * lam + shuffled_data * (1 - lam)\n        new_targets = targets * lam + shuffled_targets * (1 - lam)\n    \n        return new_data, new_targets","metadata":{"papermill":{"duration":0.040647,"end_time":"2024-05-27T08:24:05.864611","exception":false,"start_time":"2024-05-27T08:24:05.823964","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-07-04T07:31:07.411257Z","iopub.execute_input":"2024-07-04T07:31:07.411780Z","iopub.status.idle":"2024-07-04T07:31:07.429821Z","shell.execute_reply.started":"2024-07-04T07:31:07.411736Z","shell.execute_reply":"2024-07-04T07:31:07.428223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if isTrain== True:\n    spec_xymasking = albumentations.XYMasking(num_masks_x=2, num_masks_y=1, \n                                              mask_x_length=cfg.size_x // 10, mask_y_length=cfg.n_mels // 10,\n                                              fill_value=0, mask_fill_value=0, p=1)\n\ndef spec_mixup(data, targets):\n    type = data.dtype\n\n    #只是改变顺序，不改变对应关系\n    indices = torch.randperm(data.size(0))\n    shuffled_data = data[indices]\n    shuffled_targets = targets[indices]\n\n    data = np.array(data)\n    data_transposed = np.transpose(data, (2, 3, 1, 0))\n    data_transposed = spec_xymasking(image=data_transposed)[\"image\"]\n    data_transposed = np.transpose(data_transposed, (3, 2, 0, 1))  \n\n    #差异不为0的位置在掩码中为1，表示数据被改变；只混合被掩码覆盖的数据点\n    diff = data - data_transposed\n    mask = (diff != 0).astype(int)\n\n    shuffled_data_masked = (shuffled_data * mask)\n\n    new_data = torch.tensor(data_transposed, dtype=type) + torch.tensor(shuffled_data_masked, dtype=type)\n\n    lam = mask.sum() / len(data) / (cfg.n_mels*cfg.size_x)\n    new_targets = targets * (1-lam) + shuffled_targets *lam\n\n    return new_data, new_targets","metadata":{"papermill":{"duration":0.042092,"end_time":"2024-05-27T08:24:06.084224","exception":false,"start_time":"2024-05-27T08:24:06.042132","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-07-04T07:31:07.431493Z","iopub.execute_input":"2024-07-04T07:31:07.432051Z","iopub.status.idle":"2024-07-04T07:31:07.445275Z","shell.execute_reply.started":"2024-07-04T07:31:07.432015Z","shell.execute_reply":"2024-07-04T07:31:07.443951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#转换为Mel频谱表示\nspec_layer = torchaudio.transforms.MelSpectrogram(\n    sample_rate=cfg.sr, hop_length=cfg.hop_length, n_fft=cfg.n_fft,\n    n_mels=cfg.n_mels,f_min=cfg.fmin,f_max=cfg.fmax,mel_scale='slaney',center=True, pad_mode='reflect'\n).to(device)\n#测试集\nvalid_spec_layer = torchaudio.transforms.MelSpectrogram(\n    sample_rate=cfg.sr, hop_length=cfg.test_hop_length, n_fft=cfg.n_fft,\n    n_mels=cfg.n_mels,f_min=cfg.fmin,f_max=cfg.fmax,mel_scale='slaney',center=True, pad_mode='reflect'\n).to(device)\n#训练层\ntest_spec_layer = torchaudio.transforms.MelSpectrogram(\n    sample_rate=cfg.sr, hop_length=cfg.test_hop_length, n_fft=cfg.n_fft,\n    n_mels=cfg.n_mels,f_min=cfg.fmin,f_max=cfg.fmax,mel_scale='slaney',center=True, pad_mode='reflect'\n).cpu()","metadata":{"papermill":{"duration":0.181583,"end_time":"2024-05-27T08:24:06.34538","exception":false,"start_time":"2024-05-27T08:24:06.163797","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-07-04T07:31:07.446876Z","iopub.execute_input":"2024-07-04T07:31:07.447286Z","iopub.status.idle":"2024-07-04T07:31:07.602893Z","shell.execute_reply.started":"2024-07-04T07:31:07.447253Z","shell.execute_reply":"2024-07-04T07:31:07.601573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#处理标签、识别和删除重复的文件名\nsample_submission = pd.read_csv(cfg.dir+\"sample_submission.csv\")\nLABELS = list(sample_submission.set_index(\"row_id\").columns)\nLABELS[:5]\ntrain_csv = pd.read_csv(cfg.dir+\"train_metadata.csv\")\ntrain_csv['new_target'] = train_csv['primary_label'] + ' ' + train_csv['secondary_labels'].map(lambda x: ' '.join(ast.literal_eval(x)))\ntrain_csv['len_new_target'] =train_csv['new_target'].map(lambda x: len(x.split()))\ntrain_csv[\"len_new_target\"].value_counts().plot(kind=\"bar\", figsize=(4,2))\ntrain_csv[\"filename_tmp\"] = train_csv[\"filename\"].map(lambda x:x.split(\"/\")[1][:-4])\nduplicated_filenames = train_csv[\"filename_tmp\"].value_counts()[train_csv[\"filename_tmp\"].value_counts() > 1].index\ntrain_csv = train_csv[~train_csv[\"filename_tmp\"].isin(duplicated_filenames)]\ntrain_csv = train_csv.reset_index(drop=True)","metadata":{"papermill":{"duration":0.945215,"end_time":"2024-05-27T08:24:07.431523","exception":false,"start_time":"2024-05-27T08:24:06.486308","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-07-04T07:31:07.604764Z","iopub.execute_input":"2024-07-04T07:31:07.605208Z","iopub.status.idle":"2024-07-04T07:31:08.543178Z","shell.execute_reply.started":"2024-07-04T07:31:07.605167Z","shell.execute_reply":"2024-07-04T07:31:08.541859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BirdCLEF_Dataset(torch.utils.data.Dataset):\n    #augmentation 是否应用数据增强;是否真的有采用？\n    def __init__(self, df, augmentation=False, mode='train'):\n        if mode == 'train':\n            self.df = df.reset_index(drop=True)\n        elif mode == 'valid':\n            self.df = df.reset_index(drop=True)\n        else:\n            self.df = df\n        self.mode = mode\n        self.augmentation = augmentation\n    \n    def __len__(self):\n        return len(self.df)\n#规范化处理，数据在【0,1】之间\n    def normalize(self, x):\n        valid_values = x[x != float('-inf')]\n        mean_value = np.mean(valid_values)\n        x[x == float('-inf')] = mean_value\n        \n\n        x = x - x.min()\n        x = x / x.max()\n        return x\n#确保音频长度符合要求\n    def wave_tile_and_cutoff(self, data):\n      \n        drop_duration = cfg.sr*cfg.train_drop_duration\n        use_duration  = cfg.sr*cfg.train_duration\n        \n        if len(data[0]) > drop_duration: \n            data = data[:,drop_duration:]\n\n        if len(data[0]) < use_duration:\n            iter = 1 + (use_duration) // len(data[0])\n            data = np.tile(data, (1, iter))\n\n        data = data[:,:use_duration]\n        return data\n#避免过拟合\n    def label_smoothing(self, idx, target):\n    \n        secondary_target = target * cfg.secondary_label_value\n    \n        out_of_target_noise_intensity = cfg.smoothing_value/(len(LABELS)-1) \n        out_of_target_noise_array = torch.ones(target.shape) * out_of_target_noise_intensity\n        \n        secondary_target_with_noise = secondary_target + out_of_target_noise_array\n        secondary_target_with_noise = torch.clip(secondary_target_with_noise, min=0, max=cfg.secondary_label_value)\n    \n        primary_target = np.isin(LABELS, self.df.loc[idx, \"primary_label\"]).astype(int)\n        primary_target = torch.tensor(primary_target, dtype=torch.float32)\n\n        primary_and_secondary_target_with_noise = primary_target + secondary_target_with_noise\n        new_target = torch.clip(primary_and_secondary_target_with_noise, min=0, max=1)\n    \n        new_target = new_target - primary_target * cfg.smoothing_value\n    \n        return new_target\n\n    \n    def __getitem__(self, idx):\n#训练集\n        if self.mode == 'train':\n\n          \n            if cfg.useSecondary == True:\n                target = np.isin(LABELS, self.df.loc[idx, \"new_target\"].split()).astype(int)\n            else:\n                target = np.isin(LABELS, self.df.loc[idx, \"primary_label\"].split()).astype(int)\n            target = torch.tensor(target, dtype=torch.float32)\n          \n            target = self.label_smoothing(idx, target)\n            \n            fileID = self.df.loc[idx, 'fileID'] \n            \n            path = f\"{cfg.wave_path}{fileID}.npy\"\n            wave = np.load(path)\n            \n\n      \n            wave = self.wave_tile_and_cutoff(data=wave)\n\n            \n            input_duration = cfg.sr * cfg.slice_duration\n            \n            \n            if self.augmentation == True:\n               \n                if cfg.aug_wave_mixup > np.random.random():\n                    #train_duration -> slice_duration\n                    wave_reshape = wave.reshape(-1, input_duration)\n                    wave = mixup(data=wave_reshape, targets=target, alpha=cfg.alpha, mode=\"same_wave\")\n                    wave = wave[:1,:]\n                else:\n                    wave = wave[:, :input_duration]\n                \n     \n                wave = normal_augment(samples=wave, sample_rate=cfg.sr)\n\n    \n                wave = torch.tensor(wave).to(device)\n                mel_spec = spec_layer(wave)\n                mel_spec = np.array(mel_spec.cpu())\n\n                mel_spec = np.log(mel_spec)\n                for i in range(len(mel_spec)):\n                    mel_spec[i] = self.normalize(mel_spec[i])\n                mel_spec = torch.tensor(mel_spec)\n                mel_spec = mel_spec[:,:,:cfg.size_x]\n\n     \n                mel_spec = np.array(mel_spec.cpu())\n                mel_spec = np.transpose(mel_spec, (1, 2, 0))                \n                mel_spec = albumentations_augment(image=mel_spec)[\"image\"]\n                mel_spec = np.transpose(mel_spec, (2, 0, 1))\n\n\n                \n            else:\n                wave = wave[:, :input_duration]\n                \n                wave = torch.tensor(wave).to(device)\n                mel_spec = spec_layer(wave)\n                mel_spec = np.array(mel_spec.cpu())\n\n                mel_spec = np.log(mel_spec)\n\n                for i in range(len(mel_spec)):\n                    mel_spec[i] = self.normalize(mel_spec[i])\n                    \n\n                mel_spec = torch.tensor(mel_spec)\n                mel_spec = mel_spec[:,:,:cfg.size_x]\n\n            \n            mel_spec = torch.tensor(mel_spec)\n\n            \n            return mel_spec, target\n#验证集  用于评估模型的性能，不需要数据增强\n        elif self.mode == 'valid':\n            \n\n            if cfg.useSecondary == True:\n                target = np.isin(LABELS, self.df.loc[idx, \"new_target\"].split()).astype(int)\n            else:\n                target = np.isin(LABELS, self.df.loc[idx, \"primary_target\"].split()).astype(int)\n            target = torch.tensor(target, dtype=torch.float32)\n            \n            fileID = self.df.loc[idx, 'fileID'] \n            \n            path = f\"{cfg.wave_path}{fileID}.npy\"\n            wave = np.load(path)\n\n            wave = self.wave_tile_and_cutoff(data=wave)\n\n            input_duration = cfg.sr*cfg.test_duration\n            wave_reshape = wave.reshape(-1, input_duration)\n\n            wave_reshape = torch.tensor(wave_reshape).to(device)\n            mel_specs = valid_spec_layer(wave_reshape)\n            mel_specs = mel_specs.cpu().numpy()\n\n            mel_specs = np.log(mel_specs)\n            for i in range(len(mel_specs)):\n                mel_specs[i] = self.normalize(mel_specs[i])\n            mel_specs = torch.tensor(mel_specs)\n            \n            mel_specs = mel_specs[:,:,:cfg.size_x]\n\n            targets = torch.tile(target, dims=(mel_specs.shape[0],1))\n            return mel_specs, targets\n#测试集 使用未见过的数据测试模型的泛化能力\n        elif self.mode == 'test':\n\n            filepath = self.df[idx]\n            wave, _  = torchaudio.load(filepath)\n            wave = wave[:,:60*4*32000]\n\n            wave_reshaped = wave.reshape(-1, 1, cfg.test_duration*cfg.sr)\n            \n            mel_spec = test_spec_layer(wave_reshaped)\n            mel_spec = np.log(mel_spec)\n\n            mel_spec = np.array(mel_spec)\n            for i in range(len(mel_spec)):\n                mel_spec[i] = self.normalize(mel_spec[i])\n            mel_spec = torch.tensor(mel_spec)\n\n            mel_spec = mel_spec[:,:,:cfg.size_x]\n            return mel_spec\n#可以评估模型对未见数据的性能，尤其是在测试或验证阶段。\n        elif self.mode == 'clean':\n\n            filepath = self.df[idx]\n            wave, _  = torchaudio.load(filepath)\n\n            wave = wave[:, :6*cfg.test_duration*cfg.sr]\n\n            chunk_length = len(wave[0]) // (cfg.test_duration*cfg.sr)\n            \n            wave = wave[:,:chunk_length*cfg.test_duration*cfg.sr]\n\n            wave_reshaped = wave.reshape(-1, 1, cfg.test_duration*cfg.sr)\n            \n            mel_spec = test_spec_layer(wave_reshaped)\n            mel_spec = np.log(mel_spec)\n\n            mel_spec = np.array(mel_spec)\n            for i in range(len(mel_spec)):\n                mel_spec[i] = self.normalize(mel_spec[i])\n            mel_spec = torch.tensor(mel_spec)\n\n            return mel_spec, filepath","metadata":{"papermill":{"duration":0.078512,"end_time":"2024-05-27T08:24:07.588078","exception":false,"start_time":"2024-05-27T08:24:07.509566","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-07-04T07:31:08.545221Z","iopub.execute_input":"2024-07-04T07:31:08.545838Z","iopub.status.idle":"2024-07-04T07:31:08.596269Z","shell.execute_reply.started":"2024-07-04T07:31:08.545794Z","shell.execute_reply":"2024-07-04T07:31:08.594956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if isTrain:\n    print(\"train data\")\n    dataset = BirdCLEF_Dataset(df=train_csv, augmentation=True,  mode=\"train\")\n    data, target = dataset[270]\n    fig, ax = plt.subplots(figsize=(6,4))\n    plt.imshow(data[0], cmap=\"jet\", origin=\"lower\")\n    plt.show()\n    \n    print(\"validation data\")\n    dataset = BirdCLEF_Dataset(df=train_csv, augmentation=True,  mode=\"valid\")\n    data, target = dataset[270]\n    fig, axes = plt.subplots(figsize=(12,8), nrows=len(data), tight_layout=True)\n    for idx, ax in enumerate(axes.ravel()):\n        ax.imshow(data[idx], cmap=\"jet\", origin=\"lower\")","metadata":{"papermill":{"duration":0.040422,"end_time":"2024-05-27T08:24:07.705757","exception":false,"start_time":"2024-05-27T08:24:07.665335","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-07-04T07:31:08.601264Z","iopub.execute_input":"2024-07-04T07:31:08.601702Z","iopub.status.idle":"2024-07-04T07:31:08.615177Z","shell.execute_reply.started":"2024-07-04T07:31:08.601658Z","shell.execute_reply":"2024-07-04T07:31:08.613818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BirdModel(torch.nn.Module):\n    def __init__(self, model_name, pretrained, in_channels, num_classes, pool=\"default\"):\n        super().__init__()\n\n        self.pool = pool\n        self.normalize = transforms.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225])\n        #数值为均值以及标准差\n        if pool == \"default\":\n            self.backbone = timm.create_model(\n                model_name=model_name, pretrained=pretrained,\n                num_classes=0, in_chans=3)\n        #采用自定义的全局池化\n        else:\n            self.backbone = timm.create_model(\n                model_name=model_name, pretrained=pretrained,\n                num_classes=0, in_chans=3, global_pool=\"\")\n        #获取输出特征数量\n        in_features = self.backbone.num_features\n\n\n\n        self.max_pooling = torch.nn.Sequential(torch.nn.AdaptiveMaxPool2d(1),\n                                               torch.nn.Flatten(start_dim=1, end_dim=-1))\n        self.avg_pooling = torch.nn.Sequential(torch.nn.AdaptiveAvgPool2d(1),\n                                               torch.nn.Flatten(start_dim=1, end_dim=-1))\n        #批量归一层，线性层\n        self.both_pooling_neck = torch.nn.Sequential(torch.nn.BatchNorm1d(2*in_features),\n                                                     torch.nn.Linear(in_features=2*in_features, out_features=in_features))\n        \n        self.head = torch.nn.Sequential(\n            torch.nn.BatchNorm1d(in_features),\n            torch.nn.Linear(in_features=in_features, out_features=256),\n            torch.nn.Hardswish(inplace=True),torch.nn.Dropout(0.1),\n            torch.nn.Linear(in_features=256, out_features=len(LABELS))  \n        )\n\n\n\n        self.active = torch.nn.Sigmoid()\n    def forward(self, x):\n        x = x.expand(-1, 3, -1, -1)\n        x = self.normalize(x)\n        x = self.backbone(x)\n\n        if self.pool == \"max\":\n            x = self.max_pooling(x)\n        elif self.pool == \"avg\":\n            x = self.avg_pooling(x)\n        elif self.pool == \"both\":\n            x_max = self.max_pooling(x)\n            x_avg = self.avg_pooling(x)\n            x = x_max + x_avg\n            # x = torch.cat([x_max, x_avg], dim=1)\n            # x = self.both_pooling_neck(x)\n         #池化后的输出模型传递到模型分类头部   \n        x = self.head(x)\n        # x = self.active(x)\n        return x","metadata":{"papermill":{"duration":0.048113,"end_time":"2024-05-27T08:24:07.883643","exception":false,"start_time":"2024-05-27T08:24:07.83553","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-07-04T07:31:08.616541Z","iopub.execute_input":"2024-07-04T07:31:08.616944Z","iopub.status.idle":"2024-07-04T07:31:08.636495Z","shell.execute_reply.started":"2024-07-04T07:31:08.616913Z","shell.execute_reply":"2024-07-04T07:31:08.635226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#分层抽样的K折交叉验证，获取行索引\nskf = StratifiedKFold(n_splits=cfg.nfolds, shuffle=True, random_state=cfg.seed)\nfor fold, (train_index, valid_index) in enumerate(skf.split(train_csv, train_csv['primary_label'])):\n    train_csv.loc[valid_index, 'fold'] = int(fold)\n    \n#分组统计 以及评估分层抽样 K 折交叉验证是否成功地保持了每个折叠中类别分布的一致性。\nif isTrain:\n    train_csv.groupby(\"fold\", as_index=False)[\"primary_label\"].value_counts()   \n    \n    \ndef set_random_seed(seed: int = 42, deterministic: bool = False):\n    \"\"\"Set seeds\"\"\"\n    random.seed(seed)\n    np.random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)  # type: ignore\n    torch.backends.cudnn.deterministic = deterministic  # type: ignore    ","metadata":{"papermill":{"duration":0.082794,"end_time":"2024-05-27T08:24:08.04628","exception":false,"start_time":"2024-05-27T08:24:07.963486","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-07-04T07:31:08.637997Z","iopub.execute_input":"2024-07-04T07:31:08.638653Z","iopub.status.idle":"2024-07-04T07:31:08.698367Z","shell.execute_reply.started":"2024-07-04T07:31:08.638601Z","shell.execute_reply":"2024-07-04T07:31:08.697220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BCEFocalLoss(nn.Module):\n    def __init__(self, alpha=0.25, gamma=2.0):\n        super().__init__()\n        self.alpha = alpha\n        self.gamma = gamma\n\n    def forward(self, preds, targets):\n        bce_loss = nn.BCEWithLogitsLoss(reduction='none')(preds, targets)\n        probas = torch.sigmoid(preds)\n\n        \n\n        tmp = targets * self.alpha * (1. - probas)**self.gamma * bce_loss\n        smp = (1. - targets) * probas**self.gamma * bce_loss\n        \n        loss = tmp + smp\n        loss = loss.mean()\n        return loss","metadata":{"papermill":{"duration":0.039605,"end_time":"2024-05-27T08:24:08.164291","exception":false,"start_time":"2024-05-27T08:24:08.124686","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-07-04T07:31:08.699955Z","iopub.execute_input":"2024-07-04T07:31:08.700391Z","iopub.status.idle":"2024-07-04T07:31:08.708891Z","shell.execute_reply.started":"2024-07-04T07:31:08.700353Z","shell.execute_reply":"2024-07-04T07:31:08.707472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def initialization():\n    model = BirdModel(model_name=cfg.model_name, pretrained=True, in_channels=3, num_classes=len(LABELS), pool=cfg.pool_type)\n    \n    if cfg.optimizer=='adan':\n        optimizer = Adan(model.parameters(), lr=cfg.lr, betas=(0.02, 0.08, 0.01), weight_decay=cfg.weight_decay)\n    else:\n        optimizer = torch.optim.AdamW(params=model.parameters(), lr=cfg.lr, weight_decay=cfg.weight_decay)\n    \n    scheduler = torch.optim.lr_scheduler.OneCycleLR(\n        optimizer=optimizer, epochs=cfg.max_epoch,\n        pct_start=0.0, steps_per_epoch=len(train_dataloader),\n        max_lr=cfg.lr, div_factor=25, final_div_factor=4.0e-01\n    )\n    \n    scaler = amp.GradScaler(enabled=cfg.enable_amp)\n    if cfg.loss_type == \"BCEWithLogitsLoss\":\n        loss_func = torch.nn.CrossEntropyLoss()\n    elif cfg.loss_type == \"BCEFocalLoss\":\n        loss_func = BCEFocalLoss(alpha=1)\n    \n    \n\n\n    return model.to(device), optimizer, scheduler, scaler, loss_func.to(device)","metadata":{"papermill":{"duration":0.041677,"end_time":"2024-05-27T08:24:08.283638","exception":false,"start_time":"2024-05-27T08:24:08.241961","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-07-04T07:31:08.710709Z","iopub.execute_input":"2024-07-04T07:31:08.711180Z","iopub.status.idle":"2024-07-04T07:31:08.730211Z","shell.execute_reply.started":"2024-07-04T07:31:08.711137Z","shell.execute_reply":"2024-07-04T07:31:08.728824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_one_loop(model, optimizer, scaler, scheduler, dataloader, loss_fn):\n    trainloss = 0; model.train()\n\n    count = 0\n    for idx, (data, label) in enumerate(tqdm(dataloader,leave=False ,desc=\"[train]\")):\n        # label = label.reshape(-1, len(LABELS))\n        \n        data, label = data.to(device), label.to(device)\n        \n        optimizer.zero_grad()\n        with amp.autocast(cfg.enable_amp, dtype=torch.bfloat16):\n        # with amp.autocast(cfg.enable_amp):\n            pred = model.forward(data)\n            loss = loss_fn(pred, label)\n\n        scaler.scale(loss).backward()\n        scaler.step(optimizer)\n        scaler.update()\n\n        scheduler.step()\n        \n        trainloss += loss.item()\n        # print(idx, loss.item())\n        # if cfg.wandb == True:\n        #     wandb.log({f\"train_loss\": loss.item(), f\"lr\":scheduler.get_lr()[0]})\n        del data, label, loss\n        count += 1\n        # if count == 300:\n        # break\n    trainloss /= len(dataloader)\n    if cfg.wandb == True:\n        wandb.log({f\"train_loss\": trainloss, f\"lr\":scheduler.get_lr()[0]})\n    return model, optimizer, scaler, scheduler, trainloss\n\n\ndef mixup_one_loop(model, optimizer, scaler, scheduler, dataloader, loss_fn):\n    trainloss = 0; model.train()\n\n    count = 0\n    for idx, (data, label) in enumerate(tqdm(dataloader,leave=False ,desc=\"[train]\")):\n        if np.random.random()>cfg.aug_spec_mixup_prob:\n            data, label = mixup(data=data, targets=label, alpha=cfg.alpha, mode=\"other_wave\")\n        else:\n            data, label = spec_mixup(data=data, targets=label)\n        data, label = data.to(device), label.to(device)\n        \n        optimizer.zero_grad()\n        with amp.autocast(cfg.enable_amp, dtype=torch.bfloat16):\n        # with amp.autocast(cfg.enable_amp):\n            pred = model.forward(data)\n            loss = loss_fn(pred, label)\n\n        scaler.scale(loss).backward()\n        scaler.step(optimizer)\n        scaler.update()\n\n        scheduler.step()\n        \n        trainloss += loss.item()\n        # print(idx, loss.item())\n        # if cfg.wandb == True:\n        #     wandb.log({f\"lr\":scheduler.get_lr()[0]})\n        del data, label, loss\n        count += 1\n        # if count == 300:\n        # break\n    trainloss /= len(dataloader)\n    if cfg.wandb == True:\n        wandb.log({f\"train_loss\": trainloss, f\"lr\":scheduler.get_lr()[0]})\n    return model, optimizer, scaler, scheduler, trainloss\n\nfrom sklearn.metrics import f1_score\ndef evaluate_validation(model, dataloader, loss_fn):\n    validloss=0\n    model.eval()\n\n    preds, trues, targets = [], [], []\n    \n    for idx, (data, label) in enumerate(tqdm(dataloader,leave=False ,desc=\"[valid]\")):\n        # label = label.reshape(-1, len(LABELS))\n\n        d = data[0].unsqueeze(1)\n        label = label[0]\n        \n        d = d.to(device)\n        # with amp.autocast(cfg.enable_amp):\n        pred = model.forward(d)\n\n        preds.extend(pred.detach().cpu())\n        trues.extend(label)\n        targets.extend(label.argmax(axis=1))\n        \n    #======================== metrics ========================#\n    # y_preds = torch.stack(preds)\n    t = torch.stack(preds)\n    t = torch.sigmoid(t)\n    targets = torch.tensor(targets)\n    y_trues = torch.stack(trues)\n\n\n    validloss = loss_fn(torch.stack(preds), torch.stack(trues))\n    #     # print(idx, loss)\n    #     # wandb.log({\"valid_loss\": loss})\n\n    # validloss /= len(dataloader)\n    \n\n\n# 假设 t 是一个 PyTorch 张量，targets 是标签的列表\n# 将 targets 转换为 PyTorch 张量\n\n    targets_tensor = torch.tensor(targets)\n\n    def calculate_f1_at_threshold(y_true, predictions, threshold, average=\"micro\"):\n    # 应用阈值并计算F1分数\n        binary_predictions = (predictions > threshold).int()\n        return f1_score(y_true, binary_predictions.numpy(), average=average)\n\n# 计算不同阈值下的F1分数\n    f1_scores = {\n        \"F1_03\": calculate_f1_at_threshold(targets_tensor, t, 0.3),\n        \"F1_05\": calculate_f1_at_threshold(targets_tensor, t, 0.5)\n    }\n\n    # 计算AUC和精确度\n    auc = multiclass_auroc(t, targets_tensor, len(LABELS), \"macro\")\n    prec = multiclass_precision(t, targets_tensor, len(LABELS), \"macro\")\n\n     # 计算微观和宏观平均的F1分数\n    f1 = multiclass_f1_score(t, targets_tensor, len(LABELS), \"micro\")\n    f1_macro = multiclass_f1_score(t, targets_tensor, len(LABELS), \"macro\")\n\n# 打印结果\n    print(f\"AUC (Macro): {auc}\")\n    print(f\"Precision (Macro): {prec}\")\n    print(f\"F1 Score (Micro): {f1}\")\n    print(f\"F1 Score (Macro): {f1_macro}\")\n    for key, value in f1_scores.items():\n        print(f\"{key}: {value}\")\n\n   \n    if cfg.wandb == True:\n        wandb.log({f\"valid_loss\": validloss,\n                   f\"AUC\":auc,\n                   # \"auc_micro\":auc_micro,\n                   \"precision\":prec, \n                   # \"recall\":rec, \n                   # \"accuracy\":acc,\n                   f\"F1\":f1,\n                   \"F1_macro\":f1_macro,\n                   f\"F1 30%\":f1_03,\n                   f\"F1 50%\":f1_05})\n    return validloss, auc, f1, f1_03, f1_05, sk_f1_30, sk_f1_50","metadata":{"papermill":{"duration":0.063951,"end_time":"2024-05-27T08:24:08.425685","exception":false,"start_time":"2024-05-27T08:24:08.361734","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-07-04T07:31:08.732008Z","iopub.execute_input":"2024-07-04T07:31:08.732481Z","iopub.status.idle":"2024-07-04T07:31:08.764555Z","shell.execute_reply.started":"2024-07-04T07:31:08.732418Z","shell.execute_reply":"2024-07-04T07:31:08.763257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if isTrain == True:\n    tmp_params = dict(vars(config))\n    del tmp_params['__module__'],tmp_params['__dict__'],tmp_params['__weakref__'],tmp_params['__doc__']\n\ndef get_oversampled_df(df):\n    \n    new_df = [df]\n\n    low_sample_birds = df[\"primary_label\"].value_counts()[df[\"primary_label\"].value_counts() < cfg.oversample_threthold].index\n    for bird in low_sample_birds:\n        tmp = df[df[\"primary_label\"] == bird]\n        data_num = len(tmp)\n    \n        tiles = 1 + cfg.oversample_threthold // data_num\n    \n        tile_df = []\n        for i in range(tiles):\n            tile_df.append(tmp)\n    \n        tiled_df = pd.concat(tile_df)\n        piece = tiled_df[data_num:cfg.oversample_threthold]\n        new_df.append(piece)\n    \n    return pd.concat(new_df)","metadata":{"papermill":{"duration":0.042146,"end_time":"2024-05-27T08:24:08.546191","exception":false,"start_time":"2024-05-27T08:24:08.504045","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-07-04T07:31:08.766866Z","iopub.execute_input":"2024-07-04T07:31:08.767386Z","iopub.status.idle":"2024-07-04T07:31:08.784189Z","shell.execute_reply.started":"2024-07-04T07:31:08.767340Z","shell.execute_reply":"2024-07-04T07:31:08.782861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if isTrain == True:\n    set_random_seed(seed=42)\n    \n    \n    if cfg.wandb == True:\n        wandb.init(project='BirdCLEF_cv_ver2', name=f\"{name}\",\n                   config=tmp_params)\n        \n    # for fold in range(cfg.nfolds):\n    for fold in cfg.inference_folds:\n        train_ = train_csv.loc[train_csv[\"fold\"]!=fold]\n\n        if cfg.oversample == True:\n            train = get_oversampled_df(df=train_)\n        else:\n            train = train_\n        \n        augme_dataset = BirdCLEF_Dataset(df=train, augmentation=True, mode='train')\n        augme_dataloader = torch.utils.data.DataLoader(dataset=augme_dataset, batch_size=cfg.train_batchsize, shuffle=True)\n\n        train_dataset = BirdCLEF_Dataset(df=train, augmentation=False, mode='train')\n        train_dataloader = torch.utils.data.DataLoader(dataset=train_dataset, batch_size=cfg.train_batchsize, shuffle=True)\n        \n        valid = train_csv.loc[train_csv[\"fold\"]==fold]\n        valid_dataset = BirdCLEF_Dataset(df=valid, augmentation=False, mode='valid')\n        valid_dataloader = torch.utils.data.DataLoader(dataset=valid_dataset, batch_size=cfg.valid_batchsize, shuffle=False)\n    \n        model, optimizer, scheduler, scaler, loss_func =  initialization()\n    \n    \n        best_f1 = 0\n        best_auc = 0\n        best_loss = 1.00000\n        for e in range(cfg.max_epoch):\n            start_time = time.time()\n            if e < cfg.aug_epoch:\n                if cfg.aug_spec_mixup > np.random.random():\n                    model, optimizer, scaler, shcheduler, train_loss = mixup_one_loop(model=model,optimizer=optimizer,scaler=scaler, \n                                                                                          scheduler=scheduler,dataloader=augme_dataloader, loss_fn=loss_func)\n                else:\n                    model, optimizer, scaler, shcheduler, train_loss = train_one_loop(model=model,optimizer=optimizer,scaler=scaler, \n                                                                                          scheduler=scheduler,dataloader=augme_dataloader, loss_fn=loss_func)\n\n            else:\n                model, optimizer, scaler, shcheduler, train_loss = train_one_loop(model=model,optimizer=optimizer,scaler=scaler, \n                                                                                          scheduler=scheduler,dataloader=train_dataloader, loss_fn=loss_func)\n            \n            valid_loss, auc, f1, f1_03, f1_05, sk_f1_30, sk_f1_50 = evaluate_validation(model=model, dataloader=valid_dataloader, loss_fn=loss_func)\n            # print(f\"epoch {e} , train_loss is {train_loss}, valid_loss is {valid_loss}\")\n            \n            if best_loss > valid_loss:\n                end_time = time.time()\n                print(f\"[epoch {str(e).zfill(2)}] AUC{auc: .4f}, F1{f1: .4f}, F1_03{f1_03: .4f}, F1_05{f1_05: .4f}\")\n                print(f\"[epoch {str(e).zfill(2)}] SKF1_03{sk_f1_30: .4f}, SKF1_05{sk_f1_50: .4f}\")\n                print(f\"[epoch {str(e).zfill(2)}] valid_loss {valid_loss: .6f}\")\n                print(f\"[epoch {str(e).zfill(2)}] update loss {best_loss: .6f} --> {valid_loss: .6f} {(end_time - start_time): .1f}[s]\")\n                print(f\"[epoch {str(e).zfill(2)}] update auc score {best_auc: .6f} --> {auc: .6f} {(end_time - start_time): .1f}[s]\")\n                model_name = f'{name}/checkpoint/fold_{fold}_snapshot_epoch_{str(e).zfill(2)}.pth'\n                best_model = model\n                best_loss = valid_loss\n                best_auc = auc\n                best_f1 = f1\n            else:\n                end_time = time.time()\n                print(f\"[epoch {str(e).zfill(2)}] NOT update loss {best_loss: .6f} <-- {valid_loss: .6f} {(end_time - start_time): .1f}[s]\")\n                print(f\"[epoch {str(e).zfill(2)}] NOT update score {best_auc: .6f} <-- {auc: .6f} {(end_time - start_time): .1f}[s]\")\n\n        if cfg.wandb == True:\n            wandb.log({f\"best_loss\": best_loss,\n                       f\"best_f1\": best_f1,\n                       f\"best_auc\":best_auc})\n\n        torch.save(best_model.state_dict(), model_name)\n        \n        del model, best_model\n        gc.collect()\n        torch.cuda.empty_cache()\n        print(\"--\")\n","metadata":{"papermill":{"duration":0.05756,"end_time":"2024-05-27T08:24:08.683435","exception":false,"start_time":"2024-05-27T08:24:08.625875","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-07-04T07:31:08.785995Z","iopub.execute_input":"2024-07-04T07:31:08.786413Z","iopub.status.idle":"2024-07-04T07:31:08.811502Z","shell.execute_reply.started":"2024-07-04T07:31:08.786379Z","shell.execute_reply":"2024-07-04T07:31:08.809916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = dict()\nmodels_names = dict()\n# for fold in range(cfg.nfolds):\nfor fold in cfg.inference_folds:\n    bestmodel_path = sorted(glob.glob(f\"/kaggle/input/{name}/checkpoint/fold_{fold}*.pth\"))[-1]\n\n    print(bestmodel_path)\n    model = BirdModel(model_name=cfg.model_name, pretrained=False, in_channels=1, num_classes=len(LABELS))\n    model.load_state_dict(torch.load(bestmodel_path, map_location=torch.device('cpu')))\n    model = model.eval()\n    models[fold] = model\n\n    models_names[fold] = bestmodel_path.split(\".\")[0]+\".onnx\"\n    print(models_names[fold])","metadata":{"papermill":{"duration":0.614286,"end_time":"2024-05-27T08:24:09.375315","exception":false,"start_time":"2024-05-27T08:24:08.761029","status":"completed"},"scrolled":true,"tags":[],"execution":{"iopub.status.busy":"2024-07-04T07:31:08.813295Z","iopub.execute_input":"2024-07-04T07:31:08.813835Z","iopub.status.idle":"2024-07-04T07:31:09.617115Z","shell.execute_reply.started":"2024-07-04T07:31:08.813788Z","shell.execute_reply":"2024-07-04T07:31:09.615550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntest_audio_dir = f\"{cfg.dir}test_soundscapes/\"\nfile_list = glob.glob(test_audio_dir+\"*.ogg\")\nfile_list = sorted(file_list)\n\n\ntest_dataset = BirdCLEF_Dataset(df=file_list, mode=\"test\")\ntest_dataloader = torch.utils.data.DataLoader(dataset=test_dataset, \n                                              batch_size=1, \n                                              shuffle=False)\n\ninput_tensor = torch.randn((48, 1, cfg.n_mels, cfg.size_x+1))  # input shape\noutput_names=['output']\ninput_names=[\"x\"]\n\n\n# models_names = []\nmodels_names = dict()\n# for fold in range(cfg.nfolds):\nfor fold in cfg.inference_folds:\n    onnxmodel_path = sorted(glob.glob(f\"/kaggle/input/{name}/checkpoint/fold_{fold}*.onnx\"))[-1]\n\n    print(onnxmodel_path)\n\n    models_names[fold] = onnxmodel_path\n    \n    \nonnx_sessions = dict()\n# for fold in range(cfg.nfolds):\nfor fold in cfg.inference_folds:\n\n    onnx_model = onnx.load(models_names[fold])\n    onnx_model_graph = onnx_model.graph\n    onnx_session = ort.InferenceSession(onnx_model.SerializeToString())\n\n    onnx_sessions[fold] = onnx_session    ","metadata":{"papermill":{"duration":0.449231,"end_time":"2024-05-27T08:24:09.902461","exception":false,"start_time":"2024-05-27T08:24:09.45323","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-07-04T07:31:09.619086Z","iopub.execute_input":"2024-07-04T07:31:09.619512Z","iopub.status.idle":"2024-07-04T07:31:10.131214Z","shell.execute_reply.started":"2024-07-04T07:31:09.619470Z","shell.execute_reply":"2024-07-04T07:31:10.129620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"start_time = time.time()\n\npredictions = []\nfor data in tqdm(test_dataloader):\n    \n    preds = []\n    \n#     for fold, session in enumerate(onnx_sessions):\n    for fold in cfg.inference_folds:\n        session = onnx_sessions[fold]\n        pred = session.run(output_names, {input_names[0]: data[0].numpy()})[0]\n        \n        pred = torch.sigmoid(torch.tensor(pred))\n        preds.append(pred)\n    preds_per_batch = torch.stack(preds, axis=0).mean(axis=0)\n    \n    predictions.extend(preds_per_batch)\n    \nif len(predictions)>0:\n    predictions = torch.stack(predictions)\nelse:\n    predictions = predictions\nend_time = time.time()\nuse_time = end_time - start_time","metadata":{"papermill":{"duration":0.069935,"end_time":"2024-05-27T08:24:10.05298","exception":false,"start_time":"2024-05-27T08:24:09.983045","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-07-04T07:31:10.132903Z","iopub.execute_input":"2024-07-04T07:31:10.133394Z","iopub.status.idle":"2024-07-04T07:31:10.168134Z","shell.execute_reply.started":"2024-07-04T07:31:10.133346Z","shell.execute_reply":"2024-07-04T07:31:10.166707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bird_cols = sample_submission.columns[1:]\ndf = pd.DataFrame(columns=['row_id']+list(bird_cols))\n\n\nrow_list = []\nfor file in file_list:\n    dataname = file.split(\"/\")[-1][:-4]\n    for i in range(int(4*60/5)):\n        row = f\"{dataname}_{(i+1)*5}\"\n        row_list.append(row)\n        \n        \n        \ndf['row_id'] = row_list        \n\nif len(predictions) < 1:\n    pass\nelse:\n    df[bird_cols] = predictions\n    \n    \ndf.to_csv(\"submission1.csv\", index=False)     ","metadata":{"papermill":{"duration":0.055183,"end_time":"2024-05-27T08:24:10.186739","exception":false,"start_time":"2024-05-27T08:24:10.131556","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-07-04T07:31:10.170232Z","iopub.execute_input":"2024-07-04T07:31:10.170652Z","iopub.status.idle":"2024-07-04T07:31:10.203843Z","shell.execute_reply.started":"2024-07-04T07:31:10.170618Z","shell.execute_reply":"2024-07-04T07:31:10.202356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2024-07-04T07:31:10.205283Z","iopub.execute_input":"2024-07-04T07:31:10.205786Z","iopub.status.idle":"2024-07-04T07:31:10.234987Z","shell.execute_reply.started":"2024-07-04T07:31:10.205752Z","shell.execute_reply":"2024-07-04T07:31:10.233533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## another","metadata":{}},{"cell_type":"code","source":"#import\n\nimport random\nimport cv2\nimport json\nimport copy\nimport gc\nimport os\nimport pickle\nimport time\n\n\nfrom glob import glob\nfrom tqdm import tqdm\n\nimport pandas as pd\nimport numpy as np\nimport soundfile as sf\n\nimport torch\nimport torchaudio\nimport torch.nn as nn\nfrom torch.utils.data import DataLoader","metadata":{"execution":{"iopub.status.busy":"2024-07-04T07:31:10.236475Z","iopub.execute_input":"2024-07-04T07:31:10.236863Z","iopub.status.idle":"2024-07-04T07:31:10.555273Z","shell.execute_reply.started":"2024-07-04T07:31:10.236830Z","shell.execute_reply":"2024-07-04T07:31:10.554000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install /kaggle/input/openvino-package/openvino-2024.0.0-14509-cp310-cp310-manylinux2014_x86_64.whl --no-index --find-links /kaggle/input/openvino-package","metadata":{"execution":{"iopub.status.busy":"2024-07-04T07:31:10.557139Z","iopub.execute_input":"2024-07-04T07:31:10.557679Z","iopub.status.idle":"2024-07-04T07:31:28.276490Z","shell.execute_reply.started":"2024-07-04T07:31:10.557630Z","shell.execute_reply":"2024-07-04T07:31:28.275004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import openvino as ov\nimport openvino.properties.hint as hints\n\nclass VINOEngine:\n    def __init__(self, onnx_f):\n        core = ov.Core()\n        \n        model_onnx = core.read_model(onnx_f)        \n        \n        self.compiled_model = core.compile_model(model=model_onnx,device_name='AUTO')\n        \n        \n        self.output_layer = self.compiled_model.output(0)\n    def __call__(self, data):\n        \n        result_infer = self.compiled_model(data)[self.output_layer]\n        return result_infer\n\n","metadata":{"execution":{"iopub.status.busy":"2024-07-04T07:31:28.278753Z","iopub.execute_input":"2024-07-04T07:31:28.279172Z","iopub.status.idle":"2024-07-04T07:31:28.442300Z","shell.execute_reply.started":"2024-07-04T07:31:28.279132Z","shell.execute_reply":"2024-07-04T07:31:28.440874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CFG={\n    'batch_size':1,\n    'num_worker':4,\n    'slice_len':5,\n    'used_frame':48,\n    'sample_rate':32000,\n    'data':'/kaggle/input/birdclef-2024/test_soundscapes',\n    'weights_1d':'/kaggle/input/bird-1d',\n    'weights_spec':'/kaggle/input/bird-spec',\n    'weights_mix':'/kaggle/input/another-spec',\n    'nm2cls':{'asbfly': 0, 'ashdro1': 1, 'ashpri1': 2, 'ashwoo2': 3, 'asikoe2': 4, 'asiope1': 5,\n                     'aspfly1': 6, 'aspswi1': 7, 'barfly1': 8, 'barswa': 9, 'bcnher': 10, 'bkcbul1': 11,\n                     'bkrfla1': 12, 'bkskit1': 13, 'bkwsti': 14, 'bladro1': 15, 'blaeag1': 16, 'blakit1': 17,\n                     'blhori1': 18, 'blnmon1': 19, 'blrwar1': 20, 'bncwoo3': 21, 'brakit1': 22, 'brasta1': 23,\n                     'brcful1': 24, 'brfowl1': 25, 'brnhao1': 26, 'brnshr': 27, 'brodro1': 28, 'brwjac1': 29,\n                     'brwowl1': 30, 'btbeat1': 31, 'bwfshr1': 32, 'categr': 33, 'chbeat1': 34, 'cohcuc1': 35,\n                     'comfla1': 36, 'comgre': 37, 'comior1': 38, 'comkin1': 39, 'commoo3': 40, 'commyn': 41,\n                     'compea': 42, 'comros': 43, 'comsan': 44, 'comtai1': 45, 'copbar1': 46, 'crbsun2': 47,\n                     'cregos1': 48, 'crfbar1': 49, 'crseag1': 50, 'dafbab1': 51, 'darter2': 52, 'eaywag1': 53,\n                     'emedov2': 54, 'eucdov': 55, 'eurbla2': 56, 'eurcoo': 57, 'forwag1': 58, 'gargan': 59,\n                     'gloibi': 60, 'goflea1': 61, 'graher1': 62, 'grbeat1': 63, 'grecou1': 64, 'greegr': 65,\n                     'grefla1': 66, 'grehor1': 67, 'grejun2': 68, 'grenig1': 69, 'grewar3': 70, 'grnsan': 71,\n                     'grnwar1': 72, 'grtdro1': 73, 'gryfra': 74, 'grynig2': 75, 'grywag': 76, 'gybpri1': 77,\n                     'gyhcaf1': 78, 'heswoo1': 79, 'hoopoe': 80, 'houcro1': 81, 'houspa': 82, 'inbrob1': 83,\n                     'indpit1': 84, 'indrob1': 85, 'indrol2': 86, 'indtit1': 87, 'ingori1': 88, 'inpher1': 89,\n                     'insbab1': 90, 'insowl1': 91, 'integr': 92, 'isbduc1': 93, 'jerbus2': 94, 'junbab2': 95,\n                     'junmyn1': 96, 'junowl1': 97, 'kenplo1': 98, 'kerlau2': 99, 'labcro1': 100, 'laudov1': 101,\n                     'lblwar1': 102, 'lesyel1': 103, 'lewduc1': 104, 'lirplo': 105, 'litegr': 106, 'litgre1': 107,\n                     'litspi1': 108, 'litswi1': 109, 'lobsun2': 110, 'maghor2': 111, 'malpar1': 112, 'maltro1': 113,\n                     'malwoo1': 114, 'marsan': 115, 'mawthr1': 116, 'moipig1': 117, 'nilfly2': 118, 'niwpig1': 119,\n                     'nutman': 120, 'orihob2': 121, 'oripip1': 122, 'pabflo1': 123, 'paisto1': 124, 'piebus1': 125,\n                     'piekin1': 126, 'placuc3': 127, 'plaflo1': 128, 'plapri1': 129, 'plhpar1': 130, 'pomgrp2': 131,\n                     'purher1': 132, 'pursun3': 133, 'pursun4': 134, 'purswa3': 135, 'putbab1': 136, 'redspu1': 137,\n                     'rerswa1': 138, 'revbul': 139, 'rewbul': 140, 'rewlap1': 141, 'rocpig': 142, 'rorpar': 143,\n                     'rossta2': 144, 'rufbab3': 145, 'ruftre2': 146, 'rufwoo2': 147, 'rutfly6': 148, 'sbeowl1': 149,\n                     'scamin3': 150, 'shikra1': 151, 'smamin1': 152, 'sohmyn1': 153, 'spepic1': 154, 'spodov': 155,\n                     'spoowl1': 156, 'sqtbul1': 157, 'stbkin1': 158, 'sttwoo1': 159, 'thbwar1': 160, 'tibfly3': 161,\n                     'tilwar1': 162, 'vefnut1': 163, 'vehpar1': 164, 'wbbfly1': 165, 'wemhar1': 166, 'whbbul2': 167,\n                     'whbsho3': 168, 'whbtre1': 169, 'whbwag1': 170, 'whbwat1': 171, 'whbwoo2': 172, 'whcbar1': 173,\n                     'whiter2': 174, 'whrmun': 175, 'whtkin2': 176, 'woosan': 177, 'wynlau1': 178, 'yebbab1': 179,\n                     'yebbul3': 180, 'zitcis1': 181}\n\n}\n\n\nif len(os.listdir('/kaggle/input/birdclef-2024/test_soundscapes'))==1:\n    CFG['data'] = '/kaggle/input/birdclef-2024/unlabeled_soundscapes/'\n","metadata":{"execution":{"iopub.status.busy":"2024-07-04T07:31:28.444124Z","iopub.execute_input":"2024-07-04T07:31:28.444649Z","iopub.status.idle":"2024-07-04T07:31:28.471581Z","shell.execute_reply.started":"2024-07-04T07:31:28.444553Z","shell.execute_reply":"2024-07-04T07:31:28.469680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weights_1d=[]\n\n\nCFG['weights_1d']=[os.path.join(CFG['weights_1d'],x) for x in sorted(os.listdir(CFG['weights_1d']))]\nCFG['weights_spec']=[os.path.join(CFG['weights_spec'],x) for x in sorted(os.listdir(CFG['weights_spec']))]\nCFG['weights_mix']=[os.path.join(CFG['weights_mix'],x) for x in sorted(os.listdir(CFG['weights_mix']))]\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-07-04T07:31:28.472964Z","iopub.execute_input":"2024-07-04T07:31:28.473348Z","iopub.status.idle":"2024-07-04T07:31:28.501949Z","shell.execute_reply.started":"2024-07-04T07:31:28.473303Z","shell.execute_reply":"2024-07-04T07:31:28.500594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SPECTRUM=False\n\n\n\n\nclass Transform(nn.Module):\n    def __init__(self, ):\n        super().__init__()\n\n        \n        self.wave_transform = nn.Sequential(\n            torchaudio.transforms.MelSpectrogram(\n                32000,\n                n_mels=512,\n                f_min=0,\n                f_max=16000,\n                n_fft=2048*2,\n                hop_length=512,\n                normalized=True,\n            ),\n            torchaudio.transforms.AmplitudeToDB(top_db=80.0),\n\n        )\n        \n        self.resize = nn.UpsamplingBilinear2d(size=(256, 256))\n\n    def forward(self, x):\n        bs = x.size(0)\n        image = self.wave_transform(x)\n        \n        resized_image = torch.unsqueeze(image, dim=1)\n        resized_image = self.resize(resized_image)\n        resized_image = torch.squeeze(resized_image, dim=1)\n        \n        return image,resized_image\n\n","metadata":{"execution":{"iopub.status.busy":"2024-07-04T07:31:28.508961Z","iopub.execute_input":"2024-07-04T07:31:28.509384Z","iopub.status.idle":"2024-07-04T07:31:28.520186Z","shell.execute_reply.started":"2024-07-04T07:31:28.509348Z","shell.execute_reply":"2024-07-04T07:31:28.518795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dataiter\nclass AlaskaDataIter():\n    def __init__(self, \n                 data_dir,\n                 training_flag=False,\n                 shuffle=False,\n                 use_spec=False):\n        \n        file_list=os.listdir(data_dir)\n        \n        self.file_list=[os.path.join(CFG['data'],x) for x in file_list if 'ogg' in x]\n        self.file_list.sort()\n        if CFG['data']=='/kaggle/input/birdclef-2024/unlabeled_soundscapes/':\n            self.file_list=self.file_list[:2]\n        self.training_flag=training_flag\n        self.use_spec=use_spec\n        if self.use_spec:\n            self.wave2spec=Transform().to('cpu')\n    def __getitem__(self, item):\n        \n        return self.single_map_func(self.file_list[item], self.training_flag)\n\n    def __len__(self):\n\n        return len(self.file_list)\n    \n    def safe_pad(self,waves,valid_lenth=32000*5):\n        L=waves.shape[0]\n\n\n        if L<valid_lenth:\n            padded_array = np.zeros(valid_lenth)\n            padded_array[:L] = waves\n\n            return padded_array\n        else:\n            return waves\n    def single_map_func(self, fn, is_training):\n        \"\"\"Data augmentation function.\"\"\"\n        \n        ####customed here\n        base_name = os.path.basename(fn)\n        \n        row_id=base_name.rsplit('.',1)[0]\n        \n        \n        waves, samplerate = sf.read(fn)\n            \n        \n        waves=np.reshape(waves,newshape=[-1,CFG['sample_rate']*CFG['slice_len']])\n\n        data=waves.astype(np.float32)\n        \n        # clip\n        for i in range(data.shape[0]):\n            max_v=np.max(np.abs(data[i]))\n            if max_v>1:\n                data[i]=data[i]/max_v\n        \n        raw=data\n        if self.use_spec:\n            data_tensor = torch.from_numpy(data).to('cpu')\n            \n            spec,spec_256 = self.wave2spec(data_tensor)\n            spec = spec.cpu().numpy().astype(np.float32)\n            spec_256= spec_256.cpu().numpy().astype(np.float32)\n            \n            \n        else:\n            spec=None\n            spec_256=None\n        row_ids=[]\n\n        for i in range(int(waves.shape[0]*(CFG['slice_len']/5))):\n            row_ids.append(row_id+'_%d'%(5*(i+1)))\n        \n        data={'raw':raw,\n              'spec':spec,\n              'mix':spec_256,\n              'row_ids':row_ids}\n        return data\n        ","metadata":{"execution":{"iopub.status.busy":"2024-07-04T07:31:28.521763Z","iopub.execute_input":"2024-07-04T07:31:28.522133Z","iopub.status.idle":"2024-07-04T07:31:28.542284Z","shell.execute_reply.started":"2024-07-04T07:31:28.522101Z","shell.execute_reply":"2024-07-04T07:31:28.540984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def run_single_model(models,data):\n    output_collection=[]\n    for vino_modle in models:\n        output = vino_modle(data)\n        # new a array to avoid err,\n        output = np.array(output)\n        output_collection.append(output)\n    ans=np.mean(output_collection,axis=0)\n    \n    return ans","metadata":{"execution":{"iopub.status.busy":"2024-07-04T07:31:28.544113Z","iopub.execute_input":"2024-07-04T07:31:28.544595Z","iopub.status.idle":"2024-07-04T07:31:28.562826Z","shell.execute_reply.started":"2024-07-04T07:31:28.544561Z","shell.execute_reply":"2024-07-04T07:31:28.561523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def inference(test_loader, models):\n    \n    prediction_dict = {}\n    preds = []\n    row_ids=[]\n    with tqdm(test_loader, unit=\"test_batch\", desc='Inference') as tqdm_test_loader:\n        for step, X in enumerate(tqdm_test_loader):\n            \n            X=X[0]\n            used_frame=CFG['used_frame']\n            \n            x_raw=X['raw']\n            x_spec=X['spec']\n            x_spec_256=X['mix']\n            \n            row_id=X['row_ids']\n            tmppre=[]\n            for j in range(len(x_raw)):\n                input_raw=x_raw[j][None,]\n                input_spec=x_spec[j][None,]\n                input_spec_256=x_spec_256[j][None,]\n\n                output1=run_single_model(models['raw'],[input_raw])\n                output2=run_single_model(models['spec'],[input_spec])\n                output3=run_single_model(models['mix'],[input_raw,input_spec_256])\n                \n                ans=output1*0.5+output2*0.4+output3*0.1\n                \n                tmppre.append(ans)\n            y_preds=np.concatenate(tmppre,axis=0)\n            \n            \n            preds.append(y_preds) \n            row_ids.append(row_id) \n                \n    prediction_dict[\"predictions\"] = np.concatenate(preds) \n    prediction_dict[\"row_ids\"] = np.concatenate(row_ids) \n    return prediction_dict","metadata":{"execution":{"iopub.status.busy":"2024-07-04T07:31:28.564537Z","iopub.execute_input":"2024-07-04T07:31:28.565187Z","iopub.status.idle":"2024-07-04T07:31:28.580091Z","shell.execute_reply.started":"2024-07-04T07:31:28.565139Z","shell.execute_reply":"2024-07-04T07:31:28.578964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def run():\n    \n    t0=time.time()\n    # dataloader\n    test_dataset = AlaskaDataIter(CFG['data'], training_flag=False, shuffle=False,use_spec=True)\n    test_loader = DataLoader(test_dataset,\n                     CFG['batch_size'],\n                     num_workers=CFG['num_worker'],\n                     shuffle=False,\n                     collate_fn=lambda x :x)\n    vino_models={'raw':[],\n               'spec':[],\n                'mix':[]}\n    \n    \n    for model_weight in CFG['weights_1d'][:3]:\n        model = VINOEngine(model_weight)\n        vino_models['raw'].append(model)\n        \n    for model_weight in CFG['weights_spec'][:1]:\n        model = VINOEngine(model_weight)\n        vino_models['spec'].append(model)\n        \n    for model_weight in CFG['weights_mix'][:1]:\n        model = VINOEngine(model_weight)\n        vino_models['mix'].append(model)\n    \n    prediction_dict = inference(test_loader, vino_models)\n    \n    \n\n    print('probably %.2f seconds'%(1100/2*(time.time()-t0)))\n    \n    return prediction_dict","metadata":{"execution":{"iopub.status.busy":"2024-07-04T07:32:44.293014Z","iopub.execute_input":"2024-07-04T07:32:44.293512Z","iopub.status.idle":"2024-07-04T07:32:44.305290Z","shell.execute_reply.started":"2024-07-04T07:32:44.293474Z","shell.execute_reply":"2024-07-04T07:32:44.303942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_dict=run()","metadata":{"execution":{"iopub.status.busy":"2024-07-04T07:32:47.270925Z","iopub.execute_input":"2024-07-04T07:32:47.271516Z","iopub.status.idle":"2024-07-04T07:33:06.573691Z","shell.execute_reply.started":"2024-07-04T07:32:47.271466Z","shell.execute_reply":"2024-07-04T07:33:06.572307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(prediction_dict['row_ids'].shape)\nprint(prediction_dict['predictions'].shape)","metadata":{"execution":{"iopub.status.busy":"2024-07-04T07:43:29.918939Z","iopub.execute_input":"2024-07-04T07:43:29.919385Z","iopub.status.idle":"2024-07-04T07:43:29.925927Z","shell.execute_reply.started":"2024-07-04T07:43:29.919345Z","shell.execute_reply":"2024-07-04T07:43:29.924664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"row_ids= list(prediction_dict['row_ids'])\nrow_ids=[str(x) for x in row_ids]\n\nsub_pred = pd.DataFrame(prediction_dict['predictions'], columns=CFG[\"nm2cls\"].keys())\nsub_id = pd.DataFrame({'row_id': row_ids})\n\nsub = pd.concat([sub_id, sub_pred], axis=1)\n\nsub.to_csv('submission2.csv',index=False)\nprint(f'Submissionn shape: {sub.shape}')\nsub.head(20)","metadata":{"execution":{"iopub.status.busy":"2024-07-04T07:43:29.928238Z","iopub.execute_input":"2024-07-04T07:43:29.928724Z","iopub.status.idle":"2024-07-04T07:43:30.015577Z","shell.execute_reply.started":"2024-07-04T07:43:29.928679Z","shell.execute_reply":"2024-07-04T07:43:30.014384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 合并数据","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\n# 读取两个CSV文件\ndf1 = pd.read_csv('/kaggle/working/submission1.csv')\ndf2 = pd.read_csv('/kaggle/working/submission2.csv')\n\n# 计算新的数据框（排除表头）\nnew_df = df1.iloc[:, 1:] * 0.8 + df2.iloc[:, 1:] * 0.2\n\n# 将表头重新加入新的数据框\nnew_df.insert(0, df1.columns[0], df1[df1.columns[0]])\n\n# 保存新的数据框到CSV文件\nnew_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-07-04T07:43:30.017080Z","iopub.execute_input":"2024-07-04T07:43:30.017562Z","iopub.status.idle":"2024-07-04T07:43:30.083657Z","shell.execute_reply.started":"2024-07-04T07:43:30.017526Z","shell.execute_reply":"2024-07-04T07:43:30.082508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sub_time = (tock-tick)*550 # ~1100 recording on the test data\n# sub_time = time.gmtime(sub_time)\n# sub_time = time.strftime(\"%H hr: %M min : %S sec\", sub_time)\n# print(f\">> Time for submission: ~ {sub_time}\")","metadata":{"execution":{"iopub.status.busy":"2024-07-04T07:43:30.085163Z","iopub.execute_input":"2024-07-04T07:43:30.085647Z","iopub.status.idle":"2024-07-04T07:43:30.090876Z","shell.execute_reply.started":"2024-07-04T07:43:30.085603Z","shell.execute_reply":"2024-07-04T07:43:30.089487Z"},"trusted":true},"execution_count":null,"outputs":[]}]}