{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install efficientnet_pytorch","metadata":{"execution":{"iopub.status.busy":"2022-04-25T13:41:44.016291Z","iopub.execute_input":"2022-04-25T13:41:44.016738Z","iopub.status.idle":"2022-04-25T13:41:51.800637Z","shell.execute_reply.started":"2022-04-25T13:41:44.016701Z","shell.execute_reply":"2022-04-25T13:41:51.799742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"从下面开始↓=============================================================================================\n","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n \nimport warnings\nwarnings.filterwarnings(action='ignore')\nimport librosa\n\nfrom sklearn.utils import shuffle\nfrom PIL import Image\nfrom tqdm import tqdm\n\n# Global vars 全局变量\nRANDOM_SEED = 1337\nSAMPLE_RATE = 32000\nSIGNAL_LENGTH = 5  # seconds\nSPEC_SHAPE = (224, 224)  # height x width\nFMIN = 20\nFMAX = 16000\n\n# Load metadata file\ntrain = pd.read_csv('../input/birdclef-2022/train_metadata.csv', )\n# Second, assume that birds with the most training samples are also the most common\n# A species needs at least 200 recordings with a rating above 4 to be considered common\nbirds_count = {}\n\nfor bird_species, count in zip(train.primary_label.unique(),\n                               train.groupby('primary_label')['primary_label'].count().values):\n    birds_count[bird_species] = count\nmost_represented_birds = [key for key, value in birds_count.items()]\n \nTRAIN = train.query('primary_label in @most_represented_birds')\nLABELS = sorted(TRAIN.primary_label.unique())\n\n# Let's see how many species and samples we have left\nprint('NUMBER OF SPECIES IN TRAIN DATA:', len(LABELS))\nprint('NUMBER OF SAMPLES IN TRAIN DATA:', len(TRAIN))\nprint('LABELS:', most_represented_birds)\n# Shuffle the training data and limit the number of audio files to MAX_AUDIO_FILES\nTRAIN = shuffle(TRAIN, random_state=RANDOM_SEED)\n\n\n# Define a function that splits an audio file, 定义一个分割音频文件的函数\n# extracts spectrograms and saves them in a working directory 提取声谱图并保存在工作目录中\ndef get_spectrograms(filepath, primary_label, output_dir):\n\n    # Open the file with librosa (limited to the first 15 seconds) 使用librosa打开文件(限制为前15秒)\n    sig, rate = librosa.load(filepath, sr=SAMPLE_RATE, offset=None, duration=5)\n    \n    # Split signal into five second chunks  将信号分割成5秒的块\n    sig_splits = []\n    for i in range(0, len(sig), int(SIGNAL_LENGTH * SAMPLE_RATE)):\n        split = sig[i:i + int(SIGNAL_LENGTH * SAMPLE_RATE)]\n \n        # End of signal?\n        if len(split) < int(SIGNAL_LENGTH * SAMPLE_RATE):\n            break\n \n        sig_splits.append(split)\n \n    # Extract mel spectrograms for each audio chunk 为每个音频块提取mel谱图\n    s_cnt = 0\n    saved_samples = []\n    for chunk in sig_splits:\n \n        hop_length = int(SIGNAL_LENGTH * SAMPLE_RATE / (SPEC_SHAPE[1] - 1))\n        mel_spec = librosa.feature.melspectrogram(y=chunk,\n                                                  sr=SAMPLE_RATE,\n                                                  n_fft=2048,\n                                                  hop_length=hop_length,\n                                                  n_mels=SPEC_SHAPE[0],\n                                                  fmin=FMIN,\n                                                  fmax=FMAX)\n \n        mel_spec = librosa.power_to_db(mel_spec, ref=np.max)\n \n        # Normalize\n        mel_spec -= mel_spec.min()\n        mel_spec /= mel_spec.max()\n \n        # Save as image file\n        save_dir = os.path.join(output_dir, primary_label)\n        if not os.path.exists(save_dir):\n            os.makedirs(save_dir)\n        save_path = os.path.join(save_dir, filepath.rsplit(os.sep, 1)[-1].rsplit('.', 1)[0] +\n                                 '_' + str(s_cnt) + '.png')\n        im = Image.fromarray(mel_spec * 255.0).convert(\"L\")\n        im.save(save_path)\n \n        saved_samples.append(save_path)\n        s_cnt += 1\n \n    return saved_samples\nprint('FINAL NUMBER OF AUDIO FILES IN TRAINING DATA:', len(TRAIN))\n# Parse audio files and extract training samples\ninput_dir = '../input/birdclef-2022/train_audio/'\noutput_dir = '../working/melspectrogram_dataset/'\n\n\n ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-04-25T12:14:27.912289Z","iopub.execute_input":"2022-04-25T12:14:27.912545Z","iopub.status.idle":"2022-04-25T12:14:28.012661Z","shell.execute_reply.started":"2022-04-25T12:14:27.912519Z","shell.execute_reply":"2022-04-25T12:14:28.011931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n \nimport warnings\nwarnings.filterwarnings(action='ignore')\nimport librosa\n\nfrom sklearn.utils import shuffle\nfrom PIL import Image\nfrom tqdm import tqdm\n\n# Global vars 全局变量\nRANDOM_SEED = 1337\nSAMPLE_RATE = 32000\nSIGNAL_LENGTH = 5  # seconds\nSPEC_SHAPE = (224, 224)  # height x width\nFMIN = 20\nFMAX = 16000\n\n# Load metadata file\ntrain = pd.read_csv('../input/birdclef-2022/train_metadata.csv', )\n# Second, assume that birds with the most training samples are also the most common\n# A species needs at least 200 recordings with a rating above 4 to be considered common\nbirds_count = {}\n\nfor bird_species, count in zip(train.primary_label.unique(),\n                               train.groupby('primary_label')['primary_label'].count().values):\n    birds_count[bird_species] = count\nmost_represented_birds = [key for key, value in birds_count.items()]\n \nTRAIN = train.query('primary_label in @most_represented_birds')\nLABELS = sorted(TRAIN.primary_label.unique())\n\n# Let's see how many species and samples we have left\nprint('NUMBER OF SPECIES IN TRAIN DATA:', len(LABELS))\nprint('NUMBER OF SAMPLES IN TRAIN DATA:', len(TRAIN))\nprint('LABELS:', most_represented_birds)\n# Shuffle the training data and limit the number of audio files to MAX_AUDIO_FILES\nTRAIN = shuffle(TRAIN, random_state=RANDOM_SEED)\n\n\n# Define a function that splits an audio file, 定义一个分割音频文件的函数\n# extracts spectrograms and saves them in a working directory 提取声谱图并保存在工作目录中\ndef get_spectrograms(filepath, primary_label, output_dir):\n\n    # Open the file with librosa (limited to the first 15 seconds) 使用librosa打开文件(限制为前15秒)\n    sig, rate = librosa.load(filepath, sr=SAMPLE_RATE, offset=None, duration=5)\n    \n    # Split signal into five second chunks  将信号分割成5秒的块\n    sig_splits = []\n    for i in range(0, len(sig), int(SIGNAL_LENGTH * SAMPLE_RATE)):\n        split = sig[i:i + int(SIGNAL_LENGTH * SAMPLE_RATE)]\n \n        # End of signal?\n        if len(split) < int(SIGNAL_LENGTH * SAMPLE_RATE):\n            break\n \n        sig_splits.append(split)\n \n    # Extract mel spectrograms for each audio chunk 为每个音频块提取mel谱图\n    s_cnt = 0\n    saved_samples = []\n    for chunk in sig_splits:\n \n        hop_length = int(SIGNAL_LENGTH * SAMPLE_RATE / (SPEC_SHAPE[1] - 1))\n        mel_spec = librosa.feature.melspectrogram(y=chunk,\n                                                  sr=SAMPLE_RATE,\n                                                  n_fft=2048,\n                                                  hop_length=hop_length,\n                                                  n_mels=SPEC_SHAPE[0],\n                                                  fmin=FMIN,\n                                                  fmax=FMAX)\n \n        mel_spec = librosa.power_to_db(mel_spec, ref=np.max)\n \n        # Normalize\n        mel_spec -= mel_spec.min()\n        mel_spec /= mel_spec.max()\n \n        # Save as image file\n        save_dir = os.path.join(output_dir, primary_label)\n        if not os.path.exists(save_dir):\n            os.makedirs(save_dir)\n        save_path = os.path.join(save_dir, filepath.rsplit(os.sep, 1)[-1].rsplit('.', 1)[0] +\n                                 '_' + str(s_cnt) + '.png')\n        im = Image.fromarray(mel_spec * 255.0).convert(\"L\")\n        im.save(save_path)\n \n        saved_samples.append(save_path)\n        s_cnt += 1\n \n    return saved_samples\nprint('FINAL NUMBER OF AUDIO FILES IN TRAINING DATA:', len(TRAIN))\n# Parse audio files and extract training samples\ninput_dir = '../input/birdclef-2022/train_audio/'\noutput_dir = '../working/melspectrogram_dataset/'\n\n\n ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-04-25T12:14:32.097018Z","iopub.execute_input":"2022-04-25T12:14:32.097267Z","iopub.status.idle":"2022-04-25T12:14:32.191896Z","shell.execute_reply.started":"2022-04-25T12:14:32.09724Z","shell.execute_reply":"2022-04-25T12:14:32.191181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"samples = []\nwith tqdm(total=len(TRAIN)) as pbar:\n    for idx, row in TRAIN.iterrows():\n        pbar.update(1)\n \n        if row.primary_label in most_represented_birds:\n            audio_file_path = os.path.join(input_dir, row.filename)\n            samples += get_spectrograms(audio_file_path, row.primary_label, output_dir)\n            \nprint(samples)\n","metadata":{"execution":{"iopub.status.busy":"2022-04-25T12:14:41.716671Z","iopub.execute_input":"2022-04-25T12:14:41.71696Z","iopub.status.idle":"2022-04-25T12:28:36.7557Z","shell.execute_reply.started":"2022-04-25T12:14:41.716929Z","shell.execute_reply":"2022-04-25T12:28:36.754946Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"==========\n这里是数据打包代码↓","metadata":{}},{"cell_type":"markdown","source":"==========\n这里是数据打包代码↑","metadata":{}},{"cell_type":"markdown","source":"========这里是删除文件代码↑","metadata":{}},{"cell_type":"code","source":"\nstr_samples = ','.join(samples)\nTRAIN_SPECS = shuffle(samples, random_state=RANDOM_SEED)\nfilename = open('a.txt', 'w')\nfilename.write(str_samples)\nfilename.close()","metadata":{"execution":{"iopub.status.busy":"2022-04-25T12:35:39.101518Z","iopub.execute_input":"2022-04-25T12:35:39.10213Z","iopub.status.idle":"2022-04-25T12:35:39.112865Z","shell.execute_reply.started":"2022-04-25T12:35:39.10209Z","shell.execute_reply":"2022-04-25T12:35:39.112154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#切分训练集和验证集\nimport os\nimport warnings\n \nwarnings.filterwarnings(action='ignore')\nfrom sklearn.model_selection import train_test_split\nimport shutil\n \nfilename = open('a.txt', 'r')\nstr_samples = filename.read()\nfilename.close()\nstr_samples = str_samples.replace(\"\\\\\", \"/\")\nsamples = str_samples.split(',')\ntrainval_files, test_files = train_test_split(samples, test_size=0.3, random_state=42)\ntrain_dir = '../working/train/'\nval_dir = '../working/val/'\n \n \ndef copyfiles(file, dir):\n    filelist = file.split('/')\n    filename = filelist[-1]\n    lable = filelist[-2]\n    cpfile = dir + \"/\" + lable\n    if not os.path.exists(cpfile):\n        os.makedirs(cpfile)\n    cppath = cpfile + '/' + filename\n    shutil.copy(file, cppath)\n \n \nfor file in trainval_files:\n    copyfiles(file, train_dir)\nfor file in test_files:\n    copyfiles(file, val_dir)","metadata":{"execution":{"iopub.status.busy":"2022-04-25T12:35:42.751709Z","iopub.execute_input":"2022-04-25T12:35:42.751968Z","iopub.status.idle":"2022-04-25T12:35:44.78095Z","shell.execute_reply.started":"2022-04-25T12:35:42.751938Z","shell.execute_reply":"2022-04-25T12:35:44.780189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#训练\nimport torch.optim as optim\nimport torch\nimport torch.nn as nn\nimport torch.nn.parallel\nfrom torch.autograd import Variable\nimport torch.optim\nimport torch.utils.data\nimport torch.utils.data.distributed\nimport torchvision.transforms as transforms\nimport torchvision.datasets as datasets\nfrom efficientnet_pytorch import EfficientNet\nimport os\nimport time\n\n# 设置超参数\nmomentum = 0.9\nBATCH_SIZE = 32\nclass_num = 397\nEPOCHS = 30\nlr = 0.001\nuse_gpu = True\nnet_name = 'efficientnet-b3'\nDEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n# 数据预处理\n\ntransform = transforms.Compose([\n    transforms.Resize(224),\n    transforms.ToTensor(),\n    transforms.Normalize([0.5, 0.5, 0.5], [0.5, 0.5, 0.5])\n])\ndataset_train = datasets.ImageFolder('../working/train', transform)\ndataset_val = datasets.ImageFolder('../working/val', transform)\n# 对应文件夹的label\nprint(dataset_train.class_to_idx)\ndset_sizes = len(dataset_train)\ndset_sizes_val = len(dataset_val)\nprint(\"dset_sizes_val Length:\", dset_sizes_val)\ntrain_loader = torch.utils.data.DataLoader(dataset_train, batch_size=BATCH_SIZE, shuffle=True)\ntest_loader = torch.utils.data.DataLoader(dataset_val, batch_size=BATCH_SIZE, shuffle=True)\n \n \ndef exp_lr_scheduler(optimizer, epoch, init_lr=0.001, lr_decay_epoch=10):\n    \"\"\"Decay learning rate by a f#            model_out_path =\"./model/W_epoch_{}.pth\".format(epoch)\n#            torch.save(model_W, model_out_path) actor of 0.1 every lr_decay_epoch epochs.\"\"\"\n    lr = init_lr * (0.8 ** (epoch // lr_decay_epoch))\n    print('LR is set to {}'.format(lr))\n    for param_group in optimizer.param_groups:\n        param_group['lr'] = lr\n    return optimizer\n \n \ndef train_model(model_ft, criterion, optimizer, lr_scheduler, num_epochs=50):\n    train_loss = []\n    since = time.time()\n    best_model_wts = model_ft.state_dict()\n    best_acc = 0.0\n    model_ft.train(True)\n    \n    for epoch in range(num_epochs):\n        print('Epoch {}/{}'.format(epoch, num_epochs - 1))\n        print('-' * 10)\n        optimizer = lr_scheduler(optimizer, epoch)\n        running_loss = 0.0\n        running_corrects = 0\n        count = 0\n        for data in train_loader:\n            inputs, labels = data\n            labels = torch.squeeze(labels.type(torch.LongTensor))\n            if use_gpu:\n                inputs, labels = Variable(inputs.cuda()), Variable(labels.cuda())\n            else:\n                inputs, labels = Variable(inputs), Variable(labels)\n            outputs = model_ft(inputs)\n            loss = criterion(outputs, labels)\n            _, preds = torch.max(outputs.data, 1)\n            optimizer.zero_grad()\n            loss.backward()\n            optimizer.step()\n            count += 1\n            if count % 30 == 0 or outputs.size()[0] < BATCH_SIZE:\n                print('Epoch:{}: loss:{:.3f}'.format(epoch, loss.item()))\n                train_loss.append(loss.item())\n            running_loss += loss.item() * inputs.size(0)\n            running_corrects += torch.sum(preds == labels.data)\n        epoch_loss = running_loss / dset_sizes\n        epoch_acc = running_corrects.double() / dset_sizes\n        print('Loss: {:.4f} Acc: {:.4f}'.format(\n            epoch_loss, epoch_acc))\n        if epoch_acc > best_acc:\n            best_acc = epoch_acc\n            best_model_wts = model_ft.state_dict()\n \n    # save best model\n    save_dir = 'model'\n    os.makedirs(save_dir, exist_ok=True)\n    model_ft.load_state_dict(best_model_wts)\n    model_out_path = save_dir + \"/\" + net_name + '.pth'\n    torch.save(model_ft, model_out_path)\n    time_elapsed = time.time() - since\n    print('Training complete in {:.0f}m {:.0f}s'.format(\n        time_elapsed // 60, time_elapsed % 60))\n    return train_loss, best_model_wts\n \n \nmodel_ft = EfficientNet.from_pretrained('efficientnet-b3')\nnum_ftrs = model_ft._fc.in_features\nmodel_ft._fc = nn.Linear(num_ftrs, class_num)\ncriterion = nn.CrossEntropyLoss()\nprint(\"状态：\",use_gpu)\nif use_gpu:\n    model_ft = model_ft.cuda()\n    criterion = criterion.cuda()\n\noptimizer = optim.Adam((model_ft.parameters()), lr=lr)\ntrain_loss, best_model_wts = train_model(model_ft, criterion, optimizer, exp_lr_scheduler, num_epochs=EPOCHS)","metadata":{"execution":{"iopub.status.busy":"2022-04-25T14:04:00.769783Z","iopub.execute_input":"2022-04-25T14:04:00.770068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#测试\nimport os\nimport pandas as pd\nimport torch\nimport librosa\nimport numpy as np\n \n# Global vars\nRANDOM_SEED = 1337\nSAMPLE_RATE = 32000\nSIGNAL_LENGTH = 5  # seconds\nSPEC_SHAPE = (224, 224)  # height x width\nFMIN = 20\nFMAX = 16000\n# Load metadata file\ntrain = pd.read_csv('../input/birdclef-2022/train_metadata.csv', )\n# Second, assume that birds with the most training samples are also the most common\n# A species needs at least 200 recordings with a rating above 4 to be considered common\nbirds_count = {}\nfor bird_species, count in zip(train.primary_label.unique(),\n                               train.groupby('primary_label')['primary_label'].count().values):\n    birds_count[bird_species] = count\nmost_represented_birds = [key for key, value in birds_count.items()]\n \nTRAIN = train.query('primary_label in @most_represented_birds')\nLABELS = sorted(TRAIN.primary_label.unique())\n \n# Let's see how many species and samples we have left\nprint('NUMBER OF SPECIES IN TRAIN DATA:', len(LABELS))\nprint('NUMBER OF SAMPLES IN TRAIN DATA:', len(TRAIN))\nprint('LABELS:', most_represented_birds)\n \n \n# First, get a list of soundscape files to process.\n# We'll use the test_soundscape directory if it contains \"ogg\" files\n# (which it only does when submitting the notebook),\n# otherwise we'll use the train_soundscape folder to make predictions.\ndef list_files(path):\n    return [os.path.join(path, f) for f in os.listdir(path) if f.rsplit('.', 1)[-1] in ['ogg']]\n \n \ntest_audio = list_files('../input/birdclef-2022/test_soundscapes')\nif len(test_audio) == 0:\n    test_audio = list_files('../input/birdclef-2022/train_soundscapes')\nprint('{} FILES IN TEST SET.'.format(len(test_audio)))\npath = test_audio[0]\ndata = path.split(os.sep)[-1].rsplit('.', 1)[0].split('_')\nprint('FILEPATH:', path)\n#print('ID: {}, SITE: {}, DATE: {}'.format(data[0], data[1], data[2]))\n# This is where we will store our results\npred = {'row_id': [], 'birds': []}\nmodel = torch.load(\"./model/efficientnet-b3.pth\")\nmodel.eval()\nimport torchvision.transforms as transforms\nfrom PIL import Image\n \ntransform = transforms.Compose([\n    transforms.Resize(224),\n    transforms.ToTensor(),\n    transforms.Lambda(lambda x: x.repeat(3, 1, 1)),\n    transforms.Normalize([0.5, 0.5, 0.5], [0.5, 0.5, 0.5])\n])\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\n# Analyze each soundscape recording\n# Store results so that we can analyze them later\ndata = {'row_id': [], 'birds': []}\nfor path in test_audio:\n    path = path.replace(\"\\\\\", \"/\")\n    # Open file with Librosa\n    # Split file into 5-second chunks\n    # Extract spectrogram for each chunk\n    # Predict on spectrogram\n    # Get row_id and birds and store result\n    # (maybe using a post-filter based on location)\n    # The above steps are just placeholders, we will use mock predictions.\n    # Our \"model\" will predict \"nocall\" for each spectrogram.\n    sig, rate = librosa.load(path, sr=SAMPLE_RATE)\n    # Split signal into 5-second chunks\n    # Just like we did before (well, this could actually be a seperate function)\n    sig_splits = []\n    for i in range(0, len(sig), int(SIGNAL_LENGTH * SAMPLE_RATE)):\n        split = sig[i:i + int(SIGNAL_LENGTH * SAMPLE_RATE)]\n \n        # End of signal?\n        if len(split) < int(SIGNAL_LENGTH * SAMPLE_RATE):\n            break\n \n        sig_splits.append(split)\n    # Get the spectrograms and run inference on each of them\n    # This should be the exact same process as we used to\n    # generate training samples!\n    seconds, scnt = 0, 0\n    for chunk in sig_splits:\n        # Keep track of the end time of each chunk\n        seconds += 5\n        # Get the spectrogram\n        hop_length = int(SIGNAL_LENGTH * SAMPLE_RATE / (SPEC_SHAPE[1] - 1))\n        mel_spec = librosa.feature.melspectrogram(y=chunk,\n                                                  sr=SAMPLE_RATE,\n                                                  n_fft=2048,\n                                                  hop_length=hop_length,\n                                                  n_mels=SPEC_SHAPE[0],\n                                                  fmin=FMIN,\n                                                  fmax=FMAX)\n        mel_spec = librosa.power_to_db(mel_spec, ref=np.max)\n        # Normalize to match the value range we used during training.\n        # That's something you should always double check!\n        mel_spec -= mel_spec.min()\n        mel_spec /= mel_spec.max()\n        im = Image.fromarray(mel_spec * 255.0).convert(\"L\")\n        im = transform(im)\n        print(im.shape)\n        im.unsqueeze_(0)\n        # 没有这句话会报错\n        im = im.to(device)\n        # Predict\n        p = model(im)[0]\n        print(p.shape)\n        # Get highest scoring species\n        idx = p.argmax()\n        print(idx)\n        species = LABELS[idx]\n        print(species)\n        score = p[idx]\n        print(score)\n        # Prepare submission entry\n        spath = path.split('/')[-1].rsplit('_', 1)[0]\n        print(spath)\n        data['row_id'].append(path.split('/')[-1].rsplit('_', 1)[0] +\n                              '_' + str(seconds))\n        # Decide if it's a \"nocall\" or a species by applying a threshold\n        if score > 0.75:\n            data['birds'].append(species)\n            scnt += 1\n        else:\n            data['birds'].append('nocall')\n    print('SOUNSCAPE ANALYSIS DONE. FOUND {} BIRDS.'.format(scnt))\n# Make a new data frame and look at a few \"results\"\nresults = pd.DataFrame(data, columns=['row_id', 'birds'])\nresults.head()\n# Convert our results to csv\nresults.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-04-25T13:46:57.207477Z","iopub.execute_input":"2022-04-25T13:46:57.208036Z","iopub.status.idle":"2022-04-25T13:46:58.411888Z","shell.execute_reply.started":"2022-04-25T13:46:57.207996Z","shell.execute_reply":"2022-04-25T13:46:58.411169Z"},"trusted":true},"execution_count":null,"outputs":[]}]}