{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":6413819,"sourceType":"datasetVersion","datasetId":3699109},{"sourceId":6418204,"sourceType":"datasetVersion","datasetId":3702018},{"sourceId":6419153,"sourceType":"datasetVersion","datasetId":3702721},{"sourceId":6419284,"sourceType":"datasetVersion","datasetId":3702816},{"sourceId":6424853,"sourceType":"datasetVersion","datasetId":3706601},{"sourceId":7393031,"sourceType":"datasetVersion","datasetId":4297953},{"sourceId":7394581,"sourceType":"datasetVersion","datasetId":4299079},{"sourceId":7394627,"sourceType":"datasetVersion","datasetId":4299116},{"sourceId":7395787,"sourceType":"datasetVersion","datasetId":4299971},{"sourceId":7398583,"sourceType":"datasetVersion","datasetId":4301847},{"sourceId":7407093,"sourceType":"datasetVersion","datasetId":4307822}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\nimport tensorflow as tf\nfrom keras.optimizers import Adam\nfrom keras.layers import Add, Dropout, Input, Conv2D, DepthwiseConv2D, BatchNormalization, ReLU, GlobalAveragePooling2D, Dense\nfrom keras.models import Model\nimport numpy as np\nimport librosa\nimport matplotlib.pyplot as plt  #导入绘图工作的函数集合","metadata":{"execution":{"iopub.status.busy":"2024-03-25T13:03:23.495204Z","iopub.execute_input":"2024-03-25T13:03:23.495589Z","iopub.status.idle":"2024-03-25T13:03:23.503109Z","shell.execute_reply.started":"2024-03-25T13:03:23.495559Z","shell.execute_reply":"2024-03-25T13:03:23.501684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch.nn as nn\nimport torch.nn.functional as F\nimport torch\nfrom librosa.filters import mel as librosa_mel_fn\n\n\nclass Audio2Mel(nn.Module):\n    def __init__(\n        self,\n        n_fft=1024,\n        hop_length=256,   # 帧移\n        win_length=1024,  # 窗长\n        sampling_rate=22050,  # 采样率\n        n_mel_channels=80,  # Mel通道数\n        mel_fmin=0.0,\n        mel_fmax=None,\n    ):\n        super().__init__()\n        ##############################################\n        # FFT Parameters                              #\n        ##############################################\n        window = torch.hann_window(win_length).float()  # 加窗\n        mel_basis = librosa_mel_fn(\n            sr=sampling_rate, n_fft=n_fft, n_mels=n_mel_channels, fmin=mel_fmin, fmax=mel_fmax\n        )\n        mel_basis = torch.from_numpy(mel_basis).float()\n        self.register_buffer(\"mel_basis\", mel_basis)\n        self.register_buffer(\"window\", window)\n        self.n_fft = n_fft\n        self.hop_length = hop_length\n        self.win_length = win_length\n        self.sampling_rate = sampling_rate\n        self.n_mel_channels = n_mel_channels\n\n    def forward(self, audio):\n        p = (self.n_fft - self.hop_length) // 2\n        print('p:', p)\n        \n        audio = F.pad(audio, (p, p), \"reflect\").squeeze(1)  # 反射扩充模式，如何效果？？\n        print('audio:', audio)\n        fft = torch.stft(   # 短时傅里叶变换\n            audio,\n            n_fft=self.n_fft,\n            hop_length=self.hop_length,\n            win_length=self.win_length,\n            window=self.window,\n            center=False,\n            return_complex=False,\n        )\n        real_part, imag_part = fft.unbind(-1)  # 解除绑定？\n        magnitude = torch.sqrt(real_part ** 2 + imag_part ** 2)  # 计算平方根\n        mel_output = torch.matmul(self.mel_basis, magnitude)  # 计算矩阵乘法\n        log_mel_spec = torch.log10(torch.clamp(mel_output, min=1e-5))  # 将Mel输出限制大小并取以10为底的对数\n        return log_mel_spec","metadata":{"execution":{"iopub.status.busy":"2024-03-25T13:03:26.421565Z","iopub.execute_input":"2024-03-25T13:03:26.422000Z","iopub.status.idle":"2024-03-25T13:03:26.436993Z","shell.execute_reply.started":"2024-03-25T13:03:26.421967Z","shell.execute_reply":"2024-03-25T13:03:26.435593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# modules.py\nfrom torch.nn.utils import weight_norm\n\n\ndef weights_init(m):\n    classname = m.__class__.__name__\n    if classname.find(\"Conv\") != -1:\n        m.weight.data.normal_(0.0, 0.02)\n    elif classname.find(\"BatchNorm2d\") != -1:\n        m.weight.data.normal_(1.0, 0.02)\n        m.bias.data.fill_(0)\n\n\ndef WNConv1d(*args, **kwargs):\n    return weight_norm(nn.Conv1d(*args, **kwargs))\n\n\ndef WNConvTranspose1d(*args, **kwargs):\n    return weight_norm(nn.ConvTranspose1d(*args, **kwargs))\n\n\nclass Audio2Mel(nn.Module):\n    def __init__(\n        self,\n        n_fft=1024,\n        hop_length=256,   # 帧移\n        win_length=1024,  # 窗长\n        sampling_rate=22050,  # 采样率\n        n_mel_channels=80,  # Mel通道数\n        mel_fmin=0.0,\n        mel_fmax=None,\n    ):\n        super().__init__()\n        ##############################################\n        # FFT Parameters                              #\n        ##############################################\n        window = torch.hann_window(win_length).float()  # 加窗\n        mel_basis = librosa_mel_fn(\n            sr=sampling_rate, n_fft=n_fft, n_mels=n_mel_channels, fmin=mel_fmin, fmax=mel_fmax\n        )\n        mel_basis = torch.from_numpy(mel_basis).float()\n        self.register_buffer(\"mel_basis\", mel_basis)\n        self.register_buffer(\"window\", window)\n        self.n_fft = n_fft\n        self.hop_length = hop_length\n        self.win_length = win_length\n        self.sampling_rate = sampling_rate\n        self.n_mel_channels = n_mel_channels\n\n    def forward(self, audio):\n        p = (self.n_fft - self.hop_length) // 2\n        audio = F.pad(audio, (p, p), \"reflect\").squeeze(1)  # 反射扩充模式，如何效果？？\n        fft = torch.stft(   # 短时傅里叶变换\n            audio,\n            n_fft=self.n_fft,\n            hop_length=self.hop_length,\n            win_length=self.win_length,\n            window=self.window,\n            center=False,\n            return_complex=False,\n        )\n        real_part, imag_part = fft.unbind(-1)  # 解除绑定？\n        magnitude = torch.sqrt(real_part ** 2 + imag_part ** 2)  # 计算平方根\n        mel_output = torch.matmul(self.mel_basis, magnitude)  # 计算矩阵乘法\n        log_mel_spec = torch.log10(torch.clamp(mel_output, min=1e-5))  # 将Mel输出限制大小并取以10为底的对数\n        return log_mel_spec\n\n\nclass ResnetBlock(nn.Module):\n    def __init__(self, dim, dilation=1):\n        super().__init__()\n        self.block = nn.Sequential(\n            nn.LeakyReLU(0.2),\n            nn.ReflectionPad1d(dilation),  # 填充？\n            WNConv1d(dim, dim, kernel_size=3, dilation=dilation),  # 对一维张量进行卷积操作，并应用加权归一化\n            nn.LeakyReLU(0.2),\n            WNConv1d(dim, dim, kernel_size=1),\n        )\n        self.shortcut = WNConv1d(dim, dim, kernel_size=1)\n\n    def forward(self, x):\n        return self.shortcut(x) + self.block(x)\n\n\nclass Generator(nn.Module):\n    def __init__(self, input_size, ngf, n_residual_layers):\n        super().__init__()\n        ratios = [8, 8, 2, 2]  # 比率\n        self.hop_length = np.prod(ratios)  # 计算各个元素的乘积\n        mult = int(2 ** len(ratios))  # 16\n\n        model = [\n            nn.ReflectionPad1d(3),\n            WNConv1d(input_size, mult * ngf, kernel_size=7, padding=0),\n        ]\n\n        # ngf是生成器网络中残差块（ResBlock）的个数，每个残差块包含两个3维卷积层和一个批归一化层\n        # Upsample to raw audio scale\n        for i, r in enumerate(ratios):\n            model += [   # 分别放大 8， 8， 2， 2 倍数\n                nn.LeakyReLU(0.2),\n                WNConvTranspose1d(\n                    mult * ngf,\n                    mult * ngf // 2,\n                    kernel_size=r * 2,\n                    stride=r,\n                    padding=r // 2 + r % 2,\n                    output_padding=r % 2,\n                ),\n            ]\n\n            for j in range(n_residual_layers):\n                model += [ResnetBlock(mult * ngf // 2, dilation=3 ** j)]\n\n            mult //= 2\n\n        model += [\n            nn.LeakyReLU(0.2),\n            nn.ReflectionPad1d(3),\n            WNConv1d(ngf, 1, kernel_size=7, padding=0),\n            nn.Tanh(),\n        ]\n\n        self.model = nn.Sequential(*model)\n        self.apply(weights_init)\n\n    def forward(self, x):\n        return self.model(x)\n\n\nclass NLayerDiscriminator(nn.Module):\n    def __init__(self, ndf, n_layers, downsampling_factor):\n        super().__init__()\n        model = nn.ModuleDict()  # 一个类容器\n\n        model[\"layer_0\"] = nn.Sequential(\n            nn.ReflectionPad1d(7),\n            WNConv1d(1, ndf, kernel_size=15),\n            nn.LeakyReLU(0.2, True),\n        )\n        # \"ndf\"参数是用于指定生成器（Generator）中的噪声向量维度。\n        nf = ndf  # 16\n        stride = downsampling_factor\n        for n in range(1, n_layers + 1):\n            nf_prev = nf\n            nf = min(nf * stride, 1024)\n\n            model[\"layer_%d\" % n] = nn.Sequential(\n                WNConv1d(\n                    nf_prev,\n                    nf,\n                    kernel_size=stride * 10 + 1,\n                    stride=stride,\n                    padding=stride * 5,\n                    groups=nf_prev // 4,\n                ),\n                nn.LeakyReLU(0.2, True),\n            )\n\n        nf = min(nf * 2, 1024)\n        model[\"layer_%d\" % (n_layers + 1)] = nn.Sequential(\n            WNConv1d(nf_prev, nf, kernel_size=5, stride=1, padding=2),\n            nn.LeakyReLU(0.2, True),\n        )\n\n        model[\"layer_%d\" % (n_layers + 2)] = WNConv1d(\n            nf, 1, kernel_size=3, stride=1, padding=1\n        )\n\n        self.model = model\n\n    def forward(self, x):\n        results = []\n        for key, layer in self.model.items():\n            x = layer(x)\n            results.append(x)\n        return results\n\n\nclass Discriminator(nn.Module):\n    def __init__(self, num_D, ndf, n_layers, downsampling_factor):\n        super().__init__()\n        self.model = nn.ModuleDict()\n        for i in range(num_D):\n            self.model[f\"disc_{i}\"] = NLayerDiscriminator(\n                ndf, n_layers, downsampling_factor\n            )\n\n        self.downsample = nn.AvgPool1d(4, stride=2, padding=1, count_include_pad=False)  # 平均池化操作\n        self.apply(weights_init)\n\n    def forward(self, x):\n        results = []\n        for key, disc in self.model.items():\n            results.append(disc(x))\n            x = self.downsample(x)\n        return results\n","metadata":{"execution":{"iopub.status.busy":"2024-03-25T13:03:30.147437Z","iopub.execute_input":"2024-03-25T13:03:30.147850Z","iopub.status.idle":"2024-03-25T13:03:30.191613Z","shell.execute_reply.started":"2024-03-25T13:03:30.147817Z","shell.execute_reply":"2024-03-25T13:03:30.190154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import scipy.io.wavfile\n\n\ndef save_sample(file_path, sampling_rate, audio):\n    \"\"\"Helper function to save sample\n\n    Args:\n        file_path (str or pathlib.Path): save file path\n        sampling_rate (int): sampling rate of audio (usually 22050)\n        audio (torch.FloatTensor): torch array containing audio in [-1, 1]\n    \"\"\"\n    audio = (audio.detach().numpy() * 32768).astype(\"int16\")  # audio.numpy()\n    scipy.io.wavfile.write(file_path, sampling_rate, audio)","metadata":{"execution":{"iopub.status.busy":"2024-03-25T13:03:34.068424Z","iopub.execute_input":"2024-03-25T13:03:34.068840Z","iopub.status.idle":"2024-03-25T13:03:34.075229Z","shell.execute_reply.started":"2024-03-25T13:03:34.068807Z","shell.execute_reply":"2024-03-25T13:03:34.073962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\n\ndef generator_full(model_path, audio_name, typeof, n):\n    netG = Generator(80, 32, 3)\n    netG.load_state_dict(torch.load(model_path, map_location=torch.device('cpu'))) # 导入网络的参数\n    \n    fft = Audio2Mel(n_mel_channels=80)\n    audio_path = '/kaggle/input/leisheng-about/leisheng_wav/leisheng_wav/' + audio_name\n    for i in range(3):\n        y, sr = librosa.load(audio_path, sr=22050, mono=True, offset=0, duration=4.0)  # , offset=2.1, duration=9.0\n        y = y.reshape(1, 1, -1)\n        # print('y', type(y), y.shape)\n        s_t = fft(torch.from_numpy(y)).detach()\n        # print('s_t', s_t, type(s_t), s_t.shape)\n\n        pred_audio = netG(s_t)\n        pred_audio = pred_audio.squeeze()\n        # print('pred_audio', pred_audio, type(pred_audio), pred_audio.shape)\n        if i == 0:\n            pred_audio_part1 = pred_audio\n        if i == 1:\n            pred_audio_part2 = pred_audio\n        if i == 2:\n            pred_audio_part3 = pred_audio\n        \n    pred_audio = torch.cat((pred_audio_part1, pred_audio_part2, pred_audio_part3), dim=0)\n    pred_audio = pred_audio[:220501]\n    # pred_audio = tf.split(pred_audio, num_or_size_splits=[22050, -1])\n    save_sample('/kaggle/working/' + typeof +  '_' + str(n) + '.wav', 22050, pred_audio)  # 将wav音频文件保存\n    return pred_audio","metadata":{"execution":{"iopub.status.busy":"2024-03-25T13:08:07.900864Z","iopub.execute_input":"2024-03-25T13:08:07.901261Z","iopub.status.idle":"2024-03-25T13:08:07.913349Z","shell.execute_reply.started":"2024-03-25T13:08:07.901231Z","shell.execute_reply":"2024-03-25T13:08:07.911929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dataset.py\ndef files_to_list(filename):\n    \"\"\"\n    Takes a text file of filenames and makes a list of filenames\n    \"\"\"\n    with open(filename, encoding=\"utf-8\") as f:\n        files = f.readlines()\n\n    files = [f.rstrip() for f in files]\n    return files\n\ntest_files = '/kaggle/input/leisheng-about/leisheng_filename/leisheng_filename/leisheng_file_train.txt'\nfile = files_to_list(test_files)\n\nfor n in range(18000):\n    fn = file[n % len(file)]\n    if fn.endswith('.wav'):\n        # fd = os.path.join('/kaggle/input/niaoming-about/niaoming_about_hun/niaoming_wav/', fn)\n        model_path = '/kaggle/input/netg-s/netG/leisheng_best_netG.pt'\n        y = generator_full(model_path=model_path, audio_name=fn, typeof='leisheng', n=n)  # 进行生成","metadata":{"execution":{"iopub.status.busy":"2024-03-25T13:08:10.374127Z","iopub.execute_input":"2024-03-25T13:08:10.374547Z","iopub.status.idle":"2024-03-25T13:08:29.834423Z","shell.execute_reply.started":"2024-03-25T13:08:10.374513Z","shell.execute_reply":"2024-03-25T13:08:29.832514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\ndef generator_full(model_path, audio_name, typeof):\n    netG = Generator(80, 32, 3)\n    netG.load_state_dict(torch.load(model_path, map_location=torch.device('cpu'))) # 导入网络的参数\n    \n    fft = Audio2Mel(n_mel_channels=80)\n    audio_path = '/kaggle/input/audio-train/train/' + audio_name\n    for i in range(3):\n        y, sr = librosa.load(audio_path, sr=22050, mono=True, offset=(i * 4), duration=4.0)  # , offset=2.1, duration=9.0\n        y = y.reshape(1, 1, -1)\n        # print('y', type(y), len(y), y)\n        s_t = fft(torch.from_numpy(y)).detach()\n        # print(s_t, type(s_t), s_t.shape)\n\n        pred_audio = netG(s_t)\n        pred_audio = pred_audio.squeeze()\n        \n        if i == 0:\n            pred_audio_part1 = pred_audio\n        else:\n            pred_audio_part2 = pred_audio\n    pred_audio = torch.cat((pred_audio_part1, pred_audio_part2), dim=0)\n    save_sample('/kaggle/working/' + typeof +  '_' + audio_name, 22050, pred_audio)  # 将wav音频文件保存\n    return pred_audio\n'''","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# netG = Generator(80, 32, 3)\n# netG.load_state_dict(torch.load(\"/kaggle/input/netg-s/netG/leisheng_best_netG.pt\", map_location=torch.device('cpu'))) # 导入网络的参数","metadata":{"execution":{"iopub.status.busy":"2024-01-15T16:33:50.531378Z","iopub.execute_input":"2024-01-15T16:33:50.532199Z","iopub.status.idle":"2024-01-15T16:33:50.708695Z","shell.execute_reply.started":"2024-01-15T16:33:50.532164Z","shell.execute_reply":"2024-01-15T16:33:50.707569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fft = Audio2Mel(n_mel_channels=80)\n# audio_path = '/kaggle/input/audio-train/train/0.wav'\n# y, sr = librosa.load(audio_path, sr=22050, mono=True, offset=0, duration=4.0)  # , offset=2.1, duration=9.0\n# y = y.reshape(1, 1, -1)\n# print('y', type(y), len(y), y)\n# s_t = fft(torch.from_numpy(y)).detach()\n# print(s_t, type(s_t), s_t.shape)\n\n# pred_audio = netG(s_t)\n# pred_audio = pred_audio.squeeze()\n\n# pred_audio1 = pred_audio.detach().numpy()\n# y_hat = librosa.resample(pred_audio1, orig_sr=22050, target_sr=24000, fix=False, scale=False)  # \n# X = librosa.stft(pred_audio1, n_fft=4800, hop_length=3000, win_length=4800, window='blackmanharris')","metadata":{"execution":{"iopub.status.busy":"2024-01-15T17:32:59.452412Z","iopub.execute_input":"2024-01-15T17:32:59.453271Z","iopub.status.idle":"2024-01-15T17:32:59.957042Z","shell.execute_reply.started":"2024-01-15T17:32:59.453232Z","shell.execute_reply":"2024-01-15T17:32:59.956130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(pred_audio, type(pred_audio), pred_audio.shape)\n# result = torch.cat((pred_audio, pred_audio), dim=0)\n# print(result, type(result), result.shape)\n","metadata":{"execution":{"iopub.status.busy":"2024-01-15T16:57:30.402148Z","iopub.execute_input":"2024-01-15T16:57:30.402565Z","iopub.status.idle":"2024-01-15T16:57:30.412066Z","shell.execute_reply.started":"2024-01-15T16:57:30.402527Z","shell.execute_reply":"2024-01-15T16:57:30.410769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_label(path, filename):\n    df_diff = pd.read_csv(path, sep=\",|:|;\", header=None, names=['名称', '是否正常','类型'],\n                          usecols=['名称', '是否正常','类型'],engine='python')\n    df_diff = np.array(df_diff)\n    for item in df_diff:\n        sh = item[0]\n        if filename == sh:\n            #return item[1]\n            '''\n            if item[1] == 'normal': return 1\n            if item[1] == 'anomaly': return 0\n            if item[1] == 'A00': return 1\n            if item[1] == 'A01': return 0\n            '''\n            if item[2] == 'B01' or item[2] == 'byq_jjsd':\n                return 1\n            elif item[2] == 'B02' or item[2] == 'byq_lqqyx_A' or item[2] == 'byq_lqqyx_B':\n                return 2\n            elif item[2] == 'B03' or item[2] == 'byq_zgz':\n                return 3\n            elif item[2] == 'B04' or item[2] == 'byq_dlcj':\n                return 4\n            elif item[2] == 'B05' or item[2] == 'byq_jbfd_xffd' or item[2] == 'byq_jbfd_jxxfd' or item[2] == 'byq_jbfd_dyfd' or item[2] == 'byq_jbfd_bmfd':\n                return 5\n            elif item[2] == 'B06' or item[2] == 'byq_zlpc':\n                return 6\n            elif item[2] == 'byq_rensheng':\n                return 'rensheng'\n            elif item[2] == 'byq_leisheng':\n                return 'leisheng'\n            elif item[2] == 'byq_qidi':\n                return 'qidi'\n            elif item[2] == 'byq_niaoming':\n                return 'niaoming'\n            elif item[2] == 'byq_yusheng':\n                return 'yusheng'\n            else:\n                return 0","metadata":{"execution":{"iopub.status.busy":"2024-03-24T06:33:35.940193Z","iopub.execute_input":"2024-03-24T06:33:35.940710Z","iopub.status.idle":"2024-03-24T06:33:35.957048Z","shell.execute_reply.started":"2024-03-24T06:33:35.940675Z","shell.execute_reply":"2024-03-24T06:33:35.955739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_audio(audio_path, hop_length, win_length):    # hop_length = 3000 帧移     win_length = 4800 帧长\n    y, sr = librosa.load(audio_path, sr=24000, mono=True, offset=0.0, duration=7.95)\n    X = librosa.stft(y, n_fft=4800, hop_length=hop_length, win_length=win_length, window='blackmanharris')\n    S = librosa.feature.melspectrogram(S=((np.abs(X)) ** 2), sr=sr, n_mels=128, window='blackmanharris', center=True,\n                                       pad_mode='reflect')\n    Sdb = librosa.power_to_db(S, ref=np.max)\n    # print('Sdb.shape', Sdb.shape)\n    # Sdb = np.array(Sdb,dtype='float32')\n    Sdb = np.abs(Sdb)\n    # Sdb = Sdb.reshape(64, 64, 1)\n    Sdb = Sdb.reshape(128, 128, 1)\n    Sdb = Sdb.tolist()\n    # print(Sdb)\n    return Sdb","metadata":{"execution":{"iopub.status.busy":"2024-03-24T06:33:38.577780Z","iopub.execute_input":"2024-03-24T06:33:38.578326Z","iopub.status.idle":"2024-03-24T06:33:38.587741Z","shell.execute_reply.started":"2024-03-24T06:33:38.578291Z","shell.execute_reply":"2024-03-24T06:33:38.586777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cancha_block(inputs_b, kernel_num, strides):\n    ins_channels = inputs_b.shape[-1]\n\n    x = DepthwiseConv2D((3, 3), strides=strides, padding=\"same\")(inputs_b)\n    x = BatchNormalization()(x)\n    x = ReLU()(x)\n\n    x = Conv2D(kernel_num, (1, 1), strides=(1, 1), padding=\"same\")(x)\n    x = BatchNormalization()(x)\n    x = ReLU()(x)\n\n    if ins_channels == kernel_num and strides == 1:\n        x = Add()([x, inputs_b])\n\n    return x","metadata":{"execution":{"iopub.status.busy":"2024-03-24T06:33:40.931561Z","iopub.execute_input":"2024-03-24T06:33:40.932184Z","iopub.status.idle":"2024-03-24T06:33:40.941693Z","shell.execute_reply.started":"2024-03-24T06:33:40.932133Z","shell.execute_reply":"2024-03-24T06:33:40.939911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def Mobile_cancha(input_shape, num_classes):\n    inputs = Input(shape=input_shape)\n    # [64,64,1] --> [32,32,32]\n    x = Conv2D(32, (3, 3), strides=(2, 2), padding=\"same\")(inputs)  # 二维输入.32个卷积核的大小3*3，移动步长strides\n    x = BatchNormalization()(x)\n    x = ReLU()(x)\n    # [32,32,32] --> [32,32,64]\n    x = cancha_block(inputs_b=x, kernel_num=64, strides=1)\n    # [32,32,64] --> [16,16,128]\n    x = cancha_block(inputs_b=x, kernel_num=128, strides=2)\n    # [16,16,128] --> [16,16,128]\n    x = cancha_block(inputs_b=x, kernel_num=128, strides=1)\n    # [16,16,128] --> [8,8,256]\n    x = cancha_block(inputs_b=x, kernel_num=256, strides=2)\n    # [8,8,256] --> [8,8,256]\n    x = cancha_block(inputs_b=x, kernel_num=256, strides=1)\n    # [8,8,256] --> [4,4,512]\n    x = cancha_block(inputs_b=x, kernel_num=512, strides=2)\n    # [4,4,512] --> [4,4,512]  4次\n    for i in range(4):\n        x = cancha_block(inputs_b=x, kernel_num=512, strides=(1, 1))\n    # [4,4,256] --> [2,2,512]\n    x = DepthwiseConv2D((3, 3), strides=(2, 2), padding=\"same\")(x)\n    x = BatchNormalization()(x)\n    x = ReLU()(x)\n    x = Dropout(0.5)(x)\n    # [2,2,512] --> [2,2,1024]\n    x = Conv2D(1024, (1, 1), strides=(1, 1), padding=\"same\")(x)\n    x = BatchNormalization()(x)\n    x = ReLU()(x)\n    # [2,2,1024] --> [1,1024]\n    x = GlobalAveragePooling2D()(x)\n    x = Dropout(0.5)(x)\n    outputs = Dense(num_classes, activation=\"softmax\")(x)\n    model_b = Model(inputs=inputs, outputs=outputs)\n    \n    return model_b","metadata":{"execution":{"iopub.status.busy":"2024-03-24T06:33:43.324777Z","iopub.execute_input":"2024-03-24T06:33:43.325979Z","iopub.status.idle":"2024-03-24T06:33:43.338840Z","shell.execute_reply.started":"2024-03-24T06:33:43.325935Z","shell.execute_reply":"2024-03-24T06:33:43.337657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_loss_acc(history):\n    # 从history中提取模型训练集和验证集准确率信息和误差信息\n    acc = history.history['accuracy']\n    val_acc = history.history['val_accuracy']\n    loss = history.history['loss']\n    val_loss = history.history['val_loss']\n    # 按照上下结构将图画输出\n    plt.figure(figsize=(8, 8))\n    plt.subplot(2, 1, 1)\n    plt.plot(acc, label='Training Accuracy')\n    plt.plot(val_acc, label='Validation Accuracy')\n    plt.legend(loc='lower right')\n    plt.ylabel('Accuracy')\n    plt.ylim([min(plt.ylim()), 1])\n    plt.title('Training and Validation Accuracy')\n\n    plt.subplot(2, 1, 2)\n    plt.plot(loss, label='Training Loss')\n    plt.plot(val_loss, label='Validation Loss')\n    plt.legend(loc='upper right')\n    plt.ylabel('Cross Entropy')\n    plt.title('Training and Validation Loss')\n    plt.xlabel('epoch')\n    plt.savefig('/kaggle/working/mobilenet_test2_0907.png', dpi=200)","metadata":{"execution":{"iopub.status.busy":"2024-03-24T06:33:46.114689Z","iopub.execute_input":"2024-03-24T06:33:46.115205Z","iopub.status.idle":"2024-03-24T06:33:46.126769Z","shell.execute_reply.started":"2024-03-24T06:33:46.115165Z","shell.execute_reply":"2024-03-24T06:33:46.125252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\ndef get_gaudio(y, hop_length, win_length):\n    y_hat = librosa.resample(y, orig_sr=22050, target_sr=24000, fix=False, scale=False)  # 重新采样从orig_sr到target_sr的时间序列\n    y_hat = y_hat[:190800]\n    X = librosa.stft(y_hat, n_fft=4800, hop_length=hop_length, win_length=win_length, window='blackmanharris')\n    S = librosa.feature.melspectrogram(S=((np.abs(X)) ** 2), sr=24000, n_mels=128, window='blackmanharris', center=True,\n                                        pad_mode='reflect')\n    Sdb = librosa.power_to_db(S, ref=np.max)\n    # print('Sdb.shape', Sdb.shape)\n    # Sdb = np.array(Sdb,dtype='float32')\n    Sdb = np.abs(Sdb)\n    Sdb = Sdb.reshape(128, 128, 1)\n    Sdb = Sdb.tolist()\n    return Sdb\n    \n\naudios = []\nlabels = []\nhop_length=1500\nwin_length=4800\nfor fn in os.listdir('/kaggle/input/audio-train/train'):\n    if fn.endswith('.wav'):\n        fd = os.path.join('/kaggle/input/audio-train/train/', fn)\n        audios.append(read_audio(audio_path=fd, hop_length=hop_length, win_length=win_length))\n        if read_label('/kaggle/input/labels-1/label/answer_train.csv', fn) == 'rensheng':\n            labels.append(0)  # 首先添加一个真正常标签\n            \n            model_path = '/kaggle/input/netg-s/netG/rensheng_best_netG.pt'\n            y = generator_full(model_path=model_path, audio_name=fn, typeof='rensheng')  # 进行生成并添加标签\n            y = y.detach().numpy()\n            audios.append(get_gaudio(y=y, hop_length=hop_length, win_length=win_length))\n            labels.append(0)  # 再添加一个正常标签\n        elif read_label('/kaggle/input/labels-1/label/answer_train.csv', fn) == 'leisheng':\n            labels.append(0)  # 首先添加一个真正常标签\n            \n            model_path = '/kaggle/input/netg-s/netG/leisheng_best_netG.pt'\n            y = generator_full(model_path=model_path, audio_name=fn, typeof='leisheng')  # 进行生成并添加标签\n            y = y.detach().numpy()\n            audios.append(get_gaudio(y=y, hop_length=hop_length, win_length=win_length))\n            labels.append(0)  # 再添加一个正常标签\n        elif read_label('/kaggle/input/labels-1/label/answer_train.csv', fn) == 'qidi':\n            labels.append(0)  # 首先添加一个真正常标签\n            \n            model_path = '/kaggle/input/netg-s/netG/qidi_best_netG.pt'\n            y = generator_full(model_path=model_path, audio_name=fn, typeof='qidi')  # 进行生成并添加标签\n            y = y.detach().numpy()\n            audios.append(get_gaudio(y=y, hop_length=hop_length, win_length=win_length))\n            labels.append(0)  # 再添加一个正常标签\n        elif read_label('/kaggle/input/labels-1/label/answer_train.csv', fn) == 'yusheng':\n            labels.append(0)  # 首先添加一个真正常标签\n            \n            model_path = '/kaggle/input/netg-s/netG/yusheng_best_netG.pt'\n            y = generator_full(model_path=model_path, audio_name=fn, typeof='yusheng')  # 进行生成并添加标签\n            y = y.detach().numpy()\n            audios.append(get_gaudio(y=y, hop_length=hop_length, win_length=win_length))\n            labels.append(0)  # 再添加一个正常标签\n        elif read_label('/kaggle/input/labels-1/label/answer_train.csv', fn) == 'niaoming':\n            labels.append(0)  # 首先添加一个真正常标签\n            \n            model_path = '/kaggle/input/netg-s/netG/niaoming_best_netG.pt'\n            y = generator_full(model_path=model_path, audio_name=fn, typeof='niaoming')  # 进行生成并添加标签\n            y = y.detach().numpy()\n            audios.append(get_gaudio(y=y, hop_length=hop_length, win_length=win_length))\n            labels.append(0)  # 再添加一个正常标签\n        else:\n            labels.append(read_label('/kaggle/input/labels-1/label/answer_train.csv', fn))\n\n            \n'''","metadata":{"execution":{"iopub.status.busy":"2024-01-19T08:54:38.091744Z","iopub.execute_input":"2024-01-19T08:54:38.092085Z","iopub.status.idle":"2024-01-19T08:54:47.799907Z","shell.execute_reply.started":"2024-01-19T08:54:38.092060Z","shell.execute_reply":"2024-01-19T08:54:47.798667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# models_add = [\n#     \"/kaggle/input/gen-models-0906/models/generator_leisheng.h5\",\n#     \"/kaggle/input/gen-models-0906/models/generator_niaoming.h5\",\n#     \"/kaggle/input/gen-models-0906/models/generator_qidi.h5\",\n#     \"/kaggle/input/gen-models-0906/models/generator_rensheng.h5\",\n#     \"/kaggle/input/gen-models-0906/models/generator_yusheng.h5\"\n# ]\n# for i in range(5):\n#     generator_machine = tf.keras.models.load_model(models_add[i], compile=False)  # todo 修改模型名称\n#     for j in range(300):\n#         seed = tf.random.normal([1, 100])  # num=1, noise_dim=100\n#         outputs_1 = generator_machine(seed,  training=False)  # 将图片输入模型得到结果\n#         outputs_1 = np.array(outputs_1)\n#         outputs_1 = outputs_1.astype(\"float32\") * 127.5\n#         outputs_1 = outputs_1.reshape(64, 64, 1)\n#         audios.append(outputs_1)\n#         labels.append(0)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nx_train = np.array(audios)\nx_train = x_train.astype(\"float32\") / 127.5\n\n\ny_train = np.array(labels)\ny_train = y_train.reshape(-1, 1)\n'''","metadata":{"execution":{"iopub.status.busy":"2024-01-16T02:06:44.724270Z","iopub.status.idle":"2024-01-16T02:06:44.724648Z","shell.execute_reply.started":"2024-01-16T02:06:44.724463Z","shell.execute_reply":"2024-01-16T02:06:44.724482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(x_train.shape, y_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-01-16T02:06:44.726163Z","iopub.status.idle":"2024-01-16T02:06:44.726516Z","shell.execute_reply.started":"2024-01-16T02:06:44.726332Z","shell.execute_reply":"2024-01-16T02:06:44.726357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\naudios_test = []\nlabels_test = []\nfor fn in os.listdir('/kaggle/input/audio-test2/test2'):\n    if fn.endswith('.wav'):\n        fd = os.path.join('/kaggle/input/audio-test2/test2/', fn)\n        audios_test.append(read_audio(audio_path=fd, hop_length=hop_length, win_length=win_length))\n        # print(read_audio(audio_path=fd, hop_length=3000, win_length=4800))\n        labels_test.append(read_label('/kaggle/input/labels-1/label/answer_test2.csv', fn))\n'''","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nx_test = np.array(audios_test, dtype='float32')\nx_test = x_test.astype(\"float32\") / 127.5\n\ny_test = np.array(labels_test)\ny_test = y_test.reshape(-1, 1)\n\nprint(x_test.shape, y_test.shape)\n'''","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nnum_classe = 7\n\nprint('x_train.shape,y_train.shape,x_test.shape,y_test.shape', x_train.shape, y_train.shape, x_test.shape, y_test.shape)\n\ny_train = tf.keras.utils.to_categorical(y_train, num_classes=num_classe)\ny_test = tf.keras.utils.to_categorical(y_test, num_classes=num_classe)\n'''","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train(epochs):\n    model = Mobile_cancha((128, 128, 1), num_classes=num_classe)\n    # model = MobileNet((128, 128, 1), num_classe)\n    adam = Adam(learning_rate=0.0001, beta_1=0.999, beta_2=0.999)\n    model.summary()\n    model.compile(loss=\"categorical_crossentropy\", optimizer=adam, metrics=[\"accuracy\"])\n    \n    callbacks = [tf.keras.callbacks.ModelCheckpoint(filepath='/kaggle/working/best.h5',\n                                                    save_best_only=True,\n                                                    save_weights_only=False,\n                                                    monitor='val_accuracy')]\n    \n    history = model.fit(x_train, y_train, batch_size=4, epochs=epochs, validation_data=(x_test, y_test), verbose=1, callbacks=callbacks)\n    # todo 保存模型， 修改为你要保存的模型的名称\n    model.save(\"/kaggle/working/0116.h5\")\n    # 测试\n    test_loss, test_acc = model.evaluate(x_test, y_test)\n    print(\"Test Loss: {}, Test Accuracy: {}\".format(test_loss, test_acc))\n    show_loss_acc(history)","metadata":{"execution":{"iopub.status.busy":"2024-03-24T06:33:51.428618Z","iopub.execute_input":"2024-03-24T06:33:51.429447Z","iopub.status.idle":"2024-03-24T06:33:51.439523Z","shell.execute_reply.started":"2024-03-24T06:33:51.429406Z","shell.execute_reply":"2024-03-24T06:33:51.437961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if __name__ == '__main__':\n    # train(epochs=100)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}