{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30683,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Train data and test data distribution\nIn this notebook, we will explore the following questions:\n\n- Introduce a deep denoising model to estimate the Signal-to-Noise Ratio (SNR) of audio.\n- Review the distribution of SNR in the training data and observe its correlation with rating.\n- Review the distribution of SNR in the test data (unlabeled_soundscapes).\n\nAreas for improvement:\nThe denoising model is designed for human voices, maybe we should train one specifically for enhancing bird sounds. If you know of a source for clean bird sound data, please let me know.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"# use this repo's denoising model, star it maybe if you think useful\n# may need update models for bird sounds enhancement \n!git clone https://github.com/lhwcv/DTLN_pytorch","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport glob\nimport random\nimport torch\nimport sys\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\n\nimport matplotlib.pyplot as plt\nplt.style.use('ggplot')\nimport librosa\nimport librosa.display\nfrom matplotlib.colorbar import Colorbar\nimport IPython.display as ipd\n\nsys.path.append('/kaggle/working/DTLN_pytorch/')\nfrom DTLN_model import Pytorch_DTLN\nfrom audio_io import wav_read","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:02:34.375684Z","iopub.execute_input":"2024-04-11T07:02:34.376008Z","iopub.status.idle":"2024-04-11T07:02:34.383929Z","shell.execute_reply.started":"2024-04-11T07:02:34.375978Z","shell.execute_reply":"2024-04-11T07:02:34.382738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Pytorch_DTLN().cuda()\nprint('==> load model.. ', )\nmodel.load_state_dict(torch.load('/kaggle/working/DTLN_pytorch/pretrained/model.pth'))\nmodel = model.eval()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:02:34.384968Z","iopub.execute_input":"2024-04-11T07:02:34.385220Z","iopub.status.idle":"2024-04-11T07:02:34.423253Z","shell.execute_reply.started":"2024-04-11T07:02:34.385198Z","shell.execute_reply":"2024-04-11T07:02:34.422294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calculate_snr_by_model(filename):\n    def snr(clean, noisy):\n        noise = noisy - clean\n        snr = 10 * np.log10(np.sum(clean**2) / np.sum(noise**2))\n        return snr\n\n    x, fs = wav_read(filename, tgt_fs=16000)\n    xt = torch.from_numpy(x).unsqueeze(0).cuda()\n    \n    with torch.no_grad():\n        clean = model(xt).squeeze().cpu().numpy()\n    \n    snr_value = snr(clean, x)\n    \n    return snr_value, clean, fs","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:02:34.424516Z","iopub.execute_input":"2024-04-11T07:02:34.424806Z","iopub.status.idle":"2024-04-11T07:02:34.431431Z","shell.execute_reply.started":"2024-04-11T07:02:34.424782Z","shell.execute_reply":"2024-04-11T07:02:34.430518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def visualize_audio(filename, audio_data=None, sample_rate=None):\n    if audio_data is None:\n        audio_data, sample_rate = librosa.load(filename)\n\n    ipd.display(ipd.Audio(audio_data, rate=sample_rate))\n\n    fig, axs = plt.subplots(nrows=2, ncols=1, figsize=(16, 12))\n\n    axs[0].plot(audio_data)\n    axs[0].set_title('Waveform')\n\n    spectrogram = librosa.amplitude_to_db(librosa.stft(audio_data), ref=np.max)\n    librosa.display.specshow(spectrogram, sr=sample_rate, x_axis='time', y_axis='hz', ax=axs[1])\n    axs[1].set_title('Spectrogram')","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:02:34.434082Z","iopub.execute_input":"2024-04-11T07:02:34.434366Z","iopub.status.idle":"2024-04-11T07:02:34.442414Z","shell.execute_reply.started":"2024-04-11T07:02:34.434342Z","shell.execute_reply":"2024-04-11T07:02:34.441486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filepath = '/kaggle/input/birdclef-2024/train_audio/asbfly/XC134896.ogg'\n","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:02:34.443574Z","iopub.execute_input":"2024-04-11T07:02:34.443830Z","iopub.status.idle":"2024-04-11T07:02:34.451748Z","shell.execute_reply.started":"2024-04-11T07:02:34.443808Z","shell.execute_reply":"2024-04-11T07:02:34.450904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_audio(filepath)","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:02:34.452795Z","iopub.execute_input":"2024-04-11T07:02:34.453066Z","iopub.status.idle":"2024-04-11T07:02:36.718212Z","shell.execute_reply.started":"2024-04-11T07:02:34.453043Z","shell.execute_reply":"2024-04-11T07:02:36.717029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"snr, clean, sample_rate = calculate_snr_by_model(filepath)\nprint('snr: ', snr)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:02:36.719577Z","iopub.execute_input":"2024-04-11T07:02:36.719926Z","iopub.status.idle":"2024-04-11T07:02:36.834987Z","shell.execute_reply.started":"2024-04-11T07:02:36.719896Z","shell.execute_reply":"2024-04-11T07:02:36.833996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_audio(filepath, clean, sample_rate)","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:02:36.836085Z","iopub.execute_input":"2024-04-11T07:02:36.836368Z","iopub.status.idle":"2024-04-11T07:02:38.479983Z","shell.execute_reply.started":"2024-04-11T07:02:36.836345Z","shell.execute_reply":"2024-04-11T07:02:38.478966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/birdclef-2024/train_metadata.csv')\ndf['filename'] = '/kaggle/input/birdclef-2024/train_audio/' + df['filename']","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:02:38.481306Z","iopub.execute_input":"2024-04-11T07:02:38.481645Z","iopub.status.idle":"2024-04-11T07:02:38.662150Z","shell.execute_reply.started":"2024-04-11T07:02:38.481619Z","shell.execute_reply":"2024-04-11T07:02:38.661013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"part_df = df.sample(1000, random_state=42).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:02:38.663691Z","iopub.execute_input":"2024-04-11T07:02:38.664036Z","iopub.status.idle":"2024-04-11T07:02:38.677473Z","shell.execute_reply.started":"2024-04-11T07:02:38.664008Z","shell.execute_reply":"2024-04-11T07:02:38.676637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tqdm\nsnrs = []\nfilepaths = part_df['filename'].tolist()\nfor filepath in tqdm.tqdm(filepaths):\n    snr, _, _ = calculate_snr_by_model(filepath)\n    snrs.append(snr)\n\npart_df['snr'] = snrs","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:02:38.678879Z","iopub.execute_input":"2024-04-11T07:02:38.679270Z","iopub.status.idle":"2024-04-11T07:05:20.319781Z","shell.execute_reply.started":"2024-04-11T07:02:38.679242Z","shell.execute_reply":"2024-04-11T07:05:20.318791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.pairplot(part_df[['rating', 'snr']])\nplt.title('Correlation between rating and SNR')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:05:20.321172Z","iopub.execute_input":"2024-04-11T07:05:20.321538Z","iopub.status.idle":"2024-04-11T07:05:21.669261Z","shell.execute_reply.started":"2024-04-11T07:05:20.321505Z","shell.execute_reply":"2024-04-11T07:05:21.668254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let's look rating > 4 but snr < -40\ncheck_df = part_df[part_df['rating']>4]\ncheck_df = check_df[check_df['snr']<-40]\nprint(len(check_df))\n","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:14:55.780919Z","iopub.execute_input":"2024-04-11T07:14:55.781826Z","iopub.status.idle":"2024-04-11T07:14:55.789329Z","shell.execute_reply.started":"2024-04-11T07:14:55.781793Z","shell.execute_reply":"2024-04-11T07:14:55.788274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_audio(check_df['filename'].iloc[0])","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:15:48.207982Z","iopub.execute_input":"2024-04-11T07:15:48.208749Z","iopub.status.idle":"2024-04-11T07:15:50.077977Z","shell.execute_reply.started":"2024-04-11T07:15:48.208714Z","shell.execute_reply":"2024-04-11T07:15:50.077027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### It seems sound good, we should develop a denoise model for bird sound","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"sns.histplot(part_df['snr'])\nplt.title('train data SNR distribution')  # 添加标题\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:05:21.673278Z","iopub.execute_input":"2024-04-11T07:05:21.673597Z","iopub.status.idle":"2024-04-11T07:05:22.021970Z","shell.execute_reply.started":"2024-04-11T07:05:21.673570Z","shell.execute_reply":"2024-04-11T07:05:22.021019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## let's look unlabeled_soundscapes SNR distribution\n","metadata":{}},{"cell_type":"code","source":"test_files = glob.glob('/kaggle/input/birdclef-2024/unlabeled_soundscapes/*.ogg')\npart_test_files = random.sample(test_files, 500)\ntest_snrs = []\nfor filepath in tqdm.tqdm(part_test_files):\n    snr, _, _ = calculate_snr_by_model(filepath)\n    test_snrs.append(snr)","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:05:22.023331Z","iopub.execute_input":"2024-04-11T07:05:22.023715Z","iopub.status.idle":"2024-04-11T07:12:41.555801Z","shell.execute_reply.started":"2024-04-11T07:05:22.023683Z","shell.execute_reply":"2024-04-11T07:12:41.554828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(part_df['snr'])\nplt.title('unlabeled_soundscapes data(part) SNR distribution')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:12:44.201316Z","iopub.execute_input":"2024-04-11T07:12:44.202348Z","iopub.status.idle":"2024-04-11T07:12:44.550381Z","shell.execute_reply.started":"2024-04-11T07:12:44.202313Z","shell.execute_reply":"2024-04-11T07:12:44.549433Z"},"trusted":true},"execution_count":null,"outputs":[]}]}