{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Identifying Duplicates","metadata":{}},{"cell_type":"markdown","source":"As Matt OP [posted in the discussions](https://www.kaggle.com/competitions/birdclef-2023/discussion/396506), he discovered that there is duplicates in the dataset with different XC ID. My initial approach was to hash the first five seconds of the audio and see if they are identical or not, but suprisingly none of the audios have the same hash. It turned out that even if they're identical, their values still differ slightly (maybe due to some compression reasons?).\n\nThe solution is to use dynamic time warping (DTW) distance as the similarity metric. DTW is an common metric used to measure time series similarity since it is robust to time shift.  \nManual validation is still needed since there exists false positives, for example, being completely silent in the first five seconds. This can be done be observing the spectrogram after subtracting the two signals.\n\nIn the end, there are 18 duplicates that are discovered. They all seem to be reuploads from the same author, but some of them have differences in call types, ratings, and secondary labels. If you are using those fields maybe it is worth thinking about which row to keep.\n\n*Do note that this is not guaranteed to find all duplicates. Adjusting the threshold of similarity may find more duplicates.*","metadata":{}},{"cell_type":"code","source":"!pip install tslearn","metadata":{"execution":{"iopub.status.busy":"2023-03-29T03:30:04.977954Z","iopub.execute_input":"2023-03-29T03:30:04.978492Z","iopub.status.idle":"2023-03-29T03:30:20.095401Z","shell.execute_reply.started":"2023-03-29T03:30:04.978436Z","shell.execute_reply":"2023-03-29T03:30:20.094105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torchaudio\nfrom torchaudio import transforms\nfrom pathlib import Path\nfrom tslearn import metrics\nfrom tqdm.notebook import tqdm\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport librosa","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-29T03:30:20.098839Z","iopub.execute_input":"2023-03-29T03:30:20.099329Z","iopub.status.idle":"2023-03-29T03:30:27.335007Z","shell.execute_reply.started":"2023-03-29T03:30:20.099263Z","shell.execute_reply":"2023-03-29T03:30:27.333334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    sr = 32000\n    input_base_path = Path(\"/kaggle/input/birdclef-2023/train_audio\")\nmeta_df = pd.read_csv('/kaggle/input/birdclef-2023/train_metadata.csv')\n# meta_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-29T03:34:27.838044Z","iopub.execute_input":"2023-03-29T03:34:27.839769Z","iopub.status.idle":"2023-03-29T03:34:27.947259Z","shell.execute_reply.started":"2023-03-29T03:34:27.839697Z","shell.execute_reply":"2023-03-29T03:34:27.945675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot mel spectrogram of all given audio, and plot signal difference of sig[0] and sig[1]\ndef compare_spectrogram(sig, sr=Config.sr, n_fft=2048, hop_length = None, vmin=-60, vmax=60, plt_subtract=True):     \n    if not hop_length:\n        hop_length = n_fft//2\n    num_sig = len(sig)\n    \n    fig, axs = plt.subplots(num_sig+plt_subtract, 1, figsize=(24, 6*(num_sig+plt_subtract)))\n    if num_sig==1:\n        axs = [axs]\n        sig = [sig]\n    for i in range(num_sig):\n        _sig = sig[i]\n        ax = axs[i]\n        mel_spec = librosa.power_to_db(librosa.feature.melspectrogram(y=_sig,sr=sr, n_mels=128,n_fft=n_fft,hop_length=hop_length))\n        img=librosa.display.specshow(mel_spec[0], sr = sr, hop_length = hop_length, x_axis = 'time',y_axis = 'mel', ax=ax, vmin=vmin, vmax=vmax)\n        ax.set_title('Mel Spectrogram', fontsize=10) \n        fig.colorbar(img,ax=ax,boundaries=np.linspace(-80,80,160))\n    \n    # Subtract two signals and plot Mel spectrogram\n    if plt_subtract and num_sig>1:\n        _sig = sig[1]-sig[0]\n        ax = axs[-1]\n        mel_spec = librosa.power_to_db(librosa.feature.melspectrogram(y=_sig,sr=sr, n_mels=128,n_fft=n_fft,hop_length=hop_length))\n        img=librosa.display.specshow(mel_spec[0], sr = sr, hop_length = hop_length, x_axis = 'time',y_axis = 'mel', ax=ax, vmin=vmin, vmax=vmax)\n        ax.set_title('Subtract plot', fontsize=10) \n        fig.colorbar(img,ax=ax,boundaries=np.linspace(-80,80,160))\n    plt.show()\n    \ndef load_audio_sample(row, num_frames = 32000*5):\n    try:\n        sig, sr = torchaudio.load(Config.input_base_path / Path(row.filename),frame_offset=0,num_frames=num_frames)\n    except:\n        sig, sr = torchaudio.load(Config.input_base_path / Path(row.filename.item()),frame_offset=0,num_frames=num_frames)\n    sig = torch.nn.functional.pad(sig, (0,num_frames - sig.shape[1]), \"constant\", 0)\n    return sig","metadata":{"execution":{"iopub.status.busy":"2023-03-29T03:41:32.990402Z","iopub.execute_input":"2023-03-29T03:41:32.990871Z","iopub.status.idle":"2023-03-29T03:41:33.007355Z","shell.execute_reply.started":"2023-03-29T03:41:32.990827Z","shell.execute_reply":"2023-03-29T03:41:33.006364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Quick demonstration of what we are expecting","metadata":{}},{"cell_type":"code","source":"sig1, _ = torchaudio.load('/kaggle/input/birdclef-2023/train_audio/abethr1/XC128013.ogg',frame_offset=0,num_frames=32000*5)\nsig2_1 = load_audio_sample(meta_df[meta_df.filename.str.contains(\"XC748220\")])\nsig2_2 = load_audio_sample(meta_df[meta_df.filename.str.contains(\"XC748221\")])","metadata":{"execution":{"iopub.status.busy":"2023-03-29T03:41:38.296411Z","iopub.execute_input":"2023-03-29T03:41:38.29769Z","iopub.status.idle":"2023-03-29T03:41:38.37024Z","shell.execute_reply.started":"2023-03-29T03:41:38.297633Z","shell.execute_reply":"2023-03-29T03:41:38.368718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Not similar signal\nprint(\"The DTW distance is {}\".format(metrics.dtw(sig1, sig2_1)))\ncompare_spectrogram([sig1.numpy(),sig2_1.numpy()])","metadata":{"execution":{"iopub.status.busy":"2023-03-29T03:41:38.80845Z","iopub.execute_input":"2023-03-29T03:41:38.808908Z","iopub.status.idle":"2023-03-29T03:41:40.099552Z","shell.execute_reply.started":"2023-03-29T03:41:38.80887Z","shell.execute_reply":"2023-03-29T03:41:40.0985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Similar signal\nprint(\"The DTW distance is {}\".format(metrics.dtw(sig2_1, sig2_2)))\ncompare_spectrogram([sig2_1.numpy(),sig2_2.numpy()])","metadata":{"execution":{"iopub.status.busy":"2023-03-29T03:41:46.241856Z","iopub.execute_input":"2023-03-29T03:41:46.243274Z","iopub.status.idle":"2023-03-29T03:41:47.49466Z","shell.execute_reply.started":"2023-03-29T03:41:46.243197Z","shell.execute_reply":"2023-03-29T03:41:47.493633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Calculate DTW distance\n\nIf we calculate DTW across all audio pairs, we will have a massive similarity matrix which will take a long time to compute.  \nBelow I grouped the audios with ['latitude','longitude'] and computed the similarity matrix.","metadata":{}},{"cell_type":"code","source":"dtw_threshold = 0.2\nduplicate_list = []\n\nfor group_name, df_group in tqdm(meta_df.groupby(['latitude','longitude'])):\n    if len(df_group) > 1:\n        audio_list = []\n        df_group.apply(lambda row:audio_list.append(load_audio_sample(row)),axis=1)\n        similarity_matrix = metrics.cdist_dtw(audio_list)\n        highly_similar = np.asarray(similarity_matrix<dtw_threshold).nonzero()\n        for sig1_id, sig2_id in zip(*highly_similar):\n            if sig1_id<sig2_id:\n                row1, row2 = df_group.iloc[[sig1_id]],df_group.iloc[[sig2_id]]\n                duplicate_list.append([row1.filename.item(),row2.filename.item()])\n                print(\"DTW distance is {}\".format(similarity_matrix[sig1_id,sig2_id]))\n                display(row1,row2)\n                compare_spectrogram((load_audio_sample(row1).numpy(),load_audio_sample(row2).numpy()))\n                print(\"=\" * 150 + \"\\n\")","metadata":{"execution":{"iopub.status.busy":"2023-03-29T04:03:05.600553Z","iopub.execute_input":"2023-03-29T04:03:05.601052Z","iopub.status.idle":"2023-03-29T04:11:04.670649Z","shell.execute_reply.started":"2023-03-29T04:03:05.601009Z","shell.execute_reply":"2023-03-29T04:11:04.669402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Remove manual inspected false positives\nfalse_positive_identical = [['nubwoo1/XC398476.ogg', 'reedov1/XC398475.ogg'],\n['combul2/XC397816.ogg', 'gybfis1/XC397814.ogg'],\n['combul2/XC397816.ogg', 'ratcis1/XC397815.ogg'],\n['combul2/XC397816.ogg', 'wbrcha2/XC397798.ogg']]\nfor fpi in false_positive_identical:\n    duplicate_list.remove(fpi)\nduplicate_list","metadata":{"execution":{"iopub.status.busy":"2023-03-29T04:19:45.264749Z","iopub.execute_input":"2023-03-29T04:19:45.265212Z","iopub.status.idle":"2023-03-29T04:19:45.276876Z","shell.execute_reply.started":"2023-03-29T04:19:45.265171Z","shell.execute_reply":"2023-03-29T04:19:45.275288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Hashing and checking if there are duplicates\n## In birdclef-2023, there are no audio files that are exactly the same\n\n# def extract_start_feature(row):\n#     sig, sr = torchaudio.load(Config.input_base_path / Path(row.filename),frame_offset=0,num_frames=32000*5)\n#     return hash(sig.numpy().data.tobytes())\n\n# tqdm.pandas(desc='Hashing Audio')\n# meta_df[\"hash\"] = meta_df.progress_apply(lambda row: extract_start_feature(row),axis=1)\n# print(\"Finished hashing. Showing duplicate audios.\")\n\n# meta_df[meta_df.duplicated(keep=False,subset=['hash'])]","metadata":{"execution":{"iopub.status.busy":"2023-03-29T02:24:00.368331Z","iopub.execute_input":"2023-03-29T02:24:00.368942Z","iopub.status.idle":"2023-03-29T02:24:00.37522Z","shell.execute_reply.started":"2023-03-29T02:24:00.368902Z","shell.execute_reply":"2023-03-29T02:24:00.373903Z"},"trusted":true},"execution_count":null,"outputs":[]}]}