{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\n\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport tensorflow as tf\n\n\nimport librosa.display\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nimport IPython.display as ipd","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Check files format"},{"metadata":{"trusted":true},"cell_type":"code","source":"BASE_INPUT_DIR = '/kaggle/input/rfcx-species-audio-detection/'\nTRAIN_INPUT_DIR = os.path.join(BASE_INPUT_DIR, 'train')\nTEST_INPUT_DIR = os.path.join(BASE_INPUT_DIR, 'test')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train_tp = pd.read_csv(os.path.join(BASE_INPUT_DIR, 'train_tp.csv'))\ntrain_fp = pd.read_csv(os.path.join(BASE_INPUT_DIR, 'train_tp.csv'))\nsubmission = pd.read_csv(os.path.join(BASE_INPUT_DIR, 'sample_submission.csv'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_files = os.listdir(TRAIN_INPUT_DIR)\ntrain_files = [os.path.join(TRAIN_INPUT_DIR, f) for f in train_files]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"FMIN = 40.0\nFMAX = 24000.0\n\nSR = 48000\nN_MELS = 224\n\nIMG_SIZE = (224, 512)\nIMG_HEIGHT = IMG_SIZE[0]\nIMG_WIDTH = IMG_SIZE[1]\nFRAME_MAX = 5626\n\nSEGMENT_DURATION = 10","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_tp.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_fp.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"SPECIES_ID = [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Summary"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_tp.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_tp.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_tp.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_fp.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_fp.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_files = os.listdir('/kaggle/input/rfcx-species-audio-detection/test/')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(test_files)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Time Window"},{"metadata":{"trusted":true},"cell_type":"code","source":"data, _ = librosa.load(train_files[0], sr=SR)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(data)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ipd.Audio(data, rate=sr)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_file_info = []\nfor train_file in train_files:\n    data, sr = librosa.load(train_file, sr=None)\n    train_file_info.append((len(data), sr))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(train_file_info)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_tp['recording_id'].nunique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_info = np.array(train_file_info)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(np.unique(train_info[:, 0]))\nprint(np.unique(train_info[:, 1]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(data)/sr","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_tp_tdiff = train_tp ['t_max'] - train_tp['t_min']\nprint(train_tp_tdiff.describe())\ntrain_tp_tdiff.hist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_fp_tdiff = train_fp ['t_max'] - train_fp['t_min']\nprint(train_fp_tdiff.describe())\ntrain_fp_tdiff.hist()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Overlapping time windows"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_tp_rec_grps = train_tp.sort_values(['recording_id', 't_min', 't_max']).groupby('recording_id')\n\nfor name, grp in train_tp_rec_grps:\n    prev_window = None\n    \n    for _, row in grp.iterrows():\n        if prev_window is None:\n            prev_window = (row['t_min'], row['t_max'])\n        else:\n            if row['t_min'] < prev_window[0]:\n                cur_window = (row['t_min'], row['t_max'])\n                print(f'Overlap in {name}: {prev_window} and {cur_window}')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Note: No overlaps in TP**"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_fp_rec_grps = train_fp.sort_values(['recording_id', 't_min', 't_max']).groupby('recording_id')\n\nfor name, grp in train_fp_rec_grps:\n    prev_window = None\n    \n    for _, row in grp.iterrows():\n        if prev_window is None:\n            prev_window = (row['t_min'], row['t_max'])\n        else:\n            if row['t_min'] < prev_window[0]:\n                cur_window = (row['t_min'], row['t_max'])\n                print(f'Overlap in {name}: {prev_window} and {cur_window}')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Note: No overlaps in FP**"},{"metadata":{},"cell_type":"markdown","source":"# Univariate"},{"metadata":{"trusted":true},"cell_type":"code","source":"np.sort(train_tp.species_id.unique())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"np.sort(train_fp.species_id.unique())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_tp.groupby('species_id').size().plot(kind='bar')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tp_obs_cnts = train_tp.groupby(['species_id', 'songtype_id']).size().reset_index()\ntp_obs_cnts = tp_obs_cnts.rename(columns={0: 'obs'})\n\nplt.figure(figsize=(10, 7))\nsns.barplot(x='species_id', y='obs', hue='songtype_id', data=tp_obs_cnts)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_fp.groupby('species_id').size().plot(kind='bar')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fp_obs_cnts = train_fp.groupby(['species_id', 'songtype_id']).size().reset_index()\nfp_obs_cnts = fp_obs_cnts.rename(columns={0: 'obs'})\n\nplt.figure(figsize=(10, 7))\nsns.barplot(x='species_id', y='obs', hue='songtype_id', data=fp_obs_cnts)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Spectrogram"},{"metadata":{"trusted":true},"cell_type":"code","source":"data, sr = librosa.load(train_files[0], sr=None)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"mel_spec = librosa.power_to_db(librosa.feature.melspectrogram(data, sr=sr, n_mels=256, fmin=F_MIN, fmax=F_MAX))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"mel_spec.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"librosa.display.specshow(mel_spec, x_axis='time', y_axis='mel', sr=sr, fmin=F_MIN, fmax=F_MAX)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"librosa.display.waveplot(data, sr=sr)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"stft = librosa.stft(data, hop_length=512)\nstft = librosa.power_to_db(np.abs(stft))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"librosa.display.specshow(stft, x_axis='time', y_axis='log', sr=sr, fmin=F_MIN, fmax=F_MAX)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Spectrogram with targets"},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_audio(recording_id, train=True):\n    filepath = os.path.join(TRAIN_INPUT_DIR if train else TEST_INPUT_DIR, recording_id + '.flac')\n    data, _ = librosa.load(filepath, sr=SR)\n    return data\n\n\ndef cut_audio(audio_data, tmin, tmax, sr=SR, segment_duration=SEGMENT_DURATION):\n    clip_duration = len(audio_data)/sr\n    extra_time = max(0, segment_duration - (tmax - tmin)) / 2\n    tmin = max(0, tmin - extra_time)\n    tmax = min(clip_duration, tmax + extra_time)\n    \n    min_sample, max_sample = librosa.time_to_samples([tmin, tmax], sr=sr)\n    return audio_data[min_sample:(max_sample + 1)]\n    \n\ndef get_mel_spec_img(audio_data):\n    mel_spec = librosa.feature.melspectrogram(audio_data, sr=SR, n_mels=N_MELS)\n    log_mel_spec = librosa.power_to_db(mel_spec)\n    img = tf.expand_dims(log_mel_spec, -1)\n    img = tf.image.resize(img, IMG_SIZE)\n    img = tf.image.per_image_standardization(img)\n    return img, log_mel_spec\n\n\ndef get_displayable_img(spec_img):\n    img_min = np.min(spec_img)\n    img_max = np.max(spec_img)\n    img = (spec_img - img_min)/(img_max - img_min)\n    return np.stack([np.squeeze(img.numpy())]*3, axis=-1)\n\n\ndef freq_to_mel_bin(freqs, n_mels=N_MELS, fmin=FMIN, fmax=FMAX):\n    min_mel = librosa.hz_to_mel(fmin)\n    max_mel = librosa.hz_to_mel(fmax)\n    mel_step = (max_mel - min_mel)/n_mels\n    mel_freqs = librosa.hz_to_mel(freqs)\n    return [int(np.floor(f/mel_step)) for f in mel_freqs]\n\n\ndef time_to_img_bin(times):\n    times = librosa.time_to_frames(times, sr=SR)*IMG_WIDTH/FRAME_MAX\n    return [int(np.floor(t)) for t in times]\n\n\ndef show_spectrogram(sample, ax, is_tp=True, showlabel=False):\n    audio_data = load_audio(sample['recording_id'])\n    _, mel_spec = get_mel_spec_img(audio_data)\n    librosa.display.specshow(mel_spec, x_axis='time', y_axis='mel', sr=SR, fmin=FMIN, fmax=FMAX)\n    ax.set(title=f'Mel-frequency spectrogram of {sample[\"recording_id\"]}')\n\n    sid, fmin, fmax, tmin, tmax = (sample[\"species_id\"], sample[\"f_min\"], sample[\"f_max\"], sample[\"t_min\"], sample[\"t_max\"])\n    ec = '#00ff00' if is_tp == 1 else '#0000ff'\n    ax.add_patch(\n        patches.Rectangle(xy=(tmin, fmin), width=tmax-tmin, height=fmax-fmin, ec=ec, fill=False)\n    )\n\n    if showlabel:\n        ax.text(tmin, fmax, \n        f\"{sid} {'tp' if is_tp else 'fp'}\",\n        horizontalalignment='left', verticalalignment='bottom', color=ec, fontsize=16)\n\n    \ndef show_cut_spec_img(sample, ax, cut_duration=10):\n    audio_data = load_audio(sample['recording_id'])\n    sid, fmin, fmax, tmin, tmax = (sample[\"species_id\"], sample[\"f_min\"], sample[\"f_max\"], sample[\"t_min\"], sample[\"t_max\"])\n    print(sample)\n    \n    audio_data = cut_audio(audio_data, tmin, tmax)\n    img, _ = get_mel_spec_img(audio_data)\n    ax.imshow(get_displayable_img(img))\n\n    fmin, fmax = freq_to_mel_bin([fmin, fmax])\n    ax.axhline(fmin, color='r', lw=0.5)\n    ax.axhline(fmax, color='g', lw=0.5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(train_tp.loc[2, :])\nfig, ax = plt.subplots(figsize=(15, 3))\nshow_spectrogram(train_tp.loc[2, :], ax, is_tp=True, showlabel=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots()\nshow_cut_spec_img(train_tp.loc[2, :], ax)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}