{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Before reading this Notebook, you can read my previous Notebook [[Fully Pipeline]|Stage 1: Training](https://www.kaggle.com/code/bibanh/0-71-fully-pipeline-resnet34-stage-1-training).","metadata":{}},{"cell_type":"markdown","source":"# 1. Imports","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torch.utils.data import DataLoader, Dataset, random_split\nimport torch.nn.functional as F\nimport torchaudio\nfrom torchaudio import transforms\nfrom IPython.display import Audio\nimport torchvision","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-10T01:31:41.087414Z","iopub.execute_input":"2023-03-10T01:31:41.087916Z","iopub.status.idle":"2023-03-10T01:31:45.527755Z","shell.execute_reply.started":"2023-03-10T01:31:41.087871Z","shell.execute_reply":"2023-03-10T01:31:45.525568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder, LabelEncoder\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import roc_auc_score\nimport numpy as np\nimport pandas as pd\nimport os\nimport glob\nimport math, random","metadata":{"execution":{"iopub.status.busy":"2023-03-10T01:31:45.529697Z","iopub.execute_input":"2023-03-10T01:31:45.531156Z","iopub.status.idle":"2023-03-10T01:31:46.466439Z","shell.execute_reply.started":"2023-03-10T01:31:45.531113Z","shell.execute_reply":"2023-03-10T01:31:46.464880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n\nseed = 1999\nseed_everything(seed)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T01:31:46.469407Z","iopub.execute_input":"2023-03-10T01:31:46.470442Z","iopub.status.idle":"2023-03-10T01:31:46.481845Z","shell.execute_reply.started":"2023-03-10T01:31:46.470371Z","shell.execute_reply":"2023-03-10T01:31:46.480780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    isOneHot = False\n    rate = 32000\n    num_classes = 264","metadata":{"execution":{"iopub.status.busy":"2023-03-10T01:31:46.485648Z","iopub.execute_input":"2023-03-10T01:31:46.486227Z","iopub.status.idle":"2023-03-10T01:31:46.495166Z","shell.execute_reply.started":"2023-03-10T01:31:46.486168Z","shell.execute_reply":"2023-03-10T01:31:46.493561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Define Audio Class to get Vector Embedding by MelSpectrogram","metadata":{}},{"cell_type":"markdown","source":"## 2.1 Audio Class with some necessary function","metadata":{}},{"cell_type":"code","source":"class AudioUtil():\n  @staticmethod\n  def open(audio_file):\n    sig, sr = torchaudio.load(audio_file)\n    return (sig, sr)\n\n  @staticmethod\n  def rechannel(aud, new_channel):\n    sig, sr = aud\n\n    if (sig.shape[0] == new_channel):\n      # Nothing to do\n      return aud\n\n    if (new_channel == 1):\n      # Convert from stereo to mono by selecting only the first channel\n      resig = sig[:1, :]\n    else:\n      # Convert from mono to stereo by duplicating the first channel\n      resig = torch.cat([sig, sig, sig])\n\n    return ((resig, sr))\n\n  @staticmethod\n  def resample(aud, newsr):\n    sig, sr = aud\n\n    if (sr == newsr):\n      # Nothing to do\n      return aud\n\n    num_channels = sig.shape[0]\n    # Resample first channel\n    resig = torchaudio.transforms.Resample(sr, newsr)(sig[:1,:])\n    if (num_channels > 1):\n      # Resample the second channel and merge both channels\n      retwo = torchaudio.transforms.Resample(sr, newsr)(sig[1:,:])\n      resig = torch.cat([resig, retwo])\n\n    return ((resig, newsr))\n\n  @staticmethod\n  def pad_trunc(aud, max_ms):\n    sig, sr = aud\n    num_rows, sig_len = sig.shape\n    max_len = sr//1000 * max_ms\n\n    if (sig_len > max_len):\n      # Truncate the signal to the given length\n      sig = sig[:,:max_len]\n\n    elif (sig_len < max_len):\n      # Length of padding to add at the beginning and end of the signal\n      pad_begin_len = random.randint(0, max_len - sig_len)\n      pad_end_len = max_len - sig_len - pad_begin_len\n\n      # Pad with 0s\n      pad_begin = torch.zeros((num_rows, pad_begin_len))\n      pad_end = torch.zeros((num_rows, pad_end_len))\n\n      sig = torch.cat((pad_begin, sig, pad_end), 1)\n      \n    return (sig, sr)\n\n  @staticmethod\n  def time_shift(aud, shift_limit):\n    sig,sr = aud\n    _, sig_len = sig.shape\n    shift_amt = int(random.random() * shift_limit * sig_len)\n    return (sig.roll(shift_amt), sr)\n\n  @staticmethod\n  def spectro_gram(aud, n_mels=64, n_fft=1024, hop_len=None):\n    sig,sr = aud\n    top_db = 80\n\n    # spec has shape [channel, n_mels, time], where channel is mono, stereo etc\n    spec = torchaudio.transforms.MelSpectrogram(sr, n_fft=n_fft, hop_length=hop_len, n_mels=n_mels)(sig)\n\n    # Convert to decibels\n    spec = torchaudio.transforms.AmplitudeToDB(top_db=top_db)(spec)\n    return (spec)\n\n  @staticmethod\n  def spectro_augment(spec, max_mask_pct=0.1, n_freq_masks=1, n_time_masks=1):\n    _, n_mels, n_steps = spec.shape\n    mask_value = spec.mean()\n    aug_spec = spec\n\n    freq_mask_param = max_mask_pct * n_mels\n    for _ in range(n_freq_masks):\n      aug_spec = torchaudio.transforms.FrequencyMasking(freq_mask_param)(aug_spec, mask_value)\n\n    time_mask_param = max_mask_pct * n_steps\n    for _ in range(n_time_masks):\n      aug_spec = torchaudio.transforms.TimeMasking(time_mask_param)(aug_spec, mask_value)\n\n    return aug_spec","metadata":{"execution":{"iopub.status.busy":"2023-03-10T01:31:46.497995Z","iopub.execute_input":"2023-03-10T01:31:46.498748Z","iopub.status.idle":"2023-03-10T01:31:46.522943Z","shell.execute_reply.started":"2023-03-10T01:31:46.498678Z","shell.execute_reply":"2023-03-10T01:31:46.521123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.2 Preprocessing Data Before Put into Model","metadata":{}},{"cell_type":"code","source":"def preprocessing(aud):\n    duration = 8000\n    sr = 32000\n    channel = 3\n    shift_pct = 0.4\n    reaud = AudioUtil.resample(aud, sr)\n    rechan = AudioUtil.rechannel(reaud, channel)\n    dur_aud = AudioUtil.pad_trunc(rechan, duration)\n    shift_aud = AudioUtil.time_shift(dur_aud, shift_pct)\n    sgram = AudioUtil.spectro_gram(shift_aud, n_mels=64, n_fft=1024, hop_len=None)\n    aug_sgram = AudioUtil.spectro_augment(sgram, max_mask_pct=0.1, n_freq_masks=2, n_time_masks=2)\n    aug_sgram_m, aug_sgram_s = aug_sgram.mean(), aug_sgram.std()\n    aug_sgram = (aug_sgram - aug_sgram_m) / aug_sgram_s\n    return aug_sgram","metadata":{"execution":{"iopub.status.busy":"2023-03-10T01:31:46.526458Z","iopub.execute_input":"2023-03-10T01:31:46.528716Z","iopub.status.idle":"2023-03-10T01:31:46.543568Z","shell.execute_reply.started":"2023-03-10T01:31:46.528666Z","shell.execute_reply":"2023-03-10T01:31:46.542147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.3 Test function with a sample","metadata":{}},{"cell_type":"code","source":"test_samples = list(glob.glob(\"/kaggle/input/birdclef-2023/test_soundscapes/*.ogg\"))\ntest_samples","metadata":{"execution":{"iopub.status.busy":"2023-03-10T01:31:46.545458Z","iopub.execute_input":"2023-03-10T01:31:46.546201Z","iopub.status.idle":"2023-03-10T01:31:46.567204Z","shell.execute_reply.started":"2023-03-10T01:31:46.546146Z","shell.execute_reply":"2023-03-10T01:31:46.565291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preprocessing(torchaudio.load(test_samples[0])).shape","metadata":{"execution":{"iopub.status.busy":"2023-03-10T01:31:46.569128Z","iopub.execute_input":"2023-03-10T01:31:46.569888Z","iopub.status.idle":"2023-03-10T01:31:47.927585Z","shell.execute_reply.started":"2023-03-10T01:31:46.569842Z","shell.execute_reply":"2023-03-10T01:31:47.925932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Load ResNet34 Model (CPU Only)","metadata":{}},{"cell_type":"code","source":"PATH = '/kaggle/input/birdclef-resnet34-20epochs/BirdSound_ResNet34_fold0_epoch19.pth'\nmodel = torchvision.models.resnet34(pretrained=False)\n\nnum_features = model.fc.in_features\nmodel.fc = nn.Linear(num_features, CFG.num_classes)\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\nmodel.load_state_dict(torch.load(PATH, map_location=torch.device('cpu') ))\nmyModel = model.to(device)\nnext(myModel.parameters()).device","metadata":{"execution":{"iopub.status.busy":"2023-03-10T01:31:47.929607Z","iopub.execute_input":"2023-03-10T01:31:47.930144Z","iopub.status.idle":"2023-03-10T01:31:49.304176Z","shell.execute_reply.started":"2023-03-10T01:31:47.930091Z","shell.execute_reply":"2023-03-10T01:31:49.302845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Mapping Classes to Column Name  ","metadata":{}},{"cell_type":"code","source":"meta_df = pd.read_csv('/kaggle/input/birdclef-2023/train_metadata.csv')\nprint('data shape:',meta_df.shape)\nmeta_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T01:31:49.308662Z","iopub.execute_input":"2023-03-10T01:31:49.309066Z","iopub.status.idle":"2023-03-10T01:31:49.471691Z","shell.execute_reply.started":"2023-03-10T01:31:49.309028Z","shell.execute_reply":"2023-03-10T01:31:49.470255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"le = LabelEncoder().fit(meta_df['primary_label'])\ncompetition_classes = le.classes_\ncompetition_classes","metadata":{"execution":{"iopub.status.busy":"2023-03-10T01:31:49.473117Z","iopub.execute_input":"2023-03-10T01:31:49.473517Z","iopub.status.idle":"2023-03-10T01:31:49.489532Z","shell.execute_reply.started":"2023-03-10T01:31:49.473481Z","shell.execute_reply":"2023-03-10T01:31:49.487627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub = pd.read_csv(\"/kaggle/input/birdclef-2023/sample_submission.csv\")\nsample_sub[competition_classes] = sample_sub[competition_classes].astype(np.float32)\nsample_sub.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-10T01:31:49.500193Z","iopub.execute_input":"2023-03-10T01:31:49.500550Z","iopub.status.idle":"2023-03-10T01:31:49.618647Z","shell.execute_reply.started":"2023-03-10T01:31:49.500517Z","shell.execute_reply":"2023-03-10T01:31:49.617345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for idx, col in enumerate(sample_sub.drop(columns = ['row_id'], axis = 1).columns):\n    if col != competition_classes[idx]:\n        print('Not fit class!')\nprint('If dont have any log, all fit')","metadata":{"execution":{"iopub.status.busy":"2023-03-10T01:31:49.620146Z","iopub.execute_input":"2023-03-10T01:31:49.620542Z","iopub.status.idle":"2023-03-10T01:31:49.633948Z","shell.execute_reply.started":"2023-03-10T01:31:49.620504Z","shell.execute_reply":"2023-03-10T01:31:49.632260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. Make Predictions","metadata":{}},{"cell_type":"markdown","source":"## 5.1 Function to predict each data sample in test set","metadata":{}},{"cell_type":"code","source":"def predict_for_sample(filename, sample_submission):\n    file_id = filename.split(\".ogg\")[0].split(\"/\")[-1]\n    lstFile = list(sample_submission[sample_submission.row_id.str.contains( file_id + \"_\")].row_id.unique())\n    data, rat = torchaudio.load(filename)\n    for i in range(1):\n        for file in lstFile:\n            end = int(file.split('_')[-1])\n            start = end - 5\n            chunk = (data[:, start*rat:end*rat - 1], rat)\n            samples = preprocessing(chunk).unsqueeze(0)\n            probabilities = myModel(samples)\n            probabilities = torch.nn.functional.softmax(probabilities).detach().numpy()\n\n            # set the appropriate row in the sample submission\n            sample_submission.loc[sample_submission.row_id == file, competition_classes] = probabilities","metadata":{"execution":{"iopub.status.busy":"2023-03-10T01:31:49.635785Z","iopub.execute_input":"2023-03-10T01:31:49.636219Z","iopub.status.idle":"2023-03-10T01:31:49.650249Z","shell.execute_reply.started":"2023-03-10T01:31:49.636180Z","shell.execute_reply":"2023-03-10T01:31:49.648608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5.2 PUT IT ALL TOGETHER","metadata":{}},{"cell_type":"code","source":"%%time\nfor sample_filename in test_samples:\n    predict_for_sample(sample_filename, sample_sub)\nsample_sub","metadata":{"execution":{"iopub.status.busy":"2023-03-10T01:31:49.652384Z","iopub.execute_input":"2023-03-10T01:31:49.652751Z","iopub.status.idle":"2023-03-10T01:31:51.012748Z","shell.execute_reply.started":"2023-03-10T01:31:49.652717Z","shell.execute_reply":"2023-03-10T01:31:51.011617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub.to_csv(\"submission.csv\", index=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-10T01:31:51.014134Z","iopub.execute_input":"2023-03-10T01:31:51.014767Z","iopub.status.idle":"2023-03-10T01:31:51.028994Z","shell.execute_reply.started":"2023-03-10T01:31:51.014727Z","shell.execute_reply":"2023-03-10T01:31:51.027577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}