{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h1 style=\"color: #6cb4e4;  text-align: center;  padding: 0.25em;  border-top: solid 2.5px #6cb4e4;  border-bottom: solid 2.5px #6cb4e4;  background: -webkit-repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);  background: repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);height:45px;\">\n<b>Inf Kernel</b></h1> ","metadata":{}},{"cell_type":"markdown","source":"* based [Simayi Kernel(Inf&Train)](https://www.kaggle.com/code/bibanh/lb-0-72-resnet34-melspectrogram-stage-2-inference)\n* Changed timm model","metadata":{}},{"cell_type":"markdown","source":"<a id=#cbb></a>\n<h2 style=\"color: #6cb4e4; background: #dfefff;  box-shadow: 0px 0px 0px 5px #dfefff;  border: dashed 4px white;  padding: 0.2em 0.5em;\">\n<b>Libs Timm</b></h2> \n","metadata":{}},{"cell_type":"code","source":"import sys\nsys.path.append('/kaggle/input/bird-lib-20230309095737/')\nimport timm","metadata":{"execution":{"iopub.status.busy":"2023-03-12T10:35:19.526193Z","iopub.execute_input":"2023-03-12T10:35:19.526841Z","iopub.status.idle":"2023-03-12T10:35:23.686946Z","shell.execute_reply.started":"2023-03-12T10:35:19.526803Z","shell.execute_reply":"2023-03-12T10:35:23.685464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=#cbb></a>\n<h2 style=\"color: #6cb4e4; background: #dfefff;  box-shadow: 0px 0px 0px 5px #dfefff;  border: dashed 4px white;  padding: 0.2em 0.5em;\">\n<b>Libs</b></h2> \n","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torch.utils.data import DataLoader, Dataset, random_split\nimport torch.nn.functional as F\nimport torchaudio\nfrom torchaudio import transforms\nfrom IPython.display import Audio\nimport torchvision\nfrom sklearn.preprocessing import OneHotEncoder, LabelEncoder\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import roc_auc_score\nimport numpy as np\nimport pandas as pd\nimport os\nimport glob\nimport math, random","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-12T10:35:23.689279Z","iopub.execute_input":"2023-03-12T10:35:23.689933Z","iopub.status.idle":"2023-03-12T10:35:25.039898Z","shell.execute_reply.started":"2023-03-12T10:35:23.689888Z","shell.execute_reply":"2023-03-12T10:35:25.038762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n\nseed = 42\nseed_everything(seed)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T10:35:25.041502Z","iopub.execute_input":"2023-03-12T10:35:25.042551Z","iopub.status.idle":"2023-03-12T10:35:25.053005Z","shell.execute_reply.started":"2023-03-12T10:35:25.042510Z","shell.execute_reply":"2023-03-12T10:35:25.051823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=#cbb></a>\n<h2 style=\"color: #6cb4e4; background: #dfefff;  box-shadow: 0px 0px 0px 5px #dfefff;  border: dashed 4px white;  padding: 0.2em 0.5em;\">\n<b>CFG</b></h2> \n","metadata":{}},{"cell_type":"code","source":"class CFG:\n    isOneHot = False\n    rate = 32000\n    num_classes = 264","metadata":{"execution":{"iopub.status.busy":"2023-03-12T10:35:25.055910Z","iopub.execute_input":"2023-03-12T10:35:25.056350Z","iopub.status.idle":"2023-03-12T10:35:25.071259Z","shell.execute_reply.started":"2023-03-12T10:35:25.056315Z","shell.execute_reply":"2023-03-12T10:35:25.069981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"color: #6cb4e4;  text-align: center;  padding: 0.25em;  border-top: solid 2.5px #6cb4e4;  border-bottom: solid 2.5px #6cb4e4;  background: -webkit-repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);  background: repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);height:45px;\">\n<b>Preprocess</b></h1> ","metadata":{}},{"cell_type":"code","source":"class AudioUtil():\n  @staticmethod\n  def open(audio_file):\n    sig, sr = torchaudio.load(audio_file)\n    return (sig, sr)\n\n  @staticmethod\n  def rechannel(aud, new_channel):\n    sig, sr = aud\n\n    if (sig.shape[0] == new_channel):\n      # Nothing to do\n      return aud\n\n    if (new_channel == 1):\n      # Convert from stereo to mono by selecting only the first channel\n      resig = sig[:1, :]\n    else:\n      # Convert from mono to stereo by duplicating the first channel\n      resig = torch.cat([sig, sig, sig])\n\n    return ((resig, sr))\n\n  @staticmethod\n  def resample(aud, newsr):\n    sig, sr = aud\n\n    if (sr == newsr):\n      # Nothing to do\n      return aud\n\n    num_channels = sig.shape[0]\n    # Resample first channel\n    resig = torchaudio.transforms.Resample(sr, newsr)(sig[:1,:])\n    if (num_channels > 1):\n      # Resample the second channel and merge both channels\n      retwo = torchaudio.transforms.Resample(sr, newsr)(sig[1:,:])\n      resig = torch.cat([resig, retwo])\n\n    return ((resig, newsr))\n\n  @staticmethod\n  def pad_trunc(aud, max_ms):\n    sig, sr = aud\n    num_rows, sig_len = sig.shape\n    max_len = sr//1000 * max_ms\n\n    if (sig_len > max_len):\n      # Truncate the signal to the given length\n      sig = sig[:,:max_len]\n\n    elif (sig_len < max_len):\n      # Length of padding to add at the beginning and end of the signal\n      pad_begin_len = random.randint(0, max_len - sig_len)\n      pad_end_len = max_len - sig_len - pad_begin_len\n\n      # Pad with 0s\n      pad_begin = torch.zeros((num_rows, pad_begin_len))\n      pad_end = torch.zeros((num_rows, pad_end_len))\n\n      sig = torch.cat((pad_begin, sig, pad_end), 1)\n      \n    return (sig, sr)\n\n  @staticmethod\n  def time_shift(aud, shift_limit):\n    sig,sr = aud\n    _, sig_len = sig.shape\n    shift_amt = int(random.random() * shift_limit * sig_len)\n    return (sig.roll(shift_amt), sr)\n\n  @staticmethod\n  def spectro_gram(aud, n_mels=64, n_fft=1024, hop_len=None):\n    sig,sr = aud\n    top_db = 80\n\n    # spec has shape [channel, n_mels, time], where channel is mono, stereo etc\n    spec = torchaudio.transforms.MelSpectrogram(sr, n_fft=n_fft, hop_length=hop_len, n_mels=n_mels)(sig)\n\n    # Convert to decibels\n    spec = torchaudio.transforms.AmplitudeToDB(top_db=top_db)(spec)\n    return (spec)\n\n  @staticmethod\n  def spectro_augment(spec, max_mask_pct=0.1, n_freq_masks=1, n_time_masks=1):\n    _, n_mels, n_steps = spec.shape\n    mask_value = spec.mean()\n    aug_spec = spec\n\n    freq_mask_param = max_mask_pct * n_mels\n    for _ in range(n_freq_masks):\n      aug_spec = torchaudio.transforms.FrequencyMasking(freq_mask_param)(aug_spec, mask_value)\n\n    time_mask_param = max_mask_pct * n_steps\n    for _ in range(n_time_masks):\n      aug_spec = torchaudio.transforms.TimeMasking(time_mask_param)(aug_spec, mask_value)\n\n    return aug_spec","metadata":{"execution":{"iopub.status.busy":"2023-03-12T10:35:25.073088Z","iopub.execute_input":"2023-03-12T10:35:25.073494Z","iopub.status.idle":"2023-03-12T10:35:25.093320Z","shell.execute_reply.started":"2023-03-12T10:35:25.073434Z","shell.execute_reply":"2023-03-12T10:35:25.092094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocessing(aud):\n    duration = 8000\n    sr = 32000\n    channel = 3\n    shift_pct = 0.4\n    reaud = AudioUtil.resample(aud, sr)\n    rechan = AudioUtil.rechannel(reaud, channel)\n    dur_aud = AudioUtil.pad_trunc(rechan, duration)\n    shift_aud = AudioUtil.time_shift(dur_aud, shift_pct)\n    sgram = AudioUtil.spectro_gram(shift_aud, n_mels=64, n_fft=1024, hop_len=None)\n    aug_sgram = AudioUtil.spectro_augment(sgram, max_mask_pct=0.1, n_freq_masks=2, n_time_masks=2)\n    aug_sgram_m, aug_sgram_s = aug_sgram.mean(), aug_sgram.std()\n    aug_sgram = (aug_sgram - aug_sgram_m) / aug_sgram_s\n    return aug_sgram","metadata":{"execution":{"iopub.status.busy":"2023-03-12T10:35:25.095276Z","iopub.execute_input":"2023-03-12T10:35:25.095698Z","iopub.status.idle":"2023-03-12T10:35:25.109890Z","shell.execute_reply.started":"2023-03-12T10:35:25.095647Z","shell.execute_reply":"2023-03-12T10:35:25.108707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_samples = list(glob.glob(\"/kaggle/input/birdclef-2023/test_soundscapes/*.ogg\"))\ntest_samples","metadata":{"execution":{"iopub.status.busy":"2023-03-12T10:35:25.111771Z","iopub.execute_input":"2023-03-12T10:35:25.112098Z","iopub.status.idle":"2023-03-12T10:35:25.130557Z","shell.execute_reply.started":"2023-03-12T10:35:25.112069Z","shell.execute_reply":"2023-03-12T10:35:25.129221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preprocessing(torchaudio.load(test_samples[0])).shape","metadata":{"execution":{"iopub.status.busy":"2023-03-12T10:35:25.132478Z","iopub.execute_input":"2023-03-12T10:35:25.132947Z","iopub.status.idle":"2023-03-12T10:35:26.425073Z","shell.execute_reply.started":"2023-03-12T10:35:25.132900Z","shell.execute_reply":"2023-03-12T10:35:26.423967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_df = pd.read_csv('/kaggle/input/birdclef-2023/train_metadata.csv')\nprint('data shape:',meta_df.shape)\nmeta_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T10:35:26.426444Z","iopub.execute_input":"2023-03-12T10:35:26.426815Z","iopub.status.idle":"2023-03-12T10:35:26.568582Z","shell.execute_reply.started":"2023-03-12T10:35:26.426779Z","shell.execute_reply":"2023-03-12T10:35:26.567498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"le = LabelEncoder().fit(meta_df['primary_label'])\ncompetition_classes = le.classes_","metadata":{"execution":{"iopub.status.busy":"2023-03-12T10:35:26.572516Z","iopub.execute_input":"2023-03-12T10:35:26.572868Z","iopub.status.idle":"2023-03-12T10:35:26.582710Z","shell.execute_reply.started":"2023-03-12T10:35:26.572837Z","shell.execute_reply":"2023-03-12T10:35:26.581328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"forced_defaults = 0\ncompetition_class_map = []\nfor c in competition_classes:\n    try:\n        i = classes.index(c)\n        competition_class_map.append(i)\n    except:\n        competition_class_map.append(0)\n        forced_defaults += 1","metadata":{"execution":{"iopub.status.busy":"2023-03-12T10:35:26.583996Z","iopub.execute_input":"2023-03-12T10:35:26.584839Z","iopub.status.idle":"2023-03-12T10:35:26.593342Z","shell.execute_reply.started":"2023-03-12T10:35:26.584802Z","shell.execute_reply":"2023-03-12T10:35:26.592274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub = pd.read_csv(\"/kaggle/input/birdclef-2023/sample_submission.csv\")\nsample_sub[competition_classes] = sample_sub[competition_classes].astype(np.float32)\nsample_sub.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-12T10:35:26.595001Z","iopub.execute_input":"2023-03-12T10:35:26.595323Z","iopub.status.idle":"2023-03-12T10:35:26.706120Z","shell.execute_reply.started":"2023-03-12T10:35:26.595293Z","shell.execute_reply":"2023-03-12T10:35:26.704981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for idx, col in enumerate(sample_sub.drop(columns = ['row_id'], axis = 1).columns):\n    if col != competition_classes[idx]:\n        print('Not fit class!')\nprint('If dont have any log, all fit')","metadata":{"execution":{"iopub.status.busy":"2023-03-12T10:35:26.707721Z","iopub.execute_input":"2023-03-12T10:35:26.708050Z","iopub.status.idle":"2023-03-12T10:35:26.718200Z","shell.execute_reply.started":"2023-03-12T10:35:26.708019Z","shell.execute_reply":"2023-03-12T10:35:26.716909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"color: #6cb4e4;  text-align: center;  padding: 0.25em;  border-top: solid 2.5px #6cb4e4;  border-bottom: solid 2.5px #6cb4e4;  background: -webkit-repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);  background: repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);height:45px;\">\n<b>Inference</b></h1> ","metadata":{}},{"cell_type":"markdown","source":"<a id=#cbb></a>\n<h2 style=\"color: #6cb4e4; background: #dfefff;  box-shadow: 0px 0px 0px 5px #dfefff;  border: dashed 4px white;  padding: 0.2em 0.5em;\">\n<b>Model Load & Selection</b></h2> \n","metadata":{}},{"cell_type":"code","source":"import glob\nmodel_file_list = glob.glob(\"/kaggle/input/bird-train-a-121-20230309095332/*\")\nmodel_file_list","metadata":{"execution":{"iopub.status.busy":"2023-03-12T10:35:26.719912Z","iopub.execute_input":"2023-03-12T10:35:26.720236Z","iopub.status.idle":"2023-03-12T10:35:26.737780Z","shell.execute_reply.started":"2023-03-12T10:35:26.720206Z","shell.execute_reply":"2023-03-12T10:35:26.736838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_file_list = [\n '/kaggle/input/bird-train-a-121-20230309095332/model_fold3_epoch9.pth',\n '/kaggle/input/bird-train-a-121-20230309095332/model_fold2_epoch9.pth',\n#  '/kaggle/input/bird-train-a-121-20230309095332/model_fold4_epoch9.pth',\n#  '/kaggle/input/bird-train-a-121-20230309095332/model_fold0_epoch9.pth',\n#  '/kaggle/input/bird-train-a-121-20230309095332/model_fold1_epoch9.pth',\n]\nmodel_file_list","metadata":{"execution":{"iopub.status.busy":"2023-03-12T10:35:26.739282Z","iopub.execute_input":"2023-03-12T10:35:26.739618Z","iopub.status.idle":"2023-03-12T10:35:26.747236Z","shell.execute_reply.started":"2023-03-12T10:35:26.739567Z","shell.execute_reply":"2023-03-12T10:35:26.745882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_list = []\nfor _model_path in model_file_list:\n    model = timm.create_model(\"tf_efficientnetv2_s\", \n                  pretrained=False, \n                  num_classes=264)\n    model.load_state_dict(torch.load(_model_path, map_location=torch.device('cpu')))\n    model_list.append(model)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T10:35:26.748500Z","iopub.execute_input":"2023-03-12T10:35:26.748826Z","iopub.status.idle":"2023-03-12T10:35:32.821112Z","shell.execute_reply.started":"2023-03-12T10:35:26.748795Z","shell.execute_reply":"2023-03-12T10:35:32.820097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=#cbb></a>\n<h2 style=\"color: #6cb4e4; background: #dfefff;  box-shadow: 0px 0px 0px 5px #dfefff;  border: dashed 4px white;  padding: 0.2em 0.5em;\">\n<b>Pred</b></h2> \n","metadata":{}},{"cell_type":"code","source":"def predict_for_sample(filename, sample_submission):\n    file_id = filename.split(\".ogg\")[0].split(\"/\")[-1]\n    lstFile = list(sample_submission[sample_submission.row_id.str.contains( file_id + \"_\")].row_id.unique())\n    data, rat = torchaudio.load(filename)\n    for file in lstFile:\n        end = int(file.split('_')[-1])\n        start = end - 5\n        chunk = (data[:, start*rat:end*rat - 1], rat)\n        samples = preprocessing(chunk).unsqueeze(0)\n        \n        \"\"\" Iter Inf Fold \"\"\"\n        proba_list = []\n        for _model in model_list:\n            device = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\n            myModel = _model.to(device)\n            myModel = myModel.eval()\n            probabilities = myModel(samples)\n            probabilities = torch.nn.functional.softmax(probabilities).detach().numpy()\n            proba_list.append(probabilities)\n        \n        \"\"\" Merge \"\"\"\n        probabilities = [sum(column)/len(model_file_list) for column in zip(*proba_list)]\n        sample_submission.loc[sample_submission.row_id == file, competition_classes] = probabilities","metadata":{"execution":{"iopub.status.busy":"2023-03-12T10:35:32.822550Z","iopub.execute_input":"2023-03-12T10:35:32.823335Z","iopub.status.idle":"2023-03-12T10:35:32.832761Z","shell.execute_reply.started":"2023-03-12T10:35:32.823300Z","shell.execute_reply":"2023-03-12T10:35:32.831893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=#cbb></a>\n<h2 style=\"color: #6cb4e4; background: #dfefff;  box-shadow: 0px 0px 0px 5px #dfefff;  border: dashed 4px white;  padding: 0.2em 0.5em;\">\n<b>CreateSub</b></h2> \n","metadata":{}},{"cell_type":"code","source":"for sample_filename in test_samples:\n    predict_for_sample(sample_filename, sample_sub)\nsample_sub","metadata":{"execution":{"iopub.status.busy":"2023-03-12T10:35:32.833974Z","iopub.execute_input":"2023-03-12T10:35:32.834542Z","iopub.status.idle":"2023-03-12T10:35:35.009315Z","shell.execute_reply.started":"2023-03-12T10:35:32.834509Z","shell.execute_reply":"2023-03-12T10:35:35.008078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub.to_csv(\"submission.csv\", index=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-12T10:35:35.010757Z","iopub.execute_input":"2023-03-12T10:35:35.011110Z","iopub.status.idle":"2023-03-12T10:35:35.023221Z","shell.execute_reply.started":"2023-03-12T10:35:35.011078Z","shell.execute_reply":"2023-03-12T10:35:35.022102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}