{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Birdclef2022 simple starter code error analysis\n\n- This notebook includes some error analysis methods and insight from it.\n\n- This notebook is based on inference results on validation set  from [\"PyTorch Simple Starter Using only 21 classes\"](https://www.kaggle.com/code/myso1987/pytorch-simple-starter-using-only-21-classes) \n- Similarly, the model also used the code from the link above.\n\n","metadata":{}},{"cell_type":"code","source":"import torch\nimport torchaudio\nimport IPython.display as ipd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom matplotlib.patches import Rectangle\nimport pandas as pd\nimport ipywidgets as widgets\nimport json\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom tqdm import tqdm\nfrom pathlib import Path","metadata":{"execution":{"iopub.status.busy":"2022-04-27T12:43:27.926748Z","iopub.execute_input":"2022-04-27T12:43:27.927207Z","iopub.status.idle":"2022-04-27T12:43:30.457081Z","shell.execute_reply.started":"2022-04-27T12:43:27.927121Z","shell.execute_reply":"2022-04-27T12:43:30.455993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bird_dir = Path(\"../input/birdclef-2022/\")\ncsv_file = Path(\"../input/birdclef2022-val-analysis/val_infer.csv\")\ndf = pd.read_csv(csv_file)","metadata":{"execution":{"iopub.status.busy":"2022-04-27T12:43:30.459045Z","iopub.execute_input":"2022-04-27T12:43:30.459303Z","iopub.status.idle":"2022-04-27T12:43:30.489740Z","shell.execute_reply.started":"2022-04-27T12:43:30.459273Z","shell.execute_reply":"2022-04-27T12:43:30.489113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"`csv_file` is the output of using the validation set and model of the above code.\n\nWe will manipulate the `pd.Dataframe` in a \"multi hot encoding\" format to easily analyze the results.","metadata":{}},{"cell_type":"code","source":"df.columns =['filename','TF']\n#add a True/False column.","metadata":{"execution":{"iopub.status.busy":"2022-04-27T12:43:30.490682Z","iopub.execute_input":"2022-04-27T12:43:30.491389Z","iopub.status.idle":"2022-04-27T12:43:30.495065Z","shell.execute_reply.started":"2022-04-27T12:43:30.491344Z","shell.execute_reply":"2022-04-27T12:43:30.494225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The vocab of the scored bird to be classified in 21 species","metadata":{}},{"cell_type":"code","source":"#21\nvocab = np.array(['akiapo', 'aniani', 'apapan', 'barpet', 'crehon', 'elepai', 'ercfra', 'hawama',\n                   'hawcre', 'hawgoo', 'hawhaw', 'hawpet1', 'houfin', 'iiwi', 'jabwar', 'maupar',\n                   'omao', 'puaioh', 'skylar', 'warwhe1', 'yefcan'])","metadata":{"execution":{"iopub.status.busy":"2022-04-27T12:43:30.496887Z","iopub.execute_input":"2022-04-27T12:43:30.497682Z","iopub.status.idle":"2022-04-27T12:43:30.506647Z","shell.execute_reply.started":"2022-04-27T12:43:30.497648Z","shell.execute_reply":"2022-04-27T12:43:30.506034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"class\"] = df.filename.apply(lambda x: x.split(\"_\")[-1])\ndf['filename']=df.filename.apply(lambda x: \"_\".join(x.split(\"_\")[:-1]))","metadata":{"execution":{"iopub.status.busy":"2022-04-27T12:43:30.507906Z","iopub.execute_input":"2022-04-27T12:43:30.508341Z","iopub.status.idle":"2022-04-27T12:43:30.540359Z","shell.execute_reply.started":"2022-04-27T12:43:30.508304Z","shell.execute_reply":"2022-04-27T12:43:30.539617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp = {}\n\nfor idx, x in df.iterrows()    :    \n    if x.filename not in tmp.keys():\n        tmp[x.filename]={\"classes\":[]}\n    else :\n        if x.TF ==True :\n            tmp[x.filename][\"classes\"].append(x[\"class\"])      \n\nmulticlass = pd.DataFrame.from_dict(tmp,orient=\"index\")            \n\ndel(tmp)   \n    ","metadata":{"execution":{"iopub.status.busy":"2022-04-27T12:43:30.542052Z","iopub.execute_input":"2022-04-27T12:43:30.542653Z","iopub.status.idle":"2022-04-27T12:43:30.984270Z","shell.execute_reply.started":"2022-04-27T12:43:30.542607Z","shell.execute_reply":"2022-04-27T12:43:30.983456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"multiclass.index.name=\"filename\"","metadata":{"execution":{"iopub.status.busy":"2022-04-27T12:43:30.985219Z","iopub.execute_input":"2022-04-27T12:43:30.985441Z","iopub.status.idle":"2022-04-27T12:43:30.990039Z","shell.execute_reply.started":"2022-04-27T12:43:30.985414Z","shell.execute_reply":"2022-04-27T12:43:30.989296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mlb = MultiLabelBinarizer()\nmlb.classes = vocab\ny_data = mlb.fit_transform(multiclass['classes'])","metadata":{"execution":{"iopub.status.busy":"2022-04-27T12:43:30.991349Z","iopub.execute_input":"2022-04-27T12:43:30.991633Z","iopub.status.idle":"2022-04-27T12:43:31.003504Z","shell.execute_reply.started":"2022-04-27T12:43:30.991594Z","shell.execute_reply":"2022-04-27T12:43:31.002647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## multi hot encoded result","metadata":{}},{"cell_type":"code","source":"multi_hot = pd.DataFrame(y_data,columns=mlb.classes_)\nmulti_hot.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-04-27T12:43:31.004680Z","iopub.execute_input":"2022-04-27T12:43:31.005486Z","iopub.status.idle":"2022-04-27T12:43:31.041475Z","shell.execute_reply.started":"2022-04-27T12:43:31.005448Z","shell.execute_reply":"2022-04-27T12:43:31.040627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"multiclass =multiclass.drop(\"classes\",axis=1)\nmulticlass = multiclass.reset_index(level=0)\nmulticlass =multiclass.join(multi_hot)","metadata":{"execution":{"iopub.status.busy":"2022-04-27T12:43:31.044735Z","iopub.execute_input":"2022-04-27T12:43:31.045193Z","iopub.status.idle":"2022-04-27T12:43:31.061929Z","shell.execute_reply.started":"2022-04-27T12:43:31.045146Z","shell.execute_reply":"2022-04-27T12:43:31.061108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"multiclass.to_csv(\"multi_hot.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-04-27T12:43:31.063043Z","iopub.execute_input":"2022-04-27T12:43:31.063388Z","iopub.status.idle":"2022-04-27T12:43:31.072025Z","shell.execute_reply.started":"2022-04-27T12:43:31.063353Z","shell.execute_reply":"2022-04-27T12:43:31.071360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = multiclass.iloc[:,1:]","metadata":{"execution":{"iopub.status.busy":"2022-04-27T12:43:31.073186Z","iopub.execute_input":"2022-04-27T12:43:31.073706Z","iopub.status.idle":"2022-04-27T12:43:31.080579Z","shell.execute_reply.started":"2022-04-27T12:43:31.073664Z","shell.execute_reply":"2022-04-27T12:43:31.079755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"multiclass[multiclass.filename==\"skylar/XC636223_140\"].iloc[:,1:].values","metadata":{"execution":{"iopub.status.busy":"2022-04-27T12:43:31.081849Z","iopub.execute_input":"2022-04-27T12:43:31.082677Z","iopub.status.idle":"2022-04-27T12:43:31.097881Z","shell.execute_reply.started":"2022-04-27T12:43:31.082636Z","shell.execute_reply":"2022-04-27T12:43:31.096808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Converter from `torchaudio.transfomrs`\n\n`n_fft`, `hop_length`, and `n_mels` are different from the base code. \n\nThese are just my preferred parameters.","metadata":{}},{"cell_type":"code","source":"PATH = bird_dir/Path(\"train_audio\")\nSR = 32000\nSEC = 5\nmel_converter = torchaudio.transforms.MelSpectrogram(sample_rate=SR,n_fft=1024,hop_length=256,n_mels=128) \nspec_converter = torchaudio.transforms.Spectrogram(n_fft=1024)\ndb_converter = torchaudio.transforms.AmplitudeToDB()","metadata":{"execution":{"iopub.status.busy":"2022-04-27T12:43:31.099462Z","iopub.execute_input":"2022-04-27T12:43:31.100358Z","iopub.status.idle":"2022-04-27T12:43:31.198864Z","shell.execute_reply.started":"2022-04-27T12:43:31.100315Z","shell.execute_reply":"2022-04-27T12:43:31.198068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict_classes(x):  \n    return vocab[torch.where(torch.tensor(multiclass[multiclass.filename==x].iloc[:,1:].values).squeeze())]\ndef len_predict(x):        \n    return 1 if type(predict_classes(x))==np.str_ else len(predict_classes(x))\n\n\noptions =[f\"{x}_{predict_classes(x)}\" for x in list(multiclass.filename)] #option list for interact","metadata":{"execution":{"iopub.status.busy":"2022-04-27T12:43:31.200837Z","iopub.execute_input":"2022-04-27T12:43:31.201249Z","iopub.status.idle":"2022-04-27T12:43:31.434957Z","shell.execute_reply.started":"2022-04-27T12:43:31.201199Z","shell.execute_reply":"2022-04-27T12:43:31.434020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def frequency_bin_to_hz(bin_index, sr, n_fft):\n    return bin_index*(sr/n_fft)\n\ndef time_bin_to_second(bin_index, sr, n_fft, hop_size):  \n    return  bin_index*hop_size/sr\n\ndef change_ytick_to_frequency(sr, n_fft):\n    #if you wanna use a spectrogram, use it for changing to y_tick\n    prev_yticks = plt.yticks()[0][1:-1] \n    ytick_labels= [frequency_bin_to_hz(bin_index, sr=sr, n_fft=n_fft) for bin_index in prev_yticks]\n    plt.yticks(ticks=prev_yticks, labels=ytick_labels)\n\ndef change_xtick_to_seconds(sr, n_fft, hop_size):\n    prev_xticks = plt.xticks()[0][1:-1]\n    xtick_labels= [time_bin_to_second(bin_index, sr=sr, n_fft=n_fft, hop_size=hop_size) for bin_index in prev_xticks]\n    plt.xticks(ticks=prev_xticks, labels=xtick_labels)\n    \ndef change_wav_xticks_to_seconde(sr):\n    prev_xticks = plt.xticks()[0][1:-1]\n    xtick_labels= [frame/sr for frame in prev_xticks]\n    plt.xticks(ticks=prev_xticks, labels=xtick_labels)","metadata":{"execution":{"iopub.status.busy":"2022-04-27T12:43:31.436354Z","iopub.execute_input":"2022-04-27T12:43:31.436680Z","iopub.status.idle":"2022-04-27T12:43:31.447191Z","shell.execute_reply.started":"2022-04-27T12:43:31.436639Z","shell.execute_reply":"2022-04-27T12:43:31.446348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Mel-spectrogram plotting with prediction\n\nI prefer interactive analysis using ipywidget.\n\nWhen you select an item from the dropdown menu, you can see the corresponding outputs.\n\nThe outputs are as follows.\n- Full mel-spectrogram plot of \".ogg\" file\n    - The red box indicates that part.\n- A sound with a length of 5 seconds (sound of selected item. i.e. area of the red box)\n- A mel-spectrogram with a length of 5 seconds(same as above)\n","metadata":{}},{"cell_type":"code","source":"HOP_LENGTH=256\nN_FFT= 1024\nSR =32000\ndef file_select(filename:str):\n    file_path = Path(PATH) /Path(filename.split(\"_\")[0]).with_suffix(\".ogg\")\n    start = int(filename.split(\"_\")[1])\n    filename =\"_\".join(filename.split(\"_\")[:2])\n    y,sr = torchaudio.load(file_path)\n    clip = y[:,start:start+SR*5]\n    ipd.display(ipd.Audio(clip,rate=sr))\n    prediction = vocab[torch.where(torch.tensor(multiclass[multiclass.filename==filename].iloc[:,1:].values).squeeze())]\n    print(f\"prediction : {prediction} filename : {filename}\")\n    mono_clip = clip.mean(dim=0)\n    mel = db_converter(mel_converter(mono_clip))    \n    #mel = db_converter(spec_converter(mono_clip))\n    plt.figure(figsize=(16,6),dpi=120)\n    plt.subplot(311)\n    plt.imshow(db_converter(mel_converter(y.mean(dim=0))),interpolation='nearest', aspect='auto',origin=\"lower\")    \n    change_xtick_to_seconds(SR, N_FFT, HOP_LENGTH)\n    #change_ytick_to_frequency(SR,N_FFT)\n    plt.gca().add_patch(Rectangle(((start*SR)//HOP_LENGTH+1,1),625,125,linewidth=1,edgecolor='r',facecolor='none'))\n    plt.title(\"full mel-spectrogram\")\n    plt.subplot(312)\n    plt.plot(mono_clip)\n    change_wav_xticks_to_seconde(SR)\n    plt.margins(0)\n    plt.title(f\"5-sec wavform {start}-{start+5}sec\")\n    plt.subplot(313)\n    plt.imshow(mel,interpolation='nearest', aspect='auto',origin=\"lower\")    \n    change_xtick_to_seconds(SR, N_FFT, HOP_LENGTH)    \n    #change_ytick_to_frequency(SR,N_FFT)\n    plt.title(f\"5-sec mel-spectrogram {start}-{start+5}sec\")\n    plt.suptitle(f\"{filename}      [prediction : {prediction}]\")\n    plt.tight_layout()\n    \n    \n     \nwidgets.interact(file_select, filename=options);","metadata":{"execution":{"iopub.status.busy":"2022-04-27T12:43:31.448725Z","iopub.execute_input":"2022-04-27T12:43:31.448970Z","iopub.status.idle":"2022-04-27T12:43:34.681030Z","shell.execute_reply.started":"2022-04-27T12:43:31.448940Z","shell.execute_reply":"2022-04-27T12:43:34.680164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# birds\n\nThe informations of birds from this notebook.\n### skylar\nhttps://ebird.org/species/skylar#\n\n<img src=\"https://cdn.download.ams.birds.cornell.edu/api/v1/asset/311380331/1800\" width=\"400\">\n\n### warwhe1\nhttps://ebird.org/species/warwhe1#\n\n<img src=\"https://cdn.download.ams.birds.cornell.edu/api/v1/asset/188765311/1800\" width=\"400\">\n\n### yefcan\nhttps://ebird.org/species/yefcan#\n\n<img src=\"https://cdn.download.ams.birds.cornell.edu/api/v1/asset/97651761/1800\" width=\"400\">\n\n### houfin\nhttps://ebird.org/species/houfin#\n\n<img src=\"https://cdn.download.ams.birds.cornell.edu/api/v1/asset/306327341/1800\" width=\"400\">","metadata":{}},{"cell_type":"markdown","source":"# Results and insights\n## Results\nFor some files, it is determined that there is a bird even though it is nocall.\n\n## insights\nBased on the index of the mel bin, less than 40 is not worth it (For these examples only).\n\nGeneral noise is distributed in the low-frequency band. Therefore, cutting out the part corresponding to the low frequency is also an option.\n\nIf you want to use specaug, set the mel bin range(frequncy range) to 40 or less. Otherwise, it will affect the bird call.\n\n\n","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}