{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":8096443,"sourceType":"datasetVersion","datasetId":4780521},{"sourceId":8318251,"sourceType":"datasetVersion","datasetId":4940719},{"sourceId":8416940,"sourceType":"datasetVersion","datasetId":5010206},{"sourceId":8426650,"sourceType":"datasetVersion","datasetId":5017621},{"sourceId":174384054,"sourceType":"kernelVersion"},{"sourceId":174415292,"sourceType":"kernelVersion"},{"sourceId":175005679,"sourceType":"kernelVersion"},{"sourceId":175993701,"sourceType":"kernelVersion"}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import gc\nimport os\nimport sys\nimport random\nimport time\nimport warnings\nimport re\nimport copy\n\nimport librosa\nimport numpy as np\nimport pandas as pd\nimport soundfile as sf\n\n\n\nfrom contextlib import contextmanager\nfrom joblib import Parallel, delayed\nfrom pathlib import Path\nfrom tqdm import tqdm\nfrom glob import glob\n\nimport matplotlib.pyplot as plt\nimport librosa.display","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:09.826606Z","iopub.execute_input":"2024-05-30T08:50:09.827769Z","iopub.status.idle":"2024-05-30T08:50:10.964921Z","shell.execute_reply.started":"2024-05-30T08:50:09.827718Z","shell.execute_reply":"2024-05-30T08:50:10.964020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import Audio\nSR=32000\nless_than = 50","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:10.968638Z","iopub.execute_input":"2024-05-30T08:50:10.969285Z","iopub.status.idle":"2024-05-30T08:50:10.974324Z","shell.execute_reply.started":"2024-05-30T08:50:10.969241Z","shell.execute_reply":"2024-05-30T08:50:10.973035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seed(seed=43):\n    random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    np.random.seed(seed)\n    \nset_seed(43)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:10.975522Z","iopub.execute_input":"2024-05-30T08:50:10.975869Z","iopub.status.idle":"2024-05-30T08:50:10.986438Z","shell.execute_reply.started":"2024-05-30T08:50:10.975835Z","shell.execute_reply":"2024-05-30T08:50:10.985408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_paths = glob(\"/kaggle/input/1negative-samples-npy-birdclef2024/noise/*.npy\")\nprint(len(n_paths))\nNP=[]\nfor path in n_paths:\n    y = np.load(path)\n    length = len(y)/32000\n    if length==5.:\n#         print(path,len(y)/32000)\n        NP.append(path)\nprint(len(NP))","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:10.987675Z","iopub.execute_input":"2024-05-30T08:50:10.988097Z","iopub.status.idle":"2024-05-30T08:50:11.879716Z","shell.execute_reply.started":"2024-05-30T08:50:10.988066Z","shell.execute_reply":"2024-05-30T08:50:11.878670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dur = pd.read_csv(\"/kaggle/input/2duration-log/bc2024_duration.csv\")\nprint(dur.shape)\ndur.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:11.882151Z","iopub.execute_input":"2024-05-30T08:50:11.882478Z","iopub.status.idle":"2024-05-30T08:50:12.115253Z","shell.execute_reply.started":"2024-05-30T08:50:11.882452Z","shell.execute_reply":"2024-05-30T08:50:12.114182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def find_audio_paths_birdcode(bird_code):\n    audio_paths = glob(f'/kaggle/input/1-5second-npy-birdclef2024/train_npy0/{bird_code}/*.npy')\\\n                    + glob(f'/kaggle/input/2-5second-npy-birdclef2024/train_npy1/{bird_code}/*.npy')\n    return audio_paths\n\ndef find_audio_paths_birdcode_filetags(bird_code,file_tag):\n    audio_paths = glob(f'/kaggle/input/1-5second-npy-birdclef2024/train_npy0/{bird_code}/{file_tag}.npy')\\\n                    + glob(f'/kaggle/input/2-5second-npy-birdclef2024/train_npy1/{bird_code}/{file_tag}.npy')\n    return audio_paths","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:12.116714Z","iopub.execute_input":"2024-05-30T08:50:12.117569Z","iopub.status.idle":"2024-05-30T08:50:12.123086Z","shell.execute_reply.started":"2024-05-30T08:50:12.117538Z","shell.execute_reply":"2024-05-30T08:50:12.121964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"txt_file = \"/kaggle/input/duplicates-calls-bc2024/duplicates_bird_calls.txt\"\nDL =[]\nwith open(txt_file,'r') as file:\n    lines = file.readlines()\n    for line in lines:\n        DL.append(line)\nfile.close()\nprint(len(DL))\nprint(DL[0])\n\ndiff_birds=[]\nsame_birds=[]\nfor item in DL:\n    a,b = item.split(\",\")[0],item.split(\",\")[1]\n    if a.split(\"/\")[0] != b.split(\"/\")[0]:\n        fla = a.split(\"/\")[1].split(\".\")[0]\n        flb = b.split(\"/\")[1].split(\".\")[0]\n        if fla!=flb:\n            diff_birds.append((a,b.strip()))\n    else:\n#         same_birds.append((a,b.strip()))\n        same_birds.append(a)\nprint(len(diff_birds),len(same_birds))\nprint(diff_birds[0],same_birds[0])","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:12.124635Z","iopub.execute_input":"2024-05-30T08:50:12.124997Z","iopub.status.idle":"2024-05-30T08:50:12.140783Z","shell.execute_reply.started":"2024-05-30T08:50:12.124961Z","shell.execute_reply":"2024-05-30T08:50:12.139620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #remove the same bird calls from paths\n# val = len(all_paths)\n# print(len(all_paths))\n# temp_FL=[]\n# for n,filename in enumerate(same_birds):\n#     bird_name = filename.split(\"/\")[0]\n#     file_tag = filename.split(\"/\")[1].split(\".\")[0]\n#     path = find_audio_paths_birdcode_filetags(bird_name,file_tag)\n# #     print(n,path[0])\n#     if file_tag not in temp_FL:\n#         all_paths.remove(path[0])\n#     temp_FL.append(file_tag)\n    \n# print(len(all_paths))\n# print(\"diff:\", val - len(all_paths))","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:12.142187Z","iopub.execute_input":"2024-05-30T08:50:12.142594Z","iopub.status.idle":"2024-05-30T08:50:12.151333Z","shell.execute_reply.started":"2024-05-30T08:50:12.142556Z","shell.execute_reply":"2024-05-30T08:50:12.150314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FOLDER = \"/kaggle/input/birdclef-2024\"\ntrain_dir =\"/kaggle/input/birdclef-2024/train_audio\"\nAUDIO_DURATION=5.","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:12.152658Z","iopub.execute_input":"2024-05-30T08:50:12.153000Z","iopub.status.idle":"2024-05-30T08:50:12.163952Z","shell.execute_reply.started":"2024-05-30T08:50:12.152974Z","shell.execute_reply":"2024-05-30T08:50:12.162931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import ast\n\n\ntrain = pd.read_csv(os.path.join(FOLDER,\"train_metadata.csv\"))\n\n\ntrain['new_target'] = train['primary_label'] + ' ' + train['secondary_labels'].map(lambda x: ' '.join(ast.literal_eval(x)))\n# train['len_new_target'] = train['new_target'].map(lambda x: len(x.split()))\n# train['len_new_target'].value_counts()\ntrain['file_tag'] = train['filename'].map(lambda x: x.split(\".\")[0].split(\"/\")[-1])\nprint(train.shape)\ntrain.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:12.165392Z","iopub.execute_input":"2024-05-30T08:50:12.165785Z","iopub.status.idle":"2024-05-30T08:50:12.552239Z","shell.execute_reply.started":"2024-05-30T08:50:12.165746Z","shell.execute_reply":"2024-05-30T08:50:12.551192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter=0\nfor n in range(len(train)):\n    pl = train.iloc[n]['primary_label']\n    bird_list = train.iloc[n]['new_target'].split()\n    temp_birdlist=[]\n    temp_birdlist.append(pl)\n    if len(bird_list)>1:\n        for idx in range(len(bird_list)):\n            if idx>0:\n                bird_name = bird_list[idx]\n                if bird_name!=pl:\n                    temp_birdlist.append(bird_name)\n        names = \" \".join(temp_birdlist)\n#         print(pl,names)\n        counter +=1\n        train.loc[n,'new_target']= names\nprint(counter)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:12.553384Z","iopub.execute_input":"2024-05-30T08:50:12.553685Z","iopub.status.idle":"2024-05-30T08:50:16.110338Z","shell.execute_reply.started":"2024-05-30T08:50:12.553659Z","shell.execute_reply":"2024-05-30T08:50:16.109005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for a,b in diff_birds:\n    fla = a.split(\"/\")[1].split(\".\")[0]\n    flb = b.split(\"/\")[1].split(\".\")[0]\n    idx = train[train.file_tag==flb].index\n    train.at[idx.values[0],'file_tag']=fla","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:16.111631Z","iopub.execute_input":"2024-05-30T08:50:16.112004Z","iopub.status.idle":"2024-05-30T08:50:16.151563Z","shell.execute_reply.started":"2024-05-30T08:50:16.111975Z","shell.execute_reply":"2024-05-30T08:50:16.150491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique, counts = np.unique(train.file_tag, return_counts=True)\ncounts_dict = dict(zip(unique, counts))\nprint(len(counts_dict), len(train), len(train)-len(counts_dict))\n\ntemp = {k: v for k, v in sorted(counts_dict.items(), key=lambda item: item[1],reverse=True)}\ntemp = dict(list(temp.items())[:19])\ntags = temp.keys()\nprint(list(tags))\n\nfor tag in tags:\n    temp = list(train[train.file_tag==tag].new_target.values)\n    indxs = train[train.file_tag==tag].index\n#     print(temp)\n    names = \" \".join(temp)\n    for indx in indxs:\n        train.at[indx,'new_target']= names\n        \nlist(train[train.file_tag=='XC574864'].new_target.values)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:16.153049Z","iopub.execute_input":"2024-05-30T08:50:16.153378Z","iopub.status.idle":"2024-05-30T08:50:16.331916Z","shell.execute_reply.started":"2024-05-30T08:50:16.153351Z","shell.execute_reply":"2024-05-30T08:50:16.330866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val = len(train)\nprint(len(train))\ntrain.drop_duplicates(subset='file_tag',inplace=True)\ntrain.reset_index(drop=True, inplace=True)\ntrain['len_new_target'] = train['new_target'].map(lambda x: len(x.split()))\nprint(len(train))\nprint(\"diff:\",val-len(train))","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:16.337711Z","iopub.execute_input":"2024-05-30T08:50:16.338098Z","iopub.status.idle":"2024-05-30T08:50:16.370007Z","shell.execute_reply.started":"2024-05-30T08:50:16.338067Z","shell.execute_reply":"2024-05-30T08:50:16.368864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"secondary_labels = ['asfblu1','indwhe1','bltmun1','magrob','lotshr1','orhthr1']\nss = pd.read_csv(\"/kaggle/input/birdclef-2024/sample_submission.csv\")\nbird_target_names = list(ss.columns)\nbird_target_names.pop(0)\nlen(bird_target_names)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:16.371652Z","iopub.execute_input":"2024-05-30T08:50:16.372560Z","iopub.status.idle":"2024-05-30T08:50:16.390133Z","shell.execute_reply.started":"2024-05-30T08:50:16.372523Z","shell.execute_reply":"2024-05-30T08:50:16.388924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"common_names = list(pd.unique(train.common_name))\nlen(common_names)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:16.391427Z","iopub.execute_input":"2024-05-30T08:50:16.391777Z","iopub.status.idle":"2024-05-30T08:50:16.402193Z","shell.execute_reply.started":"2024-05-30T08:50:16.391720Z","shell.execute_reply":"2024-05-30T08:50:16.401153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tax = pd.read_csv(\"/kaggle/input/birdclef-2024/eBird_Taxonomy_v2021.csv\")\nprint(tax.shape)\ntax.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:16.403873Z","iopub.execute_input":"2024-05-30T08:50:16.404749Z","iopub.status.idle":"2024-05-30T08:50:16.499232Z","shell.execute_reply.started":"2024-05-30T08:50:16.404681Z","shell.execute_reply":"2024-05-30T08:50:16.498240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SL= tax[tax.SPECIES_CODE.isin(secondary_labels)][['SPECIES_CODE','PRIMARY_COM_NAME']]\nSL.reset_index(drop=True, inplace=True)\nSL = SL.rename(columns={'SPECIES_CODE':'primary_label','PRIMARY_COM_NAME':'common_name'})\nSL","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:16.500602Z","iopub.execute_input":"2024-05-30T08:50:16.500977Z","iopub.status.idle":"2024-05-30T08:50:16.518254Z","shell.execute_reply.started":"2024-05-30T08:50:16.500949Z","shell.execute_reply":"2024-05-30T08:50:16.517244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final = train.merge(dur[['filename','duration']])\nprint(final.shape)\nfinal.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:16.519795Z","iopub.execute_input":"2024-05-30T08:50:16.520189Z","iopub.status.idle":"2024-05-30T08:50:16.575697Z","shell.execute_reply.started":"2024-05-30T08:50:16.520148Z","shell.execute_reply":"2024-05-30T08:50:16.574657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_birdlist = train[['common_name','primary_label']]\ndf_birdlist.drop_duplicates(inplace=True)\ndf_birdlist.reset_index(drop=True, inplace=True)\ndf_birdlist","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:16.577368Z","iopub.execute_input":"2024-05-30T08:50:16.577796Z","iopub.status.idle":"2024-05-30T08:50:16.599208Z","shell.execute_reply.started":"2024-05-30T08:50:16.577759Z","shell.execute_reply":"2024-05-30T08:50:16.597952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_audio = {}\nfor species in bird_target_names:\n    num_audio_files = os.listdir(os.path.join(train_dir,species))\n#     print(species,len(num_audio_files))\n    num_audio[species]=len(num_audio_files)\n    \nnew_dict = {k: v for k, v in sorted(num_audio.items(), key=lambda item: item[1])}\nprint(len(new_dict))\n\nnum_audio = pd.DataFrame(new_dict.items(), columns=['primary_label', 'NUM_AUDIO_FILES',])\nnum_audio = num_audio.merge(df_birdlist,)\nprint(num_audio.shape)\n# num_audio = num_audio.rename(columns={'SPECIES_CODE':'primary_label'})\nnum_audio.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:16.600601Z","iopub.execute_input":"2024-05-30T08:50:16.600975Z","iopub.status.idle":"2024-05-30T08:50:20.462381Z","shell.execute_reply.started":"2024-05-30T08:50:16.600943Z","shell.execute_reply":"2024-05-30T08:50:20.461056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"group_duration = final.groupby('primary_label')['duration'].sum()\ngroup_dur = pd.merge(num_audio,pd.DataFrame(group_duration).reset_index())\nprint(group_dur.shape)\n\ngroup_dur['5_second_duration'] = np.round(group_dur['duration']/AUDIO_DURATION,0)\ngroup_dur['5_second_duration'].describe()\n\ngroup_dur.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:20.463557Z","iopub.execute_input":"2024-05-30T08:50:20.463875Z","iopub.status.idle":"2024-05-30T08:50:20.492215Z","shell.execute_reply.started":"2024-05-30T08:50:20.463849Z","shell.execute_reply":"2024-05-30T08:50:20.491139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### NOISE","metadata":{}},{"cell_type":"code","source":"n_paths = glob(\"/kaggle/input/1negative-samples-npy-birdclef2024/noise/*.npy\")\nprint(len(n_paths))\nNOISE_PATHS=[]\nfor path in n_paths:\n    y = np.load(path)\n    length = len(y)/32000\n    if length==5.:\n#         print(path,len(y)/32000)\n        NOISE_PATHS.append(path)\nprint(len(NOISE_PATHS))","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:20.493616Z","iopub.execute_input":"2024-05-30T08:50:20.494056Z","iopub.status.idle":"2024-05-30T08:50:20.608475Z","shell.execute_reply.started":"2024-05-30T08:50:20.494019Z","shell.execute_reply":"2024-05-30T08:50:20.607297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bird_matrix = np.zeros((len(bird_target_names),len(bird_target_names)),dtype=np.int16)\nbird_matrix2 = np.zeros((len(bird_target_names),len(bird_target_names)),dtype=np.int16)\nprint(bird_matrix.shape)\n\nfor n in tqdm(range(len(train))):\n    pl = train.iloc[n]['primary_label']\n    bird_list = train.iloc[n]['new_target'].split()\n    pl_indx= bird_target_names.index(pl)\n    bird_matrix[pl_indx,pl_indx] +=1\n    bird_matrix2[pl_indx,pl_indx] +=1\n    if len(bird_list)>1:\n        for idx in range(len(bird_list)):\n            if idx>0:\n                sl = bird_list[idx]\n                if sl!=pl:\n                    if sl not in secondary_labels:\n                        sl_indx = bird_target_names.index(sl)\n                        bird_matrix[pl_indx,sl_indx] +=1\n                        bird_matrix2[sl_indx,pl_indx] +=1\n                        bird_matrix2[pl_indx,sl_indx] +=1                        ","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:20.609730Z","iopub.execute_input":"2024-05-30T08:50:20.610059Z","iopub.status.idle":"2024-05-30T08:50:24.528098Z","shell.execute_reply.started":"2024-05-30T08:50:20.610034Z","shell.execute_reply":"2024-05-30T08:50:24.526781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PL = final[final.len_new_target==1]\nPL.reset_index(drop=True, inplace=True)\nprint(PL.shape)\nPL.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:24.529383Z","iopub.execute_input":"2024-05-30T08:50:24.529703Z","iopub.status.idle":"2024-05-30T08:50:24.555047Z","shell.execute_reply.started":"2024-05-30T08:50:24.529672Z","shell.execute_reply":"2024-05-30T08:50:24.553830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### WATER BASED BIRDS","metadata":{}},{"cell_type":"code","source":"water = pd.read_csv(\"/kaggle/input/bc2024-water-tagged-bird-list/watertagged_bird_list2 - final_bird_list.csv\")\nwater.rename(columns={'Unnamed: 2': 'water_tag',},inplace=True)\nwater.fillna('n',inplace=True)\nprint(water.shape)\nwater.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:24.556429Z","iopub.execute_input":"2024-05-30T08:50:24.556744Z","iopub.status.idle":"2024-05-30T08:50:24.575439Z","shell.execute_reply.started":"2024-05-30T08:50:24.556719Z","shell.execute_reply":"2024-05-30T08:50:24.574409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"water_birds = list(water[water.water_tag=='w']['PRIMARY_COM_NAME'].values)\nprint(len(water_birds))\n\nwater_group = group_dur[group_dur.common_name.isin(water_birds)].merge(df_birdlist)\nwater_group.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:24.576992Z","iopub.execute_input":"2024-05-30T08:50:24.577898Z","iopub.status.idle":"2024-05-30T08:50:24.597204Z","shell.execute_reply.started":"2024-05-30T08:50:24.577856Z","shell.execute_reply":"2024-05-30T08:50:24.596052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bird_groups = { 'low_elevation': ['Zitting Cisticola','Plain Prinia','Rufous Treepie','Small Minivet','Gray-headed Swamphen',\n                   'Asian Koel','Laughing Dove','Gray Francolin',],\n  'low-mid_elevation':['Paddyfield Pipit','Common Iora','White-throated Kingfisher','Spotted Owlet',\n                      'Painted Stork','Asian Openbill','Spotted Dove','Red Spurfowl'],\n 'unlikely':['houspa','Brahminy Kite','Eurasian Marsh-Harrier','Eurasian Collared-Dove',\n            'Rock Pigeon','Gray Francolin'],}","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:24.598469Z","iopub.execute_input":"2024-05-30T08:50:24.598772Z","iopub.status.idle":"2024-05-30T08:50:24.607616Z","shell.execute_reply.started":"2024-05-30T08:50:24.598747Z","shell.execute_reply":"2024-05-30T08:50:24.606595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### LOW ELEVATION","metadata":{}},{"cell_type":"code","source":"low_elevation_birds = list(pd.unique(PL[PL.common_name.isin(bird_groups['low_elevation'])]['common_name']))\nprint(len(low_elevation_birds))\nprint(\"total # audio files:\",group_dur[group_dur.common_name.isin(low_elevation_birds)]['NUM_AUDIO_FILES'].sum())\nprint(\"Possible total # audio files:\",group_dur[group_dur.common_name.\\\n                                                isin(low_elevation_birds)]['5_second_duration'].sum())","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:24.608921Z","iopub.execute_input":"2024-05-30T08:50:24.609294Z","iopub.status.idle":"2024-05-30T08:50:24.624034Z","shell.execute_reply.started":"2024-05-30T08:50:24.609258Z","shell.execute_reply":"2024-05-30T08:50:24.622979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### LOW-MID ELEVATION","metadata":{}},{"cell_type":"code","source":"low_mid_birds  = list(pd.unique(PL[PL.common_name.isin(bird_groups['low-mid_elevation'])]['common_name']))\nprint(len(low_mid_birds))\nprint(\"total # audio files:\",group_dur[group_dur.common_name.isin(low_mid_birds)]['NUM_AUDIO_FILES'].sum())\nprint(\"Possible total # audio files:\",group_dur[group_dur.common_name.\\\n                                                isin(low_mid_birds)]['5_second_duration'].sum())","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:24.625139Z","iopub.execute_input":"2024-05-30T08:50:24.625463Z","iopub.status.idle":"2024-05-30T08:50:24.640712Z","shell.execute_reply.started":"2024-05-30T08:50:24.625432Z","shell.execute_reply":"2024-05-30T08:50:24.639469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### UNLIKELY","metadata":{}},{"cell_type":"code","source":"unlikely_birds = list(pd.unique(PL[PL.common_name.isin(bird_groups['unlikely'])]['common_name']))\nprint(len(unlikely_birds))\nprint(\"total # audio files:\",group_dur[group_dur.common_name.isin(unlikely_birds)]['NUM_AUDIO_FILES'].sum())\nprint(\"Possible total # audio files:\",group_dur[group_dur.common_name.\\\n                                                isin(unlikely_birds)]['5_second_duration'].sum())","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:24.642256Z","iopub.execute_input":"2024-05-30T08:50:24.642585Z","iopub.status.idle":"2024-05-30T08:50:24.657656Z","shell.execute_reply.started":"2024-05-30T08:50:24.642560Z","shell.execute_reply":"2024-05-30T08:50:24.656520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# eucdov - low\n# Eurasian Marsh-Harrier - open \n# rock pigeon - jog falls, humans\n# gray francolin - low grasslands,entry of naraikadu, low\n# brahminy kite - open, waterbody,","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:24.659354Z","iopub.execute_input":"2024-05-30T08:50:24.659740Z","iopub.status.idle":"2024-05-30T08:50:24.665429Z","shell.execute_reply.started":"2024-05-30T08:50:24.659703Z","shell.execute_reply":"2024-05-30T08:50:24.664290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### HIGH PRIORITY BIRDS","metadata":{}},{"cell_type":"code","source":"high_priority_birds = ['Gray Junglefowl', 'Malabar Whistling-Thrush', 'Malabar Barbet', 'White-cheeked Barbet',\n                       'Vernal Hanging-Parrot', 'Rufous Babbler', 'Southern Hill Myna', 'Dark-fronted Babbler', \n                       'Nilgiri Wood-Pigeon', 'Malabar Parakeet', 'Crimson-backed Sunbird', 'Orange Minivet',\n                       'White-bellied Blue Flycatcher', 'Malabar Woodshrike', 'Gray-fronted Green-Pigeon',\n                       'Nilgiri Flycatcher', 'Great Hornbill', 'Square-tailed Bulbul', 'Black-and-orange Flycatcher',\n                       'Nilgiri Flowerpecker', 'Yellow-browed Bulbul', 'Indian Yellow Tit', 'Malabar Trogon', \n                       'Jungle Myna', \"Loten's Sunbird\", 'Palani Laughingthrush', 'Common Flameback', 'White-bellied Woodpecker', \n                       'White-bellied Treepie', 'White-bellied Sholakili', 'Malabar Gray Hornbill', 'Wayanad Laughingthrush', \n                       'Flame-throated Bulbul',\n                        'Brown Wood-Owl','Spot-bellied Eagle-Owl','Brown Fish-Owl','Jungle Owlet','Brown Boobook',\n                       'Great Eared-Nightjar', 'Jungle Nightjar',\n                      ]\nprint(len(high_priority_birds))\nhpb = list(pd.unique(PL[PL.common_name.isin(high_priority_birds)]['primary_label']))\nprint(\"total # audio files:\",group_dur[group_dur.primary_label.isin(hpb)]['NUM_AUDIO_FILES'].sum())\nprint(\"possible total # audio files:\",group_dur[group_dur.primary_label.isin(hpb)]['5_second_duration'].sum())\nprint()\ngroup_dur[group_dur.primary_label.isin(hpb)].shape","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:24.667179Z","iopub.execute_input":"2024-05-30T08:50:24.667519Z","iopub.status.idle":"2024-05-30T08:50:24.687323Z","shell.execute_reply.started":"2024-05-30T08:50:24.667491Z","shell.execute_reply":"2024-05-30T08:50:24.686072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mid_priority_birds = ['Brown-capped Pygmy Woodpecker', 'Chestnut-headed Bee-eater','Crested Goshawk', 'Velvet-fronted Nuthatch', \n                      \"Jerdon's Bushlark\", 'Indian Scimitar-Babbler', 'Plum-headed Parakeet', 'Speckled Piculet', \n                      'Rufous Woodpecker', 'Asian Emerald Dove', 'Golden-fronted Leafbird', 'Green Warbler', \n                      'Indian Blackbird', 'Heart-spotted Woodpecker', 'Little Spiderhunter', 'Rusty-tailed Flycatcher', \n                      'Red-whiskered Bulbul', 'White-browed Bulbul', 'Streak-throated Woodpecker', 'Stork-billed Kingfisher', \n                      'White-rumped Munia', 'Large-billed Leaf Warbler', 'Yellow-billed Babbler', 'Bar-winged Flycatcher-shrike', \n                      'Indian Blue Robin', \"Tickell's Leaf Warbler\"]\nprint(len(mid_priority_birds))\nmpb = list(pd.unique(PL[PL.common_name.isin(mid_priority_birds)]['primary_label']))\nprint(\"total # audio files:\",group_dur[group_dur.primary_label.isin(mpb)]['NUM_AUDIO_FILES'].sum())\nprint(\"possible total # audio files:\",group_dur[group_dur.primary_label.isin(mpb)]['5_second_duration'].sum())\nprint()\ngroup_dur[group_dur.primary_label.isin(mpb)].shape\n","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:24.688722Z","iopub.execute_input":"2024-05-30T08:50:24.689053Z","iopub.status.idle":"2024-05-30T08:50:24.705335Z","shell.execute_reply.started":"2024-05-30T08:50:24.689020Z","shell.execute_reply":"2024-05-30T08:50:24.704080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"low_priority_birds = [\"Forest Wagtail\",\"Greater Racket-tailed Drongo\",\"Gray-headed Canary-Flycatcher\",\n                     \"Indian Pitta\",\"Jungle Babbler\",\"Lesser Yellownape\",\"Pale-billed Flowerpecker\",\n                     \"Gray-bellied Cuckoo\",\"Purple-rumped Sunbird\",\"Thick-billed Warbler\",\n                     \"Tickell's Blue Flycatcher\",'Indian Scops-Owl']\n\nprint(len(low_priority_birds))\nlpb = list(pd.unique(PL[PL.common_name.isin(low_priority_birds)]['primary_label']))\nprint(\"total # audio files:\",group_dur[group_dur.primary_label.isin(lpb)]['NUM_AUDIO_FILES'].sum())\nprint(\"possible total # audio files:\",group_dur[group_dur.primary_label.isin(lpb)]['5_second_duration'].sum())\nprint()\ngroup_dur[group_dur.primary_label.isin(lpb)]","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:24.707129Z","iopub.execute_input":"2024-05-30T08:50:24.707514Z","iopub.status.idle":"2024-05-30T08:50:24.731034Z","shell.execute_reply.started":"2024-05-30T08:50:24.707487Z","shell.execute_reply":"2024-05-30T08:50:24.729794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### NOCTURNAL BIRDS","metadata":{}},{"cell_type":"code","source":"nb = pd.read_csv(\"/kaggle/input/bc2024-nocturnal-dirunal-birds/Nocturnal_bird_list - final_bird_list.csv\")\nprint(nb.shape)\nnocturnal_birds = list(nb[nb['Dirunal/Nocturnal']=='n']['PRIMARY_COM_NAME'].values)\nnocturnal_birds.remove('Black-crowned Night-Heron')\nprint(nocturnal_birds)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:24.732447Z","iopub.execute_input":"2024-05-30T08:50:24.732769Z","iopub.status.idle":"2024-05-30T08:50:24.744812Z","shell.execute_reply.started":"2024-05-30T08:50:24.732743Z","shell.execute_reply":"2024-05-30T08:50:24.743534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bird_namex = list(pd.unique(PL[PL.common_name.isin(nocturnal_birds)]['primary_label']))\nprint(len(bird_namex))\nprint(\"total # audio files:\",group_dur[group_dur.primary_label.isin(bird_namex)]['NUM_AUDIO_FILES'].sum())\nprint(\"possible total # audio files:\",group_dur[group_dur.primary_label.isin(bird_namex)]['5_second_duration'].sum())\nprint()\ngroup_dur[group_dur.primary_label.isin(bird_namex)]","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:24.746278Z","iopub.execute_input":"2024-05-30T08:50:24.746618Z","iopub.status.idle":"2024-05-30T08:50:24.767606Z","shell.execute_reply.started":"2024-05-30T08:50:24.746590Z","shell.execute_reply":"2024-05-30T08:50:24.766394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"UNLIKELY_P = unlikely_birds+water_birds\nLOW_P      = list(set(low_priority_birds  + low_elevation_birds))\nMID_P      = list(set(mid_priority_birds  + low_mid_birds))\nHIGH_P     = high_priority_birds","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:24.769239Z","iopub.execute_input":"2024-05-30T08:50:24.769640Z","iopub.status.idle":"2024-05-30T08:50:24.778803Z","shell.execute_reply.started":"2024-05-30T08:50:24.769608Z","shell.execute_reply":"2024-05-30T08:50:24.777631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Low Priority Birds:\",len(LOW_P))\nprint(\"Mid Priority Birds:\",len(MID_P))\nprint(\"High Priority Birds:\",len(HIGH_P))\nprint(\"unlikely Birds:\", len(UNLIKELY_P))","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:24.780191Z","iopub.execute_input":"2024-05-30T08:50:24.780535Z","iopub.status.idle":"2024-05-30T08:50:24.792521Z","shell.execute_reply.started":"2024-05-30T08:50:24.780506Z","shell.execute_reply":"2024-05-30T08:50:24.791182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ALL_PATHS=[]\nhigh_counter=0\nprev_counter = -1\nfor bird_name in tqdm(HIGH_P):\n    bird_code = df_birdlist[df_birdlist.common_name==bird_name]['primary_label'].values[0]\n    audio_paths = glob(f'{train_dir}/{bird_code}/*.ogg')\n    high_counter += len(audio_paths)\nprint(\"HIGH_P:\",high_counter)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:24.794023Z","iopub.execute_input":"2024-05-30T08:50:24.794478Z","iopub.status.idle":"2024-05-30T08:50:24.858013Z","shell.execute_reply.started":"2024-05-30T08:50:24.794445Z","shell.execute_reply":"2024-05-30T08:50:24.856775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mid_counter=0\nprev_counter = -1\nfor bird_name in tqdm(MID_P):\n    bird_code = df_birdlist[df_birdlist.common_name==bird_name]['primary_label'].values[0]\n    audio_paths = glob(f'{train_dir}/{bird_code}/*.ogg')\n    mid_counter += len(audio_paths)\nprint(\"MID_P:\",mid_counter)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:24.864729Z","iopub.execute_input":"2024-05-30T08:50:24.865123Z","iopub.status.idle":"2024-05-30T08:50:24.915610Z","shell.execute_reply.started":"2024-05-30T08:50:24.865088Z","shell.execute_reply":"2024-05-30T08:50:24.914555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"low_counter=0\nprev_counter = -1\nfor bird_name in tqdm(LOW_P):\n    bird_code = df_birdlist[df_birdlist.common_name==bird_name]['primary_label'].values[0]\n    audio_paths = glob(f'{train_dir}/{bird_code}/*.ogg')\n    low_counter += len(audio_paths)\nprint(\"LOW_P:\",low_counter)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:24.917060Z","iopub.execute_input":"2024-05-30T08:50:24.917465Z","iopub.status.idle":"2024-05-30T08:50:24.955248Z","shell.execute_reply.started":"2024-05-30T08:50:24.917428Z","shell.execute_reply":"2024-05-30T08:50:24.954060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Total # audio files:\",high_counter+mid_counter+low_counter)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:24.956634Z","iopub.execute_input":"2024-05-30T08:50:24.957002Z","iopub.status.idle":"2024-05-30T08:50:24.962237Z","shell.execute_reply.started":"2024-05-30T08:50:24.956973Z","shell.execute_reply":"2024-05-30T08:50:24.961085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUDIO_COUNT_DICT =  {bird_code:0 for bird_code in (LOW_P+MID_P+HIGH_P)}\nprint(len(AUDIO_COUNT_DICT))\nPATHS_DICT =  {bird_code:[] for bird_code in (LOW_P+MID_P+HIGH_P)}\n\naudio_counter=0\nfor bird_name in tqdm(HIGH_P+MID_P+LOW_P):\n    bird_code = df_birdlist[df_birdlist.common_name==bird_name]['primary_label'].values[0]\n    audio_paths = glob(f'{train_dir}/{bird_code}/*.ogg')\n    audio_counter += len(audio_paths)\n    AUDIO_COUNT_DICT[bird_name] += len(audio_paths)\n    PATHS_DICT[bird_name].extend(audio_paths)\nprint(\"Audio Counter:\",audio_counter)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:24.963626Z","iopub.execute_input":"2024-05-30T08:50:24.964461Z","iopub.status.idle":"2024-05-30T08:50:25.122628Z","shell.execute_reply.started":"2024-05-30T08:50:24.964424Z","shell.execute_reply":"2024-05-30T08:50:25.121607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## SAVE AUDIO","metadata":{}},{"cell_type":"code","source":"def normalize_equal_shapes(signal,SR=32000,USE_SEC=5):\n    signal = signal/ np.linalg.norm(signal)\n    if len(signal)>USE_SEC*SR:\n                signal = signal[:USE_SEC*SR]\n    else:\n        diff = SR*USE_SEC - len(signal)\n        signal = np.pad(signal, (0,diff), 'constant',)\n    return signal\n\ndef change_speed(data,speed_factor=0.8,use_sec=5,sr=32000):\n    #If rate > 1, then the signal is sped up. If rate < 1, then the signal is slowed down.\n    signal = librosa.effects.time_stretch(data, rate=speed_factor)\n    if len(signal)>USE_SEC*SR:\n        signal = signal[:USE_SEC*SR]\n    else:\n        diff = SR*USE_SEC - len(signal)\n        signal = np.pad(signal, (0,diff), 'constant',)\n    return signal\n\ndef pitch_shift(data, sr=32000, pitch_factor=1.8):\n    #how many (fractional) steps to shift y\n    return librosa.effects.pitch_shift(data, sr=sr, n_steps=pitch_factor)\n\ndef audio_shift(data, sampling_rate=32000, shift_max=1, shift_direction='right'):\n    shift = np.random.randint(sampling_rate * shift_max)\n    if shift_direction == 'right':\n        shift = -shift\n    elif shift_direction == 'both':\n        direction = np.random.randint(0, 2)\n        if direction == 1:\n            shift = -shift\n    augmented_data = np.roll(data, shift)\n    # Set to silence for heading/ tailing\n    if shift > 0:\n        augmented_data[:shift] = 0\n    else:\n        augmented_data[shift:] = 0\n    return augmented_data\n\ndef noise_injection(data, noise_factor=0.05):\n    \n    noise_path = np.random.choice(NP)\n    noise = np.load(noise_path)\n    noise = noise[:len(data)]\n    augmented_data = data + noise_factor * noise\n    # Cast back to same data type\n    augmented_data = augmented_data.astype(data.dtype)\n    return augmented_data\n\ndef normalized_custom_noise_injection(data, noise_factor=2):\n    \n    noise_path = np.random.choice(NP)\n    noise = np.load(noise_path)\n    \n    noise = normalize_equal_shapes(noise,)\n    data = normalize_equal_shapes(data,)\n    augmented_data = data + noise_factor * noise\n    # Cast back to same data type\n    augmented_data =augmented_data.astype(data.dtype)\n    return augmented_data\n        \n\n# This code runs only in python 3.10 or above versions\ndef apply_augmentations(rand,signal):\n    match rand:\n        case 0:\n            return change_speed(signal,speed_factor=random.choice([0.4,0.6,0.8,1.0,1.2,1.6,1.8]))\n        case 1:\n            return pitch_shift(signal, sr=32000, pitch_factor=random.choice([0.5,1.5,2.0,2.5,3.0,3.5,4.0]))\n        case 2:\n            return audio_shift(signal, sampling_rate=32000, shift_max=random.choice([0.2,0.4,0.5,0.6,0.8,1.0]), \n                               shift_direction=random.choice(['right','both']))\n        case 3:\n            return normalized_custom_noise_injection(signal, noise_factor=random.choice([2.,3.,4.]))\n        case default:\n            return signal","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:25.124067Z","iopub.execute_input":"2024-05-30T08:50:25.124392Z","iopub.status.idle":"2024-05-30T08:50:25.140355Z","shell.execute_reply.started":"2024-05-30T08:50:25.124365Z","shell.execute_reply":"2024-05-30T08:50:25.139073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SR = 32000\n # 60 # 90 # 60\nCAP_MIX = 10\nOUT_DIR = \"/kaggle/working/\"\nMIN_DUR = 0.47 \nUSE_SEC=5","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:25.141597Z","iopub.execute_input":"2024-05-30T08:50:25.141931Z","iopub.status.idle":"2024-05-30T08:50:25.156516Z","shell.execute_reply.started":"2024-05-30T08:50:25.141905Z","shell.execute_reply":"2024-05-30T08:50:25.155313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Low Priority Birds:\",len(LOW_P))\nprint(\"Mid Priority Birds:\",len(MID_P))\nprint(\"High Priority Birds:\",len(HIGH_P))\nprint(\"unlikely Birds:\", len(UNLIKELY_P))","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:50:25.157945Z","iopub.execute_input":"2024-05-30T08:50:25.158289Z","iopub.status.idle":"2024-05-30T08:50:25.169918Z","shell.execute_reply.started":"2024-05-30T08:50:25.158261Z","shell.execute_reply":"2024-05-30T08:50:25.168650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def MIX_SAVE(FILE1,FILE2,LABEL_LIST):\n    USE_RATING=3.\n    USE_SEC=5\n    SR=32000\n    MIN_SEC=0.47\n\n    NEW_TARGETS=[]\n    NEW_PATHS=[]\n    MIXED_TAGS=[]\n    FILE_LESS,FILE_MORE=[],[]\n\n    for n in tqdm(range(len(FILE1))):\n        filea,fileb = FILE1[n],FILE2[n]\n\n\n        filea_tag = filea.split(\"/\")[-1].split(\".\")[0]\n        fileb_tag = fileb.split(\"/\")[-1].split(\".\")[0]\n\n        filea_signal = np.load(filea)\n        fileb_signal = np.load(fileb)\n\n        if len(filea_signal)>=int(MIN_SEC*SR) or len(fileb_signal)>int(MIN_SEC*SR): #MINSEC=1\n\n            FILE_LESS.append(filea)\n            FILE_MORE.append(fileb)\n\n            #FILEA\n            if len(filea_signal)>USE_SEC*SR:\n                    filea_signal = filea_signal[:USE_SEC*SR]\n            else:\n                diff = SR*USE_SEC - len(filea_signal)\n                filea_signal = np.pad(filea_signal, (0,diff), 'constant',)\n\n            #FILEB USE_STAGE\n        #         start,stop = stage*SR*USE_SEC,(stage+1)*SR*USE_SEC\n        #         fileb_signal = fileb_signal[start:stop]\n            if len(fileb_signal)>USE_SEC*SR:\n                    fileb_signal = fileb_signal[:USE_SEC*SR]\n            else:\n                diff = SR*USE_SEC - len(fileb_signal)\n                fileb_signal = np.pad(fileb_signal, (0,diff), 'constant',)\n\n\n#             bird_target = filea.split(\"/\")[-2]\n            mixed_file_tag = LABEL_LIST[n][0]+\"_\"+LABEL_LIST[n][1]+\"_\"+filea_tag + \"_\" + fileb_tag + \"_\" + str(n)\n\n            rand = np.random.randint(4)\n\n            #FILE B\n            fileb_signal = apply_augmentations(rand,fileb_signal)\n\n            with np.errstate(invalid='raise'):\n                try:\n                    mixed_signal = (filea_signal / np.linalg.norm(filea_signal))\\\n                                 + (fileb_signal / np.linalg.norm(fileb_signal))\n                    save_path = os.path.join(save_folder_path,mixed_file_tag)\n\n                    np.save(save_path, mixed_signal)\n                except FloatingPointError:\n                    print('Error: Division by Zero')\n                    print('bird name:',bird_name)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:51:27.815248Z","iopub.execute_input":"2024-05-30T08:51:27.815658Z","iopub.status.idle":"2024-05-30T08:51:27.828985Z","shell.execute_reply.started":"2024-05-30T08:51:27.815624Z","shell.execute_reply":"2024-05-30T08:51:27.827656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Nocturna - Nocturnal Priority","metadata":{}},{"cell_type":"code","source":"CAP_MIX = 50\noutput_dir = f'nocturnal_nocturnal/'\n\nsave_folder_path = output_dir\n_ = os.makedirs(save_folder_path, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:51:28.873239Z","iopub.execute_input":"2024-05-30T08:51:28.873615Z","iopub.status.idle":"2024-05-30T08:51:28.879208Z","shell.execute_reply.started":"2024-05-30T08:51:28.873590Z","shell.execute_reply":"2024-05-30T08:51:28.878016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"USE_RATING=3.\nUSE_SEC=5\nSR=32000\nMIN_SEC=3\n\n# CURB_NUM_AUDIO = 100\nMAX_STAGE=6\nCOUNTER=0\nFILE1,FILE2 =[],[]\nLABEL_LIST=[]\n\nNUM_PER_COMBINATION =1\n\nTOTAL_FILE_COUNTER = 0\n\nFILES_USED = []\n\nfor bird_name1 in tqdm(nocturnal_birds):\n    bird_code1 = df_birdlist[df_birdlist.common_name==bird_name1]['primary_label'].values[0]\n    r = re.compile(f\".*{bird_code1}\")\n    audio_paths1 = find_audio_paths_birdcode(bird_code1)\n    PATHS_USED1=[]\n    PATHS_USED2 = []\n    BIRDS_USED2=[]\n    file_counter = 0\n    HIGH_P_COPY = copy.deepcopy(nocturnal_birds)\n    HIGH_P_COPY.remove(bird_name1)\n    for _ in range(CAP_MIX):\n        path1 = random.choice(audio_paths1)\n        if path1 not in PATHS_USED1:\n            PATHS_USED1.append(path1)\n            bird_name2 = random.choice(HIGH_P_COPY)\n            if bird_name2 not in BIRDS_USED2:\n                BIRDS_USED2.append(bird_name2)\n                bird_code2 = df_birdlist[df_birdlist.common_name==bird_name2]['primary_label'].values[0]\n                audio_paths2 = find_audio_paths_birdcode(bird_code2)\n                path2 = random.choice(audio_paths2)\n                if path2 not in PATHS_USED2:\n                    PATHS_USED2.append(path2)\n                    FILE1.append(path1)\n                    FILE2.append(path2)\n                    LABEL_LIST.append([bird_code1,bird_code2])\n                    TOTAL_FILE_COUNTER+=1\n                    file_counter +=1\n                else:\n                    continue\n                    \n            else:\n                continue\n        \n        else:\n            continue\n    print(bird_name1,len(audio_paths1),file_counter)\n                \nprint(\"TOTAL_FILE_COUNTER:\",TOTAL_FILE_COUNTER )","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:51:30.531597Z","iopub.execute_input":"2024-05-30T08:51:30.532031Z","iopub.status.idle":"2024-05-30T08:51:30.649841Z","shell.execute_reply.started":"2024-05-30T08:51:30.531999Z","shell.execute_reply":"2024-05-30T08:51:30.648777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(FILE1),len(FILE2),len(LABEL_LIST)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:51:32.326277Z","iopub.execute_input":"2024-05-30T08:51:32.326668Z","iopub.status.idle":"2024-05-30T08:51:32.334323Z","shell.execute_reply.started":"2024-05-30T08:51:32.326638Z","shell.execute_reply":"2024-05-30T08:51:32.333089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FILE1[0],FILE2[0],LABEL_LIST[0]","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:51:32.514967Z","iopub.execute_input":"2024-05-30T08:51:32.515764Z","iopub.status.idle":"2024-05-30T08:51:32.522588Z","shell.execute_reply.started":"2024-05-30T08:51:32.515732Z","shell.execute_reply":"2024-05-30T08:51:32.521506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MIX_SAVE(FILE1,FILE2,LABEL_LIST)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:51:33.295398Z","iopub.execute_input":"2024-05-30T08:51:33.295798Z","iopub.status.idle":"2024-05-30T08:51:41.352224Z","shell.execute_reply.started":"2024-05-30T08:51:33.295767Z","shell.execute_reply":"2024-05-30T08:51:41.351115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selfmix_paths = glob('/kaggle/working/*/*.npy')\n\nprint(selfmix_paths[0])\nprint(\"total selfmix audio files:\", len(selfmix_paths))","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:51:48.525562Z","iopub.execute_input":"2024-05-30T08:51:48.526796Z","iopub.status.idle":"2024-05-30T08:51:48.533114Z","shell.execute_reply.started":"2024-05-30T08:51:48.526747Z","shell.execute_reply":"2024-05-30T08:51:48.531995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter=0\nfor path in selfmix_paths:\n    x = np.load(path)\n    if np.isnan(x).any():\n        counter +=1\ncounter","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:51:48.726777Z","iopub.execute_input":"2024-05-30T08:51:48.727210Z","iopub.status.idle":"2024-05-30T08:51:48.769190Z","shell.execute_reply.started":"2024-05-30T08:51:48.727179Z","shell.execute_reply":"2024-05-30T08:51:48.768118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_df = pd.DataFrame(selfmix_paths, columns=['file_path'])\npath_df['new_target'] = path_df['file_path'].map(lambda x: x.split(\"/\")[-1].split(\"_\")[0])\nprint(path_df.shape)\npath_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:51:50.263314Z","iopub.execute_input":"2024-05-30T08:51:50.263716Z","iopub.status.idle":"2024-05-30T08:51:50.276864Z","shell.execute_reply.started":"2024-05-30T08:51:50.263687Z","shell.execute_reply":"2024-05-30T08:51:50.275873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for bird_name in nocturnal_birds:\n    bird_code = df_birdlist[df_birdlist.common_name==bird_name]['primary_label'].values[0]\n    num_paths = len(os.listdir(f\"/kaggle/input/birdclef-2024/train_audio/{bird_code}\"))\n    num_files = len(path_df[path_df.new_target==bird_code])\n    print(bird_name,num_paths,num_files)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:51:50.443777Z","iopub.execute_input":"2024-05-30T08:51:50.444751Z","iopub.status.idle":"2024-05-30T08:51:50.466895Z","shell.execute_reply.started":"2024-05-30T08:51:50.444716Z","shell.execute_reply":"2024-05-30T08:51:50.465743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for bird_name in list(df_birdlist.common_name.values):\n    bird_code = df_birdlist[df_birdlist.common_name==bird_name]['primary_label'].values[0]\n    num_paths = len(os.listdir(f\"/kaggle/input/birdclef-2024/train_audio/{bird_code}\"))\n    num_files = len(path_df[path_df.new_target==bird_code])\n    if num_files !=0:\n#         if num_files>100:\n        print(bird_name,num_paths,num_files)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T08:51:50.616862Z","iopub.execute_input":"2024-05-30T08:51:50.617946Z","iopub.status.idle":"2024-05-30T08:51:50.874540Z","shell.execute_reply.started":"2024-05-30T08:51:50.617904Z","shell.execute_reply":"2024-05-30T08:51:50.873533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}