{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":8096443,"sourceType":"datasetVersion","datasetId":4780521},{"sourceId":8318251,"sourceType":"datasetVersion","datasetId":4940719},{"sourceId":8416940,"sourceType":"datasetVersion","datasetId":5010206},{"sourceId":8426650,"sourceType":"datasetVersion","datasetId":5017621},{"sourceId":170596805,"sourceType":"kernelVersion"},{"sourceId":170597150,"sourceType":"kernelVersion"},{"sourceId":170597237,"sourceType":"kernelVersion"},{"sourceId":170597938,"sourceType":"kernelVersion"},{"sourceId":170598083,"sourceType":"kernelVersion"},{"sourceId":170598279,"sourceType":"kernelVersion"},{"sourceId":170598684,"sourceType":"kernelVersion"},{"sourceId":175005679,"sourceType":"kernelVersion"},{"sourceId":175993701,"sourceType":"kernelVersion"}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import gc\nimport os\nimport sys\n# sys.path.append('../input/pytorch-image-models/pytorch-image-models-master')\nimport random\nimport time\nimport warnings\nimport re\nimport copy\n\nimport librosa\nimport numpy as np\nimport pandas as pd\nimport soundfile as sf\n\n\n\nfrom contextlib import contextmanager\nfrom joblib import Parallel, delayed\nfrom pathlib import Path\nfrom tqdm import tqdm\nfrom glob import glob\n\nimport matplotlib.pyplot as plt\nimport librosa.display","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:27.709693Z","iopub.execute_input":"2024-06-01T06:12:27.710117Z","iopub.status.idle":"2024-06-01T06:12:29.101341Z","shell.execute_reply.started":"2024-06-01T06:12:27.710083Z","shell.execute_reply":"2024-06-01T06:12:29.100203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import Audio\nSR=32000\nless_than = 50","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:29.103604Z","iopub.execute_input":"2024-06-01T06:12:29.104098Z","iopub.status.idle":"2024-06-01T06:12:29.109355Z","shell.execute_reply.started":"2024-06-01T06:12:29.104064Z","shell.execute_reply":"2024-06-01T06:12:29.107936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seed(seed=43):\n    random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    np.random.seed(seed)\n    \nset_seed(43)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:29.112008Z","iopub.execute_input":"2024-06-01T06:12:29.112867Z","iopub.status.idle":"2024-06-01T06:12:29.124371Z","shell.execute_reply.started":"2024-06-01T06:12:29.112831Z","shell.execute_reply":"2024-06-01T06:12:29.122765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_paths = glob(\"/kaggle/input/1negative-samples-npy-birdclef2024/noise/*.npy\")\nprint(len(n_paths))\nNP=[]\nfor path in n_paths:\n    y = np.load(path)\n    length = len(y)/32000\n    if length==5.:\n#         print(path,len(y)/32000)\n        NP.append(path)\nprint(len(NP))","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:29.125893Z","iopub.execute_input":"2024-06-01T06:12:29.126381Z","iopub.status.idle":"2024-06-01T06:12:30.035315Z","shell.execute_reply.started":"2024-06-01T06:12:29.126338Z","shell.execute_reply":"2024-06-01T06:12:30.034156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dur = pd.read_csv(\"/kaggle/input/2duration-log/bc2024_duration.csv\")\nprint(dur.shape)\ndur.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:30.038616Z","iopub.execute_input":"2024-06-01T06:12:30.038980Z","iopub.status.idle":"2024-06-01T06:12:30.312757Z","shell.execute_reply.started":"2024-06-01T06:12:30.038950Z","shell.execute_reply":"2024-06-01T06:12:30.311331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def find_audio_paths_birdcode(bird_code):\n    audio_paths = glob(f'/kaggle/input/1npy-birdclef2024/train_npy0/{bird_code}/*.npy')\\\n                    + glob(f'/kaggle/input/2npy-birdclef2024/train_npy1/{bird_code}/*.npy')\\\n                    + glob(f'/kaggle/input/3npy-birdclef2024/train_npy2/{bird_code}/*.npy')\\\n                    + glob(f'/kaggle/input/4npy-birdclef2024/train_npy3/{bird_code}/*.npy')\\\n                    + glob(f'/kaggle/input/5npy-birdclef2024/train_npy4/{bird_code}/*.npy')\\\n                    + glob(f'/kaggle/input/6npy-birdclef2024/train_npy5/{bird_code}/*.npy')\\\n                    + glob(f'/kaggle/input/7npy-birdclef2024/train_npy6/{bird_code}/*.npy')\n    return audio_paths\n\ndef find_audio_paths_birdcode_filetags(bird_code,file_tag):\n    audio_paths = glob(f'/kaggle/input/1npy-birdclef2024/train_npy0/{bird_code}/{file_tag}.npy')\\\n                    + glob(f'/kaggle/input/2npy-birdclef2024/train_npy1/{bird_code}/{file_tag}.npy')\\\n                    + glob(f'/kaggle/input/3npy-birdclef2024/train_npy2/{bird_code}/{file_tag}.npy')\\\n                    + glob(f'/kaggle/input/4npy-birdclef2024/train_npy3/{bird_code}/{file_tag}.npy')\\\n                    + glob(f'/kaggle/input/5npy-birdclef2024/train_npy4/{bird_code}/{file_tag}.npy')\\\n                    + glob(f'/kaggle/input/6npy-birdclef2024/train_npy5/{bird_code}/{file_tag}.npy')\\\n                    + glob(f'/kaggle/input/7npy-birdclef2024/train_npy6/{bird_code}/{file_tag}.npy')\n    return audio_paths","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:30.316744Z","iopub.execute_input":"2024-06-01T06:12:30.317108Z","iopub.status.idle":"2024-06-01T06:12:30.326245Z","shell.execute_reply.started":"2024-06-01T06:12:30.317082Z","shell.execute_reply":"2024-06-01T06:12:30.324818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_paths = glob('/kaggle/input/1npy-birdclef2024/train_npy0/*/*.npy')\\\n                + glob('/kaggle/input/2npy-birdclef2024/train_npy1/*/*.npy')\\\n                + glob('/kaggle/input/3npy-birdclef2024/train_npy2/*/*.npy')\\\n                + glob('/kaggle/input/4npy-birdclef2024/train_npy3/*/*.npy')\\\n                + glob('/kaggle/input/5npy-birdclef2024/train_npy4/*/*.npy')\\\n                + glob('/kaggle/input/6npy-birdclef2024/train_npy5/*/*.npy')\\\n                + glob('/kaggle/input/7npy-birdclef2024/train_npy6/*/*.npy')\n\n# all_path = glob.glob(\"/kaggle/input/birdclef-2024/train_audio/*/*.ogg\")\nprint(all_paths[0])\nprint(len(all_paths))","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:30.327897Z","iopub.execute_input":"2024-06-01T06:12:30.328258Z","iopub.status.idle":"2024-06-01T06:12:42.095028Z","shell.execute_reply.started":"2024-06-01T06:12:30.328230Z","shell.execute_reply":"2024-06-01T06:12:42.094214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"txt_file = \"/kaggle/input/duplicates-calls-bc2024/duplicates_bird_calls.txt\"\nDL =[]\nwith open(txt_file,'r') as file:\n    lines = file.readlines()\n    for line in lines:\n        DL.append(line)\nfile.close()\nprint(len(DL))\nprint(DL[0])\n\ndiff_birds=[]\nsame_birds=[]\nfor item in DL:\n    a,b = item.split(\",\")[0],item.split(\",\")[1]\n    if a.split(\"/\")[0] != b.split(\"/\")[0]:\n        fla = a.split(\"/\")[1].split(\".\")[0]\n        flb = b.split(\"/\")[1].split(\".\")[0]\n        if fla!=flb:\n            diff_birds.append((a,b.strip()))\n    else:\n#         same_birds.append((a,b.strip()))\n        same_birds.append(a)\nprint(len(diff_birds),len(same_birds))\nprint(diff_birds[0],same_birds[0])","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:42.096358Z","iopub.execute_input":"2024-06-01T06:12:42.097030Z","iopub.status.idle":"2024-06-01T06:12:42.111621Z","shell.execute_reply.started":"2024-06-01T06:12:42.097000Z","shell.execute_reply":"2024-06-01T06:12:42.110217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#remove the same bird calls from paths\nval = len(all_paths)\nprint(len(all_paths))\ntemp_FL=[]\nfor n,filename in enumerate(same_birds):\n    bird_name = filename.split(\"/\")[0]\n    file_tag = filename.split(\"/\")[1].split(\".\")[0]\n    path = find_audio_paths_birdcode_filetags(bird_name,file_tag)\n#     print(n,path[0])\n    if file_tag not in temp_FL:\n        all_paths.remove(path[0])\n    temp_FL.append(file_tag)\n    \nprint(len(all_paths))\nprint(\"diff:\", val - len(all_paths))","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:42.115118Z","iopub.execute_input":"2024-06-01T06:12:42.115515Z","iopub.status.idle":"2024-06-01T06:12:42.518319Z","shell.execute_reply.started":"2024-06-01T06:12:42.115485Z","shell.execute_reply":"2024-06-01T06:12:42.515949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FOLDER = \"/kaggle/input/birdclef-2024\"\ntrain_dir =\"/kaggle/input/birdclef-2024/train_audio\"\nAUDIO_DURATION=5.","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:42.519665Z","iopub.execute_input":"2024-06-01T06:12:42.520516Z","iopub.status.idle":"2024-06-01T06:12:42.525202Z","shell.execute_reply.started":"2024-06-01T06:12:42.520486Z","shell.execute_reply":"2024-06-01T06:12:42.523823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import ast\n\n\ntrain = pd.read_csv(os.path.join(FOLDER,\"train_metadata.csv\"))\n\n\ntrain['new_target'] = train['primary_label'] + ' ' + train['secondary_labels'].map(lambda x: ' '.join(ast.literal_eval(x)))\n# train['len_new_target'] = train['new_target'].map(lambda x: len(x.split()))\n# train['len_new_target'].value_counts()\ntrain['file_tag'] = train['filename'].map(lambda x: x.split(\".\")[0].split(\"/\")[-1])\nprint(train.shape)\ntrain.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:42.527206Z","iopub.execute_input":"2024-06-01T06:12:42.527984Z","iopub.status.idle":"2024-06-01T06:12:42.984971Z","shell.execute_reply.started":"2024-06-01T06:12:42.527952Z","shell.execute_reply":"2024-06-01T06:12:42.983848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter=0\nfor n in range(len(train)):\n    pl = train.iloc[n]['primary_label']\n    bird_list = train.iloc[n]['new_target'].split()\n    temp_birdlist=[]\n    temp_birdlist.append(pl)\n    if len(bird_list)>1:\n        for idx in range(len(bird_list)):\n            if idx>0:\n                bird_name = bird_list[idx]\n                if bird_name!=pl:\n                    temp_birdlist.append(bird_name)\n        names = \" \".join(temp_birdlist)\n#         print(pl,names)\n        counter +=1\n        train.loc[n,'new_target']= names\nprint(counter)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:42.986510Z","iopub.execute_input":"2024-06-01T06:12:42.987536Z","iopub.status.idle":"2024-06-01T06:12:47.354892Z","shell.execute_reply.started":"2024-06-01T06:12:42.987495Z","shell.execute_reply":"2024-06-01T06:12:47.353618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for a,b in diff_birds:\n    fla = a.split(\"/\")[1].split(\".\")[0]\n    flb = b.split(\"/\")[1].split(\".\")[0]\n    idx = train[train.file_tag==flb].index\n    train.at[idx.values[0],'file_tag']=fla","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:47.356352Z","iopub.execute_input":"2024-06-01T06:12:47.356729Z","iopub.status.idle":"2024-06-01T06:12:47.429306Z","shell.execute_reply.started":"2024-06-01T06:12:47.356699Z","shell.execute_reply":"2024-06-01T06:12:47.427966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique, counts = np.unique(train.file_tag, return_counts=True)\ncounts_dict = dict(zip(unique, counts))\nprint(len(counts_dict), len(train), len(train)-len(counts_dict))\n\ntemp = {k: v for k, v in sorted(counts_dict.items(), key=lambda item: item[1],reverse=True)}\ntemp = dict(list(temp.items())[:19])\ntags = temp.keys()\nprint(list(tags))\n\nfor tag in tags:\n    temp = list(train[train.file_tag==tag].new_target.values)\n    indxs = train[train.file_tag==tag].index\n#     print(temp)\n    names = \" \".join(temp)\n    for indx in indxs:\n        train.at[indx,'new_target']= names\n        \nlist(train[train.file_tag=='XC574864'].new_target.values)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:47.436211Z","iopub.execute_input":"2024-06-01T06:12:47.436617Z","iopub.status.idle":"2024-06-01T06:12:47.741403Z","shell.execute_reply.started":"2024-06-01T06:12:47.436584Z","shell.execute_reply":"2024-06-01T06:12:47.740189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val = len(train)\nprint(len(train))\ntrain.drop_duplicates(subset='file_tag',inplace=True)\ntrain.reset_index(drop=True, inplace=True)\ntrain['len_new_target'] = train['new_target'].map(lambda x: len(x.split()))\nprint(len(train))\nprint(\"diff:\",val-len(train))","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:47.743172Z","iopub.execute_input":"2024-06-01T06:12:47.743545Z","iopub.status.idle":"2024-06-01T06:12:47.786665Z","shell.execute_reply.started":"2024-06-01T06:12:47.743514Z","shell.execute_reply":"2024-06-01T06:12:47.785202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"secondary_labels = ['asfblu1','indwhe1','bltmun1','magrob','lotshr1','orhthr1']\nss = pd.read_csv(\"/kaggle/input/birdclef-2024/sample_submission.csv\")\nbird_target_names = list(ss.columns)\nbird_target_names.pop(0)\nlen(bird_target_names)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:47.788275Z","iopub.execute_input":"2024-06-01T06:12:47.788839Z","iopub.status.idle":"2024-06-01T06:12:47.811357Z","shell.execute_reply.started":"2024-06-01T06:12:47.788808Z","shell.execute_reply":"2024-06-01T06:12:47.810119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"common_names = list(pd.unique(train.common_name))\nlen(common_names)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:47.813912Z","iopub.execute_input":"2024-06-01T06:12:47.814321Z","iopub.status.idle":"2024-06-01T06:12:47.826721Z","shell.execute_reply.started":"2024-06-01T06:12:47.814288Z","shell.execute_reply":"2024-06-01T06:12:47.825608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tax = pd.read_csv(\"/kaggle/input/birdclef-2024/eBird_Taxonomy_v2021.csv\")\nprint(tax.shape)\ntax.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:47.828053Z","iopub.execute_input":"2024-06-01T06:12:47.828434Z","iopub.status.idle":"2024-06-01T06:12:47.940402Z","shell.execute_reply.started":"2024-06-01T06:12:47.828387Z","shell.execute_reply":"2024-06-01T06:12:47.939193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SL= tax[tax.SPECIES_CODE.isin(secondary_labels)][['SPECIES_CODE','PRIMARY_COM_NAME']]\nSL.reset_index(drop=True, inplace=True)\nSL = SL.rename(columns={'SPECIES_CODE':'primary_label','PRIMARY_COM_NAME':'common_name'})\nSL","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:47.942053Z","iopub.execute_input":"2024-06-01T06:12:47.942396Z","iopub.status.idle":"2024-06-01T06:12:47.963047Z","shell.execute_reply.started":"2024-06-01T06:12:47.942368Z","shell.execute_reply":"2024-06-01T06:12:47.961793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_df = pd.DataFrame(all_paths, columns=['file_path'])\n#path_df['filename'] = path_df['file_path'].map(lambda x: x.split('/')[-2]+'/'+x.split('/')[-1][:-4])\npath_df[\"filename\"] = path_df['file_path'].map(lambda x: x.split(\"/\")[-2] + \"/\" +\\\n                                               x.split(\"/\")[-1].split('.')[-2] + \".ogg\")\n\npath_df['file_tag2'] = path_df['filename'].map(lambda x: x.split(\".\")[0].split(\"/\")[-1])\nprint(path_df.shape)\npath_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:47.964702Z","iopub.execute_input":"2024-06-01T06:12:47.965271Z","iopub.status.idle":"2024-06-01T06:12:48.053816Z","shell.execute_reply.started":"2024-06-01T06:12:47.965229Z","shell.execute_reply":"2024-06-01T06:12:48.052724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for a,b in diff_birds:\n    fla = a.split(\"/\")[1].split(\".\")[0]\n    flb = b.split(\"/\")[1].split(\".\")[0]\n    idx = path_df[path_df.file_tag2==flb].index\n    path_df.at[idx.values[0],'file_tag2']=fla","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:48.055330Z","iopub.execute_input":"2024-06-01T06:12:48.055705Z","iopub.status.idle":"2024-06-01T06:12:48.122422Z","shell.execute_reply.started":"2024-06-01T06:12:48.055675Z","shell.execute_reply":"2024-06-01T06:12:48.121212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique, counts = np.unique(path_df.file_tag2, return_counts=True)\ncounts_dict = dict(zip(unique, counts))\nprint(len(counts_dict), len(path_df), len(path_df)-len(counts_dict))\n\ntemp = {k: v for k, v in sorted(counts_dict.items(), key=lambda item: item[1],reverse=True)}\ntemp = dict(list(temp.items())[:19])\ntags = temp.keys()\nprint(temp)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:48.123741Z","iopub.execute_input":"2024-06-01T06:12:48.124077Z","iopub.status.idle":"2024-06-01T06:12:48.199332Z","shell.execute_reply.started":"2024-06-01T06:12:48.124048Z","shell.execute_reply":"2024-06-01T06:12:48.197954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val = len(path_df)\nprint(path_df.shape)\npath_df.drop_duplicates(subset='file_tag2',inplace=True)\npath_df.reset_index(drop=True, inplace=True)\nprint(path_df.shape)\nprint(\"diff:\",val-len(path_df))","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:48.201287Z","iopub.execute_input":"2024-06-01T06:12:48.201745Z","iopub.status.idle":"2024-06-01T06:12:48.216396Z","shell.execute_reply.started":"2024-06-01T06:12:48.201705Z","shell.execute_reply":"2024-06-01T06:12:48.214806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_df['file_tag'] = path_df['file_tag2'].map(lambda x: x.split(\"_\")[0])\npath_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:48.217667Z","iopub.execute_input":"2024-06-01T06:12:48.218022Z","iopub.status.idle":"2024-06-01T06:12:48.247664Z","shell.execute_reply.started":"2024-06-01T06:12:48.217992Z","shell.execute_reply":"2024-06-01T06:12:48.246437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.merge(dur[['filename','duration']])\nprint(train.shape)\ntrain.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:48.248835Z","iopub.execute_input":"2024-06-01T06:12:48.249734Z","iopub.status.idle":"2024-06-01T06:12:48.314682Z","shell.execute_reply.started":"2024-06-01T06:12:48.249699Z","shell.execute_reply":"2024-06-01T06:12:48.313574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final = path_df.merge(train, on=['file_tag'])\nprint(final.shape)\nfinal.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:48.316206Z","iopub.execute_input":"2024-06-01T06:12:48.316643Z","iopub.status.idle":"2024-06-01T06:12:48.384670Z","shell.execute_reply.started":"2024-06-01T06:12:48.316603Z","shell.execute_reply":"2024-06-01T06:12:48.383460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_birdlist = train[['common_name','primary_label']]\ndf_birdlist.drop_duplicates(inplace=True)\ndf_birdlist.reset_index(drop=True, inplace=True)\ndf_birdlist","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:48.386272Z","iopub.execute_input":"2024-06-01T06:12:48.386735Z","iopub.status.idle":"2024-06-01T06:12:48.412325Z","shell.execute_reply.started":"2024-06-01T06:12:48.386694Z","shell.execute_reply":"2024-06-01T06:12:48.411045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_audio = {}\nfor species in bird_target_names:\n    num_audio_files = os.listdir(os.path.join(train_dir,species))\n#     print(species,len(num_audio_files))\n    num_audio[species]=len(num_audio_files)\n    \nnew_dict = {k: v for k, v in sorted(num_audio.items(), key=lambda item: item[1])}\nprint(len(new_dict))\n\nnum_audio = pd.DataFrame(new_dict.items(), columns=['primary_label', 'NUM_AUDIO_FILES',])\nnum_audio = num_audio.merge(df_birdlist,)\nprint(num_audio.shape)\n# num_audio = num_audio.rename(columns={'SPECIES_CODE':'primary_label'})\nnum_audio.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:48.413606Z","iopub.execute_input":"2024-06-01T06:12:48.413921Z","iopub.status.idle":"2024-06-01T06:12:51.302791Z","shell.execute_reply.started":"2024-06-01T06:12:48.413894Z","shell.execute_reply":"2024-06-01T06:12:51.301652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"group_duration = final.groupby('primary_label')['duration'].sum()\ngroup_dur = pd.merge(num_audio,pd.DataFrame(group_duration).reset_index())\nprint(group_dur.shape)\n\ngroup_dur['5_second_duration'] = np.round(group_dur['duration']/AUDIO_DURATION,0)\ngroup_dur['5_second_duration'].describe()\n\ngroup_dur.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:51.304459Z","iopub.execute_input":"2024-06-01T06:12:51.305213Z","iopub.status.idle":"2024-06-01T06:12:51.338009Z","shell.execute_reply.started":"2024-06-01T06:12:51.305172Z","shell.execute_reply":"2024-06-01T06:12:51.336831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### NOISE","metadata":{}},{"cell_type":"code","source":"n_paths = glob(\"/kaggle/input/1negative-samples-npy-birdclef2024/noise/*.npy\")\nprint(len(n_paths))\nNOISE_PATHS=[]\nfor path in n_paths:\n    y = np.load(path)\n    length = len(y)/32000\n    if length==5.:\n#         print(path,len(y)/32000)\n        NOISE_PATHS.append(path)\nprint(len(NOISE_PATHS))","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:51.339457Z","iopub.execute_input":"2024-06-01T06:12:51.339795Z","iopub.status.idle":"2024-06-01T06:12:51.446500Z","shell.execute_reply.started":"2024-06-01T06:12:51.339766Z","shell.execute_reply":"2024-06-01T06:12:51.445393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bird_matrix = np.zeros((len(bird_target_names),len(bird_target_names)),dtype=np.int16)\nbird_matrix2 = np.zeros((len(bird_target_names),len(bird_target_names)),dtype=np.int16)\nprint(bird_matrix.shape)\n\nfor n in tqdm(range(len(train))):\n    pl = train.iloc[n]['primary_label']\n    bird_list = train.iloc[n]['new_target'].split()\n    pl_indx= bird_target_names.index(pl)\n    bird_matrix[pl_indx,pl_indx] +=1\n    bird_matrix2[pl_indx,pl_indx] +=1\n    if len(bird_list)>1:\n        for idx in range(len(bird_list)):\n            if idx>0:\n                sl = bird_list[idx]\n                if sl!=pl:\n                    if sl not in secondary_labels:\n                        sl_indx = bird_target_names.index(sl)\n                        bird_matrix[pl_indx,sl_indx] +=1\n                        bird_matrix2[sl_indx,pl_indx] +=1\n                        bird_matrix2[pl_indx,sl_indx] +=1                        ","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:51.448208Z","iopub.execute_input":"2024-06-01T06:12:51.448744Z","iopub.status.idle":"2024-06-01T06:12:56.115687Z","shell.execute_reply.started":"2024-06-01T06:12:51.448701Z","shell.execute_reply":"2024-06-01T06:12:56.114461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PL = final[final.len_new_target==1]\nPL.reset_index(drop=True, inplace=True)\nprint(PL.shape)\nPL.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.116836Z","iopub.execute_input":"2024-06-01T06:12:56.117165Z","iopub.status.idle":"2024-06-01T06:12:56.149304Z","shell.execute_reply.started":"2024-06-01T06:12:56.117123Z","shell.execute_reply":"2024-06-01T06:12:56.147681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### WATER BASED BIRDS","metadata":{}},{"cell_type":"code","source":"water = pd.read_csv(\"/kaggle/input/bc2024-water-tagged-bird-list/watertagged_bird_list2 - final_bird_list.csv\")\nwater.rename(columns={'Unnamed: 2': 'water_tag',},inplace=True)\nwater.fillna('n',inplace=True)\nprint(water.shape)\nwater.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.150983Z","iopub.execute_input":"2024-06-01T06:12:56.151473Z","iopub.status.idle":"2024-06-01T06:12:56.171155Z","shell.execute_reply.started":"2024-06-01T06:12:56.151432Z","shell.execute_reply":"2024-06-01T06:12:56.170318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"water_birds = list(water[water.water_tag=='w']['PRIMARY_COM_NAME'].values)\nprint(len(water_birds))\n\nwater_group = group_dur[group_dur.common_name.isin(water_birds)].merge(df_birdlist)\nwater_group.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.172051Z","iopub.execute_input":"2024-06-01T06:12:56.172409Z","iopub.status.idle":"2024-06-01T06:12:56.194722Z","shell.execute_reply.started":"2024-06-01T06:12:56.172380Z","shell.execute_reply":"2024-06-01T06:12:56.193419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bird_groups = { 'low_elevation': ['Zitting Cisticola','Plain Prinia','Rufous Treepie','Small Minivet','Gray-headed Swamphen',\n                   'Asian Koel','Laughing Dove','Gray Francolin',],\n  'low-mid_elevation':['Paddyfield Pipit','Common Iora','White-throated Kingfisher',\n                       'Spotted Owlet',\n                      'Painted Stork','Asian Openbill','Spotted Dove','Red Spurfowl'],\n 'unlikely':['houspa','Brahminy Kite','Eurasian Marsh-Harrier','Eurasian Collared-Dove',\n            'Rock Pigeon','Gray Francolin'],}","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.196692Z","iopub.execute_input":"2024-06-01T06:12:56.197096Z","iopub.status.idle":"2024-06-01T06:12:56.203712Z","shell.execute_reply.started":"2024-06-01T06:12:56.197065Z","shell.execute_reply":"2024-06-01T06:12:56.202337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### LOW ELEVATION","metadata":{}},{"cell_type":"code","source":"low_elevation_birds = list(pd.unique(PL[PL.common_name.isin(bird_groups['low_elevation'])]['common_name']))\nprint(len(low_elevation_birds))\nprint(\"total # audio files:\",group_dur[group_dur.common_name.isin(low_elevation_birds)]['NUM_AUDIO_FILES'].sum())\nprint(\"Possible total # audio files:\",group_dur[group_dur.common_name.\\\n                                                isin(low_elevation_birds)]['5_second_duration'].sum())","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.205301Z","iopub.execute_input":"2024-06-01T06:12:56.205706Z","iopub.status.idle":"2024-06-01T06:12:56.221866Z","shell.execute_reply.started":"2024-06-01T06:12:56.205674Z","shell.execute_reply":"2024-06-01T06:12:56.220607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### LOW-MID ELEVATION","metadata":{}},{"cell_type":"code","source":"low_mid_birds  = list(pd.unique(PL[PL.common_name.isin(bird_groups['low-mid_elevation'])]['common_name']))\nprint(len(low_mid_birds))\nprint(\"total # audio files:\",group_dur[group_dur.common_name.isin(low_mid_birds)]['NUM_AUDIO_FILES'].sum())\nprint(\"Possible total # audio files:\",group_dur[group_dur.common_name.\\\n                                                isin(low_mid_birds)]['5_second_duration'].sum())","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.223309Z","iopub.execute_input":"2024-06-01T06:12:56.223808Z","iopub.status.idle":"2024-06-01T06:12:56.241752Z","shell.execute_reply.started":"2024-06-01T06:12:56.223776Z","shell.execute_reply":"2024-06-01T06:12:56.240787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### UNLIKELY","metadata":{}},{"cell_type":"code","source":"unlikely_birds = list(pd.unique(PL[PL.common_name.isin(bird_groups['unlikely'])]['common_name']))\nprint(len(unlikely_birds))\nprint(\"total # audio files:\",group_dur[group_dur.common_name.isin(unlikely_birds)]['NUM_AUDIO_FILES'].sum())\nprint(\"Possible total # audio files:\",group_dur[group_dur.common_name.\\\n                                                isin(unlikely_birds)]['5_second_duration'].sum())","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.243226Z","iopub.execute_input":"2024-06-01T06:12:56.243806Z","iopub.status.idle":"2024-06-01T06:12:56.256768Z","shell.execute_reply.started":"2024-06-01T06:12:56.243775Z","shell.execute_reply":"2024-06-01T06:12:56.255836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# eucdov - low\n# Eurasian Marsh-Harrier - open \n# rock pigeon - jog falls, humans\n# gray francolin - low grasslands,entry of naraikadu, low\n# brahminy kite - open, waterbody,","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.258267Z","iopub.execute_input":"2024-06-01T06:12:56.258793Z","iopub.status.idle":"2024-06-01T06:12:56.265359Z","shell.execute_reply.started":"2024-06-01T06:12:56.258763Z","shell.execute_reply":"2024-06-01T06:12:56.264225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### HIGH PRIORITY BIRDS","metadata":{}},{"cell_type":"code","source":"high_priority_birds = ['Gray Junglefowl', 'Malabar Whistling-Thrush', 'Malabar Barbet', 'White-cheeked Barbet',\n                       'Vernal Hanging-Parrot', 'Rufous Babbler', 'Southern Hill Myna', 'Dark-fronted Babbler', \n                       'Nilgiri Wood-Pigeon', 'Malabar Parakeet', 'Crimson-backed Sunbird', 'Orange Minivet',\n                       'White-bellied Blue Flycatcher', 'Malabar Woodshrike', 'Gray-fronted Green-Pigeon',\n                       'Nilgiri Flycatcher', 'Great Hornbill', 'Square-tailed Bulbul', 'Black-and-orange Flycatcher',\n                       'Nilgiri Flowerpecker', 'Yellow-browed Bulbul', 'Indian Yellow Tit', 'Malabar Trogon', \n                       'Jungle Myna', \"Loten's Sunbird\", 'Palani Laughingthrush', 'Common Flameback', 'White-bellied Woodpecker', \n                       'White-bellied Treepie', 'White-bellied Sholakili', 'Malabar Gray Hornbill', 'Wayanad Laughingthrush', \n                       'Flame-throated Bulbul',\n                        'Brown Wood-Owl','Spot-bellied Eagle-Owl','Brown Fish-Owl','Jungle Owlet','Brown Boobook',\n                       'Indian Scops-Owl',\n                       'Great Eared-Nightjar', 'Jungle Nightjar',\n                      ]\nprint(len(high_priority_birds))\nhpb = list(pd.unique(PL[PL.common_name.isin(high_priority_birds)]['primary_label']))\nprint(\"total # audio files:\",group_dur[group_dur.primary_label.isin(hpb)]['NUM_AUDIO_FILES'].sum())\nprint(\"possible total # audio files:\",group_dur[group_dur.primary_label.isin(hpb)]['5_second_duration'].sum())\nprint()\ngroup_dur[group_dur.primary_label.isin(hpb)].shape","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.266771Z","iopub.execute_input":"2024-06-01T06:12:56.267380Z","iopub.status.idle":"2024-06-01T06:12:56.289331Z","shell.execute_reply.started":"2024-06-01T06:12:56.267349Z","shell.execute_reply":"2024-06-01T06:12:56.288236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mid_priority_birds = ['Brown-capped Pygmy Woodpecker', 'Chestnut-headed Bee-eater','Crested Goshawk', 'Velvet-fronted Nuthatch', \n                      \"Jerdon's Bushlark\", 'Indian Scimitar-Babbler', 'Plum-headed Parakeet', 'Speckled Piculet', \n                      'Rufous Woodpecker', 'Asian Emerald Dove', 'Golden-fronted Leafbird', 'Green Warbler', \n                      'Indian Blackbird', 'Heart-spotted Woodpecker', 'Little Spiderhunter', 'Rusty-tailed Flycatcher', \n                      'Red-whiskered Bulbul', 'White-browed Bulbul', 'Streak-throated Woodpecker', 'Stork-billed Kingfisher', \n                      'White-rumped Munia', 'Large-billed Leaf Warbler', 'Yellow-billed Babbler', 'Bar-winged Flycatcher-shrike', \n                      'Indian Blue Robin', \"Tickell's Leaf Warbler\"]\nprint(len(mid_priority_birds))\nmpb = list(pd.unique(PL[PL.common_name.isin(mid_priority_birds)]['primary_label']))\nprint(\"total # audio files:\",group_dur[group_dur.primary_label.isin(mpb)]['NUM_AUDIO_FILES'].sum())\nprint(\"possible total # audio files:\",group_dur[group_dur.primary_label.isin(mpb)]['5_second_duration'].sum())\nprint()\ngroup_dur[group_dur.primary_label.isin(mpb)].shape\n","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.299612Z","iopub.execute_input":"2024-06-01T06:12:56.299999Z","iopub.status.idle":"2024-06-01T06:12:56.317514Z","shell.execute_reply.started":"2024-06-01T06:12:56.299971Z","shell.execute_reply":"2024-06-01T06:12:56.316738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"low_priority_birds = [\"Forest Wagtail\",\"Greater Racket-tailed Drongo\",\"Gray-headed Canary-Flycatcher\",\n                     \"Indian Pitta\",\"Jungle Babbler\",\"Lesser Yellownape\",\"Pale-billed Flowerpecker\",\n                     \"Gray-bellied Cuckoo\",\"Purple-rumped Sunbird\",\"Thick-billed Warbler\",\n                     \"Tickell's Blue Flycatcher\",]\n\nprint(len(low_priority_birds))\nlpb = list(pd.unique(PL[PL.common_name.isin(low_priority_birds)]['primary_label']))\nprint(\"total # audio files:\",group_dur[group_dur.primary_label.isin(lpb)]['NUM_AUDIO_FILES'].sum())\nprint(\"possible total # audio files:\",group_dur[group_dur.primary_label.isin(lpb)]['5_second_duration'].sum())\nprint()\ngroup_dur[group_dur.primary_label.isin(lpb)]","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.318565Z","iopub.execute_input":"2024-06-01T06:12:56.319189Z","iopub.status.idle":"2024-06-01T06:12:56.346544Z","shell.execute_reply.started":"2024-06-01T06:12:56.319159Z","shell.execute_reply":"2024-06-01T06:12:56.345231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### NOCTURNAL BIRDS","metadata":{}},{"cell_type":"code","source":"nb = pd.read_csv(\"/kaggle/input/bc2024-nocturnal-dirunal-birds/Nocturnal_bird_list - final_bird_list.csv\")\nprint(nb.shape)\nnocturnal_birds = list(nb[nb['Dirunal/Nocturnal']=='n']['PRIMARY_COM_NAME'].values)\n# nocturnal_birds.remove('Black-crowned Night-Heron')\nprint(nocturnal_birds)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.348544Z","iopub.execute_input":"2024-06-01T06:12:56.348944Z","iopub.status.idle":"2024-06-01T06:12:56.367315Z","shell.execute_reply.started":"2024-06-01T06:12:56.348910Z","shell.execute_reply":"2024-06-01T06:12:56.366041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bird_namex = list(pd.unique(PL[PL.common_name.isin(nocturnal_birds)]['primary_label']))\nprint(len(bird_namex))\nprint(\"total # audio files:\",group_dur[group_dur.primary_label.isin(bird_namex)]['NUM_AUDIO_FILES'].sum())\nprint(\"possible total # audio files:\",group_dur[group_dur.primary_label.isin(bird_namex)]['5_second_duration'].sum())\nprint()\ngroup_dur[group_dur.primary_label.isin(bird_namex)]","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.368853Z","iopub.execute_input":"2024-06-01T06:12:56.369568Z","iopub.status.idle":"2024-06-01T06:12:56.394763Z","shell.execute_reply.started":"2024-06-01T06:12:56.369535Z","shell.execute_reply":"2024-06-01T06:12:56.393642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"UNLIKELY_P = unlikely_birds+water_birds\nLOW_P      = list(set(low_priority_birds  + low_elevation_birds))\nMID_P      = list(set(mid_priority_birds  + low_mid_birds))\nHIGH_P     = high_priority_birds","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.396763Z","iopub.execute_input":"2024-06-01T06:12:56.397242Z","iopub.status.idle":"2024-06-01T06:12:56.402616Z","shell.execute_reply.started":"2024-06-01T06:12:56.397200Z","shell.execute_reply":"2024-06-01T06:12:56.401598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Low Priority Birds:\",len(LOW_P))\nprint(\"Mid Priority Birds:\",len(MID_P))\nprint(\"High Priority Birds:\",len(HIGH_P))\nprint(\"unlikely Birds:\", len(UNLIKELY_P))","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.403714Z","iopub.execute_input":"2024-06-01T06:12:56.404087Z","iopub.status.idle":"2024-06-01T06:12:56.416253Z","shell.execute_reply.started":"2024-06-01T06:12:56.404057Z","shell.execute_reply":"2024-06-01T06:12:56.415068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_paths = list(PL.file_path.values)\nprint(len(final_paths))","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.417901Z","iopub.execute_input":"2024-06-01T06:12:56.418301Z","iopub.status.idle":"2024-06-01T06:12:56.429565Z","shell.execute_reply.started":"2024-06-01T06:12:56.418269Z","shell.execute_reply":"2024-06-01T06:12:56.428207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### SELF MIX","metadata":{}},{"cell_type":"code","source":"def normalize_equal_shapes(signal,SR=32000,USE_SEC=5):\n    signal = signal/ np.linalg.norm(signal)\n    if len(signal)>USE_SEC*SR:\n                signal = signal[:USE_SEC*SR]\n    else:\n        diff = SR*USE_SEC - len(signal)\n        signal = np.pad(signal, (0,diff), 'constant',)\n    return signal","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.431203Z","iopub.execute_input":"2024-06-01T06:12:56.431752Z","iopub.status.idle":"2024-06-01T06:12:56.442822Z","shell.execute_reply.started":"2024-06-01T06:12:56.431711Z","shell.execute_reply":"2024-06-01T06:12:56.441507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def change_speed(data,speed_factor=0.8,use_sec=5,sr=32000):\n    #If rate > 1, then the signal is sped up. If rate < 1, then the signal is slowed down.\n    signal = librosa.effects.time_stretch(data, rate=speed_factor)\n    if len(signal)>USE_SEC*SR:\n        signal = signal[:USE_SEC*SR]\n    else:\n        diff = SR*USE_SEC - len(signal)\n        signal = np.pad(signal, (0,diff), 'constant',)\n    return signal\n\ndef pitch_shift(data, sr=32000, pitch_factor=1.8):\n    #how many (fractional) steps to shift y\n    return librosa.effects.pitch_shift(data, sr=sr, n_steps=pitch_factor)\n\ndef audio_shift(data, sampling_rate=32000, shift_max=1, shift_direction='right'):\n    shift = np.random.randint(sampling_rate * shift_max)\n    if shift_direction == 'right':\n        shift = -shift\n    elif shift_direction == 'both':\n        direction = np.random.randint(0, 2)\n        if direction == 1:\n            shift = -shift\n    augmented_data = np.roll(data, shift)\n    # Set to silence for heading/ tailing\n    if shift > 0:\n        augmented_data[:shift] = 0\n    else:\n        augmented_data[shift:] = 0\n    return augmented_data\n\ndef noise_injection(data, noise_factor=0.05):\n    \n    noise_path = np.random.choice(NP)\n    noise = np.load(noise_path)\n    noise = noise[:len(data)]\n    augmented_data = data + noise_factor * noise\n    # Cast back to same data type\n    augmented_data = augmented_data.astype(data.dtype)\n    return augmented_data\n\ndef normalized_custom_noise_injection(data, noise_factor=2):\n    \n    noise_path = np.random.choice(NP)\n    noise = np.load(noise_path)\n    \n    noise = normalize_equal_shapes(noise,)\n    data = normalize_equal_shapes(data,)\n    augmented_data = data + noise_factor * noise\n    # Cast back to same data type\n    augmented_data =augmented_data.astype(data.dtype)\n    return augmented_data\n        ","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.444108Z","iopub.execute_input":"2024-06-01T06:12:56.444517Z","iopub.status.idle":"2024-06-01T06:12:56.462738Z","shell.execute_reply.started":"2024-06-01T06:12:56.444481Z","shell.execute_reply":"2024-06-01T06:12:56.461363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This code runs only in python 3.10 or above versions\ndef apply_augmentations(rand,signal):\n    match rand:\n        case 0:\n            return change_speed(signal,speed_factor=random.choice([0.4,0.6,0.8,1.0,1.2,1.6,1.8]))\n        case 1:\n            return pitch_shift(signal, sr=32000, pitch_factor=random.choice([0.5,1.5,2.0,2.5,3.0,3.5,4.0]))\n        case 2:\n            return audio_shift(signal, sampling_rate=32000, shift_max=random.choice([0.2,0.4,0.5,0.6,0.8,1.0]), \n                               shift_direction=random.choice(['right','both']))\n        case 3:\n            return normalized_custom_noise_injection(signal, noise_factor=random.choice([2.,3.,4.]))\n        case default:\n            return signal","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.464055Z","iopub.execute_input":"2024-06-01T06:12:56.464570Z","iopub.status.idle":"2024-06-01T06:12:56.476826Z","shell.execute_reply.started":"2024-06-01T06:12:56.464534Z","shell.execute_reply":"2024-06-01T06:12:56.475550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def SAVE_MIX(FILE1,FILE2,STAGE_LIST,BIRD_CODES):\n    USE_RATING=3.\n    USE_SEC=5\n    SR=32000\n    MIN_SEC=3\n\n\n    for n in tqdm(range(len(FILE1))):\n        filea,fileb = FILE1[n],FILE2[n]\n        stage = STAGE_LIST[n]\n        bird_code = BIRD_CODES[n]\n        bird_name = df_birdlist[df_birdlist.primary_label==bird_code]['common_name'].values[0]\n        \n\n        filea_tag = filea.split(\"/\")[-1].split(\".\")[0]\n        fileb_tag = fileb.split(\"/\")[-1].split(\".\")[0]\n\n        filea_signal = np.load(filea)\n        fileb_signal = np.load(fileb)\n\n        if len(filea_signal)>=int(MIN_SEC*SR) or len(fileb_signal)>int(MIN_SEC*SR): #MINSEC=1\n\n            #FILEA\n            if len(filea_signal)>USE_SEC*SR:\n                    filea_signal = filea_signal[:USE_SEC*SR]\n            else:\n                diff = SR*USE_SEC - len(filea_signal)\n                filea_signal = np.pad(filea_signal, (0,diff), 'constant',)\n\n            #FILEB USE_STAGE\n            start,stop = stage*SR*USE_SEC,(stage+1)*SR*USE_SEC\n            fileb_signal = fileb_signal[start:stop]\n            if len(fileb_signal)>USE_SEC*SR:\n                    fileb_signal = fileb_signal[:USE_SEC*SR]\n            else:\n                diff = SR*USE_SEC - len(fileb_signal)\n                fileb_signal = np.pad(fileb_signal, (0,diff), 'constant',)\n\n\n#             bird_target = filea.split(\"/\")[-2]\n            mixed_file_tag = bird_code + \"_\" + filea_tag + \"_\" + fileb_tag + \"_\" + str(stage)\n\n            rand = np.random.randint(4)\n            filea_signal = apply_augmentations(rand,filea_signal)\n            \n            with np.errstate(invalid='raise'):\n                try:\n                    mixed_signal = (filea_signal / np.linalg.norm(filea_signal))\\\n                                 + (fileb_signal / np.linalg.norm(fileb_signal))\n                    save_path = os.path.join(save_folder_path,mixed_file_tag)\n\n                    np.save(save_path, mixed_signal)\n                except FloatingPointError:\n                    print('Error: Division by Zero')\n                    print('bird name:',bird_name)\n            ","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.478406Z","iopub.execute_input":"2024-06-01T06:12:56.478898Z","iopub.status.idle":"2024-06-01T06:12:56.496845Z","shell.execute_reply.started":"2024-06-01T06:12:56.478853Z","shell.execute_reply":"2024-06-01T06:12:56.495320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Nocturnal Birds","metadata":{}},{"cell_type":"code","source":"DEFAULT_ADD_FILES = 40\nADD_FILES=50\nCURB_FILES = 100","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.498082Z","iopub.execute_input":"2024-06-01T06:12:56.498468Z","iopub.status.idle":"2024-06-01T06:12:56.510480Z","shell.execute_reply.started":"2024-06-01T06:12:56.498437Z","shell.execute_reply":"2024-06-01T06:12:56.509199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"USE_RATING=3.\nUSE_SEC=5\nSR=32000\nMIN_SEC=3\n\n# CURB_NUM_AUDIO = 100\nMAX_STAGE=6\n\nCOUNTER=0\nFILE1,FILE2 =[],[]\nSTAGE_LIST=[]\nBIRD_CODES = []\n\nbird_dict = {bird_name:[] for bird_name in bird_target_names}\nfor_counter = 0\nfor bird_name in tqdm(nocturnal_birds):\n    bird_code = df_birdlist[df_birdlist.common_name==bird_name]['primary_label'].values[0]\n    r = re.compile(f\".*{bird_code}\")\n    audio_paths = list(filter(r.match, final_paths))\n    needed_num_files=0\n    mix_count=0\n    \n    if len(audio_paths)>CURB_FILES:\n        CURB_NUM_AUDIO = DEFAULT_ADD_FILES\n    else:\n        CURB_NUM_AUDIO = (CURB_FILES - len(audio_paths)) + DEFAULT_ADD_FILES\n        if CURB_NUM_AUDIO>ADD_FILES:\n            CURB_NUM_AUDIO=ADD_FILES\n    \n    if bird_name ==\"Black-crowned Night-Heron\":\n        CURB_NUM_AUDIO=10\n    if bird_name ==\"Spotted Owlet\":\n        CURB_NUM_AUDIO=10\n\n        \n#     if len(audio_paths)<CURB_NUM_AUDIO:\n    needed_num_files = CURB_NUM_AUDIO #- len(audio_paths)\n    mix_count= 0\n    stage_count = 0\n    mixed_files = []\n    audio_pathsx = copy.deepcopy(audio_paths)\n\n    while (mix_count<needed_num_files) and stage_count<MAX_STAGE:\n        if len(audio_pathsx)>1:\n#                 print(\"\\t\\trandom_choices\",len(audio_pathsx))\n            filea, fileb = random.choices(audio_pathsx,k=2)\n            if filea!=fileb:\n#                 if filea not in mixed_files and fileb not in mixed_files:\n                if len(mixed_files)<=(len(audio_paths)):\n                    mixed_files.append(filea)\n                    mixed_files.append(fileb)\n                    FILE1.append(filea)\n                    FILE2.append(fileb)\n                    STAGE_LIST.append(stage_count)\n                    BIRD_CODES.append(bird_code)\n                    mix_count +=1\n                    audio_pathsx.remove(filea)\n                    audio_pathsx.remove(fileb)\n                else:\n                    audio_pathsx = copy.deepcopy(audio_paths)\n                    stage_count +=1\n                    mixed_files=[]\n\n            else:\n                continue\n        else:\n            audio_pathsx = copy.deepcopy(audio_paths)\n            stage_count +=1\n            mixed_files=[]\n        \n        COUNTER +=1\n    bird_dict[bird_code]=[len(audio_paths),needed_num_files,mix_count]\n    for_counter +=1\nprint(\"\\n\",COUNTER)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.512101Z","iopub.execute_input":"2024-06-01T06:12:56.512493Z","iopub.status.idle":"2024-06-01T06:12:56.654124Z","shell.execute_reply.started":"2024-06-01T06:12:56.512464Z","shell.execute_reply":"2024-06-01T06:12:56.652908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# USE_RATING=3.\n# USE_SEC=5\n# SR=32000\n# MIN_SEC=3\n\n# # CURB_NUM_AUDIO = 100\n# MAX_STAGE=6\n\n# COUNTER=0\n# FILE1,FILE2 =[],[]\n# STAGE_LIST=[]\n# BIRD_CODES = []\n\n# bird_dict = {bird_name:[] for bird_name in bird_target_names}\n# for_counter = 0\n# for bird_name in tqdm(nocturnal_birds):\n#     bird_code = df_birdlist[df_birdlist.common_name==bird_name]['primary_label'].values[0]\n#     r = re.compile(f\".*{bird_code}\")\n#     audio_paths = list(filter(r.match, final_paths))\n#     needed_num_files=0\n#     mix_count=0\n# #     print(for_counter,bird_code,len(audio_paths))\n#     if bird_name==\"Spotted Owlet\":\n#         HIGH_CURB_NUM_AUDIO = len(audio_paths) + 20\n#     elif bird_name == 'Black-crowned Night-Heron':\n#         HIGH_CURB_NUM_AUDIO=20\n        \n#     else:\n#         HIGH_CURB_NUM_AUDIO = len(audio_paths) + ADD_FILES\n#     if len(audio_paths)<HIGH_CURB_NUM_AUDIO:\n#         needed_num_files = HIGH_CURB_NUM_AUDIO - len(audio_paths)\n#         mix_count= 0\n#         stage_count = 0\n#         mixed_files = []\n#         audio_pathsx = copy.deepcopy(audio_paths)\n        \n#         while (mix_count<needed_num_files) and stage_count<MAX_STAGE:\n#             if len(audio_pathsx)>1:\n# #                 print(\"\\t\\trandom_choices\",len(audio_pathsx))\n#                 filea, fileb = random.choices(audio_pathsx,k=2)\n#                 if filea!=fileb:\n#     #                 if filea not in mixed_files and fileb not in mixed_files:\n#                     if len(mixed_files)<=(len(audio_paths)):\n#                         mixed_files.append(filea)\n#                         mixed_files.append(fileb)\n#                         FILE1.append(filea)\n#                         FILE2.append(fileb)\n#                         STAGE_LIST.append(stage_count)\n#                         BIRD_CODES.append(bird_code)\n#                         mix_count +=1\n#                         audio_pathsx.remove(filea)\n#                         audio_pathsx.remove(fileb)\n#                     else:\n#                         audio_pathsx = copy.deepcopy(audio_paths)\n#                         stage_count +=1\n#                         mixed_files=[]\n            \n#                 else:\n#                     continue\n#             else:\n#                 audio_pathsx = copy.deepcopy(audio_paths)\n#                 stage_count +=1\n#                 mixed_files=[]\n        \n#         COUNTER +=1\n#     bird_dict[bird_code]=[len(audio_paths),needed_num_files,mix_count]\n#     for_counter +=1\n# print(\"\\n\",COUNTER)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.655420Z","iopub.execute_input":"2024-06-01T06:12:56.655764Z","iopub.status.idle":"2024-06-01T06:12:56.664417Z","shell.execute_reply.started":"2024-06-01T06:12:56.655735Z","shell.execute_reply":"2024-06-01T06:12:56.663121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_needed_files = 0\ngot_files=0\nfor bird_name in tqdm(nocturnal_birds):\n    bird_code = df_birdlist[df_birdlist.common_name==bird_name]['primary_label'].values[0]\n    paths,need_files,actual_files = bird_dict[bird_code]\n    total_needed_files += need_files\n    got_files += actual_files\n\nprint(\"TOTAL NEEDED FILES:\",total_needed_files)\nprint(\"ACTUAL GOT FILES:\",got_files)\nprint(\"DIFF:\",total_needed_files-got_files)\nprint(\"%:\",got_files/total_needed_files)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.666476Z","iopub.execute_input":"2024-06-01T06:12:56.666826Z","iopub.status.idle":"2024-06-01T06:12:56.688921Z","shell.execute_reply.started":"2024-06-01T06:12:56.666799Z","shell.execute_reply":"2024-06-01T06:12:56.687407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(FILE1),len(FILE2),len(STAGE_LIST),len(BIRD_CODES)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.690766Z","iopub.execute_input":"2024-06-01T06:12:56.691246Z","iopub.status.idle":"2024-06-01T06:12:56.699774Z","shell.execute_reply.started":"2024-06-01T06:12:56.691204Z","shell.execute_reply":"2024-06-01T06:12:56.698510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save_dir = 'nocturnal_mixed_self_signals'\nworking_dir = '/kaggle/working/'\n\nsave_folder_path = os.path.join(working_dir,save_dir)\nprint(save_folder_path)\n\n_ = os.makedirs(save_folder_path, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.701525Z","iopub.execute_input":"2024-06-01T06:12:56.701968Z","iopub.status.idle":"2024-06-01T06:12:56.712578Z","shell.execute_reply.started":"2024-06-01T06:12:56.701926Z","shell.execute_reply":"2024-06-01T06:12:56.711358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SAVE_MIX(FILE1,FILE2,STAGE_LIST,BIRD_CODES)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:12:56.713725Z","iopub.execute_input":"2024-06-01T06:12:56.714183Z","iopub.status.idle":"2024-06-01T06:14:18.125717Z","shell.execute_reply.started":"2024-06-01T06:12:56.714125Z","shell.execute_reply":"2024-06-01T06:14:18.122293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_noc = len(os.listdir(\"/kaggle/working/nocturnal_mixed_self_signals\"))\n\nprint(\"# Total Fles:\", num_noc)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:14:18.127099Z","iopub.execute_input":"2024-06-01T06:14:18.128072Z","iopub.status.idle":"2024-06-01T06:14:18.139423Z","shell.execute_reply.started":"2024-06-01T06:14:18.128029Z","shell.execute_reply":"2024-06-01T06:14:18.137770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selfmix_paths = glob('/kaggle/working/*/*.npy')\n\nprint(selfmix_paths[0])\nprint(\"total selfmix audio files:\", len(selfmix_paths))","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:14:18.141113Z","iopub.execute_input":"2024-06-01T06:14:18.141683Z","iopub.status.idle":"2024-06-01T06:14:18.155308Z","shell.execute_reply.started":"2024-06-01T06:14:18.141642Z","shell.execute_reply":"2024-06-01T06:14:18.153810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter=0\nfor path in selfmix_paths:\n    x = np.load(path)\n    if np.isnan(x).any():\n        counter +=1\ncounter","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:14:18.157766Z","iopub.execute_input":"2024-06-01T06:14:18.158801Z","iopub.status.idle":"2024-06-01T06:14:18.370837Z","shell.execute_reply.started":"2024-06-01T06:14:18.158759Z","shell.execute_reply":"2024-06-01T06:14:18.369825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_df = pd.DataFrame(selfmix_paths, columns=['file_path'])\npath_df['new_target'] = path_df['file_path'].map(lambda x: x.split(\"/\")[-1].split(\"_\")[0])\nprint(path_df.shape)\npath_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:14:18.373949Z","iopub.execute_input":"2024-06-01T06:14:18.374773Z","iopub.status.idle":"2024-06-01T06:14:18.389917Z","shell.execute_reply.started":"2024-06-01T06:14:18.374729Z","shell.execute_reply":"2024-06-01T06:14:18.388688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for bird_name in nocturnal_birds:\n    bird_code = df_birdlist[df_birdlist.common_name==bird_name]['primary_label'].values[0]\n    num_paths = len(os.listdir(f\"/kaggle/input/birdclef-2024/train_audio/{bird_code}\"))\n    num_files = len(path_df[path_df.new_target==bird_code])\n    print(bird_name,num_paths,num_files)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:14:18.391470Z","iopub.execute_input":"2024-06-01T06:14:18.392226Z","iopub.status.idle":"2024-06-01T06:14:18.419407Z","shell.execute_reply.started":"2024-06-01T06:14:18.392184Z","shell.execute_reply":"2024-06-01T06:14:18.418573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for bird_name in list(df_birdlist.common_name.values):\n    bird_code = df_birdlist[df_birdlist.common_name==bird_name]['primary_label'].values[0]\n    num_paths = len(os.listdir(f\"/kaggle/input/birdclef-2024/train_audio/{bird_code}\"))\n    num_files = len(path_df[path_df.new_target==bird_code])\n    if num_files !=0:\n#         if num_files>100:\n        print(bird_name,num_paths,num_files)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T06:14:18.420676Z","iopub.execute_input":"2024-06-01T06:14:18.421597Z","iopub.status.idle":"2024-06-01T06:14:18.722317Z","shell.execute_reply.started":"2024-06-01T06:14:18.421565Z","shell.execute_reply":"2024-06-01T06:14:18.721446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}