{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":8096443,"sourceType":"datasetVersion","datasetId":4780521},{"sourceId":8318251,"sourceType":"datasetVersion","datasetId":4940719},{"sourceId":8416940,"sourceType":"datasetVersion","datasetId":5010206},{"sourceId":8426650,"sourceType":"datasetVersion","datasetId":5017621},{"sourceId":170596805,"sourceType":"kernelVersion"},{"sourceId":170597150,"sourceType":"kernelVersion"},{"sourceId":170597237,"sourceType":"kernelVersion"},{"sourceId":170597938,"sourceType":"kernelVersion"},{"sourceId":170598083,"sourceType":"kernelVersion"},{"sourceId":170598279,"sourceType":"kernelVersion"},{"sourceId":170598684,"sourceType":"kernelVersion"},{"sourceId":175005679,"sourceType":"kernelVersion"},{"sourceId":175993701,"sourceType":"kernelVersion"}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### No nocturnal birds,mixed them separately","metadata":{}},{"cell_type":"code","source":"import gc\nimport os\nimport sys\n# sys.path.append('../input/pytorch-image-models/pytorch-image-models-master')\nimport random\nimport time\nimport warnings\nimport re\nimport copy\n\nimport librosa\nimport numpy as np\nimport pandas as pd\nimport soundfile as sf\n\n\n\nfrom contextlib import contextmanager\nfrom joblib import Parallel, delayed\nfrom pathlib import Path\nfrom tqdm import tqdm\nfrom glob import glob\n\nimport matplotlib.pyplot as plt\nimport librosa.display","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:12:47.161723Z","iopub.execute_input":"2024-05-30T10:12:47.162133Z","iopub.status.idle":"2024-05-30T10:12:48.472984Z","shell.execute_reply.started":"2024-05-30T10:12:47.162101Z","shell.execute_reply":"2024-05-30T10:12:48.471718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import Audio\nSR=32000\nless_than = 50","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:12:48.475469Z","iopub.execute_input":"2024-05-30T10:12:48.476613Z","iopub.status.idle":"2024-05-30T10:12:48.481841Z","shell.execute_reply.started":"2024-05-30T10:12:48.476568Z","shell.execute_reply":"2024-05-30T10:12:48.480573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seed(seed=43):\n    random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    np.random.seed(seed)\n    \nset_seed(43)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:12:48.483341Z","iopub.execute_input":"2024-05-30T10:12:48.484190Z","iopub.status.idle":"2024-05-30T10:12:48.494860Z","shell.execute_reply.started":"2024-05-30T10:12:48.484148Z","shell.execute_reply":"2024-05-30T10:12:48.493579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_paths = glob(\"/kaggle/input/1negative-samples-npy-birdclef2024/noise/*.npy\")\nprint(len(n_paths))\nNP=[]\nfor path in n_paths:\n    y = np.load(path)\n    length = len(y)/32000\n    if length==5.:\n#         print(path,len(y)/32000)\n        NP.append(path)\nprint(len(NP))","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:12:48.496490Z","iopub.execute_input":"2024-05-30T10:12:48.496983Z","iopub.status.idle":"2024-05-30T10:12:49.569800Z","shell.execute_reply.started":"2024-05-30T10:12:48.496943Z","shell.execute_reply":"2024-05-30T10:12:49.568711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dur = pd.read_csv(\"/kaggle/input/2duration-log/bc2024_duration.csv\")\nprint(dur.shape)\ndur.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:12:49.573924Z","iopub.execute_input":"2024-05-30T10:12:49.574284Z","iopub.status.idle":"2024-05-30T10:12:49.827666Z","shell.execute_reply.started":"2024-05-30T10:12:49.574255Z","shell.execute_reply":"2024-05-30T10:12:49.826491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def find_audio_paths_birdcode(bird_code):\n    audio_paths = glob(f'/kaggle/input/1npy-birdclef2024/train_npy0/{bird_code}/*.npy')\\\n                    + glob(f'/kaggle/input/2npy-birdclef2024/train_npy1/{bird_code}/*.npy')\\\n                    + glob(f'/kaggle/input/3npy-birdclef2024/train_npy2/{bird_code}/*.npy')\\\n                    + glob(f'/kaggle/input/4npy-birdclef2024/train_npy3/{bird_code}/*.npy')\\\n                    + glob(f'/kaggle/input/5npy-birdclef2024/train_npy4/{bird_code}/*.npy')\\\n                    + glob(f'/kaggle/input/6npy-birdclef2024/train_npy5/{bird_code}/*.npy')\\\n                    + glob(f'/kaggle/input/7npy-birdclef2024/train_npy6/{bird_code}/*.npy')\n    return audio_paths\n\ndef find_audio_paths_birdcode_filetags(bird_code,file_tag):\n    audio_paths = glob(f'/kaggle/input/1npy-birdclef2024/train_npy0/{bird_code}/{file_tag}.npy')\\\n                    + glob(f'/kaggle/input/2npy-birdclef2024/train_npy1/{bird_code}/{file_tag}.npy')\\\n                    + glob(f'/kaggle/input/3npy-birdclef2024/train_npy2/{bird_code}/{file_tag}.npy')\\\n                    + glob(f'/kaggle/input/4npy-birdclef2024/train_npy3/{bird_code}/{file_tag}.npy')\\\n                    + glob(f'/kaggle/input/5npy-birdclef2024/train_npy4/{bird_code}/{file_tag}.npy')\\\n                    + glob(f'/kaggle/input/6npy-birdclef2024/train_npy5/{bird_code}/{file_tag}.npy')\\\n                    + glob(f'/kaggle/input/7npy-birdclef2024/train_npy6/{bird_code}/{file_tag}.npy')\n    return audio_paths","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:12:49.829002Z","iopub.execute_input":"2024-05-30T10:12:49.829372Z","iopub.status.idle":"2024-05-30T10:12:49.837948Z","shell.execute_reply.started":"2024-05-30T10:12:49.829344Z","shell.execute_reply":"2024-05-30T10:12:49.836745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_paths = glob('/kaggle/input/1npy-birdclef2024/train_npy0/*/*.npy')\\\n                + glob('/kaggle/input/2npy-birdclef2024/train_npy1/*/*.npy')\\\n                + glob('/kaggle/input/3npy-birdclef2024/train_npy2/*/*.npy')\\\n                + glob('/kaggle/input/4npy-birdclef2024/train_npy3/*/*.npy')\\\n                + glob('/kaggle/input/5npy-birdclef2024/train_npy4/*/*.npy')\\\n                + glob('/kaggle/input/6npy-birdclef2024/train_npy5/*/*.npy')\\\n                + glob('/kaggle/input/7npy-birdclef2024/train_npy6/*/*.npy')\n\n# all_path = glob.glob(\"/kaggle/input/birdclef-2024/train_audio/*/*.ogg\")\nprint(all_paths[0])\nprint(len(all_paths))","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:12:49.839571Z","iopub.execute_input":"2024-05-30T10:12:49.839994Z","iopub.status.idle":"2024-05-30T10:12:59.209392Z","shell.execute_reply.started":"2024-05-30T10:12:49.839957Z","shell.execute_reply":"2024-05-30T10:12:59.208299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"txt_file = \"/kaggle/input/duplicates-calls-bc2024/duplicates_bird_calls.txt\"\nDL =[]\nwith open(txt_file,'r') as file:\n    lines = file.readlines()\n    for line in lines:\n        DL.append(line)\nfile.close()\nprint(len(DL))\nprint(DL[0])\n\ndiff_birds=[]\nsame_birds=[]\nfor item in DL:\n    a,b = item.split(\",\")[0],item.split(\",\")[1]\n    if a.split(\"/\")[0] != b.split(\"/\")[0]:\n        fla = a.split(\"/\")[1].split(\".\")[0]\n        flb = b.split(\"/\")[1].split(\".\")[0]\n        if fla!=flb:\n            diff_birds.append((a,b.strip()))\n    else:\n#         same_birds.append((a,b.strip()))\n        same_birds.append(a)\nprint(len(diff_birds),len(same_birds))\nprint(diff_birds[0],same_birds[0])","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:12:59.210766Z","iopub.execute_input":"2024-05-30T10:12:59.211105Z","iopub.status.idle":"2024-05-30T10:12:59.227792Z","shell.execute_reply.started":"2024-05-30T10:12:59.211077Z","shell.execute_reply":"2024-05-30T10:12:59.226429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#remove the same bird calls from paths\nval = len(all_paths)\nprint(len(all_paths))\ntemp_FL=[]\nfor n,filename in enumerate(same_birds):\n    bird_name = filename.split(\"/\")[0]\n    file_tag = filename.split(\"/\")[1].split(\".\")[0]\n    path = find_audio_paths_birdcode_filetags(bird_name,file_tag)\n#     print(n,path[0])\n    if file_tag not in temp_FL:\n        all_paths.remove(path[0])\n    temp_FL.append(file_tag)\n    \nprint(len(all_paths))\nprint(\"diff:\", val - len(all_paths))","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:12:59.229230Z","iopub.execute_input":"2024-05-30T10:12:59.229630Z","iopub.status.idle":"2024-05-30T10:12:59.646643Z","shell.execute_reply.started":"2024-05-30T10:12:59.229599Z","shell.execute_reply":"2024-05-30T10:12:59.645210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FOLDER = \"/kaggle/input/birdclef-2024\"\ntrain_dir =\"/kaggle/input/birdclef-2024/train_audio\"\nAUDIO_DURATION=5.","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:12:59.648371Z","iopub.execute_input":"2024-05-30T10:12:59.648801Z","iopub.status.idle":"2024-05-30T10:12:59.654004Z","shell.execute_reply.started":"2024-05-30T10:12:59.648767Z","shell.execute_reply":"2024-05-30T10:12:59.652866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import ast\n\n\ntrain = pd.read_csv(os.path.join(FOLDER,\"train_metadata.csv\"))\n\n\ntrain['new_target'] = train['primary_label'] + ' ' + train['secondary_labels'].map(lambda x: ' '.join(ast.literal_eval(x)))\n# train['len_new_target'] = train['new_target'].map(lambda x: len(x.split()))\n# train['len_new_target'].value_counts()\ntrain['file_tag'] = train['filename'].map(lambda x: x.split(\".\")[0].split(\"/\")[-1])\nprint(train.shape)\ntrain.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:12:59.655271Z","iopub.execute_input":"2024-05-30T10:12:59.655703Z","iopub.status.idle":"2024-05-30T10:13:00.100576Z","shell.execute_reply.started":"2024-05-30T10:12:59.655674Z","shell.execute_reply":"2024-05-30T10:13:00.099575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter=0\nfor n in range(len(train)):\n    pl = train.iloc[n]['primary_label']\n    bird_list = train.iloc[n]['new_target'].split()\n    temp_birdlist=[]\n    temp_birdlist.append(pl)\n    if len(bird_list)>1:\n        for idx in range(len(bird_list)):\n            if idx>0:\n                bird_name = bird_list[idx]\n                if bird_name!=pl:\n                    temp_birdlist.append(bird_name)\n        names = \" \".join(temp_birdlist)\n#         print(pl,names)\n        counter +=1\n        train.loc[n,'new_target']= names\nprint(counter)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:00.101915Z","iopub.execute_input":"2024-05-30T10:13:00.102257Z","iopub.status.idle":"2024-05-30T10:13:04.422874Z","shell.execute_reply.started":"2024-05-30T10:13:00.102220Z","shell.execute_reply":"2024-05-30T10:13:04.421792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for a,b in diff_birds:\n    fla = a.split(\"/\")[1].split(\".\")[0]\n    flb = b.split(\"/\")[1].split(\".\")[0]\n    idx = train[train.file_tag==flb].index\n    train.at[idx.values[0],'file_tag']=fla","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:04.424201Z","iopub.execute_input":"2024-05-30T10:13:04.424506Z","iopub.status.idle":"2024-05-30T10:13:04.495711Z","shell.execute_reply.started":"2024-05-30T10:13:04.424481Z","shell.execute_reply":"2024-05-30T10:13:04.494602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique, counts = np.unique(train.file_tag, return_counts=True)\ncounts_dict = dict(zip(unique, counts))\nprint(len(counts_dict), len(train), len(train)-len(counts_dict))\n\ntemp = {k: v for k, v in sorted(counts_dict.items(), key=lambda item: item[1],reverse=True)}\ntemp = dict(list(temp.items())[:19])\ntags = temp.keys()\nprint(list(tags))\n\nfor tag in tags:\n    temp = list(train[train.file_tag==tag].new_target.values)\n    indxs = train[train.file_tag==tag].index\n#     print(temp)\n    names = \" \".join(temp)\n    for indx in indxs:\n        train.at[indx,'new_target']= names\n        \nlist(train[train.file_tag=='XC574864'].new_target.values)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:04.501055Z","iopub.execute_input":"2024-05-30T10:13:04.501730Z","iopub.status.idle":"2024-05-30T10:13:04.796641Z","shell.execute_reply.started":"2024-05-30T10:13:04.501692Z","shell.execute_reply":"2024-05-30T10:13:04.795571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val = len(train)\nprint(len(train))\ntrain.drop_duplicates(subset='file_tag',inplace=True)\ntrain.reset_index(drop=True, inplace=True)\ntrain['len_new_target'] = train['new_target'].map(lambda x: len(x.split()))\nprint(len(train))\nprint(\"diff:\",val-len(train))","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:04.798132Z","iopub.execute_input":"2024-05-30T10:13:04.798533Z","iopub.status.idle":"2024-05-30T10:13:04.842262Z","shell.execute_reply.started":"2024-05-30T10:13:04.798483Z","shell.execute_reply":"2024-05-30T10:13:04.840984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"secondary_labels = ['asfblu1','indwhe1','bltmun1','magrob','lotshr1','orhthr1']\nss = pd.read_csv(\"/kaggle/input/birdclef-2024/sample_submission.csv\")\nbird_target_names = list(ss.columns)\nbird_target_names.pop(0)\nlen(bird_target_names)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:04.844165Z","iopub.execute_input":"2024-05-30T10:13:04.844666Z","iopub.status.idle":"2024-05-30T10:13:04.864874Z","shell.execute_reply.started":"2024-05-30T10:13:04.844626Z","shell.execute_reply":"2024-05-30T10:13:04.863798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"common_names = list(pd.unique(train.common_name))\nlen(common_names)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:04.866667Z","iopub.execute_input":"2024-05-30T10:13:04.867140Z","iopub.status.idle":"2024-05-30T10:13:04.879793Z","shell.execute_reply.started":"2024-05-30T10:13:04.867063Z","shell.execute_reply":"2024-05-30T10:13:04.878644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tax = pd.read_csv(\"/kaggle/input/birdclef-2024/eBird_Taxonomy_v2021.csv\")\nprint(tax.shape)\ntax.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:04.882560Z","iopub.execute_input":"2024-05-30T10:13:04.883038Z","iopub.status.idle":"2024-05-30T10:13:04.986265Z","shell.execute_reply.started":"2024-05-30T10:13:04.882999Z","shell.execute_reply":"2024-05-30T10:13:04.984880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SL= tax[tax.SPECIES_CODE.isin(secondary_labels)][['SPECIES_CODE','PRIMARY_COM_NAME']]\nSL.reset_index(drop=True, inplace=True)\nSL = SL.rename(columns={'SPECIES_CODE':'primary_label','PRIMARY_COM_NAME':'common_name'})\nSL","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:04.987859Z","iopub.execute_input":"2024-05-30T10:13:04.988272Z","iopub.status.idle":"2024-05-30T10:13:05.009963Z","shell.execute_reply.started":"2024-05-30T10:13:04.988232Z","shell.execute_reply":"2024-05-30T10:13:05.008600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_df = pd.DataFrame(all_paths, columns=['file_path'])\n#path_df['filename'] = path_df['file_path'].map(lambda x: x.split('/')[-2]+'/'+x.split('/')[-1][:-4])\npath_df[\"filename\"] = path_df['file_path'].map(lambda x: x.split(\"/\")[-2] + \"/\" +\\\n                                               x.split(\"/\")[-1].split('.')[-2] + \".ogg\")\n\npath_df['file_tag2'] = path_df['filename'].map(lambda x: x.split(\".\")[0].split(\"/\")[-1])\nprint(path_df.shape)\npath_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:05.011926Z","iopub.execute_input":"2024-05-30T10:13:05.012301Z","iopub.status.idle":"2024-05-30T10:13:05.102495Z","shell.execute_reply.started":"2024-05-30T10:13:05.012264Z","shell.execute_reply":"2024-05-30T10:13:05.101342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for a,b in diff_birds:\n    fla = a.split(\"/\")[1].split(\".\")[0]\n    flb = b.split(\"/\")[1].split(\".\")[0]\n    idx = path_df[path_df.file_tag2==flb].index\n    path_df.at[idx.values[0],'file_tag2']=fla","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:05.104096Z","iopub.execute_input":"2024-05-30T10:13:05.104559Z","iopub.status.idle":"2024-05-30T10:13:05.169835Z","shell.execute_reply.started":"2024-05-30T10:13:05.104497Z","shell.execute_reply":"2024-05-30T10:13:05.168709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique, counts = np.unique(path_df.file_tag2, return_counts=True)\ncounts_dict = dict(zip(unique, counts))\nprint(len(counts_dict), len(path_df), len(path_df)-len(counts_dict))\n\ntemp = {k: v for k, v in sorted(counts_dict.items(), key=lambda item: item[1],reverse=True)}\ntemp = dict(list(temp.items())[:19])\ntags = temp.keys()\nprint(temp)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:05.171041Z","iopub.execute_input":"2024-05-30T10:13:05.171362Z","iopub.status.idle":"2024-05-30T10:13:05.247979Z","shell.execute_reply.started":"2024-05-30T10:13:05.171335Z","shell.execute_reply":"2024-05-30T10:13:05.246584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val = len(path_df)\nprint(path_df.shape)\npath_df.drop_duplicates(subset='file_tag2',inplace=True)\npath_df.reset_index(drop=True, inplace=True)\nprint(path_df.shape)\nprint(\"diff:\",val-len(path_df))","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:05.249232Z","iopub.execute_input":"2024-05-30T10:13:05.249587Z","iopub.status.idle":"2024-05-30T10:13:05.262131Z","shell.execute_reply.started":"2024-05-30T10:13:05.249557Z","shell.execute_reply":"2024-05-30T10:13:05.261018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_df['file_tag'] = path_df['file_tag2'].map(lambda x: x.split(\"_\")[0])\npath_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:05.263718Z","iopub.execute_input":"2024-05-30T10:13:05.264039Z","iopub.status.idle":"2024-05-30T10:13:05.293409Z","shell.execute_reply.started":"2024-05-30T10:13:05.264013Z","shell.execute_reply":"2024-05-30T10:13:05.292308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.merge(dur[['filename','duration']])\nprint(train.shape)\ntrain.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:05.294795Z","iopub.execute_input":"2024-05-30T10:13:05.295111Z","iopub.status.idle":"2024-05-30T10:13:05.357393Z","shell.execute_reply.started":"2024-05-30T10:13:05.295083Z","shell.execute_reply":"2024-05-30T10:13:05.356405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final = path_df.merge(train, on=['file_tag'])\nprint(final.shape)\nfinal.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:05.358863Z","iopub.execute_input":"2024-05-30T10:13:05.359219Z","iopub.status.idle":"2024-05-30T10:13:05.424958Z","shell.execute_reply.started":"2024-05-30T10:13:05.359188Z","shell.execute_reply":"2024-05-30T10:13:05.423929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_birdlist = train[['common_name','primary_label']]\ndf_birdlist.drop_duplicates(inplace=True)\ndf_birdlist.reset_index(drop=True, inplace=True)\ndf_birdlist","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:05.426436Z","iopub.execute_input":"2024-05-30T10:13:05.426821Z","iopub.status.idle":"2024-05-30T10:13:05.451767Z","shell.execute_reply.started":"2024-05-30T10:13:05.426790Z","shell.execute_reply":"2024-05-30T10:13:05.450563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_audio = {}\nfor species in bird_target_names:\n    num_audio_files = os.listdir(os.path.join(train_dir,species))\n#     print(species,len(num_audio_files))\n    num_audio[species]=len(num_audio_files)\n    \nnew_dict = {k: v for k, v in sorted(num_audio.items(), key=lambda item: item[1])}\nprint(len(new_dict))\n\nnum_audio = pd.DataFrame(new_dict.items(), columns=['primary_label', 'NUM_AUDIO_FILES',])\nnum_audio = num_audio.merge(df_birdlist,)\nprint(num_audio.shape)\n# num_audio = num_audio.rename(columns={'SPECIES_CODE':'primary_label'})\nnum_audio.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:05.453161Z","iopub.execute_input":"2024-05-30T10:13:05.453505Z","iopub.status.idle":"2024-05-30T10:13:10.720995Z","shell.execute_reply.started":"2024-05-30T10:13:05.453476Z","shell.execute_reply":"2024-05-30T10:13:10.719839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"group_duration = final.groupby('primary_label')['duration'].sum()\ngroup_dur = pd.merge(num_audio,pd.DataFrame(group_duration).reset_index())\nprint(group_dur.shape)\n\ngroup_dur['5_second_duration'] = np.round(group_dur['duration']/AUDIO_DURATION,0)\ngroup_dur['5_second_duration'].describe()\n\ngroup_dur.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:10.722440Z","iopub.execute_input":"2024-05-30T10:13:10.722811Z","iopub.status.idle":"2024-05-30T10:13:10.756388Z","shell.execute_reply.started":"2024-05-30T10:13:10.722782Z","shell.execute_reply":"2024-05-30T10:13:10.755364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### NOISE","metadata":{}},{"cell_type":"code","source":"n_paths = glob(\"/kaggle/input/1negative-samples-npy-birdclef2024/noise/*.npy\")\nprint(len(n_paths))\nNOISE_PATHS=[]\nfor path in n_paths:\n    y = np.load(path)\n    length = len(y)/32000\n    if length==5.:\n#         print(path,len(y)/32000)\n        NOISE_PATHS.append(path)\nprint(len(NOISE_PATHS))","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:10.757817Z","iopub.execute_input":"2024-05-30T10:13:10.758148Z","iopub.status.idle":"2024-05-30T10:13:10.868257Z","shell.execute_reply.started":"2024-05-30T10:13:10.758119Z","shell.execute_reply":"2024-05-30T10:13:10.866883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# bird_matrix = np.zeros((len(bird_target_names),len(bird_target_names)),dtype=np.int16)\n# bird_matrix2 = np.zeros((len(bird_target_names),len(bird_target_names)),dtype=np.int16)\n# print(bird_matrix.shape)\n\n# for n in tqdm(range(len(train))):\n#     pl = train.iloc[n]['primary_label']\n#     bird_list = train.iloc[n]['new_target'].split()\n#     pl_indx= bird_target_names.index(pl)\n#     bird_matrix[pl_indx,pl_indx] +=1\n#     bird_matrix2[pl_indx,pl_indx] +=1\n#     if len(bird_list)>1:\n#         for idx in range(len(bird_list)):\n#             if idx>0:\n#                 sl = bird_list[idx]\n#                 if sl!=pl:\n#                     if sl not in secondary_labels:\n#                         sl_indx = bird_target_names.index(sl)\n#                         bird_matrix[pl_indx,sl_indx] +=1\n#                         bird_matrix2[sl_indx,pl_indx] +=1\n#                         bird_matrix2[pl_indx,sl_indx] +=1                        ","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:10.869721Z","iopub.execute_input":"2024-05-30T10:13:10.870077Z","iopub.status.idle":"2024-05-30T10:13:10.875326Z","shell.execute_reply.started":"2024-05-30T10:13:10.870046Z","shell.execute_reply":"2024-05-30T10:13:10.874235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PL = final[final.len_new_target==1]\nPL.reset_index(drop=True, inplace=True)\nprint(PL.shape)\nPL.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:10.876696Z","iopub.execute_input":"2024-05-30T10:13:10.877062Z","iopub.status.idle":"2024-05-30T10:13:10.919604Z","shell.execute_reply.started":"2024-05-30T10:13:10.877034Z","shell.execute_reply":"2024-05-30T10:13:10.918391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### WATER BASED BIRDS","metadata":{}},{"cell_type":"code","source":"water = pd.read_csv(\"/kaggle/input/bc2024-water-tagged-bird-list/watertagged_bird_list2 - final_bird_list.csv\")\nwater.rename(columns={'Unnamed: 2': 'water_tag',},inplace=True)\nwater.fillna('n',inplace=True)\nprint(water.shape)\nwater.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:10.920990Z","iopub.execute_input":"2024-05-30T10:13:10.921318Z","iopub.status.idle":"2024-05-30T10:13:10.942961Z","shell.execute_reply.started":"2024-05-30T10:13:10.921286Z","shell.execute_reply":"2024-05-30T10:13:10.941602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"water_birds = list(water[water.water_tag=='w']['PRIMARY_COM_NAME'].values)\nprint(len(water_birds))\n\nwater_group = group_dur[group_dur.common_name.isin(water_birds)].merge(df_birdlist)\nwater_group.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:10.944713Z","iopub.execute_input":"2024-05-30T10:13:10.945426Z","iopub.status.idle":"2024-05-30T10:13:10.967779Z","shell.execute_reply.started":"2024-05-30T10:13:10.945385Z","shell.execute_reply":"2024-05-30T10:13:10.966605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bird_groups = { 'low_elevation': ['Zitting Cisticola','Plain Prinia','Rufous Treepie','Small Minivet','Gray-headed Swamphen',\n                   'Asian Koel','Laughing Dove','Gray Francolin',],\n  'low-mid_elevation':['Paddyfield Pipit','Common Iora','White-throated Kingfisher',\n#                        'Spotted Owlet',\n                      'Painted Stork','Asian Openbill','Spotted Dove','Red Spurfowl'],\n 'unlikely':['houspa','Brahminy Kite','Eurasian Marsh-Harrier','Eurasian Collared-Dove',\n            'Rock Pigeon','Gray Francolin'],}","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:10.969574Z","iopub.execute_input":"2024-05-30T10:13:10.970020Z","iopub.status.idle":"2024-05-30T10:13:10.978476Z","shell.execute_reply.started":"2024-05-30T10:13:10.969982Z","shell.execute_reply":"2024-05-30T10:13:10.977542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### LOW ELEVATION","metadata":{}},{"cell_type":"code","source":"low_elevation_birds = list(pd.unique(PL[PL.common_name.isin(bird_groups['low_elevation'])]['common_name']))\nprint(len(low_elevation_birds))\nprint(\"total # audio files:\",group_dur[group_dur.common_name.isin(low_elevation_birds)]['NUM_AUDIO_FILES'].sum())\nprint(\"Possible total # audio files:\",group_dur[group_dur.common_name.\\\n                                                isin(low_elevation_birds)]['5_second_duration'].sum())","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:10.979824Z","iopub.execute_input":"2024-05-30T10:13:10.980158Z","iopub.status.idle":"2024-05-30T10:13:11.001413Z","shell.execute_reply.started":"2024-05-30T10:13:10.980129Z","shell.execute_reply":"2024-05-30T10:13:11.000007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### LOW-MID ELEVATION","metadata":{}},{"cell_type":"code","source":"low_mid_birds  = list(pd.unique(PL[PL.common_name.isin(bird_groups['low-mid_elevation'])]['common_name']))\nprint(len(low_mid_birds))\nprint(\"total # audio files:\",group_dur[group_dur.common_name.isin(low_mid_birds)]['NUM_AUDIO_FILES'].sum())\nprint(\"Possible total # audio files:\",group_dur[group_dur.common_name.\\\n                                                isin(low_mid_birds)]['5_second_duration'].sum())","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:11.002946Z","iopub.execute_input":"2024-05-30T10:13:11.003400Z","iopub.status.idle":"2024-05-30T10:13:11.016835Z","shell.execute_reply.started":"2024-05-30T10:13:11.003369Z","shell.execute_reply":"2024-05-30T10:13:11.015667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### UNLIKELY","metadata":{}},{"cell_type":"code","source":"unlikely_birds = list(pd.unique(PL[PL.common_name.isin(bird_groups['unlikely'])]['common_name']))\nprint(len(unlikely_birds))\nprint(\"total # audio files:\",group_dur[group_dur.common_name.isin(unlikely_birds)]['NUM_AUDIO_FILES'].sum())\nprint(\"Possible total # audio files:\",group_dur[group_dur.common_name.\\\n                                                isin(unlikely_birds)]['5_second_duration'].sum())","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:11.018202Z","iopub.execute_input":"2024-05-30T10:13:11.018615Z","iopub.status.idle":"2024-05-30T10:13:11.036638Z","shell.execute_reply.started":"2024-05-30T10:13:11.018585Z","shell.execute_reply":"2024-05-30T10:13:11.035369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# eucdov - low\n# Eurasian Marsh-Harrier - open \n# rock pigeon - jog falls, humans\n# gray francolin - low grasslands,entry of naraikadu, low\n# brahminy kite - open, waterbody,","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:11.038099Z","iopub.execute_input":"2024-05-30T10:13:11.038546Z","iopub.status.idle":"2024-05-30T10:13:11.045939Z","shell.execute_reply.started":"2024-05-30T10:13:11.038486Z","shell.execute_reply":"2024-05-30T10:13:11.044684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### HIGH PRIORITY BIRDS","metadata":{}},{"cell_type":"code","source":"high_priority_birds = ['Gray Junglefowl', 'Malabar Whistling-Thrush', 'Malabar Barbet', 'White-cheeked Barbet',\n                       'Vernal Hanging-Parrot', 'Rufous Babbler', 'Southern Hill Myna', 'Dark-fronted Babbler', \n                       'Nilgiri Wood-Pigeon', 'Malabar Parakeet', 'Crimson-backed Sunbird', 'Orange Minivet',\n                       'White-bellied Blue Flycatcher', 'Malabar Woodshrike', 'Gray-fronted Green-Pigeon',\n                       'Nilgiri Flycatcher', 'Great Hornbill', 'Square-tailed Bulbul', 'Black-and-orange Flycatcher',\n                       'Nilgiri Flowerpecker', 'Yellow-browed Bulbul', 'Indian Yellow Tit', 'Malabar Trogon', \n                       'Jungle Myna', \"Loten's Sunbird\", 'Palani Laughingthrush', 'Common Flameback', 'White-bellied Woodpecker', \n                       'White-bellied Treepie', 'White-bellied Sholakili', 'Malabar Gray Hornbill', 'Wayanad Laughingthrush', \n                       'Flame-throated Bulbul',\n#                         'Brown Wood-Owl','Spot-bellied Eagle-Owl','Brown Fish-Owl','Jungle Owlet','Brown Boobook',\n#                        'Indian Scops-Owl',\n#                        'Great Eared-Nightjar', 'Jungle Nightjar',\n                      ]\nprint(len(high_priority_birds))\nhpb = list(pd.unique(PL[PL.common_name.isin(high_priority_birds)]['primary_label']))\nprint(\"total # audio files:\",group_dur[group_dur.primary_label.isin(hpb)]['NUM_AUDIO_FILES'].sum())\nprint(\"possible total # audio files:\",group_dur[group_dur.primary_label.isin(hpb)]['5_second_duration'].sum())\nprint()\ngroup_dur[group_dur.primary_label.isin(hpb)].shape","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:11.047262Z","iopub.execute_input":"2024-05-30T10:13:11.047690Z","iopub.status.idle":"2024-05-30T10:13:11.069491Z","shell.execute_reply.started":"2024-05-30T10:13:11.047656Z","shell.execute_reply":"2024-05-30T10:13:11.068299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mid_priority_birds = ['Brown-capped Pygmy Woodpecker', 'Chestnut-headed Bee-eater','Crested Goshawk', 'Velvet-fronted Nuthatch', \n                      \"Jerdon's Bushlark\", 'Indian Scimitar-Babbler', 'Plum-headed Parakeet', 'Speckled Piculet', \n                      'Rufous Woodpecker', 'Asian Emerald Dove', 'Golden-fronted Leafbird', 'Green Warbler', \n                      'Indian Blackbird', 'Heart-spotted Woodpecker', 'Little Spiderhunter', 'Rusty-tailed Flycatcher', \n                      'Red-whiskered Bulbul', 'White-browed Bulbul', 'Streak-throated Woodpecker', 'Stork-billed Kingfisher', \n                      'White-rumped Munia', 'Large-billed Leaf Warbler', 'Yellow-billed Babbler', 'Bar-winged Flycatcher-shrike', \n                      'Indian Blue Robin', \"Tickell's Leaf Warbler\"]\nprint(len(mid_priority_birds))\nmpb = list(pd.unique(PL[PL.common_name.isin(mid_priority_birds)]['primary_label']))\nprint(\"total # audio files:\",group_dur[group_dur.primary_label.isin(mpb)]['NUM_AUDIO_FILES'].sum())\nprint(\"possible total # audio files:\",group_dur[group_dur.primary_label.isin(mpb)]['5_second_duration'].sum())\nprint()\ngroup_dur[group_dur.primary_label.isin(mpb)].shape\n","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:11.081303Z","iopub.execute_input":"2024-05-30T10:13:11.081710Z","iopub.status.idle":"2024-05-30T10:13:11.100945Z","shell.execute_reply.started":"2024-05-30T10:13:11.081679Z","shell.execute_reply":"2024-05-30T10:13:11.099917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"low_priority_birds = [\"Forest Wagtail\",\"Greater Racket-tailed Drongo\",\"Gray-headed Canary-Flycatcher\",\n                     \"Indian Pitta\",\"Jungle Babbler\",\"Lesser Yellownape\",\"Pale-billed Flowerpecker\",\n                     \"Gray-bellied Cuckoo\",\"Purple-rumped Sunbird\",\"Thick-billed Warbler\",\n                     \"Tickell's Blue Flycatcher\",]\n\nprint(len(low_priority_birds))\nlpb = list(pd.unique(PL[PL.common_name.isin(low_priority_birds)]['primary_label']))\nprint(\"total # audio files:\",group_dur[group_dur.primary_label.isin(lpb)]['NUM_AUDIO_FILES'].sum())\nprint(\"possible total # audio files:\",group_dur[group_dur.primary_label.isin(lpb)]['5_second_duration'].sum())\nprint()\ngroup_dur[group_dur.primary_label.isin(lpb)]","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:11.102285Z","iopub.execute_input":"2024-05-30T10:13:11.102682Z","iopub.status.idle":"2024-05-30T10:13:11.135938Z","shell.execute_reply.started":"2024-05-30T10:13:11.102646Z","shell.execute_reply":"2024-05-30T10:13:11.134661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### NOCTURNAL BIRDS","metadata":{}},{"cell_type":"code","source":"nb = pd.read_csv(\"/kaggle/input/bc2024-nocturnal-dirunal-birds/Nocturnal_bird_list - final_bird_list.csv\")\nprint(nb.shape)\nnocturnal_birds = list(nb[nb['Dirunal/Nocturnal']=='n']['PRIMARY_COM_NAME'].values)\nnocturnal_birds.remove('Black-crowned Night-Heron')\nprint(nocturnal_birds)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:11.137423Z","iopub.execute_input":"2024-05-30T10:13:11.137824Z","iopub.status.idle":"2024-05-30T10:13:11.157040Z","shell.execute_reply.started":"2024-05-30T10:13:11.137783Z","shell.execute_reply":"2024-05-30T10:13:11.155564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bird_namex = list(pd.unique(PL[PL.common_name.isin(nocturnal_birds)]['primary_label']))\nprint(len(bird_namex))\nprint(\"total # audio files:\",group_dur[group_dur.primary_label.isin(bird_namex)]['NUM_AUDIO_FILES'].sum())\nprint(\"possible total # audio files:\",group_dur[group_dur.primary_label.isin(bird_namex)]['5_second_duration'].sum())\nprint()\ngroup_dur[group_dur.primary_label.isin(bird_namex)]","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:11.159062Z","iopub.execute_input":"2024-05-30T10:13:11.159410Z","iopub.status.idle":"2024-05-30T10:13:11.182340Z","shell.execute_reply.started":"2024-05-30T10:13:11.159380Z","shell.execute_reply":"2024-05-30T10:13:11.181229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"UNLIKELY_P = unlikely_birds+water_birds\nLOW_P      = list(set(low_priority_birds  + low_elevation_birds))\nMID_P      = list(set(mid_priority_birds  + low_mid_birds))\nHIGH_P     = high_priority_birds","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:11.183883Z","iopub.execute_input":"2024-05-30T10:13:11.184297Z","iopub.status.idle":"2024-05-30T10:13:11.191552Z","shell.execute_reply.started":"2024-05-30T10:13:11.184265Z","shell.execute_reply":"2024-05-30T10:13:11.190365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Low Priority Birds:\",len(LOW_P))\nprint(\"Mid Priority Birds:\",len(MID_P))\nprint(\"High Priority Birds:\",len(HIGH_P))\nprint(\"unlikely Birds:\", len(UNLIKELY_P))","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:11.192956Z","iopub.execute_input":"2024-05-30T10:13:11.193370Z","iopub.status.idle":"2024-05-30T10:13:11.205311Z","shell.execute_reply.started":"2024-05-30T10:13:11.193340Z","shell.execute_reply":"2024-05-30T10:13:11.204143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_paths = list(PL.file_path.values)\nprint(len(final_paths))","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:11.206812Z","iopub.execute_input":"2024-05-30T10:13:11.207148Z","iopub.status.idle":"2024-05-30T10:13:11.218754Z","shell.execute_reply.started":"2024-05-30T10:13:11.207119Z","shell.execute_reply":"2024-05-30T10:13:11.217713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### SELF MIX","metadata":{}},{"cell_type":"code","source":"def normalize_equal_shapes(signal,SR=32000,USE_SEC=5):\n    signal = signal/ np.linalg.norm(signal)\n    if len(signal)>USE_SEC*SR:\n                signal = signal[:USE_SEC*SR]\n    else:\n        diff = SR*USE_SEC - len(signal)\n        signal = np.pad(signal, (0,diff), 'constant',)\n    return signal","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:11.219998Z","iopub.execute_input":"2024-05-30T10:13:11.220307Z","iopub.status.idle":"2024-05-30T10:13:11.230135Z","shell.execute_reply.started":"2024-05-30T10:13:11.220280Z","shell.execute_reply":"2024-05-30T10:13:11.229075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef change_speed(data,speed_factor=0.8,use_sec=5,sr=32000):\n    #If rate > 1, then the signal is sped up. If rate < 1, then the signal is slowed down.\n    signal = librosa.effects.time_stretch(data, rate=speed_factor)\n    if len(signal)>USE_SEC*SR:\n        signal = signal[:USE_SEC*SR]\n    else:\n        diff = SR*USE_SEC - len(signal)\n        signal = np.pad(signal, (0,diff), 'constant',)\n    return signal\n\ndef pitch_shift(data, sr=32000, pitch_factor=1.8):\n    #how many (fractional) steps to shift y\n    return librosa.effects.pitch_shift(data, sr=sr, n_steps=pitch_factor)\n\ndef audio_shift(data, sampling_rate=32000, shift_max=1, shift_direction='right'):\n    shift = np.random.randint(sampling_rate * shift_max)\n    if shift_direction == 'right':\n        shift = -shift\n    elif shift_direction == 'both':\n        direction = np.random.randint(0, 2)\n        if direction == 1:\n            shift = -shift\n    augmented_data = np.roll(data, shift)\n    # Set to silence for heading/ tailing\n    if shift > 0:\n        augmented_data[:shift] = 0\n    else:\n        augmented_data[shift:] = 0\n    return augmented_data\n\ndef noise_injection(data, noise_factor=0.05):\n    \n    noise_path = np.random.choice(NP)\n    noise = np.load(noise_path)\n    noise = noise[:len(data)]\n    augmented_data = data + noise_factor * noise\n    # Cast back to same data type\n    augmented_data = augmented_data.astype(data.dtype)\n    return augmented_data\n\ndef normalized_custom_noise_injection(data, noise_factor=2):\n    \n    noise_path = np.random.choice(NP)\n    noise = np.load(noise_path)\n    \n    noise = normalize_equal_shapes(noise,)\n    data = normalize_equal_shapes(data,)\n    augmented_data = data + noise_factor * noise\n    # Cast back to same data type\n    augmented_data =augmented_data.astype(data.dtype)\n    return augmented_data\n        ","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:11.231473Z","iopub.execute_input":"2024-05-30T10:13:11.231879Z","iopub.status.idle":"2024-05-30T10:13:11.249634Z","shell.execute_reply.started":"2024-05-30T10:13:11.231849Z","shell.execute_reply":"2024-05-30T10:13:11.248301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This code runs only in python 3.10 or above versions\ndef apply_augmentations(rand,signal):\n    match rand:\n        case 0:\n            return change_speed(signal,speed_factor=random.choice([0.4,0.6,0.8,1.0,1.2,1.6,1.8]))\n        case 1:\n            return pitch_shift(signal, sr=32000, pitch_factor=random.choice([0.5,1.5,2.0,2.5,3.0,3.5,4.0]))\n        case 2:\n            return audio_shift(signal, sampling_rate=32000, shift_max=random.choice([0.2,0.4,0.5,0.6,0.8,1.0]), \n                               shift_direction=random.choice(['right','both']))\n        case default:\n            return signal","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:11.251295Z","iopub.execute_input":"2024-05-30T10:13:11.251760Z","iopub.status.idle":"2024-05-30T10:13:11.268135Z","shell.execute_reply.started":"2024-05-30T10:13:11.251699Z","shell.execute_reply":"2024-05-30T10:13:11.266992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def SAVE_MIX(FILE1,FILE2,STAGE_LIST,BIRD_CODES,cat):\n    USE_RATING=3.\n    USE_SEC=5\n    SR=32000\n    MIN_SEC=3\n\n    NEW_TARGETS=[]\n    NEW_PATHS=[]\n    MIXED_TAGS=[]\n    \n    bird_dict = {df_birdlist[df_birdlist.primary_label==bird_code]['common_name'].values[0]:0 \\\n                            for bird_code in BIRD_CODES}\n\n    for n in tqdm(range(len(FILE1))):\n        filea,fileb = FILE1[n],FILE2[n]\n        stage = STAGE_LIST[n]\n        bird_code = BIRD_CODES[n]\n        bird_name = df_birdlist[df_birdlist.primary_label==bird_code]['common_name'].values[0]\n        \n        filea_tag = filea.split(\"/\")[-1].split(\".\")[0]\n        fileb_tag = fileb.split(\"/\")[-1].split(\".\")[0]\n\n        filea_signal = np.load(filea)\n        fileb_signal = np.load(fileb)\n\n        if len(filea_signal)>=int(MIN_SEC*SR) or len(fileb_signal)>int(MIN_SEC*SR): #MINSEC=1\n\n            #FILEA\n            if len(filea_signal)>USE_SEC*SR:\n                    filea_signal = filea_signal[:USE_SEC*SR]\n            else:\n                diff = SR*USE_SEC - len(filea_signal)\n                filea_signal = np.pad(filea_signal, (0,diff), 'constant',)\n\n            #FILEB USE_STAGE\n            start,stop = stage*SR*USE_SEC,(stage+1)*SR*USE_SEC\n            fileb_signal = fileb_signal[start:stop]\n            if len(fileb_signal)>USE_SEC*SR:\n                    fileb_signal = fileb_signal[:USE_SEC*SR]\n            else:\n                diff = SR*USE_SEC - len(fileb_signal)\n                fileb_signal = np.pad(fileb_signal, (0,diff), 'constant',)\n\n\n            bird_target = filea.split(\"/\")[-2]\n            mixed_file_tag = bird_code + \"_\" + filea_tag + \"_\" + fileb_tag + \"_\" + str(stage) + \"_\" + cat\n\n            rand = np.random.randint(4)\n            filea_signal = apply_augmentations(rand,filea_signal)\n\n\n            with np.errstate(invalid='raise'):\n                try:\n                    mixed_signal = (filea_signal / np.linalg.norm(filea_signal))\\\n                                 + (fileb_signal / np.linalg.norm(fileb_signal))\n                    save_path = os.path.join(save_folder_path,mixed_file_tag)\n\n                    np.save(save_path, mixed_signal)\n                except FloatingPointError:\n#                     print('Error: Division by Zero')\n#                     print('bird name:',bird_name)\n                    bird_dict[bird_name] +=1\n    print(bird_dict)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:11.269663Z","iopub.execute_input":"2024-05-30T10:13:11.270054Z","iopub.status.idle":"2024-05-30T10:13:11.287960Z","shell.execute_reply.started":"2024-05-30T10:13:11.270024Z","shell.execute_reply":"2024-05-30T10:13:11.286796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### High Priority ","metadata":{}},{"cell_type":"code","source":"DEFAULT_ADD_FILES = 20\nADD_FILES=70\nCURB_FILES = 100","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:11.289486Z","iopub.execute_input":"2024-05-30T10:13:11.289921Z","iopub.status.idle":"2024-05-30T10:13:11.303485Z","shell.execute_reply.started":"2024-05-30T10:13:11.289883Z","shell.execute_reply":"2024-05-30T10:13:11.302438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"USE_RATING=3.\nUSE_SEC=5\nSR=32000\nMIN_SEC=3\n\n# CURB_NUM_AUDIO = 100\nMAX_STAGE=6\n\nCOUNTER=0\nFILE1,FILE2 =[],[]\nSTAGE_LIST=[]\nBIRD_CODES = []\n\nbird_dict = {bird_name:[] for bird_name in bird_target_names}\nfor_counter = 0\nfor bird_name in tqdm(HIGH_P):\n    bird_code = df_birdlist[df_birdlist.common_name==bird_name]['primary_label'].values[0]\n    r = re.compile(f\".*{bird_code}\")\n    audio_paths = list(filter(r.match, final_paths))\n    needed_num_files=0\n    mix_count=0\n    \n    if len(audio_paths)>CURB_FILES:\n        CURB_NUM_AUDIO = DEFAULT_ADD_FILES\n    else:\n        CURB_NUM_AUDIO = (CURB_FILES - len(audio_paths)) + DEFAULT_ADD_FILES\n        if CURB_NUM_AUDIO>ADD_FILES:\n            CURB_NUM_AUDIO\n        \n    if len(audio_paths)<CURB_NUM_AUDIO:\n        needed_num_files = CURB_NUM_AUDIO - len(audio_paths)\n        mix_count= 0\n        stage_count = 0\n        mixed_files = []\n        audio_pathsx = copy.deepcopy(audio_paths)\n        \n        while (mix_count<needed_num_files) and stage_count<MAX_STAGE:\n            if len(audio_pathsx)>1:\n#                 print(\"\\t\\trandom_choices\",len(audio_pathsx))\n                filea, fileb = random.choices(audio_pathsx,k=2)\n                if filea!=fileb:\n    #                 if filea not in mixed_files and fileb not in mixed_files:\n                    if len(mixed_files)<=(len(audio_paths)):\n                        mixed_files.append(filea)\n                        mixed_files.append(fileb)\n                        FILE1.append(filea)\n                        FILE2.append(fileb)\n                        STAGE_LIST.append(stage_count)\n                        BIRD_CODES.append(bird_code)\n                        mix_count +=1\n                        audio_pathsx.remove(filea)\n                        audio_pathsx.remove(fileb)\n                    else:\n                        audio_pathsx = copy.deepcopy(audio_paths)\n                        stage_count +=1\n                        mixed_files=[]\n            \n                else:\n                    continue\n            else:\n                audio_pathsx = copy.deepcopy(audio_paths)\n                stage_count +=1\n                mixed_files=[]\n        \n        COUNTER +=1\n    bird_dict[bird_code]=[len(audio_paths),needed_num_files,mix_count]\n    for_counter +=1\nprint(\"\\n\",COUNTER)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:11.304917Z","iopub.execute_input":"2024-05-30T10:13:11.305325Z","iopub.status.idle":"2024-05-30T10:13:11.684930Z","shell.execute_reply.started":"2024-05-30T10:13:11.305288Z","shell.execute_reply":"2024-05-30T10:13:11.683892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_needed_files = 0\ngot_files=0\nfor bird_name in tqdm(HIGH_P):\n    bird_code = df_birdlist[df_birdlist.common_name==bird_name]['primary_label'].values[0]\n    paths,need_files,actual_files = bird_dict[bird_code]\n    total_needed_files += need_files\n    got_files += actual_files\n\nprint(\"TOTAL NEEDED FILES:\",total_needed_files)\nprint(\"ACTUAL GOT FILES:\",got_files)\nprint(\"DIFF:\",total_needed_files-got_files)\nprint(\"%:\",got_files/total_needed_files)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:11.686162Z","iopub.execute_input":"2024-05-30T10:13:11.686487Z","iopub.status.idle":"2024-05-30T10:13:11.717523Z","shell.execute_reply.started":"2024-05-30T10:13:11.686459Z","shell.execute_reply":"2024-05-30T10:13:11.716401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(FILE1),len(FILE2),len(STAGE_LIST),len(BIRD_CODES)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:11.718935Z","iopub.execute_input":"2024-05-30T10:13:11.719265Z","iopub.status.idle":"2024-05-30T10:13:11.726755Z","shell.execute_reply.started":"2024-05-30T10:13:11.719237Z","shell.execute_reply":"2024-05-30T10:13:11.725692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save_dir = 'high_priority_mixed_self_signals'\nworking_dir = '/kaggle/working/'\n\nsave_folder_path = os.path.join(working_dir,save_dir)\nprint(save_folder_path)\n\n_ = os.makedirs(save_folder_path, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:11.728153Z","iopub.execute_input":"2024-05-30T10:13:11.728585Z","iopub.status.idle":"2024-05-30T10:13:11.739477Z","shell.execute_reply.started":"2024-05-30T10:13:11.728549Z","shell.execute_reply":"2024-05-30T10:13:11.738340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SAVE_MIX(FILE1,FILE2,STAGE_LIST,BIRD_CODES,'high')","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:11.740782Z","iopub.execute_input":"2024-05-30T10:13:11.741122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Mid Priority ","metadata":{}},{"cell_type":"code","source":"DEFAULT_ADD_FILES = 15\nADD_FILES=40\nCURB_FILES = 100","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"USE_RATING=3.\nUSE_SEC=5\nSR=32000\nMIN_SEC=3\n\n# CURB_NUM_AUDIO = 100\nMAX_STAGE=6\n\nCOUNTER=0\nFILE1,FILE2 =[],[]\nSTAGE_LIST=[]\nBIRD_CODES = []\n\nbird_dict = {bird_name:[] for bird_name in bird_target_names}\nfor_counter = 0\nfor bird_name in tqdm(MID_P):\n    bird_code = df_birdlist[df_birdlist.common_name==bird_name]['primary_label'].values[0]\n    r = re.compile(f\".*{bird_code}\")\n    audio_paths = list(filter(r.match, final_paths))\n    needed_num_files=0\n    mix_count=0\n    \n    if len(audio_paths)>CURB_FILES:\n        CURB_NUM_AUDIO = DEFAULT_ADD_FILES\n    else:\n        CURB_NUM_AUDIO = (CURB_FILES - len(audio_paths)) + DEFAULT_ADD_FILES\n        if CURB_NUM_AUDIO>ADD_FILES:\n            CURB_NUM_AUDIO\n        \n    if len(audio_paths)<CURB_NUM_AUDIO:\n        needed_num_files = CURB_NUM_AUDIO - len(audio_paths)\n        mix_count= 0\n        stage_count = 0\n        mixed_files = []\n        audio_pathsx = copy.deepcopy(audio_paths)\n        \n        while (mix_count<needed_num_files) and stage_count<MAX_STAGE:\n            if len(audio_pathsx)>1:\n#                 print(\"\\t\\trandom_choices\",len(audio_pathsx))\n                filea, fileb = random.choices(audio_pathsx,k=2)\n                if filea!=fileb:\n    #                 if filea not in mixed_files and fileb not in mixed_files:\n                    if len(mixed_files)<=(len(audio_paths)):\n                        mixed_files.append(filea)\n                        mixed_files.append(fileb)\n                        FILE1.append(filea)\n                        FILE2.append(fileb)\n                        STAGE_LIST.append(stage_count)\n                        BIRD_CODES.append(bird_code)\n                        mix_count +=1\n                        audio_pathsx.remove(filea)\n                        audio_pathsx.remove(fileb)\n                    else:\n                        audio_pathsx = copy.deepcopy(audio_paths)\n                        stage_count +=1\n                        mixed_files=[]\n            \n                else:\n                    continue\n            else:\n                audio_pathsx = copy.deepcopy(audio_paths)\n                stage_count +=1\n                mixed_files=[]\n        \n        COUNTER +=1\n    bird_dict[bird_code]=[len(audio_paths),needed_num_files,mix_count]\n    for_counter +=1\nprint(\"\\n\",COUNTER)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_needed_files = 0\ngot_files=0\nfor bird_name in tqdm(MID_P):\n    bird_code = df_birdlist[df_birdlist.common_name==bird_name]['primary_label'].values[0]\n    paths,need_files,actual_files = bird_dict[bird_code]\n    total_needed_files += need_files\n    got_files += actual_files\n\nprint(\"TOTAL NEEDED FILES:\",total_needed_files)\nprint(\"ACTUAL GOT FILES:\",got_files)\nprint(\"DIFF:\",total_needed_files-got_files)\nprint(\"%:\",got_files/total_needed_files)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save_dir = 'mid_priority_mixed_self_signals'\nworking_dir = '/kaggle/working/'\n\nsave_folder_path = os.path.join(working_dir,save_dir)\nprint(save_folder_path)\n\n_ = os.makedirs(save_folder_path, exist_ok=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(FILE1),len(FILE2),len(STAGE_LIST),len(BIRD_CODES)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SAVE_MIX(FILE1,FILE2,STAGE_LIST,BIRD_CODES,'mid')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Low Priority ","metadata":{}},{"cell_type":"code","source":"DEFAULT_ADD_FILES = 10\nADD_FILES=30\nCURB_FILES = 100","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"USE_RATING=3.\nUSE_SEC=5\nSR=32000\nMIN_SEC=3\n\n# CURB_NUM_AUDIO = 100\nMAX_STAGE=6\n\nCOUNTER=0\nFILE1,FILE2 =[],[]\nSTAGE_LIST=[]\nBIRD_CODES = []\n\nbird_dict = {bird_name:[] for bird_name in bird_target_names}\nfor_counter = 0\nfor bird_name in tqdm(LOW_P):\n    bird_code = df_birdlist[df_birdlist.common_name==bird_name]['primary_label'].values[0]\n    r = re.compile(f\".*{bird_code}\")\n    audio_paths = list(filter(r.match, final_paths))\n    needed_num_files=0\n    mix_count=0\n    \n    if len(audio_paths)>CURB_FILES:\n        CURB_NUM_AUDIO = DEFAULT_ADD_FILES\n    else:\n        CURB_NUM_AUDIO = (CURB_FILES - len(audio_paths)) + DEFAULT_ADD_FILES\n        if CURB_NUM_AUDIO>ADD_FILES:\n            CURB_NUM_AUDIO\n        \n    if len(audio_paths)<CURB_NUM_AUDIO:\n        needed_num_files = CURB_NUM_AUDIO - len(audio_paths)\n        mix_count= 0\n        stage_count = 0\n        mixed_files = []\n        audio_pathsx = copy.deepcopy(audio_paths)\n        \n        while (mix_count<needed_num_files) and stage_count<MAX_STAGE:\n            if len(audio_pathsx)>1:\n#                 print(\"\\t\\trandom_choices\",len(audio_pathsx))\n                filea, fileb = random.choices(audio_pathsx,k=2)\n                if filea!=fileb:\n    #                 if filea not in mixed_files and fileb not in mixed_files:\n                    if len(mixed_files)<=(len(audio_paths)):\n                        mixed_files.append(filea)\n                        mixed_files.append(fileb)\n                        FILE1.append(filea)\n                        FILE2.append(fileb)\n                        STAGE_LIST.append(stage_count)\n                        BIRD_CODES.append(bird_code)\n                        mix_count +=1\n                        audio_pathsx.remove(filea)\n                        audio_pathsx.remove(fileb)\n                    else:\n                        audio_pathsx = copy.deepcopy(audio_paths)\n                        stage_count +=1\n                        mixed_files=[]\n            \n                else:\n                    continue\n            else:\n                audio_pathsx = copy.deepcopy(audio_paths)\n                stage_count +=1\n                mixed_files=[]\n        \n        COUNTER +=1\n    bird_dict[bird_code]=[len(audio_paths),needed_num_files,mix_count]\n    for_counter +=1\nprint(\"\\n\",COUNTER)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_needed_files = 0\ngot_files=0\nfor bird_name in tqdm(LOW_P):\n    bird_code = df_birdlist[df_birdlist.common_name==bird_name]['primary_label'].values[0]\n    paths,need_files,actual_files = bird_dict[bird_code]\n    total_needed_files += need_files\n    got_files += actual_files\n\nprint(\"TOTAL NEEDED FILES:\",total_needed_files)\nprint(\"ACTUAL GOT FILES:\",got_files)\nprint(\"DIFF:\",total_needed_files-got_files)\nprint(\"%:\",got_files/total_needed_files)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save_dir = 'low_priority_mixed_self_signals'\nworking_dir = '/kaggle/working/'\n\nsave_folder_path = os.path.join(working_dir,save_dir)\nprint(save_folder_path)\n\n_ = os.makedirs(save_folder_path, exist_ok=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(FILE1),len(FILE2),len(STAGE_LIST),len(BIRD_CODES)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SAVE_MIX(FILE1,FILE2,STAGE_LIST,BIRD_CODES,'low')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_high = len(os.listdir(\"/kaggle/working/high_priority_mixed_self_signals\"))\nnum_mid  = len(os.listdir(\"/kaggle/working/mid_priority_mixed_self_signals\"))\nnum_low  = len(os.listdir(\"/kaggle/working/low_priority_mixed_self_signals\"))\n\nprint(\"# of high priority files:\",num_high)\nprint(\"# of mid priority files:\",num_mid)\nprint(\"# of low priority files:\",num_low)\n\nprint()\nprint(\"# Total FIles:\", num_high+num_mid+num_low)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selfmix_paths = glob('/kaggle/working/*/*.npy')\n\nprint(selfmix_paths[0])\nprint(\"total selfmix audio files:\", len(selfmix_paths))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter=0\nfor path in selfmix_paths:\n    x = np.load(path)\n    if np.isnan(x).any():\n        counter +=1\ncounter","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_df = pd.DataFrame(selfmix_paths, columns=['file_path'])\npath_df['new_target'] = path_df['file_path'].map(lambda x: x.split(\"/\")[-1].split(\"_\")[0])\nprint(path_df.shape)\npath_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for bird_name in nocturnal_birds:\n    bird_code = df_birdlist[df_birdlist.common_name==bird_name]['primary_label'].values[0]\n    num_paths = len(os.listdir(f\"/kaggle/input/birdclef-2024/train_audio/{bird_code}\"))\n    num_files = len(path_df[path_df.new_target==bird_code])\n    print(bird_name,num_paths,num_files)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# UNLIKELY_P = unlikely_birds+water_birds\n# LOW_P      = list(set(low_priority_birds  + low_elevation_birds))\n# MID_P      = list(set(mid_priority_birds  + low_mid_birds))\n# HIGH_P     = high_priority_birds","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for bird_name in list(df_birdlist.common_name.values):\n    bird_code = df_birdlist[df_birdlist.common_name==bird_name]['primary_label'].values[0]\n    num_paths = len(os.listdir(f\"/kaggle/input/birdclef-2024/train_audio/{bird_code}\"))\n    num_files = len(path_df[path_df.new_target==bird_code])\n#     if num_files !=0:\n    if bird_name in HIGH_P:\n        print('HIGH',bird_name,num_paths,num_files)\n    elif bird_name in MID_P:\n        print('MID',bird_name,num_paths,num_files)\n    elif bird_name in LOW_P:\n        print('LOW',bird_name,num_paths,num_files)\n    else:\n        continue","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}