{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":8096443,"sourceType":"datasetVersion","datasetId":4780521},{"sourceId":8318251,"sourceType":"datasetVersion","datasetId":4940719},{"sourceId":8416940,"sourceType":"datasetVersion","datasetId":5010206},{"sourceId":8426650,"sourceType":"datasetVersion","datasetId":5017621},{"sourceId":170596805,"sourceType":"kernelVersion"},{"sourceId":170597150,"sourceType":"kernelVersion"},{"sourceId":170597237,"sourceType":"kernelVersion"},{"sourceId":170597938,"sourceType":"kernelVersion"},{"sourceId":170598083,"sourceType":"kernelVersion"},{"sourceId":170598279,"sourceType":"kernelVersion"},{"sourceId":170598684,"sourceType":"kernelVersion"},{"sourceId":175005679,"sourceType":"kernelVersion"},{"sourceId":175993701,"sourceType":"kernelVersion"}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### No nocturnal birds,mixed them separately but high & Mid","metadata":{}},{"cell_type":"code","source":"import gc\nimport os\nimport sys\n# sys.path.append('../input/pytorch-image-models/pytorch-image-models-master')\nimport random\nimport time\nimport warnings\nimport re\nimport copy\n\nimport librosa\nimport numpy as np\nimport pandas as pd\nimport soundfile as sf\n\n\n\nfrom contextlib import contextmanager\nfrom joblib import Parallel, delayed\nfrom pathlib import Path\nfrom tqdm import tqdm\nfrom glob import glob\n\nimport matplotlib.pyplot as plt\nimport librosa.display","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:19.720044Z","iopub.execute_input":"2024-05-28T10:14:19.721000Z","iopub.status.idle":"2024-05-28T10:14:20.191168Z","shell.execute_reply.started":"2024-05-28T10:14:19.720955Z","shell.execute_reply":"2024-05-28T10:14:20.190075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import Audio\nSR=32000\nless_than = 50","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:20.192444Z","iopub.execute_input":"2024-05-28T10:14:20.192878Z","iopub.status.idle":"2024-05-28T10:14:20.197717Z","shell.execute_reply.started":"2024-05-28T10:14:20.192849Z","shell.execute_reply":"2024-05-28T10:14:20.196728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seed(seed=43):\n    random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    np.random.seed(seed)\n    \nset_seed(43)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:20.198967Z","iopub.execute_input":"2024-05-28T10:14:20.199291Z","iopub.status.idle":"2024-05-28T10:14:20.207439Z","shell.execute_reply.started":"2024-05-28T10:14:20.199265Z","shell.execute_reply":"2024-05-28T10:14:20.206553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_paths = glob(\"/kaggle/input/1negative-samples-npy-birdclef2024/noise/*.npy\")\nprint(len(n_paths))\nNP=[]\nfor path in n_paths:\n    y = np.load(path)\n    length = len(y)/32000\n    if length==5.:\n#         print(path,len(y)/32000)\n        NP.append(path)\nprint(len(NP))","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:20.210249Z","iopub.execute_input":"2024-05-28T10:14:20.211015Z","iopub.status.idle":"2024-05-28T10:14:20.444439Z","shell.execute_reply.started":"2024-05-28T10:14:20.210988Z","shell.execute_reply":"2024-05-28T10:14:20.443356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dur = pd.read_csv(\"/kaggle/input/2duration-log/bc2024_duration.csv\")\nprint(dur.shape)\ndur.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:20.445666Z","iopub.execute_input":"2024-05-28T10:14:20.445976Z","iopub.status.idle":"2024-05-28T10:14:20.629847Z","shell.execute_reply.started":"2024-05-28T10:14:20.445948Z","shell.execute_reply":"2024-05-28T10:14:20.628802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def find_audio_paths_birdcode(bird_code):\n    audio_paths = glob(f'/kaggle/input/1npy-birdclef2024/train_npy0/{bird_code}/*.npy')\\\n                    + glob(f'/kaggle/input/2npy-birdclef2024/train_npy1/{bird_code}/*.npy')\\\n                    + glob(f'/kaggle/input/3npy-birdclef2024/train_npy2/{bird_code}/*.npy')\\\n                    + glob(f'/kaggle/input/4npy-birdclef2024/train_npy3/{bird_code}/*.npy')\\\n                    + glob(f'/kaggle/input/5npy-birdclef2024/train_npy4/{bird_code}/*.npy')\\\n                    + glob(f'/kaggle/input/6npy-birdclef2024/train_npy5/{bird_code}/*.npy')\\\n                    + glob(f'/kaggle/input/7npy-birdclef2024/train_npy6/{bird_code}/*.npy')\n    return audio_paths\n\ndef find_audio_paths_birdcode_filetags(bird_code,file_tag):\n    audio_paths = glob(f'/kaggle/input/1npy-birdclef2024/train_npy0/{bird_code}/{file_tag}.npy')\\\n                    + glob(f'/kaggle/input/2npy-birdclef2024/train_npy1/{bird_code}/{file_tag}.npy')\\\n                    + glob(f'/kaggle/input/3npy-birdclef2024/train_npy2/{bird_code}/{file_tag}.npy')\\\n                    + glob(f'/kaggle/input/4npy-birdclef2024/train_npy3/{bird_code}/{file_tag}.npy')\\\n                    + glob(f'/kaggle/input/5npy-birdclef2024/train_npy4/{bird_code}/{file_tag}.npy')\\\n                    + glob(f'/kaggle/input/6npy-birdclef2024/train_npy5/{bird_code}/{file_tag}.npy')\\\n                    + glob(f'/kaggle/input/7npy-birdclef2024/train_npy6/{bird_code}/{file_tag}.npy')\n    return audio_paths","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:20.631134Z","iopub.execute_input":"2024-05-28T10:14:20.631463Z","iopub.status.idle":"2024-05-28T10:14:20.719341Z","shell.execute_reply.started":"2024-05-28T10:14:20.631434Z","shell.execute_reply":"2024-05-28T10:14:20.718233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_paths = glob('/kaggle/input/1npy-birdclef2024/train_npy0/*/*.npy')\\\n                + glob('/kaggle/input/2npy-birdclef2024/train_npy1/*/*.npy')\\\n                + glob('/kaggle/input/3npy-birdclef2024/train_npy2/*/*.npy')\\\n                + glob('/kaggle/input/4npy-birdclef2024/train_npy3/*/*.npy')\\\n                + glob('/kaggle/input/5npy-birdclef2024/train_npy4/*/*.npy')\\\n                + glob('/kaggle/input/6npy-birdclef2024/train_npy5/*/*.npy')\\\n                + glob('/kaggle/input/7npy-birdclef2024/train_npy6/*/*.npy')\n\n# all_path = glob.glob(\"/kaggle/input/birdclef-2024/train_audio/*/*.ogg\")\nprint(all_paths[0])\nprint(len(all_paths))","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:20.720868Z","iopub.execute_input":"2024-05-28T10:14:20.721244Z","iopub.status.idle":"2024-05-28T10:14:21.076076Z","shell.execute_reply.started":"2024-05-28T10:14:20.721215Z","shell.execute_reply":"2024-05-28T10:14:21.074986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"txt_file = \"/kaggle/input/duplicates-calls-bc2024/duplicates_bird_calls.txt\"\nDL =[]\nwith open(txt_file,'r') as file:\n    lines = file.readlines()\n    for line in lines:\n        DL.append(line)\nfile.close()\nprint(len(DL))\nprint(DL[0])\n\ndiff_birds=[]\nsame_birds=[]\nfor item in DL:\n    a,b = item.split(\",\")[0],item.split(\",\")[1]\n    if a.split(\"/\")[0] != b.split(\"/\")[0]:\n        fla = a.split(\"/\")[1].split(\".\")[0]\n        flb = b.split(\"/\")[1].split(\".\")[0]\n        if fla!=flb:\n            diff_birds.append((a,b.strip()))\n    else:\n#         same_birds.append((a,b.strip()))\n        same_birds.append(a)\nprint(len(diff_birds),len(same_birds))\nprint(diff_birds[0],same_birds[0])","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:21.077459Z","iopub.execute_input":"2024-05-28T10:14:21.077790Z","iopub.status.idle":"2024-05-28T10:14:21.089898Z","shell.execute_reply.started":"2024-05-28T10:14:21.077763Z","shell.execute_reply":"2024-05-28T10:14:21.088849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#remove the same bird calls from paths\nval = len(all_paths)\nprint(len(all_paths))\ntemp_FL=[]\nfor n,filename in enumerate(same_birds):\n    bird_name = filename.split(\"/\")[0]\n    file_tag = filename.split(\"/\")[1].split(\".\")[0]\n    path = find_audio_paths_birdcode_filetags(bird_name,file_tag)\n#     print(n,path[0])\n    if file_tag not in temp_FL:\n        all_paths.remove(path[0])\n    temp_FL.append(file_tag)\n    \nprint(len(all_paths))\nprint(\"diff:\", val - len(all_paths))","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:21.091409Z","iopub.execute_input":"2024-05-28T10:14:21.091760Z","iopub.status.idle":"2024-05-28T10:14:21.289863Z","shell.execute_reply.started":"2024-05-28T10:14:21.091731Z","shell.execute_reply":"2024-05-28T10:14:21.288736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FOLDER = \"/kaggle/input/birdclef-2024\"\ntrain_dir =\"/kaggle/input/birdclef-2024/train_audio\"\nAUDIO_DURATION=5.","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:21.293569Z","iopub.execute_input":"2024-05-28T10:14:21.293927Z","iopub.status.idle":"2024-05-28T10:14:21.298931Z","shell.execute_reply.started":"2024-05-28T10:14:21.293897Z","shell.execute_reply":"2024-05-28T10:14:21.297842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import ast\n\n\ntrain = pd.read_csv(os.path.join(FOLDER,\"train_metadata.csv\"))\n\n\ntrain['new_target'] = train['primary_label'] + ' ' + train['secondary_labels'].map(lambda x: ' '.join(ast.literal_eval(x)))\n# train['len_new_target'] = train['new_target'].map(lambda x: len(x.split()))\n# train['len_new_target'].value_counts()\ntrain['file_tag'] = train['filename'].map(lambda x: x.split(\".\")[0].split(\"/\")[-1])\nprint(train.shape)\ntrain.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:21.300194Z","iopub.execute_input":"2024-05-28T10:14:21.300488Z","iopub.status.idle":"2024-05-28T10:14:21.692115Z","shell.execute_reply.started":"2024-05-28T10:14:21.300463Z","shell.execute_reply":"2024-05-28T10:14:21.691068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter=0\nfor n in range(len(train)):\n    pl = train.iloc[n]['primary_label']\n    bird_list = train.iloc[n]['new_target'].split()\n    temp_birdlist=[]\n    temp_birdlist.append(pl)\n    if len(bird_list)>1:\n        for idx in range(len(bird_list)):\n            if idx>0:\n                bird_name = bird_list[idx]\n                if bird_name!=pl:\n                    temp_birdlist.append(bird_name)\n        names = \" \".join(temp_birdlist)\n#         print(pl,names)\n        counter +=1\n        train.loc[n,'new_target']= names\nprint(counter)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:21.693680Z","iopub.execute_input":"2024-05-28T10:14:21.694020Z","iopub.status.idle":"2024-05-28T10:14:25.993467Z","shell.execute_reply.started":"2024-05-28T10:14:21.693992Z","shell.execute_reply":"2024-05-28T10:14:25.992335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for a,b in diff_birds:\n    fla = a.split(\"/\")[1].split(\".\")[0]\n    flb = b.split(\"/\")[1].split(\".\")[0]\n    idx = train[train.file_tag==flb].index\n    train.at[idx.values[0],'file_tag']=fla","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:25.998525Z","iopub.execute_input":"2024-05-28T10:14:25.998898Z","iopub.status.idle":"2024-05-28T10:14:26.063785Z","shell.execute_reply.started":"2024-05-28T10:14:25.998868Z","shell.execute_reply":"2024-05-28T10:14:26.062794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique, counts = np.unique(train.file_tag, return_counts=True)\ncounts_dict = dict(zip(unique, counts))\nprint(len(counts_dict), len(train), len(train)-len(counts_dict))\n\ntemp = {k: v for k, v in sorted(counts_dict.items(), key=lambda item: item[1],reverse=True)}\ntemp = dict(list(temp.items())[:19])\ntags = temp.keys()\nprint(list(tags))\n\nfor tag in tags:\n    temp = list(train[train.file_tag==tag].new_target.values)\n    indxs = train[train.file_tag==tag].index\n#     print(temp)\n    names = \" \".join(temp)\n    for indx in indxs:\n        train.at[indx,'new_target']= names\n        \nlist(train[train.file_tag=='XC574864'].new_target.values)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:26.064911Z","iopub.execute_input":"2024-05-28T10:14:26.065241Z","iopub.status.idle":"2024-05-28T10:14:26.354313Z","shell.execute_reply.started":"2024-05-28T10:14:26.065213Z","shell.execute_reply":"2024-05-28T10:14:26.353319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val = len(train)\nprint(len(train))\ntrain.drop_duplicates(subset='file_tag',inplace=True)\ntrain.reset_index(drop=True, inplace=True)\ntrain['len_new_target'] = train['new_target'].map(lambda x: len(x.split()))\nprint(len(train))\nprint(\"diff:\",val-len(train))","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:26.355512Z","iopub.execute_input":"2024-05-28T10:14:26.355838Z","iopub.status.idle":"2024-05-28T10:14:26.396640Z","shell.execute_reply.started":"2024-05-28T10:14:26.355810Z","shell.execute_reply":"2024-05-28T10:14:26.395468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"secondary_labels = ['asfblu1','indwhe1','bltmun1','magrob','lotshr1','orhthr1']\nss = pd.read_csv(\"/kaggle/input/birdclef-2024/sample_submission.csv\")\nbird_target_names = list(ss.columns)\nbird_target_names.pop(0)\nlen(bird_target_names)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:26.397931Z","iopub.execute_input":"2024-05-28T10:14:26.398290Z","iopub.status.idle":"2024-05-28T10:14:26.418711Z","shell.execute_reply.started":"2024-05-28T10:14:26.398261Z","shell.execute_reply":"2024-05-28T10:14:26.417675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"common_names = list(pd.unique(train.common_name))\nlen(common_names)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:26.419993Z","iopub.execute_input":"2024-05-28T10:14:26.420368Z","iopub.status.idle":"2024-05-28T10:14:26.429433Z","shell.execute_reply.started":"2024-05-28T10:14:26.420339Z","shell.execute_reply":"2024-05-28T10:14:26.428301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tax = pd.read_csv(\"/kaggle/input/birdclef-2024/eBird_Taxonomy_v2021.csv\")\nprint(tax.shape)\ntax.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:26.430680Z","iopub.execute_input":"2024-05-28T10:14:26.430995Z","iopub.status.idle":"2024-05-28T10:14:26.512031Z","shell.execute_reply.started":"2024-05-28T10:14:26.430968Z","shell.execute_reply":"2024-05-28T10:14:26.511037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SL= tax[tax.SPECIES_CODE.isin(secondary_labels)][['SPECIES_CODE','PRIMARY_COM_NAME']]\nSL.reset_index(drop=True, inplace=True)\nSL = SL.rename(columns={'SPECIES_CODE':'primary_label','PRIMARY_COM_NAME':'common_name'})\nSL","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:26.513576Z","iopub.execute_input":"2024-05-28T10:14:26.514353Z","iopub.status.idle":"2024-05-28T10:14:26.529655Z","shell.execute_reply.started":"2024-05-28T10:14:26.514311Z","shell.execute_reply":"2024-05-28T10:14:26.528560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_df = pd.DataFrame(all_paths, columns=['file_path'])\n#path_df['filename'] = path_df['file_path'].map(lambda x: x.split('/')[-2]+'/'+x.split('/')[-1][:-4])\npath_df[\"filename\"] = path_df['file_path'].map(lambda x: x.split(\"/\")[-2] + \"/\" +\\\n                                               x.split(\"/\")[-1].split('.')[-2] + \".ogg\")\n\npath_df['file_tag2'] = path_df['filename'].map(lambda x: x.split(\".\")[0].split(\"/\")[-1])\nprint(path_df.shape)\npath_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:26.531013Z","iopub.execute_input":"2024-05-28T10:14:26.531429Z","iopub.status.idle":"2024-05-28T10:14:26.617080Z","shell.execute_reply.started":"2024-05-28T10:14:26.531400Z","shell.execute_reply":"2024-05-28T10:14:26.616021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for a,b in diff_birds:\n    fla = a.split(\"/\")[1].split(\".\")[0]\n    flb = b.split(\"/\")[1].split(\".\")[0]\n    idx = path_df[path_df.file_tag2==flb].index\n    path_df.at[idx.values[0],'file_tag2']=fla","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:26.618508Z","iopub.execute_input":"2024-05-28T10:14:26.618922Z","iopub.status.idle":"2024-05-28T10:14:26.684721Z","shell.execute_reply.started":"2024-05-28T10:14:26.618885Z","shell.execute_reply":"2024-05-28T10:14:26.683576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique, counts = np.unique(path_df.file_tag2, return_counts=True)\ncounts_dict = dict(zip(unique, counts))\nprint(len(counts_dict), len(path_df), len(path_df)-len(counts_dict))\n\ntemp = {k: v for k, v in sorted(counts_dict.items(), key=lambda item: item[1],reverse=True)}\ntemp = dict(list(temp.items())[:19])\ntags = temp.keys()\nprint(temp)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:26.686283Z","iopub.execute_input":"2024-05-28T10:14:26.686719Z","iopub.status.idle":"2024-05-28T10:14:26.756565Z","shell.execute_reply.started":"2024-05-28T10:14:26.686682Z","shell.execute_reply":"2024-05-28T10:14:26.755494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val = len(path_df)\nprint(path_df.shape)\npath_df.drop_duplicates(subset='file_tag2',inplace=True)\npath_df.reset_index(drop=True, inplace=True)\nprint(path_df.shape)\nprint(\"diff:\",val-len(path_df))","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:26.757966Z","iopub.execute_input":"2024-05-28T10:14:26.758331Z","iopub.status.idle":"2024-05-28T10:14:26.772190Z","shell.execute_reply.started":"2024-05-28T10:14:26.758300Z","shell.execute_reply":"2024-05-28T10:14:26.770935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_df['file_tag'] = path_df['file_tag2'].map(lambda x: x.split(\"_\")[0])\npath_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:26.773647Z","iopub.execute_input":"2024-05-28T10:14:26.774291Z","iopub.status.idle":"2024-05-28T10:14:26.800819Z","shell.execute_reply.started":"2024-05-28T10:14:26.774249Z","shell.execute_reply":"2024-05-28T10:14:26.799808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.merge(dur[['filename','duration']])\nprint(train.shape)\ntrain.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:26.802166Z","iopub.execute_input":"2024-05-28T10:14:26.802572Z","iopub.status.idle":"2024-05-28T10:14:26.857896Z","shell.execute_reply.started":"2024-05-28T10:14:26.802534Z","shell.execute_reply":"2024-05-28T10:14:26.856868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final = path_df.merge(train, on=['file_tag'])\nprint(final.shape)\nfinal.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:26.859470Z","iopub.execute_input":"2024-05-28T10:14:26.859894Z","iopub.status.idle":"2024-05-28T10:14:26.920314Z","shell.execute_reply.started":"2024-05-28T10:14:26.859858Z","shell.execute_reply":"2024-05-28T10:14:26.919252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_birdlist = train[['common_name','primary_label']]\ndf_birdlist.drop_duplicates(inplace=True)\ndf_birdlist.reset_index(drop=True, inplace=True)\ndf_birdlist","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:26.921555Z","iopub.execute_input":"2024-05-28T10:14:26.921923Z","iopub.status.idle":"2024-05-28T10:14:26.944175Z","shell.execute_reply.started":"2024-05-28T10:14:26.921893Z","shell.execute_reply":"2024-05-28T10:14:26.943094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_audio = {}\nfor species in bird_target_names:\n    num_audio_files = os.listdir(os.path.join(train_dir,species))\n#     print(species,len(num_audio_files))\n    num_audio[species]=len(num_audio_files)\n    \nnew_dict = {k: v for k, v in sorted(num_audio.items(), key=lambda item: item[1])}\nprint(len(new_dict))\n\nnum_audio = pd.DataFrame(new_dict.items(), columns=['primary_label', 'NUM_AUDIO_FILES',])\nnum_audio = num_audio.merge(df_birdlist,)\nprint(num_audio.shape)\n# num_audio = num_audio.rename(columns={'SPECIES_CODE':'primary_label'})\nnum_audio.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:26.945574Z","iopub.execute_input":"2024-05-28T10:14:26.945924Z","iopub.status.idle":"2024-05-28T10:14:27.122335Z","shell.execute_reply.started":"2024-05-28T10:14:26.945895Z","shell.execute_reply":"2024-05-28T10:14:27.121292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"group_duration = final.groupby('primary_label')['duration'].sum()\ngroup_dur = pd.merge(num_audio,pd.DataFrame(group_duration).reset_index())\nprint(group_dur.shape)\n\ngroup_dur['5_second_duration'] = np.round(group_dur['duration']/AUDIO_DURATION,0)\ngroup_dur['5_second_duration'].describe()\n\ngroup_dur.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:27.123665Z","iopub.execute_input":"2024-05-28T10:14:27.123982Z","iopub.status.idle":"2024-05-28T10:14:27.148904Z","shell.execute_reply.started":"2024-05-28T10:14:27.123955Z","shell.execute_reply":"2024-05-28T10:14:27.147788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### NOISE","metadata":{}},{"cell_type":"code","source":"n_paths = glob(\"/kaggle/input/1negative-samples-npy-birdclef2024/noise/*.npy\")\nprint(len(n_paths))\nNOISE_PATHS=[]\nfor path in n_paths:\n    y = np.load(path)\n    length = len(y)/32000\n    if length==5.:\n#         print(path,len(y)/32000)\n        NOISE_PATHS.append(path)\nprint(len(NOISE_PATHS))","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:27.150104Z","iopub.execute_input":"2024-05-28T10:14:27.150429Z","iopub.status.idle":"2024-05-28T10:14:27.270964Z","shell.execute_reply.started":"2024-05-28T10:14:27.150402Z","shell.execute_reply":"2024-05-28T10:14:27.269900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bird_matrix = np.zeros((len(bird_target_names),len(bird_target_names)),dtype=np.int16)\nbird_matrix2 = np.zeros((len(bird_target_names),len(bird_target_names)),dtype=np.int16)\nprint(bird_matrix.shape)\n\nfor n in tqdm(range(len(train))):\n    pl = train.iloc[n]['primary_label']\n    bird_list = train.iloc[n]['new_target'].split()\n    pl_indx= bird_target_names.index(pl)\n    bird_matrix[pl_indx,pl_indx] +=1\n    bird_matrix2[pl_indx,pl_indx] +=1\n    if len(bird_list)>1:\n        for idx in range(len(bird_list)):\n            if idx>0:\n                sl = bird_list[idx]\n                if sl!=pl:\n                    if sl not in secondary_labels:\n                        sl_indx = bird_target_names.index(sl)\n                        bird_matrix[pl_indx,sl_indx] +=1\n                        bird_matrix2[sl_indx,pl_indx] +=1\n                        bird_matrix2[pl_indx,sl_indx] +=1                        ","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:27.272190Z","iopub.execute_input":"2024-05-28T10:14:27.272487Z","iopub.status.idle":"2024-05-28T10:14:31.818582Z","shell.execute_reply.started":"2024-05-28T10:14:27.272460Z","shell.execute_reply":"2024-05-28T10:14:31.817478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PL = final[final.len_new_target==1]\nPL.reset_index(drop=True, inplace=True)\nprint(PL.shape)\nPL.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:31.819778Z","iopub.execute_input":"2024-05-28T10:14:31.820124Z","iopub.status.idle":"2024-05-28T10:14:31.849107Z","shell.execute_reply.started":"2024-05-28T10:14:31.820095Z","shell.execute_reply":"2024-05-28T10:14:31.848111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### WATER BASED BIRDS","metadata":{}},{"cell_type":"code","source":"water = pd.read_csv(\"/kaggle/input/bc2024-water-tagged-bird-list/watertagged_bird_list2 - final_bird_list.csv\")\nwater.rename(columns={'Unnamed: 2': 'water_tag',},inplace=True)\nwater.fillna('n',inplace=True)\nprint(water.shape)\nwater.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:31.850703Z","iopub.execute_input":"2024-05-28T10:14:31.851139Z","iopub.status.idle":"2024-05-28T10:14:31.868303Z","shell.execute_reply.started":"2024-05-28T10:14:31.851101Z","shell.execute_reply":"2024-05-28T10:14:31.867244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"water_birds = list(water[water.water_tag=='w']['PRIMARY_COM_NAME'].values)\nprint(len(water_birds))\n\nwater_group = group_dur[group_dur.common_name.isin(water_birds)].merge(df_birdlist)\nwater_group.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:31.869833Z","iopub.execute_input":"2024-05-28T10:14:31.870363Z","iopub.status.idle":"2024-05-28T10:14:31.891166Z","shell.execute_reply.started":"2024-05-28T10:14:31.870325Z","shell.execute_reply":"2024-05-28T10:14:31.890132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bird_groups = { 'low_elevation': ['Zitting Cisticola','Plain Prinia','Rufous Treepie','Small Minivet','Gray-headed Swamphen',\n                   'Asian Koel','Laughing Dove','Gray Francolin',],\n  'low-mid_elevation':['Paddyfield Pipit','Common Iora','White-throated Kingfisher',\n#                        'Spotted Owlet',\n                      'Painted Stork','Asian Openbill','Spotted Dove','Red Spurfowl'],\n 'unlikely':['houspa','Brahminy Kite','Eurasian Marsh-Harrier','Eurasian Collared-Dove',\n            'Rock Pigeon','Gray Francolin'],}","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:31.892572Z","iopub.execute_input":"2024-05-28T10:14:31.892991Z","iopub.status.idle":"2024-05-28T10:14:31.899616Z","shell.execute_reply.started":"2024-05-28T10:14:31.892954Z","shell.execute_reply":"2024-05-28T10:14:31.898426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### LOW ELEVATION","metadata":{}},{"cell_type":"code","source":"low_elevation_birds = list(pd.unique(PL[PL.common_name.isin(bird_groups['low_elevation'])]['common_name']))\nprint(len(low_elevation_birds))\nprint(\"total # audio files:\",group_dur[group_dur.common_name.isin(low_elevation_birds)]['NUM_AUDIO_FILES'].sum())\nprint(\"Possible total # audio files:\",group_dur[group_dur.common_name.\\\n                                                isin(low_elevation_birds)]['5_second_duration'].sum())","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:31.900923Z","iopub.execute_input":"2024-05-28T10:14:31.901246Z","iopub.status.idle":"2024-05-28T10:14:31.915768Z","shell.execute_reply.started":"2024-05-28T10:14:31.901218Z","shell.execute_reply":"2024-05-28T10:14:31.914491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### LOW-MID ELEVATION","metadata":{}},{"cell_type":"code","source":"low_mid_birds  = list(pd.unique(PL[PL.common_name.isin(bird_groups['low-mid_elevation'])]['common_name']))\nprint(len(low_mid_birds))\nprint(\"total # audio files:\",group_dur[group_dur.common_name.isin(low_mid_birds)]['NUM_AUDIO_FILES'].sum())\nprint(\"Possible total # audio files:\",group_dur[group_dur.common_name.\\\n                                                isin(low_mid_birds)]['5_second_duration'].sum())","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:31.917138Z","iopub.execute_input":"2024-05-28T10:14:31.917503Z","iopub.status.idle":"2024-05-28T10:14:31.929312Z","shell.execute_reply.started":"2024-05-28T10:14:31.917472Z","shell.execute_reply":"2024-05-28T10:14:31.928275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### UNLIKELY","metadata":{}},{"cell_type":"code","source":"unlikely_birds = list(pd.unique(PL[PL.common_name.isin(bird_groups['unlikely'])]['common_name']))\nprint(len(unlikely_birds))\nprint(\"total # audio files:\",group_dur[group_dur.common_name.isin(unlikely_birds)]['NUM_AUDIO_FILES'].sum())\nprint(\"Possible total # audio files:\",group_dur[group_dur.common_name.\\\n                                                isin(unlikely_birds)]['5_second_duration'].sum())","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:31.930574Z","iopub.execute_input":"2024-05-28T10:14:31.930889Z","iopub.status.idle":"2024-05-28T10:14:31.943442Z","shell.execute_reply.started":"2024-05-28T10:14:31.930862Z","shell.execute_reply":"2024-05-28T10:14:31.942272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# eucdov - low\n# Eurasian Marsh-Harrier - open \n# rock pigeon - jog falls, humans\n# gray francolin - low grasslands,entry of naraikadu, low\n# brahminy kite - open, waterbody,","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:31.944876Z","iopub.execute_input":"2024-05-28T10:14:31.945573Z","iopub.status.idle":"2024-05-28T10:14:31.949978Z","shell.execute_reply.started":"2024-05-28T10:14:31.945544Z","shell.execute_reply":"2024-05-28T10:14:31.948907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### HIGH PRIORITY BIRDS","metadata":{}},{"cell_type":"code","source":"high_priority_birds = ['Gray Junglefowl', 'Malabar Whistling-Thrush', 'Malabar Barbet', 'White-cheeked Barbet',\n                       'Vernal Hanging-Parrot', 'Rufous Babbler', 'Southern Hill Myna', 'Dark-fronted Babbler', \n                       'Nilgiri Wood-Pigeon', 'Malabar Parakeet', 'Crimson-backed Sunbird', 'Orange Minivet',\n                       'White-bellied Blue Flycatcher', 'Malabar Woodshrike', 'Gray-fronted Green-Pigeon',\n                       'Nilgiri Flycatcher', 'Great Hornbill', 'Square-tailed Bulbul', 'Black-and-orange Flycatcher',\n                       'Nilgiri Flowerpecker', 'Yellow-browed Bulbul', 'Indian Yellow Tit', 'Malabar Trogon', \n                       'Jungle Myna', \"Loten's Sunbird\", 'Palani Laughingthrush', 'Common Flameback', 'White-bellied Woodpecker', \n                       'White-bellied Treepie', 'White-bellied Sholakili', 'Malabar Gray Hornbill', 'Wayanad Laughingthrush', \n                       'Flame-throated Bulbul',\n#                         'Brown Wood-Owl','Spot-bellied Eagle-Owl','Brown Fish-Owl','Jungle Owlet','Brown Boobook',\n#                        'Indian Scops-Owl',\n#                        'Great Eared-Nightjar', 'Jungle Nightjar',\n                      ]\nprint(len(high_priority_birds))\nhpb = list(pd.unique(PL[PL.common_name.isin(high_priority_birds)]['primary_label']))\nprint(\"total # audio files:\",group_dur[group_dur.primary_label.isin(hpb)]['NUM_AUDIO_FILES'].sum())\nprint(\"possible total # audio files:\",group_dur[group_dur.primary_label.isin(hpb)]['5_second_duration'].sum())\nprint()\ngroup_dur[group_dur.primary_label.isin(hpb)].shape","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:31.960211Z","iopub.execute_input":"2024-05-28T10:14:31.960536Z","iopub.status.idle":"2024-05-28T10:14:31.978069Z","shell.execute_reply.started":"2024-05-28T10:14:31.960509Z","shell.execute_reply":"2024-05-28T10:14:31.977072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mid_priority_birds = ['Brown-capped Pygmy Woodpecker', 'Chestnut-headed Bee-eater','Crested Goshawk', 'Velvet-fronted Nuthatch', \n                      \"Jerdon's Bushlark\", 'Indian Scimitar-Babbler', 'Plum-headed Parakeet', 'Speckled Piculet', \n                      'Rufous Woodpecker', 'Asian Emerald Dove', 'Golden-fronted Leafbird', 'Green Warbler', \n                      'Indian Blackbird', 'Heart-spotted Woodpecker', 'Little Spiderhunter', 'Rusty-tailed Flycatcher', \n                      'Red-whiskered Bulbul', 'White-browed Bulbul', 'Streak-throated Woodpecker', 'Stork-billed Kingfisher', \n                      'White-rumped Munia', 'Large-billed Leaf Warbler', 'Yellow-billed Babbler', 'Bar-winged Flycatcher-shrike', \n                      'Indian Blue Robin', \"Tickell's Leaf Warbler\"]\nprint(len(mid_priority_birds))\nmpb = list(pd.unique(PL[PL.common_name.isin(mid_priority_birds)]['primary_label']))\nprint(\"total # audio files:\",group_dur[group_dur.primary_label.isin(mpb)]['NUM_AUDIO_FILES'].sum())\nprint(\"possible total # audio files:\",group_dur[group_dur.primary_label.isin(mpb)]['5_second_duration'].sum())\nprint()\ngroup_dur[group_dur.primary_label.isin(mpb)].shape\n","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:31.979648Z","iopub.execute_input":"2024-05-28T10:14:31.979994Z","iopub.status.idle":"2024-05-28T10:14:31.998958Z","shell.execute_reply.started":"2024-05-28T10:14:31.979967Z","shell.execute_reply":"2024-05-28T10:14:31.997800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"low_priority_birds = [\"Forest Wagtail\",\"Greater Racket-tailed Drongo\",\"Gray-headed Canary-Flycatcher\",\n                     \"Indian Pitta\",\"Jungle Babbler\",\"Lesser Yellownape\",\"Pale-billed Flowerpecker\",\n                     \"Gray-bellied Cuckoo\",\"Purple-rumped Sunbird\",\"Thick-billed Warbler\",\n                     \"Tickell's Blue Flycatcher\",]\n\nprint(len(low_priority_birds))\nlpb = list(pd.unique(PL[PL.common_name.isin(low_priority_birds)]['primary_label']))\nprint(\"total # audio files:\",group_dur[group_dur.primary_label.isin(lpb)]['NUM_AUDIO_FILES'].sum())\nprint(\"possible total # audio files:\",group_dur[group_dur.primary_label.isin(lpb)]['5_second_duration'].sum())\nprint()\ngroup_dur[group_dur.primary_label.isin(lpb)]","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:32.000286Z","iopub.execute_input":"2024-05-28T10:14:32.000642Z","iopub.status.idle":"2024-05-28T10:14:32.026631Z","shell.execute_reply.started":"2024-05-28T10:14:32.000603Z","shell.execute_reply":"2024-05-28T10:14:32.025445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### NOCTURNAL BIRDS","metadata":{}},{"cell_type":"code","source":"nb = pd.read_csv(\"/kaggle/input/bc2024-nocturnal-dirunal-birds/Nocturnal_bird_list - final_bird_list.csv\")\nprint(nb.shape)\nnocturnal_birds = list(nb[nb['Dirunal/Nocturnal']=='n']['PRIMARY_COM_NAME'].values)\nnocturnal_birds.remove('Black-crowned Night-Heron')\nprint(nocturnal_birds)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:32.028067Z","iopub.execute_input":"2024-05-28T10:14:32.028411Z","iopub.status.idle":"2024-05-28T10:14:32.038386Z","shell.execute_reply.started":"2024-05-28T10:14:32.028382Z","shell.execute_reply":"2024-05-28T10:14:32.037270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bird_namex = list(pd.unique(PL[PL.common_name.isin(nocturnal_birds)]['primary_label']))\nprint(len(bird_namex))\nprint(\"total # audio files:\",group_dur[group_dur.primary_label.isin(bird_namex)]['NUM_AUDIO_FILES'].sum())\nprint(\"possible total # audio files:\",group_dur[group_dur.primary_label.isin(bird_namex)]['5_second_duration'].sum())\nprint()\ngroup_dur[group_dur.primary_label.isin(bird_namex)]","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:32.039719Z","iopub.execute_input":"2024-05-28T10:14:32.040095Z","iopub.status.idle":"2024-05-28T10:14:32.064445Z","shell.execute_reply.started":"2024-05-28T10:14:32.040067Z","shell.execute_reply":"2024-05-28T10:14:32.063267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"UNLIKELY_P = unlikely_birds+water_birds\nLOW_P      = list(set(low_priority_birds  + low_elevation_birds))\nMID_P      = list(set(mid_priority_birds  + low_mid_birds))\nHIGH_P     = high_priority_birds","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:32.065706Z","iopub.execute_input":"2024-05-28T10:14:32.066036Z","iopub.status.idle":"2024-05-28T10:14:32.071595Z","shell.execute_reply.started":"2024-05-28T10:14:32.066008Z","shell.execute_reply":"2024-05-28T10:14:32.070485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Low Priority Birds:\",len(LOW_P))\nprint(\"Mid Priority Birds:\",len(MID_P))\nprint(\"High Priority Birds:\",len(HIGH_P))\nprint(\"unlikely Birds:\", len(UNLIKELY_P))","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:32.073270Z","iopub.execute_input":"2024-05-28T10:14:32.073600Z","iopub.status.idle":"2024-05-28T10:14:32.082088Z","shell.execute_reply.started":"2024-05-28T10:14:32.073572Z","shell.execute_reply":"2024-05-28T10:14:32.080994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_paths = list(PL.file_path.values)\nprint(len(final_paths))","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:32.083323Z","iopub.execute_input":"2024-05-28T10:14:32.083639Z","iopub.status.idle":"2024-05-28T10:14:32.094186Z","shell.execute_reply.started":"2024-05-28T10:14:32.083614Z","shell.execute_reply":"2024-05-28T10:14:32.093027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### SELF MIX","metadata":{}},{"cell_type":"code","source":"def normalize_equal_shapes(signal,SR=32000,USE_SEC=5):\n    signal = signal/ np.linalg.norm(signal)\n    if len(signal)>USE_SEC*SR:\n                signal = signal[:USE_SEC*SR]\n    else:\n        diff = SR*USE_SEC - len(signal)\n        signal = np.pad(signal, (0,diff), 'constant',)\n    return signal","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:32.095614Z","iopub.execute_input":"2024-05-28T10:14:32.096019Z","iopub.status.idle":"2024-05-28T10:14:32.103538Z","shell.execute_reply.started":"2024-05-28T10:14:32.095987Z","shell.execute_reply":"2024-05-28T10:14:32.102525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef change_speed(data,speed_factor=0.8,use_sec=5,sr=32000):\n    #If rate > 1, then the signal is sped up. If rate < 1, then the signal is slowed down.\n    signal = librosa.effects.time_stretch(data, rate=speed_factor)\n    if len(signal)>USE_SEC*SR:\n        signal = signal[:USE_SEC*SR]\n    else:\n        diff = SR*USE_SEC - len(signal)\n        signal = np.pad(signal, (0,diff), 'constant',)\n    return signal\n\ndef pitch_shift(data, sr=32000, pitch_factor=1.8):\n    #how many (fractional) steps to shift y\n    return librosa.effects.pitch_shift(data, sr=sr, n_steps=pitch_factor)\n\ndef audio_shift(data, sampling_rate=32000, shift_max=1, shift_direction='right'):\n    shift = np.random.randint(sampling_rate * shift_max)\n    if shift_direction == 'right':\n        shift = -shift\n    elif shift_direction == 'both':\n        direction = np.random.randint(0, 2)\n        if direction == 1:\n            shift = -shift\n    augmented_data = np.roll(data, shift)\n    # Set to silence for heading/ tailing\n    if shift > 0:\n        augmented_data[:shift] = 0\n    else:\n        augmented_data[shift:] = 0\n    return augmented_data\n\ndef noise_injection(data, noise_factor=0.05):\n    \n    noise_path = np.random.choice(NP)\n    noise = np.load(noise_path)\n    noise = noise[:len(data)]\n    augmented_data = data + noise_factor * noise\n    # Cast back to same data type\n    augmented_data = augmented_data.astype(data.dtype)\n    return augmented_data\n\ndef normalized_custom_noise_injection(data, noise_factor=2):\n    \n    noise_path = np.random.choice(NP)\n    noise = np.load(noise_path)\n    \n    noise = normalize_equal_shapes(noise,)\n    data = normalize_equal_shapes(data,)\n    augmented_data = data + noise_factor * noise\n    # Cast back to same data type\n    augmented_data =augmented_data.astype(data.dtype)\n    return augmented_data\n        ","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:32.105005Z","iopub.execute_input":"2024-05-28T10:14:32.105434Z","iopub.status.idle":"2024-05-28T10:14:32.121430Z","shell.execute_reply.started":"2024-05-28T10:14:32.105398Z","shell.execute_reply":"2024-05-28T10:14:32.120386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This code runs only in python 3.10 or above versions\ndef apply_augmentations(rand,signal):\n    match rand:\n        case 0:\n            return change_speed(signal,speed_factor=random.choice([0.4,0.6,0.8,1.0,1.2,1.6,1.8]))\n        case 1:\n            return pitch_shift(signal, sr=32000, pitch_factor=random.choice([0.5,1.5,2.0,2.5,3.0,3.5,4.0]))\n        case 2:\n            return audio_shift(signal, sampling_rate=32000, shift_max=random.choice([0.2,0.4,0.5,0.6,0.8,1.0]), \n                               shift_direction=random.choice(['right','both']))\n        case 3:\n            return normalized_custom_noise_injection(signal, noise_factor=random.choice([2.,3.,4.]))\n        case default:\n            return signal","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:32.122705Z","iopub.execute_input":"2024-05-28T10:14:32.123076Z","iopub.status.idle":"2024-05-28T10:14:32.135851Z","shell.execute_reply.started":"2024-05-28T10:14:32.123027Z","shell.execute_reply":"2024-05-28T10:14:32.134593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def SAVE_MIX(FILE1,FILE2,STAGE_LIST,BIRD_CODES):\n    USE_RATING=3.\n    USE_SEC=5\n    SR=32000\n    MIN_SEC=3\n\n    NEW_TARGETS=[]\n    NEW_PATHS=[]\n    MIXED_TAGS=[]\n\n    for n in tqdm(range(len(FILE1))):\n        filea,fileb = FILE1[n],FILE2[n]\n        stage = STAGE_LIST[n]\n        bird_code = BIRD_CODES[n]\n#         bird_code = df_birdlist[df_birdlist.common_name==bird_name]['primary_label'].values[0]\n\n        filea_tag = filea.split(\"/\")[-1].split(\".\")[0]\n        fileb_tag = fileb.split(\"/\")[-1].split(\".\")[0]\n\n        filea_signal = np.load(filea)\n        fileb_signal = np.load(fileb)\n\n        if len(filea_signal)>=int(MIN_SEC*SR) or len(fileb_signal)>int(MIN_SEC*SR): #MINSEC=1\n\n            #FILEA\n            if len(filea_signal)>USE_SEC*SR:\n                    filea_signal = filea_signal[:USE_SEC*SR]\n            else:\n                diff = SR*USE_SEC - len(filea_signal)\n                filea_signal = np.pad(filea_signal, (0,diff), 'constant',)\n\n            #FILEB USE_STAGE\n            start,stop = stage*SR*USE_SEC,(stage+1)*SR*USE_SEC\n            fileb_signal = fileb_signal[start:stop]\n            if len(fileb_signal)>USE_SEC*SR:\n                    fileb_signal = fileb_signal[:USE_SEC*SR]\n            else:\n                diff = SR*USE_SEC - len(fileb_signal)\n                fileb_signal = np.pad(fileb_signal, (0,diff), 'constant',)\n\n\n            bird_target = filea.split(\"/\")[-2]\n            mixed_file_tag = bird_code + \"_\" + filea_tag + \"_\" + fileb_tag + \"_\" + str(stage)\n\n            rand = np.random.randint(4)\n            filea_signal = apply_augmentations(rand,filea_signal)\n\n\n            mixed_signal = (filea_signal / np.linalg.norm(filea_signal))\\\n                                 + (fileb_signal / np.linalg.norm(fileb_signal))\n    #         mixed_signal = filea_signal + fileb_signal\n            save_path = os.path.join(save_folder_path,mixed_file_tag)\n\n            NEW_TARGETS.append(bird_target)\n            NEW_PATHS.append(save_path)\n            MIXED_TAGS.append(mixed_file_tag)\n            np.save(save_path, mixed_signal)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:32.137906Z","iopub.execute_input":"2024-05-28T10:14:32.138373Z","iopub.status.idle":"2024-05-28T10:14:32.154666Z","shell.execute_reply.started":"2024-05-28T10:14:32.138334Z","shell.execute_reply":"2024-05-28T10:14:32.153450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### High Priority ","metadata":{}},{"cell_type":"code","source":"ADD_FILES = 50","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:32.155770Z","iopub.execute_input":"2024-05-28T10:14:32.156175Z","iopub.status.idle":"2024-05-28T10:14:32.168767Z","shell.execute_reply.started":"2024-05-28T10:14:32.156144Z","shell.execute_reply":"2024-05-28T10:14:32.167546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"USE_RATING=3.\nUSE_SEC=5\nSR=32000\nMIN_SEC=3\n\n# CURB_NUM_AUDIO = 100\nMAX_STAGE=6\n\nCOUNTER=0\nFILE1,FILE2 =[],[]\nSTAGE_LIST=[]\nBIRD_CODES = []\n\nbird_dict = {bird_name:[] for bird_name in bird_target_names}\nfor_counter = 0\nfor bird_name in tqdm(HIGH_P):\n    bird_code = df_birdlist[df_birdlist.common_name==bird_name]['primary_label'].values[0]\n    r = re.compile(f\".*{bird_code}\")\n    audio_paths = list(filter(r.match, final_paths))\n    needed_num_files=0\n    mix_count=0\n    HIGH_CURB_NUM_AUDIO = len(audio_paths) + ADD_FILES\n    if len(audio_paths)<HIGH_CURB_NUM_AUDIO:\n        needed_num_files = HIGH_CURB_NUM_AUDIO - len(audio_paths)\n        mix_count= 0\n        stage_count = 0\n        mixed_files = []\n        audio_pathsx = copy.deepcopy(audio_paths)\n        \n        while (mix_count<needed_num_files) and stage_count<MAX_STAGE:\n            if len(audio_pathsx)>1:\n#                 print(\"\\t\\trandom_choices\",len(audio_pathsx))\n                filea, fileb = random.choices(audio_pathsx,k=2)\n                if filea!=fileb:\n    #                 if filea not in mixed_files and fileb not in mixed_files:\n                    if len(mixed_files)<=(len(audio_paths)):\n                        mixed_files.append(filea)\n                        mixed_files.append(fileb)\n                        FILE1.append(filea)\n                        FILE2.append(fileb)\n                        STAGE_LIST.append(stage_count)\n                        BIRD_CODES.append(bird_code)\n                        mix_count +=1\n                        audio_pathsx.remove(filea)\n                        audio_pathsx.remove(fileb)\n                    else:\n                        audio_pathsx = copy.deepcopy(audio_paths)\n                        stage_count +=1\n                        mixed_files=[]\n            \n                else:\n                    continue\n            else:\n                audio_pathsx = copy.deepcopy(audio_paths)\n                stage_count +=1\n                mixed_files=[]\n        \n        COUNTER +=1\n    bird_dict[bird_code]=[len(audio_paths),needed_num_files,mix_count]\n    for_counter +=1\nprint(\"\\n\",COUNTER)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:32.170247Z","iopub.execute_input":"2024-05-28T10:14:32.170670Z","iopub.status.idle":"2024-05-28T10:14:32.549381Z","shell.execute_reply.started":"2024-05-28T10:14:32.170627Z","shell.execute_reply":"2024-05-28T10:14:32.548373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_needed_files = 0\ngot_files=0\nfor bird_name in tqdm(HIGH_P):\n    bird_code = df_birdlist[df_birdlist.common_name==bird_name]['primary_label'].values[0]\n    paths,need_files,actual_files = bird_dict[bird_code]\n    total_needed_files += need_files\n    got_files += actual_files\n\nprint(\"TOTAL NEEDED FILES:\",total_needed_files)\nprint(\"ACTUAL GOT FILES:\",got_files)\nprint(\"DIFF:\",total_needed_files-got_files)\nprint(\"%:\",got_files/total_needed_files)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:32.550720Z","iopub.execute_input":"2024-05-28T10:14:32.551081Z","iopub.status.idle":"2024-05-28T10:14:32.578604Z","shell.execute_reply.started":"2024-05-28T10:14:32.551030Z","shell.execute_reply":"2024-05-28T10:14:32.577649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(FILE1),len(FILE2),len(STAGE_LIST),len(BIRD_CODES)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:32.579698Z","iopub.execute_input":"2024-05-28T10:14:32.580005Z","iopub.status.idle":"2024-05-28T10:14:32.587225Z","shell.execute_reply.started":"2024-05-28T10:14:32.579979Z","shell.execute_reply":"2024-05-28T10:14:32.586186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save_dir = 'high_priority_mixed_self_signals'\nworking_dir = '/kaggle/working/'\n\nsave_folder_path = os.path.join(working_dir,save_dir)\nprint(save_folder_path)\n\n_ = os.makedirs(save_folder_path, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:32.588699Z","iopub.execute_input":"2024-05-28T10:14:32.589127Z","iopub.status.idle":"2024-05-28T10:14:32.597119Z","shell.execute_reply.started":"2024-05-28T10:14:32.589089Z","shell.execute_reply":"2024-05-28T10:14:32.596125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SAVE_MIX(FILE1,FILE2,STAGE_LIST,BIRD_CODES)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:14:32.598638Z","iopub.execute_input":"2024-05-28T10:14:32.599377Z","iopub.status.idle":"2024-05-28T10:17:20.266536Z","shell.execute_reply.started":"2024-05-28T10:14:32.599339Z","shell.execute_reply":"2024-05-28T10:17:20.263676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Mid Priority ","metadata":{}},{"cell_type":"code","source":"ADD_FILES = 30","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:17:20.270406Z","iopub.execute_input":"2024-05-28T10:17:20.274863Z","iopub.status.idle":"2024-05-28T10:17:20.290637Z","shell.execute_reply.started":"2024-05-28T10:17:20.274780Z","shell.execute_reply":"2024-05-28T10:17:20.288898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"USE_RATING=3.\nUSE_SEC=5\nSR=32000\nMIN_SEC=3\n\n# CURB_NUM_AUDIO = 100\nMAX_STAGE=6\n\nCOUNTER=0\nFILE1,FILE2 =[],[]\nSTAGE_LIST=[]\nBIRD_CODES = []\n\nbird_dict = {bird_name:[] for bird_name in bird_target_names}\nfor_counter = 0\nfor bird_name in tqdm(MID_P):\n    bird_code = df_birdlist[df_birdlist.common_name==bird_name]['primary_label'].values[0]\n    r = re.compile(f\".*{bird_code}\")\n    audio_paths = list(filter(r.match, final_paths))\n    needed_num_files=0\n    mix_count=0\n    MID_CURB_NUM_AUDIO = len(audio_paths) + ADD_FILES\n    if len(audio_paths)<MID_CURB_NUM_AUDIO:\n        needed_num_files = MID_CURB_NUM_AUDIO - len(audio_paths)\n        mix_count= 0\n        stage_count = 0\n        mixed_files = []\n        audio_pathsx = copy.deepcopy(audio_paths)\n        \n        while (mix_count<needed_num_files) and stage_count<MAX_STAGE:\n            if len(audio_pathsx)>1:\n#                 print(\"\\t\\trandom_choices\",len(audio_pathsx))\n                filea, fileb = random.choices(audio_pathsx,k=2)\n                if filea!=fileb:\n    #                 if filea not in mixed_files and fileb not in mixed_files:\n                    if len(mixed_files)<=(len(audio_paths)):\n                        mixed_files.append(filea)\n                        mixed_files.append(fileb)\n                        FILE1.append(filea)\n                        FILE2.append(fileb)\n                        STAGE_LIST.append(stage_count)\n                        BIRD_CODES.append(bird_code)\n                        mix_count +=1\n                        audio_pathsx.remove(filea)\n                        audio_pathsx.remove(fileb)\n                    else:\n                        audio_pathsx = copy.deepcopy(audio_paths)\n                        stage_count +=1\n                        mixed_files=[]\n            \n                else:\n                    continue\n            else:\n                audio_pathsx = copy.deepcopy(audio_paths)\n                stage_count +=1\n                mixed_files=[]\n        \n        COUNTER +=1\n    bird_dict[bird_code]=[len(audio_paths),needed_num_files,mix_count]\n    for_counter +=1\nprint(\"\\n\",COUNTER)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:17:20.293338Z","iopub.execute_input":"2024-05-28T10:17:20.294538Z","iopub.status.idle":"2024-05-28T10:17:20.757574Z","shell.execute_reply.started":"2024-05-28T10:17:20.294489Z","shell.execute_reply":"2024-05-28T10:17:20.756303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_needed_files = 0\ngot_files=0\nfor bird_name in tqdm(MID_P):\n    bird_code = df_birdlist[df_birdlist.common_name==bird_name]['primary_label'].values[0]\n    paths,need_files,actual_files = bird_dict[bird_code]\n    total_needed_files += need_files\n    got_files += actual_files\n\nprint(\"TOTAL NEEDED FILES:\",total_needed_files)\nprint(\"ACTUAL GOT FILES:\",got_files)\nprint(\"DIFF:\",total_needed_files-got_files)\nprint(\"%:\",got_files/total_needed_files)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:17:20.759705Z","iopub.execute_input":"2024-05-28T10:17:20.760212Z","iopub.status.idle":"2024-05-28T10:17:20.792702Z","shell.execute_reply.started":"2024-05-28T10:17:20.760180Z","shell.execute_reply":"2024-05-28T10:17:20.791482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save_dir = 'mid_priority_mixed_self_signals'\nworking_dir = '/kaggle/working/'\n\nsave_folder_path = os.path.join(working_dir,save_dir)\nprint(save_folder_path)\n\n_ = os.makedirs(save_folder_path, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:17:20.794254Z","iopub.execute_input":"2024-05-28T10:17:20.794590Z","iopub.status.idle":"2024-05-28T10:17:20.802236Z","shell.execute_reply.started":"2024-05-28T10:17:20.794561Z","shell.execute_reply":"2024-05-28T10:17:20.800981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(FILE1),len(FILE2),len(STAGE_LIST),len(BIRD_CODES)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:17:20.803843Z","iopub.execute_input":"2024-05-28T10:17:20.804363Z","iopub.status.idle":"2024-05-28T10:17:20.817290Z","shell.execute_reply.started":"2024-05-28T10:17:20.804333Z","shell.execute_reply":"2024-05-28T10:17:20.815696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SAVE_MIX(FILE1,FILE2,STAGE_LIST,BIRD_CODES)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:17:20.819425Z","iopub.execute_input":"2024-05-28T10:17:20.820216Z","iopub.status.idle":"2024-05-28T10:19:41.363090Z","shell.execute_reply.started":"2024-05-28T10:17:20.820174Z","shell.execute_reply":"2024-05-28T10:19:41.359786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_high = len(os.listdir(\"/kaggle/working/high_priority_mixed_self_signals\"))\nnum_mid  = len(os.listdir(\"/kaggle/working/mid_priority_mixed_self_signals\"))\n\nprint(\"# of high priority files:\",num_high)\nprint(\"# of mid priority files:\",num_mid)\n\nprint()\nprint(\"# Total FIles:\", num_high+num_mid)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:19:41.365113Z","iopub.execute_input":"2024-05-28T10:19:41.365912Z","iopub.status.idle":"2024-05-28T10:19:41.378849Z","shell.execute_reply.started":"2024-05-28T10:19:41.365870Z","shell.execute_reply":"2024-05-28T10:19:41.377220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selfmix_paths = glob('/kaggle/working/*/*.npy')\n\nprint(selfmix_paths[0])\nprint(\"total selfmix audio files:\", len(selfmix_paths))","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:19:41.381208Z","iopub.execute_input":"2024-05-28T10:19:41.381641Z","iopub.status.idle":"2024-05-28T10:19:41.409283Z","shell.execute_reply.started":"2024-05-28T10:19:41.381600Z","shell.execute_reply":"2024-05-28T10:19:41.408142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_df = pd.DataFrame(selfmix_paths, columns=['file_path'])\npath_df['new_target'] = path_df['file_path'].map(lambda x: x.split(\"/\")[-1].split(\"_\")[0])\nprint(path_df.shape)\npath_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:19:41.411158Z","iopub.execute_input":"2024-05-28T10:19:41.411885Z","iopub.status.idle":"2024-05-28T10:19:41.443876Z","shell.execute_reply.started":"2024-05-28T10:19:41.411846Z","shell.execute_reply":"2024-05-28T10:19:41.442803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for bird_name in nocturnal_birds:\n    bird_code = df_birdlist[df_birdlist.common_name==bird_name]['primary_label'].values[0]\n    num_paths = len(os.listdir(f\"/kaggle/input/birdclef-2024/train_audio/{bird_code}\"))\n    num_files = len(path_df[path_df.new_target==bird_code])\n    print(bird_name,num_paths,num_files)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:19:41.445661Z","iopub.execute_input":"2024-05-28T10:19:41.446397Z","iopub.status.idle":"2024-05-28T10:19:41.482816Z","shell.execute_reply.started":"2024-05-28T10:19:41.446356Z","shell.execute_reply":"2024-05-28T10:19:41.481725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for bird_name in list(df_birdlist.common_name.values):\n    bird_code = df_birdlist[df_birdlist.common_name==bird_name]['primary_label'].values[0]\n    num_paths = len(os.listdir(f\"/kaggle/input/birdclef-2024/train_audio/{bird_code}\"))\n    num_files = len(path_df[path_df.new_target==bird_code])\n    if num_files !=0:\n#         if num_files>100:\n        print(bird_name,num_paths,num_files)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:19:41.485026Z","iopub.execute_input":"2024-05-28T10:19:41.485393Z","iopub.status.idle":"2024-05-28T10:19:42.089477Z","shell.execute_reply.started":"2024-05-28T10:19:41.485362Z","shell.execute_reply":"2024-05-28T10:19:42.088441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}