{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# birdclef2023 Solution Notebook","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport librosa, glob\nfrom tqdm.notebook import tqdm\n\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-05-27T23:40:04.832552Z","iopub.execute_input":"2023-05-27T23:40:04.832941Z","iopub.status.idle":"2023-05-27T23:40:04.963912Z","shell.execute_reply.started":"2023-05-27T23:40:04.832912Z","shell.execute_reply":"2023-05-27T23:40:04.962804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Setup DataFrame","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/birdclef-2023/train_metadata.csv\")\ndf[\"secondary_labels\"] = df.secondary_labels.apply(eval)\ndf[\"sec_num\"] = df[\"secondary_labels\"].apply(len)\ndf[\"filename_id\"] = df[\"filename\"].apply(lambda x: x.split(\"/\")[-1].replace(\".ogg\",\"\"))\ndf[\"sort_index\"] = df.index\n\nsubmission = pd.read_csv(\"/kaggle/input/birdclef-2023/sample_submission.csv\")\n\nunique_key = list(submission.columns[1:])\nsec_unique_key = df.explode(\"secondary_labels\").dropna(subset=[\"secondary_labels\"])[\"secondary_labels\"].unique()\nlabel2id = {label: label_id for label_id, label in enumerate(sorted(unique_key))}\nid2label = {val: key for key,val in label2id.items()}\ndf.loc[:,\"label_id\"] = df.loc[:,\"primary_label\"].map(label2id)\ndf.loc[:,\"labels_id\"] = df.loc[:,\"secondary_labels\"].apply(lambda x: list(np.vectorize(\n    lambda s: label2id[s])(x)) if len(x)!=0 else -1)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T23:40:04.967698Z","iopub.execute_input":"2023-05-27T23:40:04.968071Z","iopub.status.idle":"2023-05-27T23:40:05.549531Z","shell.execute_reply.started":"2023-05-27T23:40:04.968042Z","shell.execute_reply":"2023-05-27T23:40:05.548506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# examine duration\npathdf = pd.DataFrame(glob.glob(\"/kaggle/input/birdclef-2023/train_audio/**/*.ogg\"),columns=[\"audio_path\"])\npathdf[\"filename_id\"] = pathdf.audio_path.apply(lambda x: x.split(\"/\")[-1].replace(\".ogg\",\"\").replace(\".npy\",\"\"))\naudiosec_df = {}\nfor index, row in tqdm(pathdf.iterrows(),total=len(pathdf)):\n    second = librosa.get_duration(filename=row.audio_path)\n    audiosec_df[row.filename_id] = second\n    \nsecdf = pd.DataFrame.from_dict(audiosec_df, orient=\"index\",columns=[\"sec\"]).reset_index().rename(columns={\"index\":\"filename_id\"})\ndf = df.merge(secdf,on=[\"filename_id\"])\ndf = pd.merge(df,pathdf,on=[\"filename_id\"])","metadata":{"execution":{"iopub.status.busy":"2023-05-27T23:40:05.550745Z","iopub.execute_input":"2023-05-27T23:40:05.551058Z","iopub.status.idle":"2023-05-27T23:51:52.938441Z","shell.execute_reply.started":"2023-05-27T23:40:05.551033Z","shell.execute_reply":"2023-05-27T23:51:52.937589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# detect duplicate\n\ndup_groups = df.groupby([\"latitude\",\"longitude\",\"sec\",\"author\"]).filename_id.apply(list).reset_index()\ndup_groups[\"group_num\"] = dup_groups.filename_id.apply(len)\ndup_filename_id = dup_groups[dup_groups[\"group_num\"] > 1].explode(\"filename_id\")[\"filename_id\"].unique()\n\npathdf = pd.DataFrame(glob.glob(\"/kaggle/input/specpg1/specnm128f514nf1024hl320s15/**.npy\"),columns=[\"spec_path\"])\npathdf[\"filename_id\"] = pathdf.spec_path.apply(lambda x: x.split(\"/\")[-1].replace(\".ogg\",\"\").replace(\".npy\",\"\").split(\"_\")[0])\ndf = pd.merge(df,pathdf, on=[\"filename_id\"])\n\ncdf = df.fillna(-1).groupby([\"latitude\",\"longitude\",\"sec\",\"author\"]).spec_path.apply(list).reset_index()\ncdf[\"group_num\"] = cdf.spec_path.apply(len)\ncdf = cdf[cdf[\"group_num\"] > 1].reset_index(drop=True)\n\ndef cos_sim(v1, v2):\n    return np.dot(v1, v2) / (np.linalg.norm(v1) * np.linalg.norm(v2))\n\ni = 0\nsim_dict = {}\nfor idx, row in cdf.iterrows():\n    check_arr = []\n    for idx, path in enumerate(row.spec_path):\n        check_arr.append(np.load(path))\n    asize = len(check_arr)\n    for jdx in range(asize):\n        for kdx in range(jdx+1, asize):\n            sim = cos_sim(check_arr[jdx].astype(float).flatten(), check_arr[kdx].astype(float).flatten())\n            if sim > 0.995:\n                pair1 = row.spec_path[jdx].split(\"/\")[-1].split(\".\")[0]\n                pair2 = row.spec_path[kdx].split(\"/\")[-1].split(\".\")[0]\n                sim_dict[pair1] = i\n                sim_dict[pair2] = i\n                i = i + 1\n                \ndup_ids = pd.DataFrame.from_dict(sim_dict,orient=\"index\",columns=[\"dup_id\"]).index.unique()\npd.DataFrame.from_dict(sim_dict,orient=\"index\",columns=[\"dup_id\"]).to_csv(\"dup.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:04.061889Z","iopub.execute_input":"2023-05-28T00:07:04.062264Z","iopub.status.idle":"2023-05-28T00:07:05.349787Z","shell.execute_reply.started":"2023-05-28T00:07:04.062238Z","shell.execute_reply":"2023-05-28T00:07:05.347432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## CV Setup","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nimport pandas as pd\n\ndef birds_stratified_split(df, target_col, test_size=0.2):\n    class_counts = df[target_col].value_counts()\n    low_count_classes = class_counts[class_counts < 4].index.tolist() ### Birds with single counts\n\n    df['train'] = df[target_col].isin(low_count_classes)\n\n    train_df, val_df = train_test_split(df[~df['train']], test_size=test_size, stratify=df[~df['train']][target_col], random_state=42)\n\n    train_df = pd.concat([train_df, df[df['train']]], axis=0).reset_index(drop=True)\n\n    # Remove the 'valid' column\n    train_df.drop('train', axis=1, inplace=True)\n    val_df.drop('train', axis=1, inplace=True)\n\n    return train_df, val_df","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:05.351705Z","iopub.status.idle":"2023-05-28T00:07:05.352621Z","shell.execute_reply.started":"2023-05-28T00:07:05.352295Z","shell.execute_reply":"2023-05-28T00:07:05.352325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df, valid_df = birds_stratified_split(df, 'primary_label', 0.2)\nprint(len(df.primary_label.unique()))\nprint(len(train_df.primary_label.unique()))\nprint(len(valid_df.primary_label.unique()))\n\ntrain_df[\"eval\"] = 0\nvalid_df[\"eval\"] = 1\n\ndf = pd.concat([train_df,valid_df]).reset_index(drop=True)\ndf.loc[df.filename_id.isin(dup_filename_id),\"eval\"] = 0\ndf.to_csv(\"train.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:05.354022Z","iopub.status.idle":"2023-05-28T00:07:05.355986Z","shell.execute_reply.started":"2023-05-28T00:07:05.355648Z","shell.execute_reply":"2023-05-28T00:07:05.355676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Preprocessing","metadata":{}},{"cell_type":"code","source":"# get random sampling\nrow = df.iloc[0]\ndisplay(row)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:05.357791Z","iopub.status.idle":"2023-05-28T00:07:05.358749Z","shell.execute_reply.started":"2023-05-28T00:07:05.358429Z","shell.execute_reply":"2023-05-28T00:07:05.358475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Inner Mixup Candidate Rule <br>\n\n1. Random Samplig offset rule <br>\n2. Weak Label to Strong Label(WtoS) Weighted Sampling offset rule <br>\n3. Geometric rule <br>","metadata":{}},{"cell_type":"code","source":"# Random sampling (Not Diverse)\nperiod = 50\nfor _ in range(2):\n    if period >= 30:\n        duration_seconds = librosa.get_duration(filename=row.audio_path,sr=None)\n        #訓練時にはランダムにスタートラインを変える(time shift augmentations)\n        if duration_seconds > max(35, period + 5):\n            offset = random.uniform(0, duration_seconds - period)\n        else:\n            offset = 0\n    else:\n        offset = 0\n        \n    print(offset)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:05.360513Z","iopub.status.idle":"2023-05-28T00:07:05.361424Z","shell.execute_reply.started":"2023-05-28T00:07:05.361146Z","shell.execute_reply":"2023-05-28T00:07:05.361173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# WtoS weight sampling\n\nboxdf = pd.read_csv(\"/kaggle/input/sodaugresult/box.csv\",index_col=0).reset_index().rename(columns={\"index\":\"unique_id\"})\nboxdf[\"filename_id\"] = boxdf[\"unique_id\"].apply(lambda x: x.split(\"_\")[0])\nboxdf[\"time_id\"] = boxdf[\"unique_id\"].apply(lambda x: x.split(\"_\")[1])\nboxdf[\"box_id\"] = boxdf[\"unique_id\"].apply(lambda x: x.split(\"_\")[2])\nboxdf[\"start_x\"] = boxdf[\"time_id\"].astype(int)*500\nboxdf[\"x1\"] = boxdf[\"x1\"] + boxdf[\"start_x\"]\nboxdf[\"y1\"] = boxdf[\"y1\"]\nboxdf[\"x2\"] = boxdf[\"x2\"] + boxdf[\"start_x\"]\nboxdf[\"y2\"] = boxdf[\"y2\"]\n\n# Group the DataFrame by 'filename_id'\ngrouped = boxdf.groupby('filename_id')\n\n# Create a dictionary where the keys are the filename_ids and the values are the corresponding sub-dataframes (gdf)\ngdf_dict = {name: group for name, group in grouped}","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:05.450123Z","iopub.execute_input":"2023-05-28T00:07:05.450835Z","iopub.status.idle":"2023-05-28T00:07:10.028045Z","shell.execute_reply.started":"2023-05-28T00:07:05.450801Z","shell.execute_reply":"2023-05-28T00:07:10.027082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Object Detection Information by YOLOv8\ngdf_dict[row.filename_id]","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:10.029922Z","iopub.execute_input":"2023-05-28T00:07:10.030730Z","iopub.status.idle":"2023-05-28T00:07:10.050194Z","shell.execute_reply.started":"2023-05-28T00:07:10.030698Z","shell.execute_reply":"2023-05-28T00:07:10.049023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gdf = gdf_dict[row.filename_id]\nmask = np.zeros((128, gdf.start_x.max() + 500),dtype=np.float64)\nfor jdx, brow in gdf.iterrows():\n    mask[int(brow.y1):int(brow.y2),int(brow.x1):int(brow.x2)] += brow.conf","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:10.052047Z","iopub.execute_input":"2023-05-28T00:07:10.052630Z","iopub.status.idle":"2023-05-28T00:07:10.072435Z","shell.execute_reply.started":"2023-05-28T00:07:10.052597Z","shell.execute_reply":"2023-05-28T00:07:10.071495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Visualization Mask Map\nplt.figure(figsize=(20,5), dpi=300)\nplt.imshow(mask)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:10.078042Z","iopub.execute_input":"2023-05-28T00:07:10.080322Z","iopub.status.idle":"2023-05-28T00:07:10.747802Z","shell.execute_reply.started":"2023-05-28T00:07:10.080267Z","shell.execute_reply":"2023-05-28T00:07:10.746535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the sampling weights each training periods\nperiod = 5\nsampling_weights = np.lib.stride_tricks.sliding_window_view(mask.max(axis=0), period*100).sum(axis=1)[::100]\nsample_prob = sampling_weights/sampling_weights.sum()\nprint(sample_prob)\n\n# period = 10\n# sampling_weights = np.lib.stride_tricks.sliding_window_view(mask.max(axis=0), period*100).sum(axis=1)[::100]\n# sample_prob = sampling_weights/sampling_weights.sum()\n# print(sample_prob)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:10.749139Z","iopub.execute_input":"2023-05-28T00:07:10.749592Z","iopub.status.idle":"2023-05-28T00:07:10.758303Z","shell.execute_reply.started":"2023-05-28T00:07:10.749550Z","shell.execute_reply":"2023-05-28T00:07:10.756937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"offset = np.random.choice(len(sample_prob),p=sample_prob)\nprint(offset)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:10.760150Z","iopub.execute_input":"2023-05-28T00:07:10.761278Z","iopub.status.idle":"2023-05-28T00:07:10.770379Z","shell.execute_reply.started":"2023-05-28T00:07:10.761237Z","shell.execute_reply":"2023-05-28T00:07:10.769019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Inner Geometric Rule\ncand_df = df[[\"filename_id\",\"latitude\",\"longitude\",\"primary_label\",\"secondary_labels\"]].astype({\"secondary_labels\":str}).dropna().drop_duplicates()\ncand_df[\"latitude\"] = cand_df[\"latitude\"].round(2)\ncand_df[\"longitude\"] = cand_df[\"longitude\"].round(2)\ncand_df = cand_df.merge(cand_df,on=[\"latitude\",\"longitude\",\"primary_label\",\"secondary_labels\"],suffixes=(\"\",\"_cand\"))\ncand_df = cand_df[cand_df.filename_id!=cand_df.filename_id_cand].reset_index(drop=True)\ncand_dict_geo_inner = cand_df.groupby(\"filename_id\").filename_id_cand.apply(list).to_dict()","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:10.772087Z","iopub.execute_input":"2023-05-28T00:07:10.772617Z","iopub.status.idle":"2023-05-28T00:07:11.183558Z","shell.execute_reply.started":"2023-05-28T00:07:10.772577Z","shell.execute_reply":"2023-05-28T00:07:11.182279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cand_dict_geo_inner[row.filename_id]","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:11.184847Z","iopub.execute_input":"2023-05-28T00:07:11.185221Z","iopub.status.idle":"2023-05-28T00:07:11.192760Z","shell.execute_reply.started":"2023-05-28T00:07:11.185191Z","shell.execute_reply":"2023-05-28T00:07:11.191413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.choice(cand_dict_geo_inner[row.filename_id])","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:11.194621Z","iopub.execute_input":"2023-05-28T00:07:11.195204Z","iopub.status.idle":"2023-05-28T00:07:11.205912Z","shell.execute_reply.started":"2023-05-28T00:07:11.195144Z","shell.execute_reply":"2023-05-28T00:07:11.204791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Outer Mixup Candidate Rule <br>\n\n1. Random Samplig rule <br>\n2. Resonance Matrix Samplig rule <br>\n3. Geometric rule <br>","metadata":{}},{"cell_type":"code","source":"row = df.iloc[100]\ndisplay(row)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:11.210738Z","iopub.execute_input":"2023-05-28T00:07:11.211101Z","iopub.status.idle":"2023-05-28T00:07:11.220207Z","shell.execute_reply.started":"2023-05-28T00:07:11.211072Z","shell.execute_reply":"2023-05-28T00:07:11.219163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pair_idx = np.random.choice(len(df))\nrow2 = df.iloc[pair_idx]\nprint(row2)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:11.221914Z","iopub.execute_input":"2023-05-28T00:07:11.222254Z","iopub.status.idle":"2023-05-28T00:07:11.231865Z","shell.execute_reply.started":"2023-05-28T00:07:11.222228Z","shell.execute_reply":"2023-05-28T00:07:11.230739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Matrix Factorization \nmfdf = df[(df.sec_num > 0)][[\"label_id\",\"labels_id\"]].explode(\"labels_id\").reset_index(drop=True)\n\n#generate label_id-list for mixup\nmixup_idlist = mfdf.groupby(\"label_id\").labels_id.apply(list).to_dict()\n\n#extract mixup candidate only single label and one primary_label\nsdf = df[(df.sec_num==0)|(df.primary_label==\"lotcor1\")]\n\n# Random sampling for record number from label_id-list\nid2record = sdf.groupby(\"label_id\").sort_index.apply(list)\n\nrow2 = df.iloc[np.random.choice(id2record[row.label_id])]\nprint(row2)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:11.233402Z","iopub.execute_input":"2023-05-28T00:07:11.233899Z","iopub.status.idle":"2023-05-28T00:07:11.287075Z","shell.execute_reply.started":"2023-05-28T00:07:11.233860Z","shell.execute_reply":"2023-05-28T00:07:11.285921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cand_df = df[[\"filename_id\",\"latitude\",\"longitude\"]].dropna().drop_duplicates()\ncand_df[\"latitude\"] = cand_df[\"latitude\"].round(2)\ncand_df[\"longitude\"] = cand_df[\"longitude\"].round(2)\ncand_df = cand_df.merge(cand_df,on=[\"latitude\",\"longitude\"],suffixes=(\"\",\"_cand\"))\ncand_df = cand_df[cand_df.filename_id!=cand_df.filename_id_cand].reset_index(drop=True)\ncand_dict = cand_df.groupby(\"filename_id\").filename_id_cand.apply(list).to_dict()\n\nnp.random.choice(cand_dict[row.filename_id])","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:11.288779Z","iopub.execute_input":"2023-05-28T00:07:11.289108Z","iopub.status.idle":"2023-05-28T00:07:12.237751Z","shell.execute_reply.started":"2023-05-28T00:07:11.289080Z","shell.execute_reply":"2023-05-28T00:07:12.236521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Example","metadata":{}},{"cell_type":"code","source":"audio1, sr = librosa.load(row.audio_path, sr= 32000)\nlibrosa.display.waveshow(audio1, sr=sr)\nplt.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:12.238936Z","iopub.execute_input":"2023-05-28T00:07:12.239259Z","iopub.status.idle":"2023-05-28T00:07:12.658573Z","shell.execute_reply.started":"2023-05-28T00:07:12.239224Z","shell.execute_reply":"2023-05-28T00:07:12.657431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#5s segment sample prob\nperiod = 5\nsampling_weights = np.lib.stride_tricks.sliding_window_view(mask.max(axis=0), period*100).sum(axis=1)[::100]\nsample_prob = sampling_weights/sampling_weights.sum()\nprint(sample_prob)\n\noffset = np.random.choice(len(sample_prob),p=sample_prob,size=2)\nprint(offset)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:12.659982Z","iopub.execute_input":"2023-05-28T00:07:12.660335Z","iopub.status.idle":"2023-05-28T00:07:12.669797Z","shell.execute_reply.started":"2023-05-28T00:07:12.660306Z","shell.execute_reply":"2023-05-28T00:07:12.668522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"librosa.display.waveshow(audio1[offset[0]*sr:(offset[0]+period)*sr], sr=sr)\nplt.axis('off')\nplt.show()\n\nlibrosa.display.waveshow(audio1[offset[1]*sr:(offset[1]+period)*sr], sr=sr)\nplt.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:12.673023Z","iopub.execute_input":"2023-05-28T00:07:12.673380Z","iopub.status.idle":"2023-05-28T00:07:13.426091Z","shell.execute_reply.started":"2023-05-28T00:07:12.673351Z","shell.execute_reply":"2023-05-28T00:07:13.424976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio1_0 = audio1[offset[0]*sr:(offset[0]+period)*sr]\naudio1_1 = audio1[offset[1]*sr:(offset[1]+period)*sr]","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:13.427643Z","iopub.execute_input":"2023-05-28T00:07:13.428752Z","iopub.status.idle":"2023-05-28T00:07:13.434878Z","shell.execute_reply.started":"2023-05-28T00:07:13.428712Z","shell.execute_reply":"2023-05-28T00:07:13.433666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio2, sr = librosa.load(row2.audio_path, sr= 32000)\n\nlibrosa.display.waveshow(audio2, sr=sr)\nplt.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:13.438221Z","iopub.execute_input":"2023-05-28T00:07:13.438855Z","iopub.status.idle":"2023-05-28T00:07:13.853961Z","shell.execute_reply.started":"2023-05-28T00:07:13.438814Z","shell.execute_reply":"2023-05-28T00:07:13.852819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gdf = gdf_dict[row2.filename_id]\nmask = np.zeros((128, gdf.start_x.max() + 500),dtype=np.float64)\nfor jdx, brow in gdf.iterrows():\n    mask[int(brow.y1):int(brow.y2),int(brow.x1):int(brow.x2)] += brow.conf\n    \n#Visualization Mask Map\nplt.figure(figsize=(20,5), dpi=300)\nplt.imshow(mask)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:13.855796Z","iopub.execute_input":"2023-05-28T00:07:13.856543Z","iopub.status.idle":"2023-05-28T00:07:14.355680Z","shell.execute_reply.started":"2023-05-28T00:07:13.856501Z","shell.execute_reply":"2023-05-28T00:07:14.354653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#5s segment sample prob\nperiod = 5\nsampling_weights = np.lib.stride_tricks.sliding_window_view(mask.max(axis=0), period*100).sum(axis=1)[::100]\nsample_prob = sampling_weights/sampling_weights.sum()\nprint(sample_prob)\n\noffset = np.random.choice(len(sample_prob),p=sample_prob,size=2)\nprint(offset)\n\nlibrosa.display.waveshow(audio2[offset[0]*sr:(offset[0]+period)*sr], sr=sr)\nplt.axis('off')\nplt.show()\nlibrosa.display.waveshow(audio2[offset[1]*sr:(offset[1]+period)*sr], sr=sr)\nplt.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:14.358012Z","iopub.execute_input":"2023-05-28T00:07:14.358332Z","iopub.status.idle":"2023-05-28T00:07:15.072309Z","shell.execute_reply.started":"2023-05-28T00:07:14.358307Z","shell.execute_reply":"2023-05-28T00:07:15.071262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio2_0 = audio2[offset[0]*sr:(offset[0]+period)*sr]\naudio2_1 = audio2[offset[1]*sr:(offset[1]+period)*sr]","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:15.074135Z","iopub.execute_input":"2023-05-28T00:07:15.074596Z","iopub.status.idle":"2023-05-28T00:07:15.080562Z","shell.execute_reply.started":"2023-05-28T00:07:15.074557Z","shell.execute_reply":"2023-05-28T00:07:15.079347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install audiomentations\nimport audiomentations as AA\ntrain_aug = AA.Compose(\n    [\n        AA.AddBackgroundNoise(\n            sounds_path=\"/kaggle/input/birdclef2021-background-noise/ff1010bird_nocall/nocall\", min_snr_in_db=5, max_snr_in_db=10, p=0.5\n        ),\n        AA.AddBackgroundNoise(\n            sounds_path=\"/kaggle/input/birdclef2021-background-noise/train_soundscapes/nocall\", min_snr_in_db=5, max_snr_in_db=10, p=0.5\n        ),\n        AA.AddBackgroundNoise(\n            sounds_path=\"/kaggle/input/birdclef2021-background-noise/aicrowd2020_noise_30sec/noise_30sec\",\n            min_snr_in_db=5,\n            max_snr_in_db=10,\n            p=0.75,\n        ),\n        AA.AddBackgroundNoise(\n            sounds_path=\"/kaggle/input/birdclef2023esc50-sample/useesc50\",\n            min_snr_in_db=5,\n            max_snr_in_db=10,\n            p=0.75,\n        ),\n        AA.AddGaussianSNR(\n            min_snr_in_db=5,max_snr_in_db=10.0,p=0.25\n        )\n    ]\n)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:15.082021Z","iopub.execute_input":"2023-05-28T00:07:15.082401Z","iopub.status.idle":"2023-05-28T00:07:29.675912Z","shell.execute_reply.started":"2023-05-28T00:07:15.082372Z","shell.execute_reply":"2023-05-28T00:07:29.674531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"librosa.display.waveshow(train_aug(samples=audio1_0, sample_rate=32000),sr=32000)\nplt.axis('off')\nplt.show()\nlibrosa.display.waveshow(train_aug(samples=audio1_1, sample_rate=32000),sr=32000)\nplt.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:29.679569Z","iopub.execute_input":"2023-05-28T00:07:29.679951Z","iopub.status.idle":"2023-05-28T00:07:31.271633Z","shell.execute_reply.started":"2023-05-28T00:07:29.679917Z","shell.execute_reply":"2023-05-28T00:07:31.270440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"librosa.display.waveshow(train_aug(samples=audio2_0, sample_rate=32000),sr=32000)\nplt.axis('off')\nplt.show()\nlibrosa.display.waveshow(train_aug(samples=audio2_1, sample_rate=32000),sr=32000)\nplt.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:31.273072Z","iopub.execute_input":"2023-05-28T00:07:31.273433Z","iopub.status.idle":"2023-05-28T00:07:32.899564Z","shell.execute_reply.started":"2023-05-28T00:07:31.273404Z","shell.execute_reply":"2023-05-28T00:07:32.898759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torchaudio, torch\nmelspec_tf = torchaudio.transforms.MelSpectrogram(\n        n_mels = 128,\n        sample_rate= 32000,\n        f_min = 40,\n        f_max = 15000,\n        n_fft = 1024,\n        hop_length= 320,\n        norm = None,\n        power = 2,\n        mel_scale = 'htk'\n)\n\nptodb = torchaudio.transforms.AmplitudeToDB(top_db=None)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:32.900731Z","iopub.execute_input":"2023-05-28T00:07:32.901417Z","iopub.status.idle":"2023-05-28T00:07:32.909041Z","shell.execute_reply.started":"2023-05-28T00:07:32.901387Z","shell.execute_reply":"2023-05-28T00:07:32.908142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spec1_0 = ptodb(melspec_tf(torch.tensor(audio1_0[None,:])))\nspec1_1 = ptodb(melspec_tf(torch.tensor(audio1_1[None,:])))\n\nspec2_0 = ptodb(melspec_tf(torch.tensor(audio2_0[None,:])))\nspec2_1 = ptodb(melspec_tf(torch.tensor(audio2_1[None,:])))","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:32.910411Z","iopub.execute_input":"2023-05-28T00:07:32.911279Z","iopub.status.idle":"2023-05-28T00:07:32.933541Z","shell.execute_reply.started":"2023-05-28T00:07:32.911239Z","shell.execute_reply":"2023-05-28T00:07:32.932409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(spec1_0.numpy()[0])\nplt.axis('off')\nplt.show()\n\nplt.imshow(spec1_1.numpy()[0])\nplt.axis('off')\nplt.show()\n\nplt.imshow(spec2_0.numpy()[0])\nplt.axis('off')\nplt.show()\n\nplt.imshow(spec2_1.numpy()[0])\nplt.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:07:38.198033Z","iopub.execute_input":"2023-05-28T00:07:38.198471Z","iopub.status.idle":"2023-05-28T00:07:38.783867Z","shell.execute_reply.started":"2023-05-28T00:07:38.198422Z","shell.execute_reply":"2023-05-28T00:07:38.782069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_state = np.random.RandomState(42)\ndef get_lambda(batch_size, mixup_alpha):\n    lams = []\n    inv_lams = []\n    for _ in range(batch_size):\n        lam = random_state.beta(mixup_alpha, mixup_alpha, 1)[0]\n        lams.append(lam)\n        inv_lams.append(1.0-lam)\n    return torch.tensor(lams, dtype=torch.float32), torch.tensor(inv_lams, dtype=torch.float32)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:08:39.905706Z","iopub.execute_input":"2023-05-28T00:08:39.906160Z","iopub.status.idle":"2023-05-28T00:08:39.915675Z","shell.execute_reply.started":"2023-05-28T00:08:39.906131Z","shell.execute_reply":"2023-05-28T00:08:39.914313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lam_inner_1, lam_inner_2 = get_lambda(1,2.0)\nspec1 = lam_inner_1[:,None,None]*spec1_0 + lam_inner_2[:,None,None]*spec1_1\n\nlam_inner_1, lam_inner_2 = get_lambda(1,2.0)\nspec2 = lam_inner_1[:,None,None]*spec2_0 + lam_inner_2[:,None,None]*spec2_1","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:13:00.175142Z","iopub.execute_input":"2023-05-28T00:13:00.175658Z","iopub.status.idle":"2023-05-28T00:13:00.189509Z","shell.execute_reply.started":"2023-05-28T00:13:00.175619Z","shell.execute_reply":"2023-05-28T00:13:00.188078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(spec1.numpy()[0])\nplt.axis('off')\nplt.show()\n\nplt.imshow(spec2.numpy()[0])\nplt.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:13:00.361115Z","iopub.execute_input":"2023-05-28T00:13:00.361583Z","iopub.status.idle":"2023-05-28T00:13:00.668509Z","shell.execute_reply.started":"2023-05-28T00:13:00.361551Z","shell.execute_reply":"2023-05-28T00:13:00.666994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lam_outer_1, lam_outer_2 = get_lambda(1,1.0)\nspec = lam_outer_1[:,None,None]*spec1 + lam_outer_2[:,None,None]*spec2","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:13:00.671622Z","iopub.execute_input":"2023-05-28T00:13:00.672311Z","iopub.status.idle":"2023-05-28T00:13:00.696873Z","shell.execute_reply.started":"2023-05-28T00:13:00.672255Z","shell.execute_reply":"2023-05-28T00:13:00.695120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(spec.numpy()[0])\nplt.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-28T00:13:00.869370Z","iopub.execute_input":"2023-05-28T00:13:00.870173Z","iopub.status.idle":"2023-05-28T00:13:01.018161Z","shell.execute_reply.started":"2023-05-28T00:13:00.870129Z","shell.execute_reply":"2023-05-28T00:13:01.016346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}