{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%%writefile get_data.sh\n\n#!/bin/bash\n\n# 配列にディレクトリ名を格納\nDIRS=(7525805 7525349 7079124 7079380 7078499 7050014)\n\n# ダウンロードするファイル名を格納\nFILES=(annotations.csv species.csv)\n\n# 配列内の各ディレクトリに対してwgetを実行\nfor dir in ${DIRS[@]}; do\n    mkdir $dir\n    for file in ${FILES[@]}; do\n        wget https://zenodo.org/record/$dir/files/$file -P ./$dir\n    done\ndone","metadata":{"execution":{"iopub.status.busy":"2023-05-05T11:22:55.196312Z","iopub.execute_input":"2023-05-05T11:22:55.197638Z","iopub.status.idle":"2023-05-05T11:22:55.241015Z","shell.execute_reply.started":"2023-05-05T11:22:55.197578Z","shell.execute_reply":"2023-05-05T11:22:55.239371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!chmod 777 get_data.sh\n!./get_data.sh >> log.txt 2>&1","metadata":{"execution":{"iopub.status.busy":"2023-05-05T11:22:55.243562Z","iopub.execute_input":"2023-05-05T11:22:55.244017Z","iopub.status.idle":"2023-05-05T11:23:39.030182Z","shell.execute_reply.started":"2023-05-05T11:22:55.243978Z","shell.execute_reply":"2023-05-05T11:23:39.028154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport random,os, glob\nimport torch, cv2\nimport pandas as pd\nfrom pathlib import Path\nfrom matplotlib import pyplot as plt\nimport torchaudio\nimport librosa\nimport itertools\nimport joblib\nfrom tqdm.notebook import tqdm\n\nimport warnings\n#warnings.filterwarnings('ignore')\nwarnings.simplefilter('ignore')","metadata":{"execution":{"iopub.status.busy":"2023-05-05T11:23:39.035938Z","iopub.execute_input":"2023-05-05T11:23:39.036573Z","iopub.status.idle":"2023-05-05T11:23:43.722098Z","shell.execute_reply.started":"2023-05-05T11:23:39.036508Z","shell.execute_reply":"2023-05-05T11:23:43.720271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"annot_paths = glob.glob(\"**/annotations.csv\")\nfor idx, path in enumerate(annot_paths):\n    df = pd.read_csv(path) if idx==0 else pd.concat([df,pd.read_csv(path)]).reset_index(drop=True)\n    \ndf[\"start_frame\"] = (df[\"Start Time (s)\"]*100).astype(int)\ndf[\"end_frame\"] = (df[\"End Time (s)\"]*100).astype(int)\n\nex_unique_keys = set(df[\"Species eBird Code\"].unique())\n\npdf = pd.DataFrame(glob.glob(\"/kaggle/input/zenodo-*/*.flac\"),columns=[\"path\"])\npdf[\"Filename\"] = pdf.path.apply(lambda x: x.split(\"/\")[-1])\ndf = pd.merge(df,pdf,on=[\"Filename\"])\n\nunique_key = df['Species eBird Code'].unique()\nlabel2id = {label: label_id for label_id, label in enumerate(sorted(unique_key))}\nid2label = {val: key for key,val in label2id.items()}\ndf[\"label_id\"] = df[\"Species eBird Code\"].map(label2id)\n\ndef hz_to_mel(freq):\n    return 2595 * np.log10(1 + freq / 700)\n\ndef freq_to_index(freq, n_mels=128, fmin=40, fmax=15000, n_fft=1024):\n    mel = hz_to_mel(freq)\n    mel_low = hz_to_mel(fmin)\n    mel_high = hz_to_mel(fmax)\n    mel_range = np.linspace(mel_low, mel_high, n_mels + 2)\n    mel_index = np.argmin(np.abs(mel_range - mel))\n    return mel_index - 1\n\ndf[\"start_index_freq\"] = df[\"Low Freq (Hz)\"].apply(freq_to_index).clip(0,128)\ndf[\"end_index_freq\"] = df[\"High Freq (Hz)\"].apply(freq_to_index).clip(0,128)\n\nenddf = df.groupby(\"Filename\").end_frame.max().reset_index()\nenddf = enddf[enddf.end_frame > 1000].reset_index(drop=True)\nval_id = enddf.sort_values(\"end_frame\")[:200].Filename.unique()\n\ntrain = df[~df.Filename.isin(val_id)].reset_index(drop=True)\ntest = df[df.Filename.isin(val_id)].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-05T12:21:42.897369Z","iopub.execute_input":"2023-05-05T12:21:42.899043Z","iopub.status.idle":"2023-05-05T12:21:58.828322Z","shell.execute_reply.started":"2023-05-05T12:21:42.898993Z","shell.execute_reply":"2023-05-05T12:21:58.826899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-05-05T11:23:44.710375Z","iopub.execute_input":"2023-05-05T11:23:44.710924Z","iopub.status.idle":"2023-05-05T11:23:44.744193Z","shell.execute_reply.started":"2023-05-05T11:23:44.710848Z","shell.execute_reply":"2023-05-05T11:23:44.742488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-05-05T11:23:44.746471Z","iopub.execute_input":"2023-05-05T11:23:44.747708Z","iopub.status.idle":"2023-05-05T11:24:00.022859Z","shell.execute_reply.started":"2023-05-05T11:23:44.747638Z","shell.execute_reply":"2023-05-05T11:24:00.021584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-05-05T12:21:22.032563Z","iopub.execute_input":"2023-05-05T12:21:22.033228Z","iopub.status.idle":"2023-05-05T12:21:22.107837Z","shell.execute_reply.started":"2023-05-05T12:21:22.033165Z","shell.execute_reply":"2023-05-05T12:21:22.106162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class WaveformDataset:\n    def __init__(self,\n                 df: pd.DataFrame,\n                 cfg,fold_type=\"train\"):\n        \n        self.df = df\n        self.cfg = cfg\n        self.sr = cfg.sr\n        \n        #wav to image helper\n        self.mel = torchaudio.transforms.MelSpectrogram(\n            n_mels = cfg.n_mel, \n            sample_rate= cfg.sr, \n            f_min = cfg.fmin, \n            f_max = cfg.fmax, \n            n_fft = cfg.n_fft, \n            hop_length=cfg.hop_len,\n            norm = None,\n            power = cfg.power,\n            mel_scale = 'htk')\n        \n        self.ptodb = torchaudio.transforms.AmplitudeToDB(top_db=cfg.top_db)\n        # 保存先のフォルダを作成\n        self.img_folder = f\"/kaggle/audio_images/{fold_type}/\"\n        self.label_folder = f\"/kaggle/audio_labels/{fold_type}/\"\n        os.makedirs(self.img_folder, exist_ok=True)\n        os.makedirs(self.label_folder, exist_ok=True)\n    \n    def make_melspec(self, wav):\n        melimg= self.mel(wav)\n        dbimg = self.ptodb(melimg)\n        img = (dbimg.to(torch.float32) + 80)/80\n        return img\n    \n    def get_xywh(self, row, start_index):\n        xs = max(row.start_frame - start_index, 0)\n        xe = min(row.end_frame- start_index, 500)\n        x = (xs + xe)*0.5 / 500\n        y = (row.start_index_freq + row.end_index_freq)*0.5 /128\n        w = (xe - xs)/500\n        h = (row.end_index_freq - row.start_index_freq)/128 \n        return x,y,w,h\n        \n\n    def __call__(self, gdf):\n        #データ読み込み\n        data, sr = librosa.load(gdf.path.values[0], sr=self.sr)\n\n        #test datasetの最大長 \n        max_sec = len(data)//sr \n        #0秒の場合は１秒として取り扱う\n        max_sec = 1 if max_sec==0 else max_sec\n\n        #データをメル周波数によって画像化\n        audio = torch.tensor(data, dtype=torch.float32)\n        images = self.make_melspec(audio)\n        \n        \n        for idx, row in gdf.sort_values(\"start_frame\").reset_index(drop=True).iterrows():\n                \n            savedata = {}\n            start_index = random.randint(max(0, row.start_frame-50), row.start_frame)\n            savedata[\"img\"] = images[:,start_index:start_index+500]\n            cnd = (gdf.start_frame > start_index)&(gdf.start_frame < start_index + 500)\n            if len(gdf[cnd]) > 0:\n                annotation_data = []\n                for jdx, row in gdf[cnd].reset_index(drop=True).iterrows():\n                    x,y,w,h = self.get_xywh(row, start_index)\n                    annotation_data.append(f\"0 {x} {y} {w} {h}\")\n                \n                #保存\n                img_path = f\"{self.img_folder}/{row.Filename}_{idx}.jpg\"\n                label_path = f\"{self.label_folder}/{row.Filename}_{idx}.txt\"\n\n                # torch.Tensorをnumpy.arrayに変換し、適切な形式に変換\n                img_numpy = savedata[\"img\"].numpy()\n                img_normalized = (img_numpy - img_numpy.min()) / (img_numpy.max() - img_numpy.min()) * 255\n                img_uint8 = img_normalized.astype(np.uint8)\n                img_3_channel = cv2.cvtColor(img_uint8, cv2.COLOR_GRAY2BGR)\n\n                # 画像データを保存\n                cv2.imwrite(img_path, img_3_channel)\n\n                # アノテーションデータを保存\n                with open(label_path, 'w') as f:\n                    f.write(\"\\n\".join(annotation_data))\n        \ndef get_audios_as_images(df,fold_type):    \n    converter = WaveformDataset(df,cfg=cfg,fold_type=fold_type)\n    tasks = [converter(gdf) for idx, gdf in tqdm(df)]","metadata":{"execution":{"iopub.status.busy":"2023-05-05T12:23:59.089551Z","iopub.execute_input":"2023-05-05T12:23:59.090073Z","iopub.status.idle":"2023-05-05T12:23:59.119122Z","shell.execute_reply.started":"2023-05-05T12:23:59.090017Z","shell.execute_reply":"2023-05-05T12:23:59.117562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class cfg:\n    \n    #切り取るサンプリング周波数 (最大周波数×2を目安として取る場合が多い。)\n    sr = 32000\n    \n    #メル周波数\n    n_mel = 128\n    \n    #最小周波数\n    fmin = 40\n    \n    #最大周波数\n    fmax = 15000\n    \n    #FFT周波数\n    n_fft = 1024\n    \n    #hoplen\n    hop_len = 320\n    \n    power = 2\n    \n    top_db = None","metadata":{"execution":{"iopub.status.busy":"2023-05-05T12:23:59.233291Z","iopub.execute_input":"2023-05-05T12:23:59.233745Z","iopub.status.idle":"2023-05-05T12:23:59.240973Z","shell.execute_reply.started":"2023-05-05T12:23:59.233708Z","shell.execute_reply":"2023-05-05T12:23:59.239363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_audios_as_images(train.reset_index(drop=True).groupby(\"Filename\"),fold_type=\"train\")\nget_audios_as_images(test.reset_index(drop=True).groupby(\"Filename\"),fold_type=\"val\")","metadata":{"execution":{"iopub.status.busy":"2023-05-05T12:24:00.624193Z","iopub.execute_input":"2023-05-05T12:24:00.624678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!zip -qjr images_train.zip /kaggle/audio_images/train\n!zip -qjr labels_train.zip /kaggle/audio_labels/train\n!zip -qjr images_val.zip /kaggle/audio_images/val\n!zip -qjr labels_val.zip /kaggle/audio_labels/val","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# torch.load(\"/kaggle/audio_images/HSN_001_20150708_061805.flac_4.pt\")\n# plt.imshow(torch.load(\"/kaggle/audio_images/HSN_001_20150708_061805.flac_4.pt\")[\"img\"].numpy())","metadata":{"execution":{"iopub.status.busy":"2023-05-05T11:24:01.524678Z","iopub.status.idle":"2023-05-05T11:24:01.525178Z","shell.execute_reply.started":"2023-05-05T11:24:01.524940Z","shell.execute_reply":"2023-05-05T11:24:01.524965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df[df.Filename==\"HSN_001_20150708_061805.flac\"]","metadata":{"execution":{"iopub.status.busy":"2023-05-05T11:24:01.526875Z","iopub.status.idle":"2023-05-05T11:24:01.527450Z","shell.execute_reply.started":"2023-05-05T11:24:01.527218Z","shell.execute_reply":"2023-05-05T11:24:01.527245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plt.imshow(tmp['mask'][9000:12000,:])","metadata":{"execution":{"iopub.status.busy":"2023-05-05T11:24:01.530749Z","iopub.status.idle":"2023-05-05T11:24:01.531357Z","shell.execute_reply.started":"2023-05-05T11:24:01.531107Z","shell.execute_reply":"2023-05-05T11:24:01.531135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv(\"train.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-05-05T11:24:01.532810Z","iopub.status.idle":"2023-05-05T11:24:01.533305Z","shell.execute_reply.started":"2023-05-05T11:24:01.533082Z","shell.execute_reply":"2023-05-05T11:24:01.533106Z"},"trusted":true},"execution_count":null,"outputs":[]}]}