{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from itertools import groupby\nimport numpy as np\nfrom tqdm.notebook import tqdm\ntqdm.pandas()\nimport pandas as pd\nimport os\nimport pickle, torch\nimport cv2, torchaudio\nfrom multiprocessing import Pool\nimport matplotlib.pyplot as plt\n# import cupy as cp\nimport ast, librosa\nimport glob, joblib\n\nimport shutil\n\nfrom joblib import Parallel, delayed\n\nfrom IPython.display import display, HTML\n\nfrom matplotlib import animation, rc\nrc('animation', html='jshtml')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-06T19:25:46.868147Z","iopub.execute_input":"2023-05-06T19:25:46.869691Z","iopub.status.idle":"2023-05-06T19:25:46.884801Z","shell.execute_reply.started":"2023-05-06T19:25:46.869650Z","shell.execute_reply":"2023-05-06T19:25:46.883575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Helper Fuction","metadata":{}},{"cell_type":"code","source":"def voc2yolo(image_height, image_width, bboxes):\n    \"\"\"\n    voc  => [x1, y1, x2, y1]\n    yolo => [xmid, ymid, w, h] (normalized)\n    \"\"\"\n    \n    bboxes = bboxes.copy().astype(float) # otherwise all value will be 0 as voc_pascal dtype is np.int\n    \n    bboxes[..., [0, 2]] = bboxes[..., [0, 2]]/ image_width\n    bboxes[..., [1, 3]] = bboxes[..., [1, 3]]/ image_height\n    \n    w = bboxes[..., 2] - bboxes[..., 0]\n    h = bboxes[..., 3] - bboxes[..., 1]\n    \n    bboxes[..., 0] = bboxes[..., 0] + w/2\n    bboxes[..., 1] = bboxes[..., 1] + h/2\n    bboxes[..., 2] = w\n    bboxes[..., 3] = h\n    \n    return bboxes\n\ndef yolo2voc(image_height, image_width, bboxes):\n    \"\"\"\n    yolo => [xmid, ymid, w, h] (normalized)\n    voc  => [x1, y1, x2, y1]\n    \n    \"\"\" \n    bboxes = bboxes.copy().astype(float) # otherwise all value will be 0 as voc_pascal dtype is np.int\n    \n    bboxes[..., [0, 2]] = bboxes[..., [0, 2]]* image_width\n    bboxes[..., [1, 3]] = bboxes[..., [1, 3]]* image_height\n    \n    bboxes[..., [0, 1]] = bboxes[..., [0, 1]] - bboxes[..., [2, 3]]/2\n    bboxes[..., [2, 3]] = bboxes[..., [0, 1]] + bboxes[..., [2, 3]]\n    \n    return bboxes\n\ndef coco2yolo(image_height, image_width, bboxes):\n    \"\"\"\n    coco => [xmin, ymin, w, h]\n    yolo => [xmid, ymid, w, h] (normalized)\n    \"\"\"\n    \n    bboxes = bboxes.copy().astype(float) # otherwise all value will be 0 as voc_pascal dtype is np.int\n    \n    # normolizinig\n    bboxes[..., [0, 2]]= bboxes[..., [0, 2]]/ image_width\n    bboxes[..., [1, 3]]= bboxes[..., [1, 3]]/ image_height\n    \n    # converstion (xmin, ymin) => (xmid, ymid)\n    bboxes[..., [0, 1]] = bboxes[..., [0, 1]] + bboxes[..., [2, 3]]/2\n    \n    return bboxes\n\ndef yolo2coco(image_height, image_width, bboxes):\n    \"\"\"\n    yolo => [xmid, ymid, w, h] (normalized)\n    coco => [xmin, ymin, w, h]\n    \n    \"\"\" \n    bboxes = bboxes.copy().astype(float) # otherwise all value will be 0 as voc_pascal dtype is np.int\n    \n    # denormalizing\n    bboxes[..., [0, 2]]= bboxes[..., [0, 2]]* image_width\n    bboxes[..., [1, 3]]= bboxes[..., [1, 3]]* image_height\n    \n    # converstion (xmid, ymid) => (xmin, ymin) \n    bboxes[..., [0, 1]] = bboxes[..., [0, 1]] - bboxes[..., [2, 3]]/2\n    \n    return bboxes\n\ndef load_image(image_path):\n    return cv2.cvtColor(cv2.imread(image_path), cv2.COLOR_BGR2RGB)\n\n\ndef plot_one_box(x, img, color=None, label=None, line_thickness=None):\n    # Plots one bounding box on image img\n    tl = line_thickness or round(0.002 * (img.shape[0] + img.shape[1]) / 2) + 1  # line/font thickness\n    color = color or [random.randint(0, 255) for _ in range(3)]\n    c1, c2 = (int(x[0]), int(x[1])), (int(x[2]), int(x[3]))\n    cv2.rectangle(img, c1, c2, color, thickness=tl, lineType=cv2.LINE_AA)\n    if label:\n        tf = max(tl - 1, 1)  # font thickness\n        t_size = cv2.getTextSize(label, 0, fontScale=tl / 3, thickness=tf)[0]\n        c2 = c1[0] + t_size[0], c1[1] - t_size[1] - 3\n        cv2.rectangle(img, c1, c2, color, -1, cv2.LINE_AA)  # filled\n        cv2.putText(img, label, (c1[0], c1[1] - 2), 0, tl / 3, [225, 255, 255], thickness=tf, lineType=cv2.LINE_AA)\n\ndef draw_bboxes(img, bboxes, classes, class_ids, colors = None, show_classes = None, bbox_format = 'yolo', class_name = False, line_thickness = 2):  \n     \n    image = img.copy()\n    show_classes = classes if show_classes is None else show_classes\n    colors = (0, 255 ,0) if colors is None else colors\n    \n    if bbox_format == 'yolo':\n        \n        for idx in range(len(bboxes)):  \n            \n            bbox  = bboxes[idx]\n            cls   = classes[idx]\n            cls_id = class_ids[idx]\n            color = colors[cls_id] if type(colors) is list else colors\n            \n            if cls in show_classes:\n            \n                x1 = round(float(bbox[0])*image.shape[1])\n                y1 = round(float(bbox[1])*image.shape[0])\n                w  = round(float(bbox[2])*image.shape[1]/2) #w/2 \n                h  = round(float(bbox[3])*image.shape[0]/2)\n\n                voc_bbox = (x1-w, y1-h, x1+w, y1+h)\n                plot_one_box(voc_bbox, \n                             image,\n                             color = color,\n                             label = cls if class_name else str(get_label(cls)),\n                             line_thickness = line_thickness)\n            \n    elif bbox_format == 'coco':\n        \n        for idx in range(len(bboxes)):  \n            \n            bbox  = bboxes[idx]\n            cls   = classes[idx]\n            cls_id = class_ids[idx]\n            color = colors[cls_id] if type(colors) is list else colors\n            \n            if cls in show_classes:            \n                x1 = int(round(bbox[0]))\n                y1 = int(round(bbox[1]))\n                w  = int(round(bbox[2]))\n                h  = int(round(bbox[3]))\n\n                voc_bbox = (x1, y1, x1+w, y1+h)\n                plot_one_box(voc_bbox, \n                             image,\n                             color = color,\n                             label = cls if class_name else str(cls_id),\n                             line_thickness = line_thickness)\n\n    elif bbox_format == 'voc_pascal':\n        \n        for idx in range(len(bboxes)):  \n            \n            bbox  = bboxes[idx]\n            cls   = classes[idx]\n            cls_id = class_ids[idx]\n            color = colors[cls_id] if type(colors) is list else colors\n            \n            if cls in show_classes: \n                x1 = int(round(bbox[0]))\n                y1 = int(round(bbox[1]))\n                x2 = int(round(bbox[2]))\n                y2 = int(round(bbox[3]))\n                voc_bbox = (x1, y1, x2, y2)\n                plot_one_box(voc_bbox, \n                             image,\n                             color = color,\n                             label = cls if class_name else str(cls_id),\n                             line_thickness = line_thickness)\n    else:\n        raise ValueError('wrong bbox format')\n\n    return image\n\ndef get_bbox(annots):\n    bboxes = [list(annot.values()) for annot in annots]\n    return bboxes\n\ndef get_imgsize(row):\n    row['width'], row['height'] = imagesize.get(row['image_path'])\n    return row\n\n\n# https://www.kaggle.com/diegoalejogm/great-barrier-reefs-eda-with-animations\ndef create_animation(ims):\n    fig = plt.figure(figsize=(16, 12))\n    plt.axis('off')\n    im = plt.imshow(ims[0])\n\n    def animate_func(i):\n        im.set_array(ims[i])\n        return [im]\n\n    return animation.FuncAnimation(fig, animate_func, frames = len(ims), interval = 1000//12)\n\nnp.random.seed(32)\ncolors = [(np.random.randint(255), np.random.randint(255), np.random.randint(255))\\\n          for idx in range(1)]","metadata":{"execution":{"iopub.status.busy":"2023-05-06T19:25:46.886976Z","iopub.execute_input":"2023-05-06T19:25:46.887286Z","iopub.status.idle":"2023-05-06T19:25:46.921142Z","shell.execute_reply.started":"2023-05-06T19:25:46.887262Z","shell.execute_reply":"2023-05-06T19:25:46.919891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class WaveformDataset:\n    def __init__(self,cfg):\n        \n        self.cfg = cfg\n        self.sr = cfg.sr\n        self.period = cfg.period\n        \n        #wav to image helper\n        self.mel = torchaudio.transforms.MelSpectrogram(\n            n_mels = cfg.n_mel, \n            sample_rate= cfg.sr, \n            f_min = cfg.fmin, \n            f_max = cfg.fmax, \n            n_fft = cfg.n_fft, \n            hop_length=cfg.hop_len,\n            norm = None,\n            power = cfg.power,\n            mel_scale = 'htk')\n        \n        self.ptodb = torchaudio.transforms.AmplitudeToDB(top_db=cfg.top_db)\n        # 保存先のフォルダを作成\n        self.img_folder = f\"/kaggle/audio_images\"\n        os.makedirs(self.img_folder, exist_ok=True)\n        \n    def crop_or_pad(self, y, length, is_train=False, start=None):\n        if len(y) < length:\n            y = np.concatenate([y, np.zeros(length - len(y))])\n\n            n_repeats = length // len(y)\n            epsilon = length % len(y)\n\n            y = np.concatenate([y]*n_repeats + [y[:epsilon]])\n\n        elif len(y) > length:\n            if not is_train:\n                start = start or 0\n            else:\n                start = start or np.random.randint(len(y) - length)\n\n            y = y[start:start + length]\n\n        return y\n    \n    def make_melspec(self, wav):\n        melimg= self.mel(wav)\n        dbimg = self.ptodb(melimg)\n        img = (dbimg.to(torch.float32) + 80)/80\n        return img\n        \n\n    def __call__(self, row):\n        #データ読み込み\n        data, sr = librosa.load(row.audio_path, sr=self.sr)\n\n        #test datasetの最大長 \n        max_sec = len(data)//sr \n        #0秒の場合は１秒として取り扱う\n        max_sec = 1 if max_sec==0 else max_sec\n        \n        #データを5秒間隔でかつ7秒幅を取って区切る\n        datas = [data[int(i * sr):int(min(max_sec, i + self.period) * sr)] for i in range(0, max_sec, self.period)]\n        datas[0] = self.crop_or_pad(datas[0] , length=sr*self.period)\n        datas[-1] = self.crop_or_pad(datas[-1] , length=sr*self.period)\n        \n        audio = torch.tensor(np.stack(datas),dtype=torch.float32)\n        images = self.make_melspec(audio).numpy()\n        \n        for idx, img_numpy in enumerate(images):\n            img_normalized = (img_numpy - img_numpy.min()) / (img_numpy.max() - img_numpy.min()) * 255\n            img_uint8 = img_normalized.astype(np.uint8)\n            img_3_channel = cv2.cvtColor(img_uint8, cv2.COLOR_GRAY2BGR)\n\n            # 画像データを保存\n            cv2.imwrite(f\"/kaggle/audio_images/{row.filename}_{idx}.jpg\", img_3_channel)\n        \ndef get_audios_as_images(df):\n    #pool = joblib.Parallel(4)\n    \n    converter = WaveformDataset(cfg= CFG)\n    #mapper = joblib.delayed(converter)\n    tasks = [converter(row) for idx, row in tqdm(df.iterrows(),total=len(df))]\n    #pool(tqdm(tasks))","metadata":{"execution":{"iopub.status.busy":"2023-05-06T19:25:46.922740Z","iopub.execute_input":"2023-05-06T19:25:46.923091Z","iopub.status.idle":"2023-05-06T19:25:46.939012Z","shell.execute_reply.started":"2023-05-06T19:25:46.923066Z","shell.execute_reply":"2023-05-06T19:25:46.938076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/birdclef-2023/train_metadata.csv\")\ntrain[\"filename\"] = train.filename.apply(lambda x: x.split(\"/\")[-1])\npdf = pd.DataFrame(glob.glob(\"/kaggle/input/birdclef-2023/train_audio/**/*.ogg\"),columns=[\"audio_path\"])\npdf[\"filename\"] = pdf.apply(lambda x: x[\"audio_path\"].split(\"/\")[-1],axis=1)\ntrain = pd.merge(train,pdf,on=[\"filename\"])","metadata":{"execution":{"iopub.status.busy":"2023-05-06T19:25:46.941213Z","iopub.execute_input":"2023-05-06T19:25:46.941716Z","iopub.status.idle":"2023-05-06T19:25:47.406292Z","shell.execute_reply.started":"2023-05-06T19:25:46.941676Z","shell.execute_reply":"2023-05-06T19:25:47.405167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    \n    #切り取る時間(validationが5秒なので5秒)\n    period = 5\n    \n    #切り取るサンプリング周波数 (最大周波数×2を目安として取る場合が多い。)\n    sr = 32000\n    \n    #メル周波数\n    n_mel = 128\n    \n    #最小周波数\n    fmin = 50\n    \n    #最大周波数\n    fmax = 14000\n    \n    power = 2\n    \n    top_db = None\n    \n    n_fft = 1024\n    \n    hop_len = 320","metadata":{"execution":{"iopub.status.busy":"2023-05-06T19:25:47.408197Z","iopub.execute_input":"2023-05-06T19:25:47.408517Z","iopub.status.idle":"2023-05-06T19:25:47.413552Z","shell.execute_reply.started":"2023-05-06T19:25:47.408490Z","shell.execute_reply":"2023-05-06T19:25:47.412862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_audios_as_images(train[:10])","metadata":{"execution":{"iopub.status.busy":"2023-05-06T19:25:47.414952Z","iopub.execute_input":"2023-05-06T19:25:47.415531Z","iopub.status.idle":"2023-05-06T19:25:56.781747Z","shell.execute_reply.started":"2023-05-06T19:25:47.415503Z","shell.execute_reply":"2023-05-06T19:25:56.780767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pip install method (recommended)\n%pip install ultralytics\nimport ultralytics\nultralytics.checks()\n\nfrom ultralytics import YOLO\n\n# Load a model\nmodel = YOLO(\"yolov8x.yaml\")  # build a new model from scratch\n#model = YOLO(\"yolov8n.pt\")\nmodel = YOLO(\"/kaggle/input/birdclef2023objectdetectionweight/version2/train15/weights/best.pt\")\n\nresults = model(\"/kaggle/audio_images/*.jpg\",imgsz=[128,512],conf=0.1)\n\nresult_df = {}\nfor result in results:\n    result_df[result.path] = result.boxes.xyxy.numpy().astype(int)\n    \n# 辞書をファイルに保存\nwith open(\"box.pkl\", \"wb\") as f:\n    pickle.dump(result_df, f)\n    \n    \nresult = {}\nfor key, boxes in result_df.items():\n    filename = key.split(\"/\")[-1].replace(\".ogg\",\"\").replace(\".jpg\",\"\")\n    for box_id, box in enumerate(boxes):\n        result[f\"{filename}_{box_id}\"] = {}\n        for b,k in zip(box,[\"x\",\"y\",\"w\",\"h\"]):\n            result[f\"{filename}_{box_id}\"][k] = b\n            \nboxdf = pd.DataFrame(result).T\nboxdf.to_csv(\"box.csv\")","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-05-06T19:25:56.783787Z","iopub.execute_input":"2023-05-06T19:25:56.784100Z","iopub.status.idle":"2023-05-06T19:31:45.653799Z","shell.execute_reply.started":"2023-05-06T19:25:56.784075Z","shell.execute_reply":"2023-05-06T19:31:45.652662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-05-06T19:45:14.618872Z","iopub.execute_input":"2023-05-06T19:45:14.619413Z","iopub.status.idle":"2023-05-06T19:45:14.652622Z","shell.execute_reply.started":"2023-05-06T19:45:14.619369Z","shell.execute_reply":"2023-05-06T19:45:14.651477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-05-06T19:45:28.353032Z","iopub.execute_input":"2023-05-06T19:45:28.353484Z","iopub.status.idle":"2023-05-06T19:45:28.483763Z","shell.execute_reply.started":"2023-05-06T19:45:28.353447Z","shell.execute_reply":"2023-05-06T19:45:28.482609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-05-06T19:45:35.541359Z","iopub.execute_input":"2023-05-06T19:45:35.542628Z","iopub.status.idle":"2023-05-06T19:45:35.558916Z","shell.execute_reply.started":"2023-05-06T19:45:35.542581Z","shell.execute_reply":"2023-05-06T19:45:35.557977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for result in results:\n#     boxes = result.boxes.xyxy\n#     image = cv2.imread(result.path)\n#     for box in boxes:\n#         x1, y1, x2, y2 = [int(coord) for coord in box]\n#         cv2.rectangle(image, (x1, y1), (x2, y2), (255, 0, 0), 2)  # 青色の線で長方形を描く\n\n#     # 画像を表示する\n#     plt.imshow(image)\n#     plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}