{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# BirdCLEF 2023 - Image Creation 128 x 256\n\nImage creation notebook for the first simple training kernel.\n\n# Image generation process\n- Compute dB scaled mel power spectrum over 5 seconds interval.\n- Use primary label for each of these intervals.\n- Pad to 5 second images if we have a minimal duration.\n- Consider a maximum duration for a maximum number of images created per file.\n- Create images independently.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import os, pathlib\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm\nfrom pydantic import BaseModel as ConfigBaseModel\nfrom joblib import delayed, Parallel\nimport librosa\nprint(\"librosa:\", librosa.__version__)\nimport tensorflow as tf\nprint(\"tensorflow:\", tf.__version__)\nimport cv2\nprint(\"opencv:\", cv2.__version__)\nfrom IPython.display import Audio","metadata":{"execution":{"iopub.status.busy":"2023-03-28T14:24:47.900891Z","iopub.execute_input":"2023-03-28T14:24:47.901289Z","iopub.status.idle":"2023-03-28T14:24:58.622630Z","shell.execute_reply.started":"2023-03-28T14:24:47.901250Z","shell.execute_reply":"2023-03-28T14:24:58.621264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Config","metadata":{}},{"cell_type":"code","source":"class Config(ConfigBaseModel):\n    # data\n    base_dir = \"/kaggle/input/birdclef-2023/\"\n    train_sound_dir = \"/kaggle/input/birdclef-2023/train_audio/\"\n    path_train = base_dir + \"train_metadata.csv\"\n    path_sample_submission = base_dir + \"sample_submission.csv\"\n    sample_rate = 32_000\n    # spec\n    img_size = (128, 256)\n    seconds = 5\n    num_offset_max = 24\n    min_duration = 0.5\n    n_fft = 2048\n    n_mels = img_size[0]\n    hop_length = (seconds * sample_rate - n_fft) // (img_size[1] - 1) \n    center = False\n    fmin = 500\n    fmax = 12_500\n    top_db = 80\n    # output\n    out_dir = \"/kaggle/working/train/\"\n    jpeg_quality = 100\n    \n\ncfg = Config()\nwith open(\"cfg.json\", \"w\") as f:\n    f.write(cfg.json(indent=2))\ncfg.dict()","metadata":{"execution":{"iopub.status.busy":"2023-03-28T14:24:58.624379Z","iopub.execute_input":"2023-03-28T14:24:58.625839Z","iopub.status.idle":"2023-03-28T14:24:58.646378Z","shell.execute_reply.started":"2023-03-28T14:24:58.625803Z","shell.execute_reply":"2023-03-28T14:24:58.645316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv(cfg.path_train)\ndata[\"path_ogg\"] = cfg.train_sound_dir + data[\"filename\"]","metadata":{"execution":{"iopub.status.busy":"2023-03-28T14:30:14.214385Z","iopub.execute_input":"2023-03-28T14:30:14.215699Z","iopub.status.idle":"2023-03-28T14:30:14.344651Z","shell.execute_reply.started":"2023-03-28T14:30:14.215634Z","shell.execute_reply":"2023-03-28T14:30:14.343524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv(cfg.path_sample_submission)\nlabels = sample_submission.columns[1:].to_list()\nassert labels == sorted(labels), \"labels are not sorted\"\nlabel_encoder = pd.Series(np.arange(len(labels)), index=labels)\ndata[\"label\"] = data[\"primary_label\"].map(label_encoder)","metadata":{"execution":{"iopub.status.busy":"2023-03-28T14:30:21.110494Z","iopub.execute_input":"2023-03-28T14:30:21.111514Z","iopub.status.idle":"2023-03-28T14:30:21.135851Z","shell.execute_reply.started":"2023-03-28T14:30:21.111395Z","shell.execute_reply":"2023-03-28T14:30:21.134360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_duration(rec):\n    return librosa.get_duration(path=rec[\"path_ogg\"])\n\ndef get_duration_df(df):\n    return df.apply(get_duration, axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-03-28T14:30:31.590397Z","iopub.execute_input":"2023-03-28T14:30:31.590865Z","iopub.status.idle":"2023-03-28T14:30:31.596836Z","shell.execute_reply.started":"2023-03-28T14:30:31.590831Z","shell.execute_reply":"2023-03-28T14:30:31.595779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"durations = Parallel(n_jobs=os.cpu_count(), verbose=1, backend='multiprocessing')(\n    delayed(get_duration_df)(sub) \n    for sub in np.array_split(data, os.cpu_count())\n)\ndata[\"duration\"] = pd.concat(durations)","metadata":{"execution":{"iopub.status.busy":"2023-03-28T14:30:33.258248Z","iopub.execute_input":"2023-03-28T14:30:33.258669Z","iopub.status.idle":"2023-03-28T14:31:04.689160Z","shell.execute_reply.started":"2023-03-28T14:30:33.258634Z","shell.execute_reply":"2023-03-28T14:31:04.687859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[\"num_offset\"] = (1 + (data[\"duration\"] - cfg.min_duration) // cfg.seconds).astype('int')\ndata[\"num_offset\"] = data[\"num_offset\"].clip(upper=cfg.num_offset_max)","metadata":{"execution":{"iopub.status.busy":"2023-03-28T14:44:20.586471Z","iopub.execute_input":"2023-03-28T14:44:20.587073Z","iopub.status.idle":"2023-03-28T14:44:20.602117Z","shell.execute_reply.started":"2023-03-28T14:44:20.587018Z","shell.execute_reply":"2023-03-28T14:44:20.600027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data","metadata":{"execution":{"iopub.status.busy":"2023-03-28T14:33:16.153300Z","iopub.execute_input":"2023-03-28T14:33:16.153723Z","iopub.status.idle":"2023-03-28T14:33:16.185053Z","shell.execute_reply.started":"2023-03-28T14:33:16.153691Z","shell.execute_reply":"2023-03-28T14:33:16.183920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Helper function\n\n## Get spectogram image\nIn short we like to use 5 second interval spectograms as input images. But what should be done with the corner cases?\n- There is a maximum number of offset considered for very long audio files.\n- Very short files should be padded with zero to get a minimal length.","metadata":{}},{"cell_type":"code","source":"def get_mel_spec_db(path_ogg, offset):\n    \"\"\"Get dB scaled mel power spectrum\"\"\"\n    required_len = cfg.seconds * cfg.sample_rate\n    sig, dr = librosa.load(path=path_ogg, sr=cfg.sample_rate, offset=(offset * cfg.seconds), duration=cfg.seconds)\n    sig = np.concatenate([sig, np.zeros((required_len - len(sig)), dtype=sig.dtype)])\n    mel_spec = librosa.feature.melspectrogram(\n        y=sig, \n        hop_length=cfg.hop_length,\n        sr=cfg.sample_rate, \n        n_fft=cfg.n_fft, \n        n_mels=cfg.n_mels,\n        center=cfg.center,\n        fmin=cfg.fmin,\n        fmax=cfg.fmax,\n    )\n    mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max, top_db=cfg.top_db)\n    return mel_spec_db","metadata":{"execution":{"iopub.status.busy":"2023-03-28T14:33:36.217994Z","iopub.execute_input":"2023-03-28T14:33:36.218467Z","iopub.status.idle":"2023-03-28T14:33:36.227112Z","shell.execute_reply.started":"2023-03-28T14:33:36.218407Z","shell.execute_reply":"2023-03-28T14:33:36.225541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def normalize_img(img):\n    \"\"\"Normalize to uint8 image range\"\"\"\n    assert img.ndim == 2, \"unexpected dimension\"\n    v_min, v_max = np.min(img), np.max(img)\n    return ((img - v_min) / (v_max - v_min) * 255).astype('uint8')","metadata":{"execution":{"iopub.status.busy":"2023-03-28T14:36:09.018988Z","iopub.execute_input":"2023-03-28T14:36:09.019439Z","iopub.status.idle":"2023-03-28T14:36:09.027105Z","shell.execute_reply.started":"2023-03-28T14:36:09.019402Z","shell.execute_reply":"2023-03-28T14:36:09.025947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_record(rec):\n    \"\"\"Process a single record\"\"\"\n    rec_dir = cfg.out_dir + rec.primary_label\n    os.makedirs(rec_dir, exist_ok=True)\n    stats = []\n    base_stat = {\"label\": rec.label, \"orig_filename\": rec.filename}\n    for offset in range(rec.num_offset):\n        mel_spec_db = get_mel_spec_db(rec.path_ogg, offset=offset)\n        img = normalize_img(mel_spec_db)\n        fname = f\"{pathlib.Path(rec.filename).stem}_{offset}.jpeg\"\n        path_img = os.path.join(rec_dir, fname)\n        ret = cv2.imwrite(path_img, img, [cv2.IMWRITE_JPEG_QUALITY, cfg.jpeg_quality])\n        stat = base_stat.copy()\n        stat.update({\n            \"offset\": offset,\n            \"ret\": ret,\n            \"filename\": \"/\".join(pathlib.Path(path_img).parts[-2:]),\n        })\n        stats.append(stat)\n    return pd.DataFrame(stats)\n\n\ndef process_data(data):\n    \"\"\"Process dataframe\"\"\"\n    errors = []\n    l_stats = []\n    for rec in data.itertuples():\n        try: \n            stats = process_record(rec)\n            l_stats.append(stats)\n        except Exception as err:\n            print(f\"Error reading {rec.filename}: {str(err)}\")\n            errors.append((rec.filename, str(err)))\n    return l_stats, errors","metadata":{"execution":{"iopub.status.busy":"2023-03-28T14:36:26.759007Z","iopub.execute_input":"2023-03-28T14:36:26.760043Z","iopub.status.idle":"2023-03-28T14:36:26.773948Z","shell.execute_reply.started":"2023-03-28T14:36:26.760006Z","shell.execute_reply":"2023-03-28T14:36:26.771679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Dev","metadata":{"execution":{"iopub.status.busy":"2023-03-28T14:02:15.503992Z","iopub.execute_input":"2023-03-28T14:02:15.504956Z","iopub.status.idle":"2023-03-28T14:02:15.513348Z","shell.execute_reply.started":"2023-03-28T14:02:15.504893Z","shell.execute_reply":"2023-03-28T14:02:15.511936Z"}}},{"cell_type":"code","source":"orig_out_dir = cfg.out_dir\nprint(\"original output directory:\", orig_out_dir)\ncfg.out_dir = \"/kaggle/temp/\"\nresults = Parallel(n_jobs=os.cpu_count(), verbose=1, backend='multiprocessing')(\n    delayed(process_data)(sub) for sub in np.array_split(data.sample(2), os.cpu_count())\n)\ncfg.out_dir = orig_out_dir","metadata":{"execution":{"iopub.status.busy":"2023-03-28T14:39:15.927508Z","iopub.execute_input":"2023-03-28T14:39:15.927917Z","iopub.status.idle":"2023-03-28T14:39:22.787247Z","shell.execute_reply.started":"2023-03-28T14:39:15.927883Z","shell.execute_reply":"2023-03-28T14:39:22.786073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tree \"/kaggle/temp/\"","metadata":{"execution":{"iopub.status.busy":"2023-03-28T14:43:26.255913Z","iopub.execute_input":"2023-03-28T14:43:26.256305Z","iopub.status.idle":"2023-03-28T14:43:26.550337Z","shell.execute_reply.started":"2023-03-28T14:43:26.256271Z","shell.execute_reply":"2023-03-28T14:43:26.549386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Run all","metadata":{}},{"cell_type":"code","source":"results = Parallel(n_jobs=os.cpu_count(), verbose=1, backend='multiprocessing')(\n    delayed(process_data)(sub) for sub in np.array_split(data, os.cpu_count())\n)","metadata":{"execution":{"iopub.status.busy":"2023-03-28T14:40:54.789835Z","iopub.execute_input":"2023-03-28T14:40:54.790321Z","iopub.status.idle":"2023-03-28T14:42:33.712799Z","shell.execute_reply.started":"2023-03-28T14:40:54.790278Z","shell.execute_reply":"2023-03-28T14:42:33.711871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"errors = [x for r in results for x in r[1]]\nimg_stats = [x for r in results for x in r[0]]\nif len(img_stats):\n    img_stats = pd.concat(img_stats).reset_index(drop=True)\nimg_stats","metadata":{"execution":{"iopub.status.busy":"2023-03-28T14:42:33.714207Z","iopub.execute_input":"2023-03-28T14:42:33.714549Z","iopub.status.idle":"2023-03-28T14:42:33.769492Z","shell.execute_reply.started":"2023-03-28T14:42:33.714514Z","shell.execute_reply":"2023-03-28T14:42:33.767835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Expected number of images:\", data[\"num_offset\"].sum())","metadata":{"execution":{"iopub.status.busy":"2023-03-28T14:44:31.125357Z","iopub.execute_input":"2023-03-28T14:44:31.125835Z","iopub.status.idle":"2023-03-28T14:44:31.134513Z","shell.execute_reply.started":"2023-03-28T14:44:31.125768Z","shell.execute_reply":"2023-03-28T14:44:31.132599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"errors","metadata":{"execution":{"iopub.status.busy":"2023-03-28T14:44:32.899548Z","iopub.execute_input":"2023-03-28T14:44:32.900585Z","iopub.status.idle":"2023-03-28T14:44:32.906669Z","shell.execute_reply.started":"2023-03-28T14:44:32.900543Z","shell.execute_reply":"2023-03-28T14:44:32.905744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_stats.to_csv(\"img_stats.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-03-28T14:44:33.961154Z","iopub.execute_input":"2023-03-28T14:44:33.961602Z","iopub.status.idle":"2023-03-28T14:44:33.978734Z","shell.execute_reply.started":"2023-03-28T14:44:33.961565Z","shell.execute_reply":"2023-03-28T14:44:33.977161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_bytes(num):\n    for x in ['bytes', 'KB', 'MB', 'GB', 'TB']:\n        if num < 1024.0:\n            return \"%3.1f %s\" % (num, x)\n        num /= 1024.0\n\n        \nbs = sum(os.stat(f).st_size for f in pathlib.Path(cfg.out_dir).glob(\"*/*\"))\nprint(cfg.out_dir, convert_bytes(bs))","metadata":{"execution":{"iopub.status.busy":"2023-03-28T14:44:35.577663Z","iopub.execute_input":"2023-03-28T14:44:35.578402Z","iopub.status.idle":"2023-03-28T14:44:35.597360Z","shell.execute_reply.started":"2023-03-28T14:44:35.578360Z","shell.execute_reply":"2023-03-28T14:44:35.596275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}