{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":19596,"databundleVersionId":1292430,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":10899657,"sourceType":"datasetVersion","datasetId":6773990},{"sourceId":10903305,"sourceType":"datasetVersion","datasetId":6776577},{"sourceId":10905088,"sourceType":"datasetVersion","datasetId":6777969}],"dockerImageVersionId":30919,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os, pathlib\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm\nfrom pydantic import BaseModel as ConfigBaseModel\nfrom joblib import delayed, Parallel\nimport librosa\nprint(\"librosa:\", librosa.__version__)\nimport tensorflow as tf\nprint(\"tensorflow:\", tf.__version__)\nimport cv2\nprint(\"opencv:\", cv2.__version__)\nfrom IPython.display import Audio","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from typing import ClassVar, Tuple, Literal\n\nclass Config(ConfigBaseModel):\n    # data\n    base_dir: ClassVar[str] = \"/kaggle/input/birdsong-recognition/\"\n    train_sound_dir: ClassVar[str] = \"/kaggle/input/birdsong-recognition/train_audio/\"\n    path_train: ClassVar[str] = base_dir + \"train.csv\"\n    path_sample_submission: ClassVar[str] = base_dir + \"sample_submission.csv\"\n    sample_rate: ClassVar[int] = 32_000\n    # spec\n    img_size: ClassVar[Tuple[int, int]] = (128, 256)\n    seconds: ClassVar[int] = 5\n    num_offset_max: ClassVar[int] = 24\n    min_duration: ClassVar[float] = 0.5\n    n_fft: ClassVar[int] = 2048\n    n_mels: ClassVar[int] = img_size[0]\n    hop_length: ClassVar[int] = (seconds * sample_rate - n_fft) // (img_size[1] - 1) \n    center: ClassVar[bool] = False\n    fmin: ClassVar[int] = 500\n    fmax: ClassVar[int] = 12_500\n    top_db: ClassVar[int] = 80\n    # output\n    out_dir: ClassVar[str] = \"/kaggle/working/train/\"\n    jpeg_quality: ClassVar[int] = 100\n    \n\ncfg = Config()\nwith open(\"cfg.json\", \"w\") as f:\n    # Use model_dump_json and indent parameter\n    f.write(cfg.model_dump_json(indent=2)) \ncfg.dict()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import glob\ndef load_audio_files(path_patterns):\n    \"\"\"\n    Returns a DataFrame with columns: 'filepath' and 'label'.\n    Assumes that the parent directory of each file is its label.\n    Accepts a single glob pattern (str) or a list of glob patterns.\n    \"\"\"\n    if isinstance(path_patterns, str):\n        path_patterns = [path_patterns]\n    \n    data = []\n    for pattern in path_patterns:\n        file_paths = glob.glob(pattern, recursive=True)\n        for fp in file_paths:\n            label = os.path.basename(os.path.dirname(fp))\n            data.append({'filepath': fp, 'label': label})\n    return pd.DataFrame(data)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":" df = load_audio_files(cfg.train_sound_dir+\"**/*.mp3\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_duration(rec):\n    return librosa.get_duration(path=rec[\"filepath\"])\n\ndef get_duration_df(df):\n    return df.apply(get_duration, axis=1)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"durations = Parallel(n_jobs=os.cpu_count(), verbose=1, backend='multiprocessing')(\n    delayed(get_duration_df)(sub) \n    for sub in np.array_split(df, os.cpu_count())\n)\ndf[\"duration\"] = pd.concat(durations)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[\"num_offset\"] = (1 + (df[\"duration\"] - cfg.min_duration) // cfg.seconds).astype('int')\ndf[\"num_offset\"] = df[\"num_offset\"].clip(upper=cfg.num_offset_max)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head(10)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_mel_spec_db(path_ogg, offset):\n    \"\"\"Get dB scaled mel power spectrum\"\"\"\n    required_len = cfg.seconds * cfg.sample_rate\n    sig, dr = librosa.load(path=path_ogg, sr=cfg.sample_rate, offset=(offset * cfg.seconds), duration=cfg.seconds)\n    sig = np.concatenate([sig, np.zeros((required_len - len(sig)), dtype=sig.dtype)])\n    mel_spec = librosa.feature.melspectrogram(\n        y=sig, \n        hop_length=cfg.hop_length,\n        sr=cfg.sample_rate, \n        n_fft=cfg.n_fft, \n        n_mels=cfg.n_mels,\n        center=cfg.center,\n        fmin=cfg.fmin,\n        fmax=cfg.fmax,\n    )\n    mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max, top_db=cfg.top_db)\n    return mel_spec_db","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def normalize_img(img):\n    \"\"\"Normalize to uint8 image range\"\"\"\n    assert img.ndim == 2, \"unexpected dimension\"\n    v_min, v_max = np.min(img), np.max(img)\n    return ((img - v_min) / (v_max - v_min) * 255).astype('uint8')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_record(rec):\n    \"\"\"Process a single record\"\"\"\n    orig_filename = rec.filepath.split(\"/\")[-1]\n    rec_dir = cfg.out_dir + rec.label\n    os.makedirs(rec_dir, exist_ok=True)\n    stats = []\n    base_stat = {\"label\": rec.label, \"orig_filename\": rec.filepath.split(\"/\")[-1]}\n    for offset in range(rec.num_offset):\n        mel_spec_db = get_mel_spec_db(rec.filepath, offset=offset)\n        img = normalize_img(mel_spec_db)\n        fname = f\"{pathlib.Path(orig_filename).stem}_{offset}.jpeg\"\n        path_img = os.path.join(rec_dir, fname)\n        ret = cv2.imwrite(path_img, img, [cv2.IMWRITE_JPEG_QUALITY, cfg.jpeg_quality])\n        stat = base_stat.copy()\n        stat.update({\n            \"offset\": offset,\n            \"ret\": ret,\n            \"filename\": \"/\".join(pathlib.Path(path_img).parts[-2:]),\n        })\n        stats.append(stat)\n    return pd.DataFrame(stats)\n\n\ndef process_data(data):\n    \"\"\"Process dataframe\"\"\"\n    errors = []\n    l_stats = []\n    orig_filename = rec.filepath.split(\"/\")[-1]\n    for rec in data.itertuples():\n        try: \n            stats = process_record(rec)\n            l_stats.append(stats)\n        except Exception as err:\n            print(f\"Error reading {orig_filename}: {str(err)}\")\n            errors.append((orig_filename, str(err)))\n    return l_stats, errors","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for rec in df.itertuples():\n    print(rec.filepath.split(\"/\")[-1])\n    break","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results = Parallel(n_jobs=os.cpu_count(), verbose=1, backend='multiprocessing')(\n    delayed(process_data)(sub) for sub in np.array_split(df, os.cpu_count())\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"errors = [x for r in results for x in r[1]]\nimg_stats = [x for r in results for x in r[0]]\nif len(img_stats):\n    img_stats = pd.concat(img_stats).reset_index(drop=True)\nimg_stats","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Expected number of images:\", df[\"num_offset\"].sum())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"errors","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img_stats.to_csv(\"img_stats.csv\", index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def convert_bytes(num):\n    for x in ['bytes', 'KB', 'MB', 'GB', 'TB']:\n        if num < 1024.0:\n            return \"%3.1f %s\" % (num, x)\n        num /= 1024.0\n\n        \nbs = sum(os.stat(f).st_size for f in pathlib.Path(cfg.out_dir).glob(\"*/*\"))\nprint(cfg.out_dir, convert_bytes(bs))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\n\n# Thư mục bạn muốn nén\nsource_folder = \"/kaggle/working/train\"\n\n# Tạo file ZIP có tên \"my_folder.zip\" trong /kaggle/working/\nshutil.make_archive(\"/kaggle/working/training\", 'zip', source_folder)\n\nprint(\"Nén xong! Kiểm tra file trong /kaggle/working/\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import FileLink\n\nFileLink(\"/kaggle/working/training.zip\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, pathlib\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3'\nfrom datetime import datetime\nimport collections\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom pydantic import BaseModel as ConfigBaseModel\nimport tensorflow as tf\nprint(\"tensorflow:\", tf.__version__)\n\n\n#tf.config.run_functions_eagerly(True)\n\nimport keras_cv\nprint(\"keras_cv:\", keras_cv.__version__)\nimport tensorflow_io as tfio\nprint(\"tfio:\", tfio.__version__)\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras import regularizers\nfrom tensorflow.keras.layers import *\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:20:05.111452Z","iopub.execute_input":"2025-03-03T10:20:05.111818Z","iopub.status.idle":"2025-03-03T10:20:10.940938Z","shell.execute_reply.started":"2025-03-03T10:20:05.111774Z","shell.execute_reply":"2025-03-03T10:20:10.940209Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"strategy = tf.distribute.MirroredStrategy()\nprint(\"Strategy:\", strategy)\nprint(\"Number of replicas:\", strategy.num_replicas_in_sync)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:20:10.941911Z","iopub.execute_input":"2025-03-03T10:20:10.942346Z","iopub.status.idle":"2025-03-03T10:20:11.107469Z","shell.execute_reply.started":"2025-03-03T10:20:10.942324Z","shell.execute_reply":"2025-03-03T10:20:11.106672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sorted(tf.config.list_logical_devices())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:20:11.109021Z","iopub.execute_input":"2025-03-03T10:20:11.109284Z","iopub.status.idle":"2025-03-03T10:20:11.125074Z","shell.execute_reply.started":"2025-03-03T10:20:11.10926Z","shell.execute_reply":"2025-03-03T10:20:11.124263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from typing import ClassVar\n\nclass Config(ConfigBaseModel):\n    ## general\n    run_ts: ClassVar[str] = datetime.now().strftime(\"%Y-%d-%m %H:%M:%S\") # Add type annotation and ClassVar\n    debug: bool = False  # Add type annotation\n    model_name: str = \"EFF-b0-LSTM\"  # Add type annotation\n    test_size: float = 0.2  # Add type annotation\n    seed: int = 887  # Add type annotation\n    fit_verbose: int = 1 if (os.environ.get('KAGGLE_KERNEL_RUN_TYPE') == \"Interactive\") else 2  # Add type annotation\n    ## data\n    dataset_dir: str = \"/kaggle/input/training-data/training/\"  # Add type annotation\n    path_data: str = \"/kaggle/input/training-data/img_stats.csv\"  # Add type annotation\n    label: str = \"label_int\"  # Add type annotation\n    n_label: int = 264  # Add type annotation\n    img_size: tuple = (128, 256)  # Add type annotation\n    channels: int = 1  # Add type annotation\n    img_shape: tuple = (*img_size, channels)  # Add type annotation\n    ## model\n    base_model_weights: str = \"imagenet\"  # Add type annotation\n    dropout: float = 0.20  # Add type annotation\n    ## training\n    label_smoothing: float = 0.05  # Add type annotation\n    shuffle_size: int = 1028  # Add type annotation\n    steps_per_epoch: int = 400  # Add type annotation\n    batch_size: int = 128  # Add type annotation\n    valid_batch_size: int = batch_size  # Add type annotation\n    epochs: int = 100  # Add type annotation\n    patience: int = 5  # Add type annotation\n    monitor: str = \"val_loss\"  # Add type annotation\n    monitor_mode: str = \"auto\"  # Add type annotation\n    lr: float = 1e-4  # Add type annotation\n    ## aug\n    aug_proba: float = 0.8  # Add type annotation\n\n\n\ncfg = Config()\nwith open(\"cfg.json\", \"w\") as f:\n    # Use model_dump_json() with indent argument\n    f.write(cfg.model_dump_json(indent=2)) \ncfg.dict()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:20:11.126198Z","iopub.execute_input":"2025-03-03T10:20:11.126464Z","iopub.status.idle":"2025-03-03T10:20:11.18967Z","shell.execute_reply.started":"2025-03-03T10:20:11.126431Z","shell.execute_reply":"2025-03-03T10:20:11.188769Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = pd.read_csv(cfg.path_data)\ndata[\"path_img\"] = cfg.dataset_dir + data[\"filename\"]\nif cfg.debug:\n    data = data.iloc[:1000]\ndata\n\ndata_label = dict()\nfor i,v in enumerate(sorted(data[\"label\"].unique())):\n    data_label[v] = i\n\ndata[\"label_int\"] = data['label'].map(data_label)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:20:11.190562Z","iopub.execute_input":"2025-03-03T10:20:11.190843Z","iopub.status.idle":"2025-03-03T10:20:11.45492Z","shell.execute_reply.started":"2025-03-03T10:20:11.190821Z","shell.execute_reply":"2025-03-03T10:20:11.453953Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:20:11.455707Z","iopub.execute_input":"2025-03-03T10:20:11.45596Z","iopub.status.idle":"2025-03-03T10:20:11.469537Z","shell.execute_reply.started":"2025-03-03T10:20:11.455938Z","shell.execute_reply":"2025-03-03T10:20:11.46865Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"AUTOTUNE = tf.data.AUTOTUNE\n\n\ndef show_img_stats(img):\n    if isinstance(img, tf.Tensor):\n        print((img.shape, img.dtype, img.numpy().min(), img.numpy().max()))\n    elif isinstance(img, np.array):\n        print((img.shape, img.dtype, img.min(), img.max()))\n    else:\n        print(f\"unexpected type: {type(img)}\")\n\n\ndef read_image(path_img):\n    img_data = tf.io.read_file(path_img)\n    img = tf.io.decode_jpeg(img_data, channels=cfg.channels)\n    img = tf.reshape(img, cfg.img_shape)\n    img = tf.cast(img, tf.float32)\n    return img\n\n\ndef decode_label(label):\n    return tf.one_hot(label, depth=cfg.n_label)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:20:11.470557Z","iopub.execute_input":"2025-03-03T10:20:11.470894Z","iopub.status.idle":"2025-03-03T10:20:11.476845Z","shell.execute_reply.started":"2025-03-03T10:20:11.470863Z","shell.execute_reply":"2025-03-03T10:20:11.476149Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# class RandomRowMask(keras_cv.layers.BaseImageAugmentationLayer):\n#     def __init__(self, param=10, num_mask=1, **kwargs):\n#         super().__init__(**kwargs)\n#         self.param = param\n#         self.num_mask = num_mask\n    \n#     @tf.function\n#     def augment_image(self, image, transformation=None, **kwargs):\n#         image_shape = tf.shape(image)\n#         num = tf.random.uniform(shape=(), minval=1, maxval=self.num_mask + 1, dtype=tf.int32)\n#         for _ in tf.range(num):\n#             image = tfio.audio.time_mask(tf.squeeze(image), param=self.param)\n#             image = tf.reshape(image, shape=image_shape)\n#         return image\n\n\n# class RandomColumnMask(keras_cv.layers.BaseImageAugmentationLayer):\n#     def __init__(self, param=10, num_mask=1, **kwargs):\n#         super().__init__(**kwargs)\n#         self.param = param\n#         self.num_mask = num_mask\n\n#     @tf.function\n#     def augment_image(self, image, transformation=None, **kwargs):\n#         image_shape = tf.shape(image)\n#         num = tf.random.uniform(shape=(), minval=1, maxval=self.num_mask + 1, dtype=tf.int32)\n#         for _ in tf.range(num):\n#             image = tfio.audio.freq_mask(tf.squeeze(image), param=self.param)\n#             image = tf.reshape(image, shape=image_shape)\n#         return image\n\n\n\n# augmenter = keras_cv.layers.Augmenter(\n#     layers=[\n#         keras_cv.layers.RandomBrightness(factor=0.2),\n#         keras_cv.layers.RandomContrast(factor=0.2,value_range=(0, 255)),\n#         keras_cv.layers.GridMask(ratio_factor=(0.05, 0.10)),\n#         keras_cv.layers.RandomGaussianBlur(kernel_size=2, factor=0.1),\n#         RandomRowMask(10, 3),\n#         RandomColumnMask(40, 2)\n#     ]\n# )\n\n\n# def augment_image(img):\n#     if tf.random.uniform([]) <= cfg.aug_proba:\n#         img = augmenter(img)\n#     return img\n\n\n# # def augment_image(img):\n# #     return tf.cond(\n# #         tf.random.uniform([]) <= cfg.aug_proba,\n# #         lambda: augmenter(img),\n# #         lambda: img\n# #     )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:20:11.479342Z","iopub.execute_input":"2025-03-03T10:20:11.479564Z","iopub.status.idle":"2025-03-03T10:20:11.490153Z","shell.execute_reply.started":"2025-03-03T10:20:11.479544Z","shell.execute_reply":"2025-03-03T10:20:11.489528Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_io as tfio\nimport keras_cv\n\nclass RandomRowMask(keras_cv.layers.BaseImageAugmentationLayer):\n    def __init__(self, param=10, num_mask=1, **kwargs):\n        super().__init__(**kwargs)\n        self.param = param\n        self.num_mask = num_mask\n\n    @tf.function\n    def augment_image(self, image, transformation=None, **kwargs):\n        image_shape = tf.shape(image)\n        num = tf.random.uniform(shape=(), minval=1, maxval=self.num_mask + 1, dtype=tf.int32)\n\n        i = tf.constant(0)\n        def cond(i, img): return tf.less(i, num)\n        def body(i, img):\n            img = tfio.audio.time_mask(tf.squeeze(img), param=self.param)\n            img = tf.reshape(img, shape=image_shape)\n            return i + 1, img\n\n        _, image = tf.while_loop(cond, body, [i, image])\n        return image\n\n\nclass RandomColumnMask(keras_cv.layers.BaseImageAugmentationLayer):\n    def __init__(self, param=10, num_mask=1, **kwargs):\n        super().__init__(**kwargs)\n        self.param = param\n        self.num_mask = num_mask\n\n    @tf.function\n    def augment_image(self, image, transformation=None, **kwargs):\n        image_shape = tf.shape(image)\n        num = tf.random.uniform(shape=(), minval=1, maxval=self.num_mask + 1, dtype=tf.int32)\n\n        i = tf.constant(0)\n        def cond(i, img): return tf.less(i, num)\n        def body(i, img):\n            img = tfio.audio.freq_mask(tf.squeeze(img), param=self.param)\n            img = tf.reshape(img, shape=image_shape)\n            return i + 1, img\n\n        _, image = tf.while_loop(cond, body, [i, image])\n        return image\n\n\naugmenter = keras_cv.layers.Augmenter(\n    layers=[\n        keras_cv.layers.RandomBrightness(factor=0.2),\n        keras_cv.layers.RandomContrast(factor=0.2, value_range=(0, 255)),\n        keras_cv.layers.GridMask(ratio_factor=(0.05, 0.10)),\n        keras_cv.layers.RandomGaussianBlur(kernel_size=2, factor=0.1),\n        RandomRowMask(10, 3),\n        RandomColumnMask(40, 2)\n    ]\n)\n\n\ndef augment_image(img):\n    return tf.cond(\n        tf.random.uniform([]) <= cfg.aug_proba,\n        lambda: augmenter(img),\n        lambda: img\n    )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:20:11.491498Z","iopub.execute_input":"2025-03-03T10:20:11.491905Z","iopub.status.idle":"2025-03-03T10:20:11.527146Z","shell.execute_reply.started":"2025-03-03T10:20:11.491883Z","shell.execute_reply":"2025-03-03T10:20:11.526511Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_dataset(data, include_label=True, repeat=False, shuffle=False, augment=False, prefetch=False, batch_size=None):\n    slices = data[\"path_img\"].values\n    read_func = read_image\n    aug_func = augment_image\n    if include_label:\n        slices = slices, decode_label(data[cfg.label].values)\n        read_func = lambda path_img, label: (read_image(path_img), label)\n        aug_func = lambda img, label: (augment_image(img), label)\n    ds = tf.data.Dataset.from_tensor_slices(slices)\n    ds = ds.map(read_func, num_parallel_calls=AUTOTUNE)\n    if repeat: ds = ds.repeat()\n    if shuffle: ds = ds.shuffle(buffer_size=cfg.shuffle_size)\n    if augment: ds = ds.map(aug_func, num_parallel_calls=AUTOTUNE)\n    if batch_size: ds = ds.batch(batch_size)\n    if prefetch: ds = ds.prefetch(AUTOTUNE)\n    return ds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:20:11.527845Z","iopub.execute_input":"2025-03-03T10:20:11.52807Z","iopub.status.idle":"2025-03-03T10:20:11.533201Z","shell.execute_reply.started":"2025-03-03T10:20:11.528049Z","shell.execute_reply":"2025-03-03T10:20:11.532436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_training_dataset(data):\n    return create_dataset(\n        data,\n        include_label=True,\n        repeat=True,\n        shuffle=True,\n        augment=True,\n        prefetch=True,\n        batch_size=cfg.batch_size,\n    )\n\n\ndef create_validation_dataset(data):\n    return create_dataset(\n        data,\n        include_label=True,\n        repeat=False,\n        shuffle=False,\n        augment=False,\n        prefetch=True,\n        batch_size=cfg.valid_batch_size,\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:20:11.534028Z","iopub.execute_input":"2025-03-03T10:20:11.534294Z","iopub.status.idle":"2025-03-03T10:20:11.548242Z","shell.execute_reply.started":"2025-03-03T10:20:11.534262Z","shell.execute_reply":"2025-03-03T10:20:11.54752Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rec = data.sample(1).iloc[0]\nrec","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:20:11.548994Z","iopub.execute_input":"2025-03-03T10:20:11.549211Z","iopub.status.idle":"2025-03-03T10:20:11.568955Z","shell.execute_reply.started":"2025-03-03T10:20:11.549192Z","shell.execute_reply":"2025-03-03T10:20:11.568202Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img = read_image(rec.path_img)\nfig, axs = plt.subplots(3, 4, sharex='all', sharey='all', figsize=(16, 7))\nfor i, ax in enumerate(axs.flat):\n    if i == 0:\n        ax.imshow(img, cmap='viridis')\n        show_img_stats(img)\n    else:\n        img1 = augmenter(img)\n        ax.imshow(img1, cmap='viridis')\n        show_img_stats(img1)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:20:11.569654Z","iopub.execute_input":"2025-03-03T10:20:11.569873Z","iopub.status.idle":"2025-03-03T10:20:17.850956Z","shell.execute_reply.started":"2025-03-03T10:20:11.569853Z","shell.execute_reply":"2025-03-03T10:20:17.850018Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dev_data = data.sample(500)\ndev_ds = create_training_dataset(dev_data)\ndev_ds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:20:17.851883Z","iopub.execute_input":"2025-03-03T10:20:17.852152Z","iopub.status.idle":"2025-03-03T10:20:18.340341Z","shell.execute_reply.started":"2025-03-03T10:20:17.852128Z","shell.execute_reply":"2025-03-03T10:20:18.339626Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# elem = next(iter(dev_ds.take(1)))\n# elem[1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:20:18.341105Z","iopub.execute_input":"2025-03-03T10:20:18.341347Z","iopub.status.idle":"2025-03-03T10:20:18.344767Z","shell.execute_reply.started":"2025-03-03T10:20:18.341326Z","shell.execute_reply":"2025-03-03T10:20:18.343712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# fig, axs = plt.subplots(3, 4, sharex='all', sharey='all', figsize=(16, 8))\n# for i, ax in enumerate(axs.flat):\n#     img = elem[0][i]\n#     show_img_stats(img)\n#     ax.imshow(img, cmap=\"viridis\")\n#     ax.set_title(f\"label:{np.argmax(elem[1][i].numpy())}\")\n# plt.tight_layout()\n# plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:20:18.345471Z","iopub.execute_input":"2025-03-03T10:20:18.345732Z","iopub.status.idle":"2025-03-03T10:20:18.362534Z","shell.execute_reply.started":"2025-03-03T10:20:18.34571Z","shell.execute_reply":"2025-03-03T10:20:18.361838Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.applications.efficientnet import EfficientNetB7 as BaseModel\nfrom tensorflow.keras.applications.efficientnet import preprocess_input\nfrom tensorflow.keras import layers, losses, metrics, callbacks","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:20:18.363421Z","iopub.execute_input":"2025-03-03T10:20:18.363765Z","iopub.status.idle":"2025-03-03T10:20:18.377459Z","shell.execute_reply.started":"2025-03-03T10:20:18.363722Z","shell.execute_reply":"2025-03-03T10:20:18.37678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def create_model(lr):\n#     inputs = layers.Input(shape=cfg.img_shape, dtype=tf.float32)\n#     x = tf.image.grayscale_to_rgb(inputs)\n#     x = layers.Lambda(preprocess_input, name=\"preprocess_input\")(x)\n#     base_model = BaseModel(include_top=False, weights=cfg.base_model_weights, pooling=\"avg\")\n#     base_model.trainable = True\n#     fine_tune_at = 200  # Specify the number of layers to fine-tune\n#     for layer in base_model.layers[:-fine_tune_at]:\n#         layer.trainable = False\n#     x = base_model(x)\n#     #x = base_model(x, training=False)\n#    # x = base_model.output\n#     x = layers.Flatten()(x)\n#     x =layers.Reshape((1, 2560))(x)\n#     x = layers.GRU(256, return_sequences=False)(x)\n#     x = layers.Dropout(0.5)(x)\n#     x = Dense(64, activation='relu')(x)\n#     outputs = Dense(cfg.n_label, name='logits')(x)\n\n# #     x = layers.Dropout(cfg.dropout, name=\"top_dropout\")(x)\n# #     outputs = layers.Dense(cfg.n_label,kernel_regularizer=regularizers.l2(0.001), name=\"logits\")(x)\n#     model = tf.keras.Model(inputs=inputs, outputs=outputs, name=cfg.model_name)\n#     model.compile(\n#         optimizer=tf.keras.optimizers.Adam(learning_rate=lr,beta_1=0.9,beta_2=0.999,epsilon=1e-08),\n#         loss=tf.keras.losses.CategoricalCrossentropy(from_logits=True, label_smoothing=cfg.label_smoothing),\n#         metrics=['acc']\n#     )\n#     return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:20:18.378306Z","iopub.execute_input":"2025-03-03T10:20:18.378625Z","iopub.status.idle":"2025-03-03T10:20:18.388336Z","shell.execute_reply.started":"2025-03-03T10:20:18.378578Z","shell.execute_reply":"2025-03-03T10:20:18.387554Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras import layers\n\ndef create_model(lr):\n    inputs = layers.Input(shape=cfg.img_shape, dtype=tf.float32)\n\n    # Dùng Lambda Layer để chuyển grayscale thành RGB\n    x = layers.Lambda(lambda img: tf.image.grayscale_to_rgb(img))(inputs)\n\n    x = layers.Lambda(preprocess_input, name=\"preprocess_input\")(x)\n\n    base_model = BaseModel(include_top=False, weights=cfg.base_model_weights, pooling=\"avg\")\n    base_model.trainable = True\n\n    fine_tune_at = 200  # Freeze các layer trước fine_tune_at\n    for layer in base_model.layers[:-fine_tune_at]:\n        layer.trainable = False\n\n    x = base_model(x)\n    x = layers.Flatten()(x)\n    x = layers.Reshape((1, 2560))(x)\n    x = layers.GRU(256, return_sequences=False)(x)\n    x = layers.Dropout(0.5)(x)\n    x = layers.Dense(64, activation='relu')(x)\n    outputs = layers.Dense(cfg.n_label, name='logits')(x)\n\n    model = tf.keras.Model(inputs=inputs, outputs=outputs, name=cfg.model_name)\n\n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(learning_rate=lr, beta_1=0.9, beta_2=0.999, epsilon=1e-08),\n        loss=tf.keras.losses.CategoricalCrossentropy(from_logits=True, label_smoothing=cfg.label_smoothing),\n        metrics=['acc']\n    )\n\n    return model\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:20:18.389206Z","iopub.execute_input":"2025-03-03T10:20:18.389437Z","iopub.status.idle":"2025-03-03T10:20:18.403344Z","shell.execute_reply.started":"2025-03-03T10:20:18.389418Z","shell.execute_reply":"2025-03-03T10:20:18.402685Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tf.keras.backend.clear_session()\nwith strategy.scope():\n    dev_model = create_model(lr=cfg.lr)\ndev_model.summary(line_length=120)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:20:18.40418Z","iopub.execute_input":"2025-03-03T10:20:18.40439Z","iopub.status.idle":"2025-03-03T10:20:24.810388Z","shell.execute_reply.started":"2025-03-03T10:20:18.404371Z","shell.execute_reply":"2025-03-03T10:20:24.809568Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"###\ndev_model.load_weights(\"/kaggle/input/bird-classification-model-weight-l2/weights_EFF-b0-LSTM_0303_L2.weights.h5\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:20:24.811218Z","iopub.execute_input":"2025-03-03T10:20:24.811522Z","iopub.status.idle":"2025-03-03T10:20:32.177972Z","shell.execute_reply.started":"2025-03-03T10:20:24.811483Z","shell.execute_reply":"2025-03-03T10:20:32.176993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dev_model.predict(dev_ds.take(1))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:20:32.179134Z","iopub.execute_input":"2025-03-03T10:20:32.179438Z","iopub.status.idle":"2025-03-03T10:20:47.251489Z","shell.execute_reply.started":"2025-03-03T10:20:32.17941Z","shell.execute_reply":"2025-03-03T10:20:47.250776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dev_model.evaluate(dev_ds.take(1), return_dict=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:20:47.254668Z","iopub.execute_input":"2025-03-03T10:20:47.254903Z","iopub.status.idle":"2025-03-03T10:21:02.969094Z","shell.execute_reply.started":"2025-03-03T10:20:47.254884Z","shell.execute_reply":"2025-03-03T10:21:02.968292Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_callbacks(filepath):\n    \"\"\"Get callbacks\"\"\"\n    cbs = [\n        callbacks.ModelCheckpoint(\n            filepath=filepath,\n            monitor=cfg.monitor,\n            mode=cfg.monitor_mode,\n            verbose=1,\n            save_best_only=True,\n            save_weights_only=True\n        ),\n        callbacks.EarlyStopping(\n            monitor=cfg.monitor,\n            mode=cfg.monitor_mode,\n            verbose=1,\n            patience=cfg.patience,\n            restore_best_weights=False,\n        ),\n    ]\n    return cbs\n\n\ndef show_history(history):\n    \"\"\"Show history\"\"\"\n    history_frame = pd.DataFrame(history.history)\n    history_frame.index = pd.RangeIndex(1, len(history_frame) + 1, name=\"epoch\")\n    display(history_frame.style\\\n        .highlight_min(color='lightgreen', subset=['val_loss'])\\\n        .highlight_max(color='lightgreen', subset=['val_acc'])\n    )\n    fig, ax = plt.subplots(1, 2, figsize=(16, 6))\n    history_frame.loc[:, ['loss', 'val_loss']].plot(ax=ax[0], title='loss')\n    history_frame.loc[:, ['acc', 'val_acc']].plot(ax=ax[1], title='acc')\n    plt.tight_layout()\n    plt.show()\n\n\ndef compute_oof(model, valid_df):\n    \"\"\"Compute OOF\"\"\"\n    valid_ds = create_validation_dataset(valid_df)\n    oof_pred = model.predict(valid_ds, verbose=False)\n    oof_pred = pd.DataFrame(tf.nn.sigmoid(oof_pred).numpy(), index=valid_df.index)\n    oof = pd.concat({\"y_true\": valid_df[cfg.label], \"y_pred\": oof_pred}, axis=1)\n    return oof","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:21:02.97027Z","iopub.execute_input":"2025-03-03T10:21:02.970575Z","iopub.status.idle":"2025-03-03T10:21:02.977625Z","shell.execute_reply.started":"2025-03-03T10:21:02.970551Z","shell.execute_reply":"2025-03-03T10:21:02.97679Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def run_training(train_df, valid_df, model_name):\n    \"\"\"Run training\"\"\"\n    # prepare dataset\n    train_ds = create_training_dataset(train_df)\n    valid_ds = create_validation_dataset(valid_df)\n    # create model\n    tf.keras.backend.clear_session()\n    with strategy.scope():\n        model = create_model(lr=cfg.lr)\n        model.load_weights(\"/kaggle/input/bird-classification-model-weight-l2/weights_EFF-b0-LSTM_0303_L2.weights.h5\")\n    # fit\n    steps_per_epoch = cfg.steps_per_epoch\n    print(\"steps_per_epoch:\", steps_per_epoch)\n    path_weight = f\"/kaggle/working/weights_{model_name}.weights.h5\"\n    print(\"path_weights:\", path_weight)\n    hist = model.fit(\n        train_ds,\n        epochs=cfg.epochs,\n        steps_per_epoch=steps_per_epoch,\n        validation_data=valid_ds,\n        callbacks=get_callbacks(path_weight),\n        verbose=cfg.fit_verbose\n    )\n    # restore\n    model.load_weights(path_weight)\n#     # save full model\n    #does not work: https://github.com/keras-team/keras/pull/17498\n    path_model = f\"/kaggle/working/{model_name}\"\n    print(\"path_model:\", path_model)\n    model.save(path_model+\".h5\")\n    # compute oof\n    oof = compute_oof(model, valid_df)\n    return hist, oof, model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:21:02.978525Z","iopub.execute_input":"2025-03-03T10:21:02.978786Z","iopub.status.idle":"2025-03-03T10:21:02.999127Z","shell.execute_reply.started":"2025-03-03T10:21:02.978762Z","shell.execute_reply":"2025-03-03T10:21:02.998229Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df, valid_df = train_test_split(data, test_size=cfg.test_size, stratify=data[cfg.label])\nvalid_df, test_df = train_test_split(valid_df, test_size=0.5, stratify=valid_df[cfg.label])\nprint(f\"Split: {len(train_df)} vs {len(valid_df)}\")\nmodel_name = f\"{cfg.model_name}\"\nprint(f\"model_name: {model_name}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:21:03.000024Z","iopub.execute_input":"2025-03-03T10:21:03.000328Z","iopub.status.idle":"2025-03-03T10:21:03.153129Z","shell.execute_reply.started":"2025-03-03T10:21:03.000297Z","shell.execute_reply":"2025-03-03T10:21:03.152164Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"hist, oof,model = run_training(train_df, valid_df, model_name)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T10:21:03.15403Z","iopub.execute_input":"2025-03-03T10:21:03.154289Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"show_history(hist)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"oof.to_csv(\"oof.csv\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_ds = create_validation_dataset(test_df)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"true_labels = []\ni=0\ntest_ds_size = test_ds.cardinality().numpy()\nprint(test_ds_size)\nfor batch in test_ds:\n    _, batch_labels = batch  # assuming that labels are the second element of the batch tuple\n    true_labels.extend(batch_labels.numpy().tolist())\n    i+=1","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# pred_labels = model.predict(test_ds, verbose=cfg.fit_verbose, workers=os.cpu_count(), use_multiprocessing=True)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred_labels = model.predict(test_ds, verbose=cfg.fit_verbose)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import average_precision_score\nmAP_score = average_precision_score(true_labels, pred_labels, average='macro')\nprint(mAP_score)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\ntrue_label = np.array(true_labels)\ntrue_label = np.argmax(true_labels, axis=1)\n\npred_label=tf.argmax(pred_labels, axis=1).numpy()\n# # assume y_true and y_pred are your true and predicted labels, respectively\nacc = accuracy_score(true_label, pred_label)\nprint(acc)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import precision_score, recall_score, f1_score\nprecision = precision_score(true_label, pred_label, average='macro')\nrecall = recall_score(true_label, pred_label,  average='macro')\nf1_score = f1_score(true_label, pred_label, average='macro')\n\nprint(\"Precision: \", precision)\nprint(\"Recall: \", recall)\nprint(\"F1_score: \", f1_score)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}