{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os, random\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom tensorflow.keras.applications import efficientnet_v2\n\ntf.keras.backend.clear_session()\nAUTO = tf.data.AUTOTUNE\nNUM_CLASSES = 104\nTFREC_SIZE = 512\nIMG_SIZE = 380\nBATCH_SIZE = 64\nVAL_PCT = 5\nMIXCUT = True\nEPOCHS_P1 = 2\nEPOCHS_P2 = 20\nLAST_EPOCHS_NO_MIX = 4\nLABEL_SMOOTH = 0.05\nBASE_LR = 1.2e-4\nWD = 2e-4\nEMA_DECAY = 0.999\nSTEPS_PER_EPOCH = 199\nDO_POLISH = True\nPOLISH_EPOCHS = 1\nPOLISH_LR = 5e-6\n\n# Фиксированный seed для единственного запуска\nSEED = 42\n\n# ---------- Пути к TFRecords ----------\ndef build_paths(tfrec_size=TFREC_SIZE):\n    base = f\"/kaggle/input/tpu-getting-started/tfrecords-jpeg-{tfrec_size}x{tfrec_size}\"\n    if os.path.exists(base):\n        return (os.path.join(base, \"train\", \"*.tfrec\"),\n                os.path.join(base, \"val\", \"*.tfrec\"),\n                os.path.join(base, \"test\", \"*.tfrec\"))\n    base = \"/kaggle/input/tpu-getting-started/tfrecords-jpeg-224x224\"\n    return (os.path.join(base, \"train\", \"*.tfrec\"),\n            os.path.join(base, \"val\", \"*.tfrec\"),\n            os.path.join(base, \"test\", \"*.tfrec\"))\n\nTRAIN_GLOB, VAL_GLOB, TEST_GLOB = build_paths(TFREC_SIZE)\nTRAINVAL_GLOBS = [TRAIN_GLOB, VAL_GLOB]\nprint(f\"Пути к данным: train={TRAIN_GLOB}, val={VAL_GLOB}, test={TEST_GLOB}\")\n\n# ---------- Парсинг TFRecords ----------\ndef _parse_with_id(ex, img_size):\n    ex = tf.io.parse_single_example(ex, {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    })\n    img = tf.image.decode_jpeg(ex[\"image\"], channels=3)\n    img = tf.image.convert_image_dtype(img, tf.float32)\n    img = tf.image.resize(img, (img_size, img_size))\n    y = tf.one_hot(tf.cast(ex[\"class\"], tf.int32), NUM_CLASSES)\n    return img, y, ex[\"id\"]\n\ndef _parse_test(ex, img_size):\n    ex = tf.io.parse_single_example(ex, {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    })\n    img = tf.image.decode_jpeg(ex[\"image\"], channels=3)\n    img = tf.image.convert_image_dtype(img, tf.float32)\n    img = tf.image.resize(img, (img_size, img_size))\n    return img, ex[\"id\"]\n\n# ---------- Holdout ----------\ndef is_holdout_id(id_):\n    bucket = tf.strings.to_hash_bucket_fast(id_, 100)\n    return bucket < VAL_PCT\n\n# ---------- Аугментации ----------\ndef aug_train(img01):\n    img01 = tf.image.random_brightness(img01, 0.10)\n    img01 = tf.image.random_contrast(img01, 0.85, 1.15)\n    img01 = tf.image.random_saturation(img01, 0.85, 1.15)\n    img01 = tf.image.random_flip_left_right(img01)\n    h = tf.shape(img01)[0]; w = tf.shape(img01)[1]\n    max_shift = tf.cast(tf.round(0.07 * tf.cast(tf.minimum(h, w), tf.float32)), tf.int32)\n    max_shift = tf.maximum(max_shift, 0)\n    dx = tf.cond(max_shift > 0, lambda: tf.random.uniform([], -max_shift, max_shift + 1, dtype=tf.int32), lambda: 0)\n    dy = tf.cond(max_shift > 0, lambda: tf.random.uniform([], -max_shift, max_shift + 1, dtype=tf.int32), lambda: 0)\n    img01 = tf.roll(img01, shift=[dy, dx], axis=[0, 1])\n    scale = tf.random.uniform([], 0.90, 1.10)\n    nh = tf.cast(tf.round(scale * tf.cast(h, tf.float32)), tf.int32)\n    nw = tf.cast(tf.round(scale * tf.cast(w, tf.float32)), tf.int32)\n    nh = tf.maximum(nh, 1); nw = tf.maximum(nw, 1)\n    img01 = tf.image.resize(img01, (nh, nw))\n    img01 = tf.image.resize_with_pad(img01, h, w)\n    return tf.clip_by_value(img01, 0.0, 1.0)\n\n# ---------- MixUp / CutMix ----------\ndef sample_beta(shape, alpha):\n    g1 = tf.random.gamma(shape, alpha, dtype=tf.float32)\n    g2 = tf.random.gamma(shape, alpha, dtype=tf.float32)\n    return g1 / (g1 + g2)\n\ndef mixup_batch(x, y, alpha=0.4):\n    b = tf.shape(x)[0]\n    idx = tf.random.shuffle(tf.range(b))\n    x2, y2 = tf.gather(x, idx), tf.gather(y, idx)\n    lam = sample_beta([b], alpha)\n    lam_x = tf.reshape(lam, [b, 1, 1, 1])\n    lam_y = tf.reshape(lam, [b, 1])\n    x = x * lam_x + x2 * (1.0 - lam_x)\n    y = y * lam_y + y2 * (1.0 - lam_y)\n    return x, y\n\ndef rand_bbox(img_h, img_w, lam):\n    cut_rat = tf.sqrt(1.0 - lam)\n    cut_w = tf.cast(tf.round(tf.cast(img_w, tf.float32) * cut_rat), tf.int32)\n    cut_h = tf.cast(tf.round(tf.cast(img_h, tf.float32) * cut_rat), tf.int32)\n    cx = tf.random.uniform([], 0, img_w, dtype=tf.int32)\n    cy = tf.random.uniform([], 0, img_h, dtype=tf.int32)\n    x1 = tf.clip_by_value(cx - cut_w // 2, 0, img_w)\n    y1 = tf.clip_by_value(cy - cut_h // 2, 0, img_h)\n    x2 = tf.clip_by_value(cx + cut_w // 2, 0, img_w)\n    y2 = tf.clip_by_value(cy + cut_h // 2, 0, img_h)\n    return x1, y1, x2, y2\n\ndef cutmix_batch(x, y, alpha=1.0):\n    b = tf.shape(x)[0]\n    h = tf.shape(x)[1]\n    w = tf.shape(x)[2]\n    idx = tf.random.shuffle(tf.range(b))\n    x2, y2 = tf.gather(x, idx), tf.gather(y, idx)\n    lam = sample_beta([b], alpha)\n    def _one(i):\n        xi, yi = x[i], y[i]\n        xj, yj = x2[i], y2[i]\n        li = tf.clip_by_value(lam[i], 1e-3, 1.0 - 1e-3)\n        x1, y1, x2b, y2b = rand_bbox(h, w, li)\n        patch = xj[y1:y2b, x1:x2b, :]\n        pad_left, pad_right = x1, w - x2b\n        pad_top, pad_bottom = y1, h - y2b\n        patch = tf.pad(patch, [[pad_top, pad_bottom],[pad_left, pad_right],[0,0]])\n        mask = tf.pad(tf.ones([y2b-y1, x2b-x1, 1], tf.float32), [[pad_top, pad_bottom],[pad_left, pad_right],[0,0]])\n        x_new = xi * (1.0 - mask) + patch * mask\n        area = tf.cast((x2b-x1)*(y2b-y1), tf.float32)\n        lam_adj = 1.0 - area / tf.cast(h*w, tf.float32)\n        y_new = yi * lam_adj + yj * (1.0 - lam_adj)\n        return x_new, y_new\n    xs, ys = tf.map_fn(_one, tf.range(b), fn_output_signature=(tf.float32, tf.float32))\n    return xs, ys\n\ndef mixcut_batch(x, y, p_cutmix=0.5, mixup_alpha=0.4, cutmix_alpha=1.0):\n    r = tf.random.uniform([])\n    return tf.cond(r < p_cutmix,\n                   lambda: cutmix_batch(x, y, alpha=cutmix_alpha),\n                   lambda: mixup_batch(x, y, alpha=mixup_alpha))\n\n# ---------- Предобработка ----------\ndef preprocess_effv2(img01):\n    return efficientnet_v2.preprocess_input(img01 * 255.0)\n\n# ---------- Датасеты ----------\ndef make_train_ds(globs, img_size, batch, seed=42, use_mixcut=True):\n    files = tf.io.gfile.glob(globs[0]) + tf.io.gfile.glob(globs[1])\n    ds = tf.data.TFRecordDataset(files, num_parallel_reads=AUTO)\n    ds = ds.shuffle(8192, seed=seed, reshuffle_each_iteration=True).repeat()\n    ds = ds.map(lambda ex: _parse_with_id(ex, img_size), num_parallel_calls=AUTO)\n    ds = ds.filter(lambda img, y, id_: tf.logical_not(is_holdout_id(id_)))\n    ds = ds.map(lambda img, y, id_: (aug_train(img), y), num_parallel_calls=AUTO)\n    ds = ds.batch(batch, drop_remainder=True)\n    if use_mixcut:\n        ds = ds.map(lambda x, y: mixcut_batch(x, y, 0.5, 0.4, 1.0), num_parallel_calls=AUTO)\n    ds = ds.map(lambda x, y: (preprocess_effv2(x), y), num_parallel_calls=AUTO)\n    return ds.prefetch(AUTO)\n\ndef make_holdout_ds(globs, img_size, batch):\n    files = tf.io.gfile.glob(globs[0]) + tf.io.gfile.glob(globs[1])\n    ds = tf.data.TFRecordDataset(files, num_parallel_reads=AUTO)\n    ds = ds.map(lambda ex: _parse_with_id(ex, img_size), num_parallel_calls=AUTO)\n    ds = ds.filter(lambda img, y, id_: is_holdout_id(id_))\n    ds = ds.map(lambda img, y, id_: (preprocess_effv2(img), y), num_parallel_calls=AUTO)\n    return ds.batch(batch).prefetch(AUTO)\n\ndef make_full_plain_ds(globs, img_size, batch, seed=42):\n    files = tf.io.gfile.glob(globs[0]) + tf.io.gfile.glob(globs[1])\n    ds = tf.data.TFRecordDataset(files, num_parallel_reads=AUTO)\n    ds = ds.shuffle(8192, seed=seed, reshuffle_each_iteration=True).repeat()\n    ds = ds.map(lambda ex: _parse_with_id(ex, img_size), num_parallel_calls=AUTO)\n    ds = ds.map(lambda img, y, id_: (preprocess_effv2(img), y), num_parallel_calls=AUTO)\n    return ds.batch(batch, drop_remainder=True).prefetch(AUTO)\n\ndef make_test_ds_raw(test_glob, img_size, batch):\n    files = tf.io.gfile.glob(test_glob)\n    ds = tf.data.TFRecordDataset(files, num_parallel_reads=AUTO)\n    ds = ds.map(lambda ex: _parse_test(ex, img_size), num_parallel_calls=AUTO)\n    return ds.batch(batch).prefetch(AUTO)\n\n# ---------- Модель ----------\ndef build_effv2s(img_size, dropout=0.5):\n    inp = layers.Input((img_size, img_size, 3))\n    base = efficientnet_v2.EfficientNetV2S(include_top=False, weights=\"imagenet\", input_tensor=inp)\n    x = layers.GlobalAveragePooling2D()(base.output)\n    x = layers.Dropout(dropout)(x)\n    out = layers.Dense(NUM_CLASSES, activation=\"softmax\", dtype=\"float32\")(x)\n    return models.Model(inp, out), base\n\ndef freeze_bn(model):\n    for layer in model.layers:\n        if isinstance(layer, tf.keras.layers.BatchNormalization):\n            layer.trainable = False\n\n# ---------- EMA ----------\nclass EMA(tf.keras.callbacks.Callback):\n    def __init__(self, decay=0.999, save_path=\"ema.weights.h5\"):\n        super().__init__()\n        self.decay = decay\n        self.save_path = save_path\n        self.shadow = None\n    def on_train_begin(self, logs=None):\n        vars_ = self.model.trainable_variables\n        self.shadow = [tf.identity(tf.cast(v, tf.float32)) for v in vars_]\n    def on_train_batch_end(self, batch, logs=None):\n        vars_ = self.model.trainable_variables\n        for i, v in enumerate(vars_):\n            self.shadow[i] = self.decay * self.shadow[i] + (1.0 - self.decay) * tf.cast(v, tf.float32)\n    def on_train_end(self, logs=None):\n        vars_ = self.model.trainable_variables\n        backup = [tf.identity(v) for v in vars_]\n        for v, s in zip(vars_, self.shadow):\n            v.assign(tf.cast(s, v.dtype))\n        self.model.save_weights(self.save_path)\n        for v, b in zip(vars_, backup):\n            v.assign(b)\n\n# ---------- LR schedule ----------\nclass WarmCos(tf.keras.optimizers.schedules.LearningRateSchedule):\n    def __init__(self, base_lr, total_steps, warm_steps, min_lr=1e-6):\n        super().__init__()\n        self.base_lr = tf.constant(base_lr, tf.float32)\n        self.total_steps = tf.constant(total_steps, tf.float32)\n        self.warm_steps = tf.constant(warm_steps, tf.float32)\n        self.min_lr = tf.constant(min_lr, tf.float32)\n    def __call__(self, step):\n        step = tf.cast(step, tf.float32)\n        warm = self.base_lr * (step / tf.maximum(self.warm_steps, 1.0))\n        t = tf.clip_by_value((step - self.warm_steps) / tf.maximum(self.total_steps - self.warm_steps, 1.0), 0.0, 1.0)\n        cos = self.min_lr + 0.5 * (self.base_lr - self.min_lr) * (1.0 + tf.cos(np.pi * t))\n        return tf.where(step < self.warm_steps, warm, cos)\n\n# ---------- Обучение модели ----------\ntf.keras.backend.clear_session()\nrandom.seed(SEED); np.random.seed(SEED); tf.random.set_seed(SEED)\ntf.keras.mixed_precision.set_global_policy(\"mixed_float16\")\n\ntrain_ds_mix = make_train_ds(TRAINVAL_GLOBS, IMG_SIZE, BATCH_SIZE, seed=SEED, use_mixcut=True)\ntrain_ds_plain = make_train_ds(TRAINVAL_GLOBS, IMG_SIZE, BATCH_SIZE, seed=SEED, use_mixcut=False)\nholdout_ds = make_holdout_ds(TRAINVAL_GLOBS, IMG_SIZE, BATCH_SIZE)\n\nmodel, base = build_effv2s(IMG_SIZE, dropout=0.5)\n\nckpt_path = f\"best_effv2s_{IMG_SIZE}_seed{SEED}.weights.h5\"\nema_path = f\"ema_effv2s_{IMG_SIZE}_seed{SEED}.weights.h5\"\nfinal_path = f\"final_effv2s_{IMG_SIZE}_seed{SEED}.weights.h5\"\n\nckpt = tf.keras.callbacks.ModelCheckpoint(ckpt_path, monitor=\"val_accuracy\", mode=\"max\", save_best_only=True, save_weights_only=True, verbose=1)\nes = tf.keras.callbacks.EarlyStopping(monitor=\"val_accuracy\", mode=\"max\", patience=6, restore_best_weights=False, verbose=1)\n\n# Фаза 1\nprint(f\"[seed={SEED}] Фаза 1: обучение головы (замороженная база), {EPOCHS_P1} эпохи\")\nbase.trainable = False\nmodel.compile(optimizer=tf.keras.optimizers.AdamW(learning_rate=3e-3, weight_decay=1e-4),\n              loss=tf.keras.losses.CategoricalCrossentropy(label_smoothing=LABEL_SMOOTH),\n              metrics=[\"accuracy\"])\nmodel.fit(train_ds_mix, steps_per_epoch=STEPS_PER_EPOCH, validation_data=holdout_ds, epochs=EPOCHS_P1, callbacks=[ckpt], verbose=2)\n\n# Фаза 2\nprint(f\"[seed={SEED}] Фаза 2: файнтюнинг всей модели, {EPOCHS_P2} эпохи\")\nbase.trainable = True\nfreeze_bn(base)\n\ntotal_steps = STEPS_PER_EPOCH * EPOCHS_P2\nsched = WarmCos(base_lr=BASE_LR, total_steps=total_steps, warm_steps=STEPS_PER_EPOCH * 1, min_lr=1e-6)\nopt2 = tf.keras.optimizers.AdamW(learning_rate=sched, weight_decay=WD, global_clipnorm=1.0)\nema_cb = EMA(decay=EMA_DECAY, save_path=ema_path)\n\nepochs_mix = max(0, EPOCHS_P2 - LAST_EPOCHS_NO_MIX)\nif epochs_mix > 0:\n    model.compile(optimizer=opt2, loss=tf.keras.losses.CategoricalCrossentropy(label_smoothing=LABEL_SMOOTH), metrics=[\"accuracy\"])\n    model.fit(train_ds_mix, steps_per_epoch=STEPS_PER_EPOCH, validation_data=holdout_ds,\n              epochs=epochs_mix, callbacks=[ckpt, es, ema_cb], verbose=2)\n\nif LAST_EPOCHS_NO_MIX > 0:\n    print(f\"[seed={SEED}] Последние {LAST_EPOCHS_NO_MIX} эпохи без MixCut и без label smoothing\")\n    model.compile(optimizer=opt2, loss=tf.keras.losses.CategoricalCrossentropy(label_smoothing=0.0), metrics=[\"accuracy\"])\n    model.fit(train_ds_plain, steps_per_epoch=STEPS_PER_EPOCH, validation_data=holdout_ds,\n              epochs=EPOCHS_P2, initial_epoch=epochs_mix, callbacks=[ckpt, es, ema_cb], verbose=2)\n\n# Выбор лучшей модели\nmodel.load_weights(ckpt_path)\nacc_ckpt = model.evaluate(holdout_ds, verbose=0)[1]\nmodel.load_weights(ema_path)\nacc_ema = model.evaluate(holdout_ds, verbose=0)[1]\nbest_path = ckpt_path if acc_ckpt >= acc_ema else ema_path\nprint(f\"[seed={SEED}] Выбор лучшей модели: {'CKPT' if best_path==ckpt_path else 'EMA'} | acc_ckpt={acc_ckpt:.5f} | acc_ema={acc_ema:.5f}\")\n\n# Полировка\nif DO_POLISH:\n    full_plain = make_full_plain_ds(TRAINVAL_GLOBS, IMG_SIZE, BATCH_SIZE, seed=SEED)\n    model.load_weights(best_path)\n    model.compile(optimizer=tf.keras.optimizers.AdamW(learning_rate=POLISH_LR, weight_decay=0.0, global_clipnorm=1.0),\n                  loss=tf.keras.losses.CategoricalCrossentropy(label_smoothing=0.0),\n                  metrics=[\"accuracy\"])\n    model.fit(full_plain, steps_per_epoch=STEPS_PER_EPOCH, epochs=POLISH_EPOCHS, verbose=2)\n    model.save_weights(final_path)\n    best_path = final_path\n\n# ---------- TTA ----------\ndef tta_log_probs_raw(model, xb_raw01, img_size):\n    views = []\n    def add(v):\n        v = tf.image.resize(v, (img_size, img_size))\n        views.append(preprocess_effv2(v))\n    add(xb_raw01)\n    add(tf.image.flip_left_right(xb_raw01))\n    add(tf.image.rot90(xb_raw01, 1))\n    add(tf.image.rot90(xb_raw01, 3))\n    c95 = tf.image.central_crop(xb_raw01, 0.95)\n    c90 = tf.image.central_crop(xb_raw01, 0.90)\n    add(c95); add(tf.image.flip_left_right(c95))\n    add(c90); add(tf.image.flip_left_right(c90))\n    acc = None\n    for v in views:\n        p = model.predict(v, verbose=0)\n        p = np.clip(p, 1e-7, 1.0)\n        lg = np.log(p)\n        acc = lg if acc is None else (acc + lg)\n    return acc / len(views)\n\n# ---------- Инференс модели ----------\ntf.keras.backend.clear_session()\ntf.keras.mixed_precision.set_global_policy(\"mixed_float16\")\n\nmodel, _ = build_effv2s(IMG_SIZE, dropout=0.0)  # dropout off на инференсе\nmodel.load_weights(best_path)\n\ntest_ds = make_test_ds_raw(TEST_GLOB, IMG_SIZE, batch=64)\n\nlogits_list = []\nall_ids = []\nfor xb_raw, ib in test_ds:\n    lg = tta_log_probs_raw(model, xb_raw, IMG_SIZE)\n    logits_list.append(lg)\n    all_ids.extend([b.numpy().decode(\"utf-8\") for b in ib])\n\nlogits = np.vstack(logits_list)\nlabels = np.argmax(logits, axis=1).astype(int)\n\n# ---------- Submission ----------\nsub = pd.DataFrame({\"id\": all_ids, \"label\": labels})\nsub.to_csv(\"submission.csv\", index=False)\nprint(\"\\nПервые 5 строк submission:\")\nprint(sub.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T17:36:51.770399Z","iopub.execute_input":"2025-12-20T17:36:51.770668Z","iopub.status.idle":"2025-12-20T19:00:03.246051Z","shell.execute_reply.started":"2025-12-20T17:36:51.770647Z","shell.execute_reply":"2025-12-20T19:00:03.245223Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}