{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"},{"sourceId":10421266,"sourceType":"datasetVersion","datasetId":6459084}],"dockerImageVersionId":31260,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Импорты и глобальные настройки","metadata":{}},{"cell_type":"code","source":"import os\nimport re\nimport random\nimport warnings\n\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\n\nfrom tensorflow.keras import layers, models, applications\nfrom sklearn.metrics import f1_score\n\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-25T07:02:09.789272Z","iopub.execute_input":"2026-01-25T07:02:09.789572Z","iopub.status.idle":"2026-01-25T07:02:09.794339Z","shell.execute_reply.started":"2026-01-25T07:02:09.789549Z","shell.execute_reply":"2026-01-25T07:02:09.793644Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Стратегия, сиды и константы","metadata":{}},{"cell_type":"code","source":"# Распределённое обучение (GPU/TPU)\ndist_strategy = tf.distribute.MirroredStrategy()\n\nAUTOTUNE = tf.data.AUTOTUNE\nGLOBAL_SEED = 42\n\nrandom.seed(GLOBAL_SEED)\nnp.random.seed(GLOBAL_SEED)\ntf.random.set_seed(GLOBAL_SEED)\n\nIMG_SHAPE = (224, 224)\nCHANNELS = 3\n\nN_CLASSES = 104\nBATCH = 16 * dist_strategy.num_replicas_in_sync","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-25T07:02:11.408841Z","iopub.execute_input":"2026-01-25T07:02:11.409131Z","iopub.status.idle":"2026-01-25T07:02:11.591290Z","shell.execute_reply.started":"2026-01-25T07:02:11.409108Z","shell.execute_reply":"2026-01-25T07:02:11.590638Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Пути к данным","metadata":{}},{"cell_type":"code","source":"ROOT_DIR = \"/kaggle/input/tpu-getting-started\"\nTFREC_DIR = f\"{ROOT_DIR}/tfrecords-jpeg-{IMG_SHAPE[0]}x{IMG_SHAPE[1]}\"\n\ntrain_files = tf.io.gfile.glob(f\"{TFREC_DIR}/train/*.tfrec\")\nvalid_files = tf.io.gfile.glob(f\"{TFREC_DIR}/val/*.tfrec\")\ntest_files  = tf.io.gfile.glob(f\"{TFREC_DIR}/test/*.tfrec\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-25T07:02:23.391578Z","iopub.execute_input":"2026-01-25T07:02:23.392220Z","iopub.status.idle":"2026-01-25T07:02:23.406824Z","shell.execute_reply.started":"2026-01-25T07:02:23.392166Z","shell.execute_reply":"2026-01-25T07:02:23.405974Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Декодирование и чтение TFRecord","metadata":{}},{"cell_type":"code","source":"def parse_image(raw_bytes):\n    img = tf.image.decode_jpeg(raw_bytes, channels=CHANNELS)\n    img = tf.image.resize(img, IMG_SHAPE)\n    img = tf.cast(img, tf.float32) / 255.0\n    return img\n\n\ndef parse_labeled(example):\n    schema = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    }\n    data = tf.io.parse_single_example(example, schema)\n    return parse_image(data[\"image\"]), tf.cast(data[\"class\"], tf.int32)\n\n\ndef parse_unlabeled(example):\n    schema = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    data = tf.io.parse_single_example(example, schema)\n    return parse_image(data[\"image\"]), data[\"id\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-25T07:02:32.271471Z","iopub.execute_input":"2026-01-25T07:02:32.272222Z","iopub.status.idle":"2026-01-25T07:02:32.278223Z","shell.execute_reply.started":"2026-01-25T07:02:32.272162Z","shell.execute_reply":"2026-01-25T07:02:32.277431Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Dataset pipeline","metadata":{}},{"cell_type":"code","source":"def build_dataset(files, labeled=True, shuffle=False):\n    opts = tf.data.Options()\n    opts.experimental_deterministic = not shuffle\n\n    ds = tf.data.TFRecordDataset(files, num_parallel_reads=AUTOTUNE)\n    ds = ds.with_options(opts)\n\n    if labeled:\n        ds = ds.map(parse_labeled, num_parallel_calls=AUTOTUNE)\n    else:\n        ds = ds.map(parse_unlabeled, num_parallel_calls=AUTOTUNE)\n\n    ds = ds.batch(BATCH)\n    ds = ds.prefetch(AUTOTUNE)\n\n    return ds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-25T07:02:42.184850Z","iopub.execute_input":"2026-01-25T07:02:42.185422Z","iopub.status.idle":"2026-01-25T07:02:42.190356Z","shell.execute_reply.started":"2026-01-25T07:02:42.185395Z","shell.execute_reply":"2026-01-25T07:02:42.189769Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Подсчёт количества объектов","metadata":{}},{"cell_type":"code","source":"def extract_count(filenames):\n    return np.sum([\n        int(re.search(r\"-([0-9]+)\\.\", fname).group(1))\n        for fname in filenames\n    ])\n\nN_VALID = extract_count(valid_files)\nN_TEST  = extract_count(test_files)\n\nprint(\"Batch size:\", BATCH)\nprint(\"Validation samples:\", N_VALID)\nprint(\"Test samples:\", N_TEST)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-25T07:02:51.070821Z","iopub.execute_input":"2026-01-25T07:02:51.071402Z","iopub.status.idle":"2026-01-25T07:02:51.076449Z","shell.execute_reply.started":"2026-01-25T07:02:51.071377Z","shell.execute_reply":"2026-01-25T07:02:51.075716Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Модель EfficientNetB7","metadata":{}},{"cell_type":"code","source":"def build_efficientnet():\n    with dist_strategy.scope():\n        backbone = applications.EfficientNetB7(\n            include_top=False,\n            weights=\"imagenet\",\n            input_shape=(*IMG_SHAPE, CHANNELS)\n        )\n\n        model = models.Sequential([\n            backbone,\n            layers.GlobalAveragePooling2D(),\n            layers.Dropout(0.3),\n            layers.Dense(N_CLASSES, activation=\"softmax\")\n        ])\n\n        model.compile(\n            optimizer=tf.keras.optimizers.AdamW(learning_rate=1e-4),\n            loss=\"sparse_categorical_crossentropy\",\n            metrics=[\"accuracy\"]\n        )\n\n        model.load_weights(\n            \"/kaggle/input/efficientnetb7-densenet201/EfficientNetB7_best.keras\",\n            skip_mismatch=True\n        )\n\n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-25T07:04:31.359988Z","iopub.execute_input":"2026-01-25T07:04:31.360749Z","iopub.status.idle":"2026-01-25T07:04:31.366291Z","shell.execute_reply.started":"2026-01-25T07:04:31.360721Z","shell.execute_reply":"2026-01-25T07:04:31.365468Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Модель DenseNet201","metadata":{}},{"cell_type":"code","source":"def build_densenet():\n    with dist_strategy.scope():\n        backbone = applications.DenseNet201(\n            include_top=False,\n            weights=\"imagenet\",\n            input_shape=(*IMG_SHAPE, CHANNELS)\n        )\n\n        model = models.Sequential([\n            backbone,\n            layers.GlobalAveragePooling2D(),\n            layers.Dropout(0.25),\n            layers.Dense(N_CLASSES, activation=\"softmax\")\n        ])\n\n        model.compile(\n            optimizer=tf.keras.optimizers.AdamW(learning_rate=1e-4),\n            loss=\"sparse_categorical_crossentropy\",\n            metrics=[\"accuracy\"]\n        )\n\n        model.load_weights(\n            \"/kaggle/input/efficientnetb7-densenet201/densenet201_best.keras\"\n        )\n\n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-25T07:04:33.213702Z","iopub.execute_input":"2026-01-25T07:04:33.214368Z","iopub.status.idle":"2026-01-25T07:04:33.219680Z","shell.execute_reply.started":"2026-01-25T07:04:33.214339Z","shell.execute_reply":"2026-01-25T07:04:33.219084Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Подбор веса ансамбля","metadata":{}},{"cell_type":"code","source":"model_a = build_efficientnet()\nmodel_b = build_densenet()\n\nval_ds = build_dataset(valid_files, labeled=True)\nval_imgs = val_ds.map(lambda x, y: x)\nval_lbls = val_ds.map(lambda x, y: y).unbatch()\n\ny_true = next(iter(val_lbls.batch(N_VALID))).numpy()\n\npred_a = model_a.predict(val_imgs, verbose=2)\npred_b = model_b.predict(val_imgs, verbose=2)\n\nmix = np.linspace(0.2, 0.8, 61)\nscores = []\n\nfor w in mix:\n    blended = w * pred_a + (1 - w) * pred_b\n    y_pred = blended.argmax(axis=1)\n    scores.append(\n        f1_score(y_true, y_pred, average=\"macro\", labels=range(N_CLASSES))\n    )\n\nbest_w = mix[np.argmax(scores)]\nbest_f1 = max(scores)\n\nprint(f\"Best weight: {best_w:.3f}\")\nprint(f\"Validation F1: {best_f1:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-25T07:04:35.153529Z","iopub.execute_input":"2026-01-25T07:04:35.153829Z","iopub.status.idle":"2026-01-25T07:06:05.514198Z","shell.execute_reply.started":"2026-01-25T07:04:35.153805Z","shell.execute_reply":"2026-01-25T07:06:05.513344Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# TTA-инференс","metadata":{}},{"cell_type":"code","source":"def infer_with_tta(model, rounds=7):\n    preds = []\n\n    for _ in range(rounds):\n        ds = build_dataset(test_files, labeled=False)\n        imgs = ds.map(lambda x, y: x)\n        preds.append(model.predict(imgs, verbose=0))\n\n    return np.mean(preds, axis=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-25T07:11:32.959206Z","iopub.execute_input":"2026-01-25T07:11:32.959922Z","iopub.status.idle":"2026-01-25T07:11:32.964284Z","shell.execute_reply.started":"2026-01-25T07:11:32.959896Z","shell.execute_reply":"2026-01-25T07:11:32.963468Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Сабмит","metadata":{}},{"cell_type":"code","source":"tta_a = infer_with_tta(model_a, rounds=7)\ntta_b = infer_with_tta(model_b, rounds=7)\n\nfinal_probs = best_w * tta_a + (1 - best_w) * tta_b\nfinal_labels = final_probs.argmax(axis=1)\n\ntest_ds = build_dataset(test_files, labeled=False)\ntest_ids = next(\n    iter(test_ds.map(lambda x, y: y).unbatch().batch(N_TEST))\n).numpy().astype(str)\n\nsubmission = pd.DataFrame({\n    \"id\": test_ids,\n    \"label\": final_labels\n})\n\nsubmission.to_csv(\"submission.csv\", index=False)\nprint(\"submission.csv saved ✔\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-25T07:11:35.185708Z","iopub.execute_input":"2026-01-25T07:11:35.186340Z","iopub.status.idle":"2026-01-25T07:22:06.068253Z","shell.execute_reply.started":"2026-01-25T07:11:35.186312Z","shell.execute_reply":"2026-01-25T07:22:06.067563Z"}},"outputs":[],"execution_count":null}]}