{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31260,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## 1. Настройка путей","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf, numpy as np, pandas as pd, os\nfrom sklearn.metrics import f1_score\nfrom kaggle_datasets import KaggleDatasets\n\nstrategy = tf.distribute.MirroredStrategy()\nprint(f\"Работаем на {strategy.num_replicas_in_sync} GPU\")\n\nGCS_DS_PATH = KaggleDatasets().get_gcs_path('tpu-getting-started')\nDATA_PATH = GCS_DS_PATH + '/tfrecords-jpeg-224x224'\n\nTRAIN_FILES = tf.io.gfile.glob(DATA_PATH + '/train/*.tfrec')\nVAL_FILES = tf.io.gfile.glob(DATA_PATH + '/val/*.tfrec')\nTEST_FILES = tf.io.gfile.glob(DATA_PATH + '/test/*.tfrec')\n\nIMAGE_SIZE = [224, 224]\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync # 32\nEPOCHS = 20 # Для предобученной сети этого хватит","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-01-22T19:52:12.080595Z","iopub.execute_input":"2026-01-22T19:52:12.080868Z","iopub.status.idle":"2026-01-22T19:52:37.349642Z","shell.execute_reply.started":"2026-01-22T19:52:12.080820Z","shell.execute_reply":"2026-01-22T19:52:37.348783Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Пайплайн данных","metadata":{}},{"cell_type":"code","source":"def read_tf(example, labeled=True):\n    fmt = {\"image\": tf.io.FixedLenFeature([], tf.string), \n           \"class\": tf.io.FixedLenFeature([], tf.int64) if labeled else tf.io.FixedLenFeature([], tf.string)}\n    res = tf.io.parse_single_example(example, fmt)\n    img = tf.image.decode_jpeg(res['image'], channels=3)\n    img = tf.reshape(img, [*IMAGE_SIZE, 3])\n    img = tf.cast(img, tf.float32) \n    if labeled:\n        return img, tf.one_hot(tf.cast(res['class'], tf.int32), 104)\n    return img, res['id']\n\ndef data_augment(img, lbl):\n    img = tf.image.random_flip_left_right(img)\n    img = tf.image.random_saturation(img, 0, 2)\n    return img, lbl\n\ndef get_train_ds():\n    ds = tf.data.TFRecordDataset(TRAIN_FILES, num_parallel_reads=tf.data.AUTOTUNE)\n    ds = ds.map(lambda x: read_tf(x, True), num_parallel_calls=tf.data.AUTOTUNE)\n    ds = ds.map(data_augment, num_parallel_calls=tf.data.AUTOTUNE)\n    return ds.repeat().shuffle(2048).batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\n\ndef get_val_ds():\n    return tf.data.TFRecordDataset(VAL_FILES).map(lambda x: read_tf(x, True)).batch(BATCH_SIZE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T19:53:30.761101Z","iopub.execute_input":"2026-01-22T19:53:30.761692Z","iopub.status.idle":"2026-01-22T19:53:30.768151Z","shell.execute_reply.started":"2026-01-22T19:53:30.761666Z","shell.execute_reply":"2026-01-22T19:53:30.767542Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Модель DenseNet201 + LR Scheduler","metadata":{}},{"cell_type":"code","source":"with strategy.scope():\n    # Загружаем \"тело\" модели DenseNet201\n    base = tf.keras.applications.DenseNet201(weights='imagenet', include_top=False, input_shape=[*IMAGE_SIZE, 3])\n    base.trainable = True # Разрешаем дообучение всех весов\n    \n    model = tf.keras.Sequential([\n        base,\n        tf.keras.layers.GlobalAveragePooling2D(),\n        tf.keras.layers.Dense(104, activation='softmax')\n    ])\n    \n    model.compile(\n        optimizer='adam',\n        loss=tf.keras.losses.CategoricalCrossentropy(label_smoothing=0.1),\n        metrics=['accuracy']\n    )\n\n# Планировщик скорости обучения\ndef lr_fn(epoch):\n    LR_START = 0.00001\n    LR_MAX = 0.00005 * strategy.num_replicas_in_sync\n    LR_MIN = 0.00001\n    LR_RAMPUP_EPOCHS = 5\n    LR_SUSTAIN_EPOCHS = 0\n    LR_EXP_DECAY = .8\n    if epoch < LR_RAMPUP_EPOCHS:\n        lr = (LR_MAX - LR_START) / LR_RAMPUP_EPOCHS * epoch + LR_START\n    elif epoch < LR_RAMPUP_EPOCHS + LR_SUSTAIN_EPOCHS:\n        lr = LR_MAX\n    else:\n        lr = (LR_MAX - LR_MIN) * LR_EXP_DECAY**(epoch - LR_RAMPUP_EPOCHS - LR_SUSTAIN_EPOCHS) + LR_MIN\n    return lr\n\nlr_callback = tf.keras.callbacks.LearningRateScheduler(lr_fn, verbose=True)\n\nprint(\"Запуск обучения...\")\nmodel.fit(get_train_ds(), steps_per_epoch=12753//BATCH_SIZE, epochs=EPOCHS, \n          validation_data=get_val_ds(), callbacks=[lr_callback])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T19:53:33.523678Z","iopub.execute_input":"2026-01-22T19:53:33.524332Z","iopub.status.idle":"2026-01-22T21:04:56.502624Z","shell.execute_reply.started":"2026-01-22T19:53:33.524298Z","shell.execute_reply":"2026-01-22T21:04:56.501723Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Итоговый расчет и Submission","metadata":{}},{"cell_type":"code","source":"# Честный расчет Macro F1\nprint(\"\\nРасчет итогового Macro F1...\")\nval_ds = get_val_ds()\ny_true, y_pred = [], []\nfor imgs, lbls in val_ds:\n    p = model.predict(imgs, verbose=0)\n    y_true.extend(np.argmax(lbls.numpy(), axis=-1))\n    y_pred.extend(np.argmax(p, axis=-1))\n\nfrom sklearn.metrics import f1_score\nscore = f1_score(y_true, y_pred, average='macro')\nprint(f\"Итоговый Macro F1: {score:.4f}\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T21:09:05.839391Z","iopub.execute_input":"2026-01-22T21:09:05.839703Z","iopub.status.idle":"2026-01-22T21:09:53.001594Z","shell.execute_reply.started":"2026-01-22T21:09:05.839676Z","shell.execute_reply":"2026-01-22T21:09:53.000806Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Предсказание для Kaggle","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport tensorflow as tf\nfrom google.colab import files\n\nprint(\" Запуск процесса предсказания...\")\n\nIMAGE_SIZE = [224, 224]\nBATCH_SIZE = 32\n\ndef read_test_tf(example):\n    fmt = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    res = tf.io.parse_single_example(example, fmt)\n    img = tf.image.decode_jpeg(res['image'], channels=3)\n    img = tf.reshape(img, [*IMAGE_SIZE, 3])\n    img = tf.cast(img, tf.float32) \n    return img, res['id']\n\ntest_ds = tf.data.TFRecordDataset(TEST_FILES, num_parallel_reads=tf.data.AUTOTUNE)\ntest_ds = test_ds.map(read_test_tf, num_parallel_calls=tf.data.AUTOTUNE).batch(BATCH_SIZE)\n\nprint(\"Сбор идентификаторов...\")\nall_ids = []\nfor _, ids in test_ds:\n    all_ids.extend([i.decode('utf-8') for i in ids.numpy()])\n\nprint(\"Запуск сетки на тестовых данных...\")\ntest_images_ds = test_ds.map(lambda img, idnum: img)\nprobabilities = model.predict(test_images_ds)\nall_preds = np.argmax(probabilities, axis=-1)\n\nprint(f\"Всего ID: {len(all_ids)}\")\nprint(f\"Всего предсказаний: {len(all_preds)}\")\n\nunique_classes = np.unique(all_preds)\nprint(f\"Модель нашла {len(unique_classes)} разных классов цветов.\")\n\nif len(all_ids) == len(all_preds):\n    submission = pd.DataFrame({'id': all_ids, 'label': all_preds})\n    submission.to_csv('submission.csv', index=False)\n    print(\"Файл submission.csv успешно создан!\")\n    \n    print(submission.head(10))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T21:59:23.155362Z","iopub.execute_input":"2026-01-22T21:59:23.155727Z","iopub.status.idle":"2026-01-22T21:59:58.503408Z","shell.execute_reply.started":"2026-01-22T21:59:23.155689Z","shell.execute_reply":"2026-01-22T21:59:58.502592Z"}},"outputs":[],"execution_count":null}]}