{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpuV5e8","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31259,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## 1. Настройка GPU","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport numpy as np\nimport pandas as pd\nimport os\nfrom sklearn.metrics import f1_score\n\n# 1. Используем сразу обе видеокарты T4\nstrategy = tf.distribute.MirroredStrategy()\nprint(f\" Работаем на {strategy.num_replicas_in_sync} GPU\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-01-22T18:26:28.482702Z","iopub.execute_input":"2026-01-22T18:26:28.483330Z","iopub.status.idle":"2026-01-22T18:26:53.033138Z","shell.execute_reply.started":"2026-01-22T18:26:28.483303Z","shell.execute_reply":"2026-01-22T18:26:53.032367Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Локальные пути к данным","metadata":{}},{"cell_type":"code","source":"from kaggle_datasets import KaggleDatasets\n\nGCS_DS_PATH = KaggleDatasets().get_gcs_path('tpu-getting-started')\nDATA_PATH = GCS_DS_PATH + '/tfrecords-jpeg-192x192'\n\nTRAIN_FILES = tf.io.gfile.glob(DATA_PATH + '/train/*.tfrec')\nVAL_FILES = tf.io.gfile.glob(DATA_PATH + '/val/*.tfrec')\nTEST_FILES = tf.io.gfile.glob(DATA_PATH + '/test/*.tfrec')\n\nIMAGE_SIZE = [192, 192]\nBATCH_SIZE = 32 * strategy.num_replicas_in_sync # 64\nEPOCHS = 40","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T19:06:06.847732Z","iopub.execute_input":"2026-01-22T19:06:06.848363Z","iopub.status.idle":"2026-01-22T19:06:07.185694Z","shell.execute_reply.started":"2026-01-22T19:06:06.848338Z","shell.execute_reply":"2026-01-22T19:06:07.185112Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Функции загрузки","metadata":{}},{"cell_type":"code","source":"def read_tf(example, labeled=True):\n    if labeled:\n        fmt = {\"image\": tf.io.FixedLenFeature([], tf.string), \"class\": tf.io.FixedLenFeature([], tf.int64)}\n        res = tf.io.parse_single_example(example, fmt)\n        img = tf.cast(tf.image.decode_jpeg(res['image'], channels=3), tf.float32) / 255.0\n        img = tf.reshape(img, [*IMAGE_SIZE, 3])\n        return img, tf.one_hot(tf.cast(res['class'], tf.int32), 104)\n    else:\n        fmt = {\"image\": tf.io.FixedLenFeature([], tf.string), \"id\": tf.io.FixedLenFeature([], tf.string)}\n        res = tf.io.parse_single_example(example, fmt)\n        img = tf.cast(tf.image.decode_jpeg(res['image'], channels=3), tf.float32) / 255.0\n        img = tf.reshape(img, [*IMAGE_SIZE, 3])\n        return img, res['id']\n\ndef data_augment(img, lbl):\n    img = tf.image.random_flip_left_right(img)\n    img = tf.image.random_brightness(img, 0.2)\n    img = tf.image.random_contrast(img, 0.8, 1.2)\n    return img, lbl\n\ndef get_train_ds():\n    ds = tf.data.TFRecordDataset(TRAIN_FILES, num_parallel_reads=tf.data.AUTOTUNE)\n    ds = ds.map(lambda x: read_tf(x, True), num_parallel_calls=tf.data.AUTOTUNE)\n    ds = ds.map(data_augment, num_parallel_calls=tf.data.AUTOTUNE)\n    return ds.repeat().shuffle(2048).batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\n\ndef get_val_ds():\n    return tf.data.TFRecordDataset(VAL_FILES).map(lambda x: read_tf(x, True)).batch(BATCH_SIZE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T19:06:09.837671Z","iopub.execute_input":"2026-01-22T19:06:09.837937Z","iopub.status.idle":"2026-01-22T19:06:09.846862Z","shell.execute_reply.started":"2026-01-22T19:06:09.837917Z","shell.execute_reply":"2026-01-22T19:06:09.846283Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Модель (самописная)","metadata":{}},{"cell_type":"code","source":"with strategy.scope():\n    model = tf.keras.Sequential([\n        tf.keras.layers.Input(shape=[192, 192, 3]),\n        \n        tf.keras.layers.Conv2D(64, 3, padding='same', activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.MaxPooling2D(),\n        \n        tf.keras.layers.Conv2D(128, 3, padding='same', activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.MaxPooling2D(),\n        \n        tf.keras.layers.Conv2D(256, 3, padding='same', activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.GlobalAveragePooling2D(),\n        \n        tf.keras.layers.Dense(512, activation='relu'),\n        tf.keras.layers.Dropout(0.4),\n        tf.keras.layers.Dense(104, activation='softmax')\n    ])\n    \n    model.compile(\n        optimizer='adam', \n        loss=tf.keras.losses.CategoricalCrossentropy(label_smoothing=0.1), \n        metrics=['accuracy']\n    )\n\ndef lr_fn(epoch):\n    return 0.001 * (0.8 ** (epoch // 5))\nlr_callback = tf.keras.callbacks.LearningRateScheduler(lr_fn, verbose=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T19:06:12.959096Z","iopub.execute_input":"2026-01-22T19:06:12.959642Z","iopub.status.idle":"2026-01-22T19:06:13.073993Z","shell.execute_reply.started":"2026-01-22T19:06:12.959615Z","shell.execute_reply":"2026-01-22T19:06:13.073462Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. Обучение и Расчет F1","metadata":{}},{"cell_type":"code","source":"model.fit(get_train_ds(), steps_per_epoch=12753//BATCH_SIZE, epochs=EPOCHS, \n          validation_data=get_val_ds(), callbacks=[lr_callback])\n\nprint(\"\\nСчитаем Macro F1...\")\nval_ds = get_val_ds()\ny_true, y_pred = [], []\nfor imgs, lbls in val_ds:\n    p = model.predict(imgs, verbose=0)\n    y_true.extend(np.argmax(lbls.numpy(), axis=-1))\n    y_pred.extend(np.argmax(p, axis=-1))\n\nprint(f\"Итоговый Macro F1 самописной сетки: {f1_score(y_true, y_pred, average='macro'):.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T19:06:16.187173Z","iopub.execute_input":"2026-01-22T19:06:16.187852Z","iopub.status.idle":"2026-01-22T19:33:02.993031Z","shell.execute_reply.started":"2026-01-22T19:06:16.187824Z","shell.execute_reply":"2026-01-22T19:33:02.992351Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 7. Сабмит","metadata":{}},{"cell_type":"code","source":"test_ds = tf.data.TFRecordDataset(TEST_FILES).map(lambda x: read_tf(x, False)).batch(BATCH_SIZE)\nall_ids, all_preds = [], []\nfor imgs, ids in test_ds:\n    p = model.predict(imgs, verbose=0)\n    all_preds.extend(np.argmax(p, axis=-1))\n    all_ids.extend([i.decode('utf-8') for i in ids.numpy()])\n\npd.DataFrame({'id': all_ids, 'label': all_preds}).to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T19:33:48.955845Z","iopub.execute_input":"2026-01-22T19:33:48.956484Z","iopub.status.idle":"2026-01-22T19:34:27.596737Z","shell.execute_reply.started":"2026-01-22T19:33:48.956445Z","shell.execute_reply":"2026-01-22T19:34:27.596136Z"}},"outputs":[],"execution_count":null}]}