{"metadata":{"colab":{"provenance":[],"gpuType":"T4"},"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"accelerator":"GPU","kaggle":{"accelerator":"none","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"},{"sourceId":7148368,"sourceType":"datasetVersion","datasetId":4126926}],"dockerImageVersionId":30615,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import tensorflow as tf\n# последовательная модель (стек слоев)\nfrom tensorflow.keras.models import Sequential, Model\n# полносвязный слой и слой выпрямляющий матрицу в вектор\nfrom tensorflow.keras.layers import Dense, Flatten, Input\n# слой выключения нейронов и слой нормализации выходных данных (нормализует данные в пределах текущей выборки)\nfrom tensorflow.keras.layers import Dropout, BatchNormalization, SpatialDropout2D, GaussianDropout\n# слои свертки и подвыборки\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, AveragePooling2D\n# работа с обратной связью от обучающейся нейронной сети\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint, ReduceLROnPlateau\n# вспомогательные инструменты\nfrom tensorflow.keras import utils\nfrom tensorflow.keras.regularizers import *\nimport numpy as np\nimport math, re, os\nfrom tensorflow.random import set_seed\ndef seed_everything(seed):\n    np.random.seed(seed)\n    set_seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    os.environ['TF_DETERMINISTIC_OPS'] = '1'\n\nseed = 42\nseed_everything(seed)\n\n# работа с изображениями\nfrom tensorflow.keras.preprocessing import image\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\nimport matplotlib.pyplot as plt\n%matplotlib inline\nprint(\"Tensorflow version \" + tf.__version__)","metadata":{"id":"EC5C_G6Vgies","outputId":"6d5b9c94-a4d8-4148-d021-c175d64e019a","execution":{"iopub.status.busy":"2023-12-07T16:29:16.832160Z","iopub.execute_input":"2023-12-07T16:29:16.832817Z","iopub.status.idle":"2023-12-07T16:29:30.555144Z","shell.execute_reply.started":"2023-12-07T16:29:16.832767Z","shell.execute_reply":"2023-12-07T16:29:30.554112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Обнаружение оборудования, возврат соответствующей стратегии распространения: TPU, GPU, CPU\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()  # Обнаружение TPU. Параметры среды не требуются, если задана переменная среды TPU_NAME. На Kaggle это всегда так.\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    strategy = tf.distribute.get_strategy() # стратегия распространения по умолчанию в Tensorflow. Работает на CPU и одном GPU.\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"id":"pjjDexyygkHI","outputId":"1eeb6a71-0d1d-4e43-aac1-7596c50ad66e","execution":{"iopub.status.busy":"2023-12-07T16:29:38.417385Z","iopub.execute_input":"2023-12-07T16:29:38.418115Z","iopub.status.idle":"2023-12-07T16:29:38.430698Z","shell.execute_reply.started":"2023-12-07T16:29:38.418079Z","shell.execute_reply":"2023-12-07T16:29:38.429808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from kaggle_datasets import KaggleDatasets\n\nGCS_DS_PATH = KaggleDatasets().get_gcs_path('tpu-getting-started') #получаем путь к наборам данных\nprint(GCS_DS_PATH)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T16:30:58.031168Z","iopub.execute_input":"2023-12-07T16:30:58.031607Z","iopub.status.idle":"2023-12-07T16:30:58.618856Z","shell.execute_reply.started":"2023-12-07T16:30:58.031572Z","shell.execute_reply":"2023-12-07T16:30:58.617651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_SIZE = [192, 192] # при таком размере графическому процессору не хватит памяти. Используйте TPU\nEPOCHS = 80\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\n\nNUM_TRAINING_IMAGES = 12753\nNUM_TEST_IMAGES = 7382\nSTEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE # находим количество шагов за эпоху","metadata":{"id":"DpVyL3pP2PTz","execution":{"iopub.status.busy":"2023-12-07T16:31:01.853814Z","iopub.execute_input":"2023-12-07T16:31:01.854995Z","iopub.status.idle":"2023-12-07T16:31:01.861648Z","shell.execute_reply.started":"2023-12-07T16:31:01.854947Z","shell.execute_reply":"2023-12-07T16:31:01.860535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def decode_image(image_data):\n    \"\"\"Декодирует изображение в vyjujvthye. vfnhbwe (тензор)\n    Нормализует данные и преобразовывает изображения к указанному размеру\"\"\"\n    image = tf.image.decode_jpeg(image_data, channels=3) # Декодирование изображения в формате JPEG в тензор uint8.\n    image = tf.cast(image, tf.float32) / 255.0  # преобразовать изображение в плавающее в диапазоне [0, 1]\n    image = tf.reshape(image, [*IMAGE_SIZE, 3]) # явный размер, необходимый для TPU\n#     image = tf.keras.applications.inception_resnet_v2.preprocess_input(image)\n    return image\n\ndef read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string), # tf.string означает байтовую строку\n        \"class\": tf.io.FixedLenFeature([], tf.int64),  # [] означает отдельный элемент\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT) # парсим отдельный пример в указанном формате\n    image = decode_image(example['image']) # преобразуем изображение к нужному нам формату\n    label = tf.cast(example['class'], tf.int32)\n    return image, label # возвращает набор данных пар (изображение, метка)\n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string), # tf.string означает байтовую строку\n        \"id\": tf.io.FixedLenFeature([], tf.string),  # [] означает отдельный элемент\n        # класс отсутствует, задача этого конкурса - предсказать классы цветов для тестового набора данных\n    }\n    example = tf.io.parse_single_example(example, UNLABELED_TFREC_FORMAT)\n    image = decode_image(example['image']) # преобразуем изображение к нужному нам формату\n    idnum = example['id']\n    return image, idnum # returns a dataset of image(s)\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    \"\"\"Читает из TFRecords. Для оптимальной производительности одновременное чтение из нескольких\n    файлов без учета порядка данных. Порядок не имеет значения, поскольку мы все равно будем перетасовывать данные\"\"\"\n\n    ignore_order = tf.data.Options() # Представляет параметры для tf.data.Dataset.\n    if not ordered:\n        ignore_order.experimental_deterministic = False # отключить порядок, увеличить скорость\n\n    dataset = tf.data.TFRecordDataset(filenames) # автоматически чередует чтение из нескольких файлов\n    dataset = dataset.with_options(ignore_order) # использует данные сразу после их поступления, а не в исходном порядке\n    dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord)\n    # возвращает набор данных пар (изображение, метка), если метка = Истина, или пар (изображение, идентификатор), если метка = Ложь\n    return dataset\n\ndef get_training_dataset():\n    dataset = load_dataset(tf.io.gfile.glob(GCS_DS_PATH  + '/tfrecords-jpeg-192x192/train/*.tfrec'), labeled=True)\n    dataset = dataset.repeat() # набор обучающих данных должен повторяться в течение нескольких эпох\n    dataset = dataset.shuffle(2048)\n    dataset = dataset.batch(BATCH_SIZE)\n    return dataset\n\ndef get_validation_dataset():\n    dataset = load_dataset(tf.io.gfile.glob(GCS_DS_PATH  + '/tfrecords-jpeg-192x192/val/*.tfrec'), labeled=True, ordered=False)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.cache() # кешируем набор\n    return dataset\n\ndef get_test_dataset(ordered=False):\n    dataset = load_dataset(tf.io.gfile.glob(GCS_DS_PATH  + '/tfrecords-jpeg-192x192/test/*.tfrec'), labeled=False, ordered=ordered)\n    dataset = dataset.batch(BATCH_SIZE)\n    return dataset\n\ntraining_dataset = get_training_dataset()\nvalidation_dataset = get_validation_dataset()","metadata":{"id":"nC8OnLHx2i8x","execution":{"iopub.status.busy":"2023-12-07T16:31:03.144489Z","iopub.execute_input":"2023-12-07T16:31:03.146157Z","iopub.status.idle":"2023-12-07T16:31:04.786165Z","shell.execute_reply.started":"2023-12-07T16:31:03.146098Z","shell.execute_reply":"2023-12-07T16:31:04.784865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTOTUNE = tf.data.AUTOTUNE\n\ndata_augmentation = tf.keras.Sequential([\n  tf.keras.layers.RandomFlip(\"horizontal\"),\n  tf.keras.layers.RandomRotation(0.2),\n  tf.keras.layers.RandomZoom(0.2),\n])\n\n\ndef prepare(ds,  augment=False):\n  # Batch all datasets.\n  #ds = ds.batch(BATCH_SIZE)\n\n  # Use data augmentation only on the training set.\n  if augment:\n    ds = ds.map(lambda x, y: (data_augmentation(x, training=True), y),\n                num_parallel_calls=AUTOTUNE)\n\n  # Use buffered prefetching on all datasets.\n  return ds.prefetch(buffer_size=AUTOTUNE)\n\ntrain_ds = prepare(training_dataset, augment=True)","metadata":{"id":"8ZxvFPfZ89mI","execution":{"iopub.status.busy":"2023-12-07T16:31:12.226879Z","iopub.execute_input":"2023-12-07T16:31:12.227399Z","iopub.status.idle":"2023-12-07T16:31:12.889149Z","shell.execute_reply.started":"2023-12-07T16:31:12.227355Z","shell.execute_reply":"2023-12-07T16:31:12.888238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow.keras as tfk\nimport tensorflow.keras.backend as K\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, Dense, Conv2D, Conv3D, DepthwiseConv2D, SeparableConv2D, Conv3DTranspose\nfrom tensorflow.keras.layers import Flatten, MaxPool2D, AvgPool2D, GlobalAvgPool2D, UpSampling2D, BatchNormalization\nfrom tensorflow.keras.layers import Concatenate, Add, Dropout, ReLU, Lambda, Activation, LeakyReLU, PReLU\nfrom time import time\nimport numpy as np\n\ndef densenet(img_shape, n_classes, f=32):\n  repetitions = 6, 12, 24, 16\n\n  def bn_rl_conv(x, f, k=1, s=1, p='same'):\n    x = BatchNormalization()(x)\n    x = ReLU()(x)\n    x = Conv2D(f, k, strides=s, padding=p)(x)\n    return x\n\n\n  def dense_block(tensor, r):\n    for _ in range(r):\n      x = bn_rl_conv(tensor, 4*f)\n      x = bn_rl_conv(x, f, 3)\n      tensor = Concatenate()([tensor, x])\n    return tensor\n\n\n  def transition_block(x):\n    x = bn_rl_conv(x, K.int_shape(x)[-1] // 2)\n    x = AvgPool2D(2, strides=2, padding='same')(x)\n    return x\n\n\n  input = Input(img_shape)\n\n  x = Conv2D(64, 7, strides=2, padding='same')(input)\n  x = MaxPool2D(3, strides=2, padding='same')(x)\n\n  for r in repetitions:\n    d = dense_block(x, r)\n    x = transition_block(d)\n\n  x = GlobalAvgPool2D()(d)\n\n  output = Dense(n_classes, activation='softmax')(x)\n\n  model = Model(input, output)\n  return model","metadata":{"id":"VJhbU_EW46LO","execution":{"iopub.status.busy":"2023-12-07T16:31:15.584317Z","iopub.execute_input":"2023-12-07T16:31:15.585300Z","iopub.status.idle":"2023-12-07T16:31:15.600177Z","shell.execute_reply.started":"2023-12-07T16:31:15.585263Z","shell.execute_reply":"2023-12-07T16:31:15.598720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_shape = 192, 192, 3\nn_classes = 104\n\nwith strategy.scope():\n    model = densenet(input_shape, n_classes)\n\nmodel.summary()","metadata":{"id":"XHvAxT5Z9I1u","outputId":"12e9f31c-64d7-48ed-f93e-9ba491cbaa95","execution":{"iopub.status.busy":"2023-12-07T16:31:18.034486Z","iopub.execute_input":"2023-12-07T16:31:18.034941Z","iopub.status.idle":"2023-12-07T16:31:23.285362Z","shell.execute_reply.started":"2023-12-07T16:31:18.034903Z","shell.execute_reply":"2023-12-07T16:31:23.284242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"callbacks_list = [EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True),\n                  ModelCheckpoint('/content/drive/MyDrive/tpu-model-{epoch:03d}-.h5', save_best_only=True, monitor='val_loss', mode='auto', verbose=1),\n                  ReduceLROnPlateau(monitor='val_loss', factor=0.1, patience=3),\n                  ]\n\nmodel.compile(\n    optimizer='nadam',\n    loss = 'sparse_categorical_crossentropy',\n    metrics=['sparse_categorical_accuracy']\n)","metadata":{"id":"u4YZcuqSlolO"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"historical = model.fit(train_ds,\n          steps_per_epoch=STEPS_PER_EPOCH,\n          epochs=EPOCHS,\n          callbacks=callbacks_list,\n          validation_data=validation_dataset)","metadata":{"id":"YEgn-JL24944","outputId":"61c258fb-9b56-4385-bcf0-bf1b83b587ba"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.load_weights('/kaggle/input/123451/tpu-model-035-.h5')","metadata":{"id":"HMohBrOy1e8q","execution":{"iopub.status.busy":"2023-12-07T16:31:54.867424Z","iopub.execute_input":"2023-12-07T16:31:54.867888Z","iopub.status.idle":"2023-12-07T16:31:56.096811Z","shell.execute_reply.started":"2023-12-07T16:31:54.867852Z","shell.execute_reply":"2023-12-07T16:31:56.095574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ds = get_test_dataset(ordered=True)\n\nprint('Вычисляем предсказания...')\ntest_images_ds = test_ds.map(lambda image, idnum: image)\nprobabilities = model.predict(test_images_ds)\npredictions = np.argmax(probabilities, axis=-1)\nprint(predictions)\n\nprint('Создание файла submission.csv...')\ntest_ids_ds = test_ds.map(lambda image, idnum: idnum).unbatch()\ntest_ids = next(iter(test_ids_ds.batch(NUM_TEST_IMAGES))).numpy().astype('U') # все в одной партии\nnp.savetxt('submission.csv', np.rec.fromarrays([test_ids, predictions]), fmt=['%s', '%d'], delimiter=',', header='id,label', comments='')","metadata":{"id":"0VVF9mTo91hZ","outputId":"6b65cfdd-d21c-46a0-91ff-8b9b5b38d621","execution":{"iopub.status.busy":"2023-12-07T16:31:58.616132Z","iopub.execute_input":"2023-12-07T16:31:58.616560Z","iopub.status.idle":"2023-12-07T16:39:42.919344Z","shell.execute_reply.started":"2023-12-07T16:31:58.616523Z","shell.execute_reply":"2023-12-07T16:39:42.918309Z"},"trusted":true},"execution_count":null,"outputs":[]}]}