{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-06-30T04:06:02.116188Z","iopub.execute_input":"2026-06-30T04:06:02.116520Z","iopub.status.idle":"2026-06-30T04:06:04.196333Z","shell.execute_reply.started":"2026-06-30T04:06:02.116480Z","shell.execute_reply":"2026-06-30T04:06:04.195395Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import re\nimport math\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix\n\nprint(\"Tensorflow version \" + tf.__version__)\n\n# Detecta se há TPU disponível; senão usa GPU/CPU\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    strategy = tf.distribute.get_strategy()\n    print('Running on CPU/GPU')\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)\n\n# Caminho do dataset da competição (ajustar conforme aparece no seu /kaggle/input)\nGCS_PATH = '/kaggle/input/competitions/tpu-getting-started/tfrecords-jpeg-192x192'\nIMAGE_SIZE = [192, 192]\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\nAUTO = tf.data.experimental.AUTOTUNE\n\nCLASSES = ['pink primrose', 'hard-leaved pocket orchid', 'canterbury bells', 'sweet pea',\n           'wild geranium', 'tiger lily', 'moon orchid', 'bird of paradise', 'monkshood',\n           'globe thistle', 'snapdragon', \"colt's foot\", 'king protea', 'spear thistle',\n           'yellow iris', 'globe-flower', 'purple coneflower', 'peruvian lily', 'balloon flower',\n           'giant white arum lily', 'fire lily', 'pincushion flower', 'fritillary', 'red ginger',\n           'grape hyacinth', 'corn poppy', 'prince of wales feathers', 'stemless gentian', 'artichoke',\n           'sweet william', 'carnation', 'garden phlox', 'love in the mist', 'cosmos',\n           'alpine sea holly', 'ruby-lipped cattleya', 'cape flower', 'great masterwort', 'siam tulip',\n           'lenten rose', 'barberton daisy', 'daffodil', 'sword lily', 'poinsettia', 'bolero deep blue',\n           'wallflower', 'marigold', 'buttercup', 'daisy', 'common dandelion', 'petunia', 'wild pansy',\n           'primula', 'sunflower', 'lilac hibiscus', 'bishop of llandaff', 'gaura', 'geranium',\n           'orange dahlia', 'pink-yellow dahlia', 'cautleya spicata', 'japanese anemone',\n           'black-eyed susan', 'silverbush', 'californian poppy', 'osteospermum', 'spring crocus',\n           'iris', 'windflower', 'tree poppy', 'gazania', 'azalea', 'water lily', 'rose',\n           'thorn apple', 'morning glory', 'passion flower', 'lotus', 'toad lily', 'anthurium',\n           'frangipani', 'clematis', 'hibiscus', 'columbine', 'desert-rose', 'tree mallow',\n           'magnolia', 'cyclamen', 'watercress', 'canna lily', 'hippeastrum', 'bee balm',\n           'pink quill', 'foxglove', 'bougainvillea', 'camellia', 'mallow', 'mexican petunia',\n           'bromelia', 'blanket flower', 'trumpet creeper', 'blackberry lily', 'common tulip',\n           'wild rose']\n\nprint(\"Número de classes:\", len(CLASSES))  # deve imprimir 104\n\nSEED = 42\nnp.random.seed(SEED)\ntf.random.set_seed(SEED)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-30T04:06:04.198197Z","iopub.execute_input":"2026-06-30T04:06:04.198748Z","iopub.status.idle":"2026-06-30T04:06:20.533190Z","shell.execute_reply.started":"2026-06-30T04:06:04.198708Z","shell.execute_reply":"2026-06-30T04:06:20.532192Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ===================== SEÇÃO 2: CARREGAMENTO E PARSING DOS TFRECORDS =====================\n\n# Lista os arquivos .tfrec de treino, validação e teste\nTRAINING_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/train/*.tfrec')\nVALIDATION_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/val/*.tfrec')\nTEST_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/test/*.tfrec')\n\ndef count_data_items(filenames):\n    # o número de imagens está embutido no nome do arquivo, ex: flowers00-230.tfrec = 230 imagens\n    n = [int(re.compile(r\"-([0-9]*)\\.\").search(filename).group(1)) for filename in filenames]\n    return np.sum(n)\n\nNUM_TRAINING_IMAGES = count_data_items(TRAINING_FILENAMES)\nNUM_VALIDATION_IMAGES = count_data_items(VALIDATION_FILENAMES)\nNUM_TEST_IMAGES = count_data_items(TEST_FILENAMES)\n\nprint('Dataset: {} imagens de treino, {} de validação, {} de teste (não rotuladas)'.format(\n    NUM_TRAINING_IMAGES, NUM_VALIDATION_IMAGES, NUM_TEST_IMAGES))\n\ndef decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0  # normaliza pixels para [0,1]\n    image = tf.reshape(image, [*IMAGE_SIZE, 3])\n    return image\n\ndef read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    label = tf.cast(example['class'], tf.int32)\n    return image, label\n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, UNLABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    idnum = example['id']\n    return image, idnum\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTO)\n    dataset = dataset.with_options(ignore_order)\n    dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord,\n                           num_parallel_calls=AUTO)\n    return dataset\n\ndef get_training_dataset():\n    dataset = load_dataset(TRAINING_FILENAMES, labeled=True)\n    dataset = dataset.repeat()\n    dataset = dataset.shuffle(2048, seed=SEED)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO)\n    return dataset\n\ndef get_validation_dataset(ordered=False):\n    dataset = load_dataset(VALIDATION_FILENAMES, labeled=True, ordered=ordered)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.cache()\n    dataset = dataset.prefetch(AUTO)\n    return dataset\n\ndef get_test_dataset(ordered=False):\n    dataset = load_dataset(TEST_FILENAMES, labeled=False, ordered=ordered)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO)\n    return dataset\n\nSTEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-30T04:06:20.534527Z","iopub.execute_input":"2026-06-30T04:06:20.535156Z","iopub.status.idle":"2026-06-30T04:06:20.564433Z","shell.execute_reply.started":"2026-06-30T04:06:20.535126Z","shell.execute_reply":"2026-06-30T04:06:20.563490Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ===================== SEÇÃO 3: ANÁLISE EXPLORATÓRIA DE DADOS (EDA) =====================\n\n# --- 3.1 Distribuição das classes no conjunto de treino ---\n# Para contar quantas imagens existem por classe, percorremos o dataset de treino (sem repetir/embaralhar)\neda_dataset = load_dataset(TRAINING_FILENAMES, labeled=True, ordered=True).batch(1)\nlabels_list = []\nfor _, label in eda_dataset:\n    labels_list.append(label.numpy()[0])\n\nlabels_array = np.array(labels_list)\nclass_counts = pd.Series(labels_array).value_counts().sort_index()\n\nprint(\"Estatísticas descritivas da quantidade de imagens por classe:\")\nprint(class_counts.describe())\n\nprint(\"\\n5 classes com MAIS imagens:\")\nfor idx in class_counts.sort_values(ascending=False).head(5).index:\n    print(f\"  {CLASSES[idx]}: {class_counts[idx]} imagens\")\n\nprint(\"\\n5 classes com MENOS imagens:\")\nfor idx in class_counts.sort_values(ascending=True).head(5).index:\n    print(f\"  {CLASSES[idx]}: {class_counts[idx]} imagens\")\n\n# --- 3.2 Gráfico de distribuição (mostra desbalanceamento entre classes) ---\nplt.figure(figsize=(16, 5))\nplt.bar(class_counts.index, class_counts.values, color='mediumseagreen')\nplt.xlabel('Classe (índice 0-103)')\nplt.ylabel('Número de imagens')\nplt.title('Distribuição do número de imagens por classe (treino)')\nplt.tight_layout()\nplt.show()\n\n# --- 3.3 Exemplos visuais de imagens ---\ndef show_examples(dataset, n=9):\n    plt.figure(figsize=(10, 10))\n    for i, (image, label) in enumerate(dataset.take(n)):\n        ax = plt.subplot(3, 3, i + 1)\n        plt.imshow(image.numpy())\n        plt.title(CLASSES[label.numpy()], fontsize=9)\n        plt.axis('off')\n    plt.tight_layout()\n    plt.show()\n\nexample_dataset = load_dataset(TRAINING_FILENAMES, labeled=True, ordered=True)\nshow_examples(example_dataset, n=9)\n\n# --- 3.4 Estatísticas de intensidade de pixel (brilho médio por imagem) ---\nbrightness_list = []\nfor image, _ in example_dataset.take(500):  # amostra de 500 imagens por questão de tempo\n    brightness_list.append(np.mean(image.numpy()))\n\nprint(\"\\nEstatísticas de brilho médio (amostra de 500 imagens, escala 0-1):\")\nprint(pd.Series(brightness_list).describe())\n\nplt.figure(figsize=(8, 4))\nsns.histplot(brightness_list, bins=30, kde=True, color='goldenrod')\nplt.xlabel('Brilho médio da imagem')\nplt.ylabel('Frequência')\nplt.title('Distribuição do brilho médio das imagens (amostra)')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-30T04:06:20.566384Z","iopub.execute_input":"2026-06-30T04:06:20.566612Z","iopub.status.idle":"2026-06-30T04:06:32.182189Z","shell.execute_reply.started":"2026-06-30T04:06:20.566591Z","shell.execute_reply":"2026-06-30T04:06:32.181214Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ===================== SEÇÃO 4: TÉCNICA 1 - RANDOM FOREST (BASELINE CLÁSSICO) =====================\n\ndef extract_features(image_np):\n    \"\"\"Extrai um vetor de características simples de uma imagem (já em [0,1], shape HxWx3).\"\"\"\n    # Imagem reduzida e achatada (captura forma geral em baixa resolução)\n    small = tf.image.resize(image_np, (16, 16)).numpy().flatten()  # 16*16*3 = 768 features\n\n    # Histograma de cor por canal (captura a distribuição de cores da imagem)\n    hist_features = []\n    for ch in range(3):\n        hist, _ = np.histogram(image_np[:, :, ch], bins=16, range=(0, 1))\n        hist_features.extend(hist / hist.sum())  # normalizado\n\n    # Estatísticas simples por canal (média e desvio padrão de intensidade)\n    stats_features = []\n    for ch in range(3):\n        stats_features.append(image_np[:, :, ch].mean())\n        stats_features.append(image_np[:, :, ch].std())\n\n    return np.concatenate([small, hist_features, stats_features])\n\ndef build_feature_matrix(filenames, max_images=None):\n    ds = load_dataset(filenames, labeled=True, ordered=True)\n    X, y = [], []\n    for i, (image, label) in enumerate(ds):\n        if max_images and i >= max_images:\n            break\n        X.append(extract_features(image.numpy()))\n        y.append(label.numpy())\n    return np.array(X), np.array(y)\n\nprint(\"Extraindo características do conjunto de treino (pode levar alguns minutos)...\")\nX_train_rf, y_train_rf = build_feature_matrix(TRAINING_FILENAMES)\n\nprint(\"Extraindo características do conjunto de validação...\")\nX_val_rf, y_val_rf = build_feature_matrix(VALIDATION_FILENAMES)\n\nprint(\"Shape treino:\", X_train_rf.shape, \"| Shape validação:\", X_val_rf.shape)\n\nrf_model = RandomForestClassifier(\n    n_estimators=200,\n    max_depth=20,\n    random_state=SEED,\n    n_jobs=-1\n)\nrf_model.fit(X_train_rf, y_train_rf)\n\ny_pred_rf = rf_model.predict(X_val_rf)\nacc_rf = accuracy_score(y_val_rf, y_pred_rf)\nprint(f\"\\nAcurácia do Random Forest na validação: {acc_rf:.4f}\")\n\nprint(\"\\nRelatório de classificação (Random Forest) - 10 primeiras classes:\")\nprint(classification_report(y_val_rf, y_pred_rf, target_names=[CLASSES[i] for i in range(10)], zero_division=0, labels=list(range(10))))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-30T04:06:32.183501Z","iopub.execute_input":"2026-06-30T04:06:32.184093Z","iopub.status.idle":"2026-06-30T04:08:56.877492Z","shell.execute_reply.started":"2026-06-30T04:06:32.184023Z","shell.execute_reply":"2026-06-30T04:08:56.876346Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ===================== SEÇÃO 5: TÉCNICA 2 - CNN TREINADA DO ZERO =====================\n\nEPOCHS_CNN = 15\n\ndef build_cnn_from_scratch():\n    model = tf.keras.Sequential([\n        tf.keras.layers.Input(shape=(*IMAGE_SIZE, 3)),\n\n        tf.keras.layers.Conv2D(32, 3, activation='relu', padding='same'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.MaxPooling2D(2),\n\n        tf.keras.layers.Conv2D(64, 3, activation='relu', padding='same'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.MaxPooling2D(2),\n\n        tf.keras.layers.Conv2D(128, 3, activation='relu', padding='same'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.MaxPooling2D(2),\n\n        tf.keras.layers.Conv2D(256, 3, activation='relu', padding='same'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.MaxPooling2D(2),\n\n        tf.keras.layers.GlobalAveragePooling2D(),\n        tf.keras.layers.Dense(256, activation='relu'),\n        tf.keras.layers.Dropout(0.5),\n        tf.keras.layers.Dense(len(CLASSES), activation='softmax')\n    ])\n    return model\n\nwith strategy.scope():\n    cnn_model = build_cnn_from_scratch()\n    cnn_model.compile(\n        optimizer='adam',\n        loss='sparse_categorical_crossentropy',\n        metrics=['sparse_categorical_accuracy']\n    )\n\ncnn_model.summary()\n\nhistory_cnn = cnn_model.fit(\n    get_training_dataset(),\n    steps_per_epoch=STEPS_PER_EPOCH,\n    epochs=EPOCHS_CNN,\n    validation_data=get_validation_dataset()\n)\n\n# Curvas de treino (úteis para os Resultados: mostram se houve overfitting)\nplt.figure(figsize=(12, 4))\nplt.subplot(1, 2, 1)\nplt.plot(history_cnn.history['sparse_categorical_accuracy'], label='Treino')\nplt.plot(history_cnn.history['val_sparse_categorical_accuracy'], label='Validação')\nplt.title('Acurácia - CNN do zero')\nplt.xlabel('Época')\nplt.legend()\n\nplt.subplot(1, 2, 2)\nplt.plot(history_cnn.history['loss'], label='Treino')\nplt.plot(history_cnn.history['val_loss'], label='Validação')\nplt.title('Loss - CNN do zero')\nplt.xlabel('Época')\nplt.legend()\nplt.tight_layout()\nplt.show()\n\nacc_cnn = max(history_cnn.history['val_sparse_categorical_accuracy'])\nprint(f\"\\nMelhor acurácia de validação (CNN do zero): {acc_cnn:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-30T04:08:56.878769Z","iopub.execute_input":"2026-06-30T04:08:56.879125Z","iopub.status.idle":"2026-06-30T04:15:17.210989Z","shell.execute_reply.started":"2026-06-30T04:08:56.879083Z","shell.execute_reply":"2026-06-30T04:15:17.209966Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ===================== SEÇÃO 6: TÉCNICA 3 - TRANSFER LEARNING (DenseNet201) =====================\n\nEPOCHS_TL = 15\n\ndef build_transfer_model():\n    base_model = tf.keras.applications.DenseNet201(\n        input_shape=(*IMAGE_SIZE, 3),\n        weights='imagenet',\n        include_top=False\n    )\n    base_model.trainable = False  # congela os pesos pré-treinados inicialmente\n\n    model = tf.keras.Sequential([\n        base_model,\n        tf.keras.layers.GlobalAveragePooling2D(),\n        tf.keras.layers.Dense(256, activation='relu'),\n        tf.keras.layers.Dropout(0.3),\n        tf.keras.layers.Dense(len(CLASSES), activation='softmax')\n    ])\n    return model, base_model\n\nwith strategy.scope():\n    tl_model, tl_base = build_transfer_model()\n    tl_model.compile(\n        optimizer='adam',\n        loss='sparse_categorical_crossentropy',\n        metrics=['sparse_categorical_accuracy']\n    )\n\ntl_model.summary()\n\nearly_stop_tl = tf.keras.callbacks.EarlyStopping(\n    monitor='val_sparse_categorical_accuracy',\n    patience=3,\n    restore_best_weights=True\n)\n\nhistory_tl = tl_model.fit(\n    get_training_dataset(),\n    steps_per_epoch=STEPS_PER_EPOCH,\n    epochs=EPOCHS_TL,\n    validation_data=get_validation_dataset(),\n    callbacks=[early_stop_tl]\n)\n\nplt.figure(figsize=(12, 4))\nplt.subplot(1, 2, 1)\nplt.plot(history_tl.history['sparse_categorical_accuracy'], label='Treino')\nplt.plot(history_tl.history['val_sparse_categorical_accuracy'], label='Validação')\nplt.title('Acurácia - Transfer Learning (DenseNet201)')\nplt.xlabel('Época')\nplt.legend()\n\nplt.subplot(1, 2, 2)\nplt.plot(history_tl.history['loss'], label='Treino')\nplt.plot(history_tl.history['val_loss'], label='Validação')\nplt.title('Loss - Transfer Learning (DenseNet201)')\nplt.xlabel('Época')\nplt.legend()\nplt.tight_layout()\nplt.show()\n\nacc_tl = max(history_tl.history['val_sparse_categorical_accuracy'])\nprint(f\"\\nMelhor acurácia de validação (Transfer Learning): {acc_tl:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-30T04:16:07.437216Z","iopub.execute_input":"2026-06-30T04:16:07.437623Z","iopub.status.idle":"2026-06-30T04:27:32.825755Z","shell.execute_reply.started":"2026-06-30T04:16:07.437589Z","shell.execute_reply":"2026-06-30T04:27:32.824660Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ===================== SEÇÃO 7: COMPARAÇÃO DAS TÉCNICAS E SUBMISSÃO =====================\n\n# --- 7.1 Tabela comparativa final ---\nresultados = pd.DataFrame({\n    'Técnica': ['Random Forest (baseline)', 'CNN do zero', 'Transfer Learning (DenseNet201)'],\n    'Acurácia (validação)': [acc_rf, acc_cnn, acc_tl]\n})\nprint(resultados)\n\nplt.figure(figsize=(7, 4))\nplt.bar(resultados['Técnica'], resultados['Acurácia (validação)'], color=['indianred', 'goldenrod', 'mediumseagreen'])\nplt.ylabel('Acurácia de validação')\nplt.title('Comparação entre as técnicas')\nplt.ylim(0, 1)\nfor i, v in enumerate(resultados['Acurácia (validação)']):\n    plt.text(i, v + 0.02, f\"{v:.4f}\", ha='center')\nplt.tight_layout()\nplt.show()\n\n# --- 7.2 Matriz de confusão do melhor modelo (Transfer Learning), para análise de erros ---\nval_ds_ordered = get_validation_dataset(ordered=True)\nimages_val = val_ds_ordered.map(lambda image, label: image)\nlabels_val = next(iter(val_ds_ordered.unbatch().map(lambda image, label: label).batch(NUM_VALIDATION_IMAGES))).numpy()\n\npred_probs = tl_model.predict(images_val)\npred_labels = np.argmax(pred_probs, axis=-1)\n\ncm = confusion_matrix(labels_val, pred_labels, labels=range(len(CLASSES)))\n\nplt.figure(figsize=(14, 12))\nsns.heatmap(cm, cmap='Blues', xticklabels=False, yticklabels=False)\nplt.xlabel('Classe prevista')\nplt.ylabel('Classe verdadeira')\nplt.title('Matriz de confusão - Transfer Learning (104 classes)')\nplt.tight_layout()\nplt.show()\n\nprint(\"\\nRelatório de classificação completo (Transfer Learning):\")\nprint(classification_report(labels_val, pred_labels, target_names=CLASSES, zero_division=0))\n\n# --- 7.3 Geração da submissão no Kaggle, usando o modelo de Transfer Learning ---\ntest_ds = get_test_dataset(ordered=True)\ntest_images_ds = test_ds.map(lambda image, idnum: image)\n\nprint(\"Gerando previsões no conjunto de teste...\")\ntest_probs = tl_model.predict(test_images_ds)\ntest_pred_labels = np.argmax(test_probs, axis=-1)\n\ntest_ids_ds = test_ds.map(lambda image, idnum: idnum).unbatch()\ntest_ids = next(iter(test_ids_ds.batch(NUM_TEST_IMAGES))).numpy().astype('U')  # decodifica bytes para string\n\nsubmission = pd.DataFrame({\n    'id': test_ids,\n    'label': test_pred_labels\n})\nsubmission.to_csv('submission.csv', index=False)\nprint(\"\\nArquivo submission.csv gerado com sucesso!\")\nprint(submission.head())\nprint(f\"\\nTotal de previsões: {len(submission)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-30T04:34:30.081583Z","iopub.execute_input":"2026-06-30T04:34:30.082469Z","iopub.status.idle":"2026-06-30T04:36:05.540666Z","shell.execute_reply.started":"2026-06-30T04:34:30.082432Z","shell.execute_reply":"2026-06-30T04:36:05.539841Z"}},"outputs":[],"execution_count":null}]}