{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":21154,"databundleVersionId":1243559}],"dockerImageVersionId":31287,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Important Tips for Data Science & AI Learners;\n* If you're interested in making career in Academia , such as Teaching,Coaching, Research Assiant professions in Data Science & AI areas, you may practice on kaggle notebooks, Google Colab Notebooks, and other IDE's platform that are user friendly for demonstations your skills.\n\n* But, if you're intersted in Industry Based Careers & Other Challanging sectors, Such as Applied Data Scientist, ML Engineer etc. Make sure you're able to deploy Data Driven AI Prototypes Projects in your local machine in a small scale, not just writing the code on Kaggle notebook, Google colab or others user friendly IDE's\n\n* Finally, Publish your work so that people can judge in open sources community.\n\n* **If you find this work/notebook insightsfull, leave a comment or share any resources for increase my knowldges for future improvement, your upvote is always welcome!**\n","metadata":{}},{"cell_type":"markdown","source":"**Import Desried things;**","metadata":{}},{"cell_type":"code","source":"import os\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n        #print(os.path.join(dirname, filename))\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T15:27:26.390173Z","iopub.execute_input":"2026-03-14T15:27:26.390386Z","iopub.status.idle":"2026-03-14T15:27:27.561527Z","shell.execute_reply.started":"2026-03-14T15:27:26.390365Z","shell.execute_reply":"2026-03-14T15:27:27.560687Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Import Necessries Libarries;**","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\nfrom pathlib import Path","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T15:27:27.563114Z","iopub.execute_input":"2026-03-14T15:27:27.563498Z","iopub.status.idle":"2026-03-14T15:27:27.567985Z","shell.execute_reply.started":"2026-03-14T15:27:27.563472Z","shell.execute_reply":"2026-03-14T15:27:27.567154Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras import applications","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T15:27:27.569007Z","iopub.execute_input":"2026-03-14T15:27:27.569191Z","iopub.status.idle":"2026-03-14T15:27:55.718234Z","shell.execute_reply.started":"2026-03-14T15:27:27.569173Z","shell.execute_reply":"2026-03-14T15:27:55.717619Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"AUTO = tf.data.AUTOTUNE\nIMG_SIZE = 331\nBATCH_SIZE = 16\nEPOCHS_STAGE1 = 1\nEPOCHS_STAGE2 = 1\nNUM_CLASSES = 104\nSEED = 123\n\nGCS_PATH = \"/kaggle/input/competitions/tpu-getting-started/tfrecords-jpeg-331x331\"\nTRAIN_GLOB = f\"{GCS_PATH}/train/*.tfrec\"\nVAL_GLOB = f\"{GCS_PATH}/val/*.tfrec\"\nTEST_GLOB = f\"{GCS_PATH}/test/*.tfrec\"\nSAMPLE_SUBMISSION_PATH = \"/kaggle/input/competitions/tpu-getting-started/sample_submission.csv\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T15:27:55.719054Z","iopub.execute_input":"2026-03-14T15:27:55.719628Z","iopub.status.idle":"2026-03-14T15:27:55.724525Z","shell.execute_reply.started":"2026-03-14T15:27:55.719591Z","shell.execute_reply":"2026-03-14T15:27:55.723744Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"strategy = tf.distribute.get_strategy()\ngpus = tf.config.list_physical_devices(\"GPU\")\nif gpus:\n    print(\"Available GPU:\", [gpu.name for gpu in gpus])\nelse:\n    print(\"GPU hardware, available CPU\")\nprint(\"Something about the hardware:\", strategy.num_replicas_in_sync)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T15:27:55.725404Z","iopub.execute_input":"2026-03-14T15:27:55.725713Z","iopub.status.idle":"2026-03-14T15:27:56.581827Z","shell.execute_reply.started":"2026-03-14T15:27:55.725690Z","shell.execute_reply":"2026-03-14T15:27:56.581111Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tf.random.set_seed(SEED)\nnp.random.seed(SEED)\n\nprint(f\"TensorFlow {tf.__version__}, Keras {keras.__version__}\")\nprint(\n    f\"IMG_SIZE={IMG_SIZE}, BATCH_SIZE={BATCH_SIZE}, \"\n    f\"EPOCHS_STAGE1={EPOCHS_STAGE1}, EPOCHS_STAGE2={EPOCHS_STAGE2}, \"\n    f\"NUM_CLASSES={NUM_CLASSES}\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T15:27:56.582717Z","iopub.execute_input":"2026-03-14T15:27:56.583004Z","iopub.status.idle":"2026-03-14T15:27:56.635984Z","shell.execute_reply.started":"2026-03-14T15:27:56.582980Z","shell.execute_reply":"2026-03-14T15:27:56.635194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"FEATURES_TRAIN = {\n    \"id\": tf.io.FixedLenFeature([], tf.string),\n    \"class\": tf.io.FixedLenFeature([], tf.int64),\n    \"image\": tf.io.FixedLenFeature([], tf.string),\n}\nFEATURES_TEST = {\n    \"id\": tf.io.FixedLenFeature([], tf.string),\n    \"image\": tf.io.FixedLenFeature([], tf.string),\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T15:27:56.638196Z","iopub.execute_input":"2026-03-14T15:27:56.638396Z","iopub.status.idle":"2026-03-14T15:27:56.648629Z","shell.execute_reply.started":"2026-03-14T15:27:56.638378Z","shell.execute_reply":"2026-03-14T15:27:56.647907Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def decode_image(img_bytes: tf.Tensor) -> tf.Tensor:\n    img = tf.io.decode_jpeg(img_bytes, channels=3)\n    img = tf.image.resize(img, [IMG_SIZE, IMG_SIZE], method=\"bilinear\")\n    img = tf.cast(img, tf.float32)  # без деления на 255: EfficientNet сама нормализует\n    return img\n\n\ndef parse_train(example: tf.Tensor):\n    parsed = tf.io.parse_single_example(example, FEATURES_TRAIN)\n    image = decode_image(parsed[\"image\"])\n    label = tf.cast(parsed[\"class\"], tf.int32)\n    return image, label\n\n\ndef parse_test(example: tf.Tensor):\n    parsed = tf.io.parse_single_example(example, FEATURES_TEST)\n    image = decode_image(parsed[\"image\"])\n    sample_id = parsed[\"id\"]\n    return image, sample_id","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T15:27:56.649573Z","iopub.execute_input":"2026-03-14T15:27:56.649870Z","iopub.status.idle":"2026-03-14T15:27:56.663020Z","shell.execute_reply.started":"2026-03-14T15:27:56.649821Z","shell.execute_reply":"2026-03-14T15:27:56.662401Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_files = sorted(tf.io.gfile.glob(TRAIN_GLOB))\nval_files = sorted(tf.io.gfile.glob(VAL_GLOB))\ntest_files = sorted(tf.io.gfile.glob(TEST_GLOB))\n\nds_train_raw = tf.data.TFRecordDataset(train_files, num_parallel_reads=AUTO)\nds_val_raw = tf.data.TFRecordDataset(val_files, num_parallel_reads=AUTO)\nds_test_raw = tf.data.TFRecordDataset(test_files, num_parallel_reads=AUTO)\n\nds_train = (\n    ds_train_raw.map(parse_train, num_parallel_calls=AUTO)\n    .shuffle(buffer_size=2048, seed=SEED)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n\nds_val = (\n    ds_val_raw.map(parse_train, num_parallel_calls=AUTO)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n\nds_test = (\n    ds_test_raw.map(parse_test, num_parallel_calls=AUTO)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n\nprint(\"class train:\", len(train_files), \"val:\", len(val_files), \"test:\", len(test_files))\nprint(\"data size: ds_train, ds_val, ds_test\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T15:27:56.664072Z","iopub.execute_input":"2026-03-14T15:27:56.664406Z","iopub.status.idle":"2026-03-14T15:27:58.994482Z","shell.execute_reply.started":"2026-03-14T15:27:56.664383Z","shell.execute_reply":"2026-03-14T15:27:58.993853Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class MacroF1(keras.metrics.Metric):\n    def __init__(self, num_classes: int, name: str = \"macro_f1\", **kwargs):\n        super().__init__(name=name, **kwargs)\n        self.num_classes = num_classes\n        self.confusion = self.add_weight(\n            shape=(num_classes, num_classes),\n            initializer=\"zeros\",\n            dtype=tf.float32,\n        )\n\n    def update_state(self, y_true, y_pred, sample_weight=None):\n        y_pred = tf.argmax(y_pred, axis=-1)\n        y_true = tf.reshape(tf.cast(y_true, tf.int32), [-1])\n        y_pred = tf.reshape(tf.cast(y_pred, tf.int32), [-1])\n        conf = tf.math.confusion_matrix(\n            y_true, y_pred, num_classes=self.num_classes, dtype=tf.float32\n        )\n        self.confusion.assign_add(conf)\n\n    def result(self):\n        tp = tf.linalg.diag_part(self.confusion)\n        fp = tf.reduce_sum(self.confusion, axis=0) - tp\n        fn = tf.reduce_sum(self.confusion, axis=1) - tp\n        f1_per_class = (2.0 * tp) / (2.0 * tp + fp + fn + 1e-7)\n        return tf.reduce_mean(f1_per_class)\n\n    def reset_state(self):\n        self.confusion.assign(tf.zeros((self.num_classes, self.num_classes)))\n\n    def get_config(self):\n        config = super().get_config()\n        config[\"num_classes\"] = self.num_classes\n        return config","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T15:27:58.995381Z","iopub.execute_input":"2026-03-14T15:27:58.995654Z","iopub.status.idle":"2026-03-14T15:27:59.003244Z","shell.execute_reply.started":"2026-03-14T15:27:58.995624Z","shell.execute_reply":"2026-03-14T15:27:59.002446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with strategy.scope():\n    data_augmentation = keras.Sequential(\n        [\n            layers.RandomFlip(\"horizontal\"),\n            layers.RandomRotation(0.1),\n            layers.RandomZoom(0.1),\n            layers.RandomContrast(0.2),\n        ],\n        name=\"data_augmentation\",\n    )\n\n    base_model = applications.EfficientNetB3(\n        include_top=False,\n        weights=\"imagenet\",\n        input_shape=(IMG_SIZE, IMG_SIZE, 3),\n        pooling=\"avg\",\n    )\n    base_model.trainable = False\n\n    preprocess_input = applications.efficientnet.preprocess_input\n\n    inputs = keras.Input(shape=(IMG_SIZE, IMG_SIZE, 3))\n    x = data_augmentation(inputs)\n    x = preprocess_input(x)\n    x = base_model(x, training=False)\n    x = layers.Dense(512, activation=\"relu\")(x)\n    x = layers.Dropout(0.5)(x)\n    outputs = layers.Dense(NUM_CLASSES, activation=\"softmax\")(x)\n\n    model = keras.Model(inputs, outputs, name=\"flowers_efficientnetb3\")\n\n    model.compile(\n        optimizer=keras.optimizers.Adam(learning_rate=1e-3),\n        loss=keras.losses.SparseCategoricalCrossentropy(),\n        metrics=[\"accuracy\", MacroF1(num_classes=NUM_CLASSES)],\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T15:27:59.004125Z","iopub.execute_input":"2026-03-14T15:27:59.004408Z","iopub.status.idle":"2026-03-14T15:28:02.122732Z","shell.execute_reply.started":"2026-03-14T15:27:59.004387Z","shell.execute_reply":"2026-03-14T15:28:02.122151Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T15:28:02.123574Z","iopub.execute_input":"2026-03-14T15:28:02.123817Z","iopub.status.idle":"2026-03-14T15:28:02.148819Z","shell.execute_reply.started":"2026-03-14T15:28:02.123796Z","shell.execute_reply":"2026-03-14T15:28:02.148269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"callbacks_stage1 = [\n    keras.callbacks.ReduceLROnPlateau(\n        monitor=\"val_loss\",\n        factor=0.5,\n        patience=2,\n        min_lr=1e-6,\n        verbose=1,\n    ),\n    keras.callbacks.ModelCheckpoint(\n        \"best_model_stage1.keras\",\n        monitor=\"val_macro_f1\",\n        mode=\"max\",\n        save_best_only=True,\n        verbose=1,\n    ),\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T15:28:02.149578Z","iopub.execute_input":"2026-03-14T15:28:02.149802Z","iopub.status.idle":"2026-03-14T15:28:02.154671Z","shell.execute_reply.started":"2026-03-14T15:28:02.149784Z","shell.execute_reply":"2026-03-14T15:28:02.154099Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history_stage1 = model.fit(\n    ds_train,\n    validation_data=ds_val,\n    epochs=EPOCHS_STAGE1,\n    callbacks=callbacks_stage1,\n    verbose=1,\n)\n\nprint(\"Something about training\")\nprint(\"best model weights best_model.keras\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T15:28:02.155626Z","iopub.execute_input":"2026-03-14T15:28:02.156365Z","iopub.status.idle":"2026-03-14T15:31:19.255941Z","shell.execute_reply.started":"2026-03-14T15:28:02.156344Z","shell.execute_reply":"2026-03-14T15:31:19.255047Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with strategy.scope():\n    base_model.trainable = True\n\n    fine_tune_at = max(0, len(base_model.layers) - 20)\n    for layer in base_model.layers[:fine_tune_at]:\n        layer.trainable = False\n\n    model.compile(\n        optimizer=keras.optimizers.Adam(learning_rate=5e-4),\n        loss=keras.losses.SparseCategoricalCrossentropy(),\n        metrics=[\"accuracy\", MacroF1(num_classes=NUM_CLASSES)],\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T15:31:19.257194Z","iopub.execute_input":"2026-03-14T15:31:19.257526Z","iopub.status.idle":"2026-03-14T15:31:19.345724Z","shell.execute_reply.started":"2026-03-14T15:31:19.257503Z","shell.execute_reply":"2026-03-14T15:31:19.344978Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"callbacks_stage2 = [\n    keras.callbacks.ReduceLROnPlateau(\n        monitor=\"val_loss\",\n        factor=0.5,\n        patience=3,\n        min_lr=1e-6,\n        verbose=1,\n    ),\n    keras.callbacks.ModelCheckpoint(\n        \"best_model_finetuned.keras\",\n        monitor=\"val_macro_f1\",\n        mode=\"max\",\n        save_best_only=True,\n        verbose=1,\n    ),\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T15:31:19.347445Z","iopub.execute_input":"2026-03-14T15:31:19.347751Z","iopub.status.idle":"2026-03-14T15:31:19.354536Z","shell.execute_reply.started":"2026-03-14T15:31:19.347716Z","shell.execute_reply":"2026-03-14T15:31:19.353656Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history_stage2 = model.fit(\n    ds_train,\n    validation_data=ds_val,\n    epochs=EPOCHS_STAGE2,\n    callbacks=callbacks_stage2,\n    verbose=1,\n)\n\nprint(\"Something about model training\")\nprint(\"best model weights best_model.keras\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T15:31:19.355485Z","iopub.execute_input":"2026-03-14T15:31:19.355846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_model = keras.models.load_model(\n    \"best_model_finetuned.keras\",\n    compile=False,\n)\n\nids_list = []\nlabels_list = []\n\nfor batch_images, batch_ids in ds_test:\n    preds = best_model.predict(batch_images, verbose=0)\n    batch_labels = tf.argmax(preds, axis=-1).numpy()\n    batch_ids_decoded = [x.numpy().decode(\"utf-8\") for x in batch_ids]\n    ids_list.extend(batch_ids_decoded)\n    labels_list.extend(batch_labels)\n\nsubmission = pd.DataFrame({\"id\": ids_list, \"label\": labels_list})\nsubmission_path = \"submission.csv\"\nsubmission.to_csv(submission_path, index=False)\n\nprint(\"Submission created successfully\", submission_path)\nprint(submission.head())","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}