{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":13836,"databundleVersionId":1718836,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":778,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":645,"modelId":55}],"dockerImageVersionId":31154,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ============================================\n# 1. CONFIGURACIÓN Y CARGA DE DATOS\n# ============================================\nimport os\nimport json\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras import layers, Model, optimizers\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import confusion_matrix, classification_report, accuracy_score\nimport seaborn as sns\n\n# Definir rutas principales\nPATHS = {\n    'TRAIN_CSV': '/kaggle/input/cassava-leaf-disease-classification/train.csv',\n    'TEST_CSV': '/kaggle/input/cassava-leaf-disease-classification/sample_submission.csv',\n    'DISEASE_MAP': '/kaggle/input/cassava-leaf-disease-classification/label_num_to_disease_map.json',\n    'TRAIN_IMAGES': '/kaggle/input/cassava-leaf-disease-classification/train_images',\n    'TEST_IMAGES': '/kaggle/input/cassava-leaf-disease-classification/test_images',\n    'OUTPUT': '/kaggle/working/submission.csv',\n    'MODEL_CACHE': '/kaggle/working/model_cache',\n    'WEIGHTS': '/kaggle/working/weights',\n    'PLOTS': '/kaggle/working/plots',\n    'SAVED_MODEL': '/kaggle/working/cassava_disease_model_tf'\n}\n\n# Crear directorios\nfor directory in ['MODEL_CACHE', 'WEIGHTS', 'PLOTS', 'SAVED_MODEL']:\n    os.makedirs(PATHS[directory], exist_ok=True)\n\n# Cargar mapeo de enfermedades\nwith open(PATHS['DISEASE_MAP'], 'r') as f:\n    disease_map = json.load(f)\ndisease_map = {int(k): v for k, v in disease_map.items()}\nnum_classes = len(disease_map)\n\nprint(f\"Número de clases: {num_classes}\")\nprint(\"Mapeo de clases:\", disease_map)\n\n# Cargar dataset de entrenamiento\ntrain_df = pd.read_csv(PATHS['TRAIN_CSV'])\nprint(f\"Datos de entrenamiento: {train_df.shape}\")\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-12T21:26:53.253976Z","iopub.execute_input":"2025-10-12T21:26:53.254508Z","iopub.status.idle":"2025-10-12T21:26:53.274918Z","shell.execute_reply.started":"2025-10-12T21:26:53.254469Z","shell.execute_reply":"2025-10-12T21:26:53.274250Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# 2. DIVISIÓN DEL DATASET Y GENERADORES\n# ============================================\ntrain_df, val_df = train_test_split(\n    train_df, test_size=0.2, stratify=train_df['label'], random_state=42\n)\n\ntrain_df['label_str'] = train_df['label'].astype(str)\nval_df['label_str'] = val_df['label'].astype(str)\n\n# Parámetros de imagen\nimg_height, img_width = 224, 224\nbatch_size = 32\n\n# Generadores\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=15,\n    width_shift_range=0.15,\n    height_shift_range=0.15,\n    shear_range=0.15,\n    zoom_range=0.15,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\nval_datagen = ImageDataGenerator(rescale=1./255)\n\ntrain_generator = train_datagen.flow_from_dataframe(\n    train_df, directory=PATHS['TRAIN_IMAGES'],\n    x_col='image_id', y_col='label_str',\n    target_size=(img_height, img_width),\n    batch_size=batch_size, class_mode='categorical', shuffle=True\n)\nval_generator = val_datagen.flow_from_dataframe(\n    val_df, directory=PATHS['TRAIN_IMAGES'],\n    x_col='image_id', y_col='label_str',\n    target_size=(img_height, img_width),\n    batch_size=batch_size, class_mode='categorical', shuffle=False\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T21:27:19.410262Z","iopub.execute_input":"2025-10-12T21:27:19.411008Z","iopub.status.idle":"2025-10-12T21:28:17.138734Z","shell.execute_reply.started":"2025-10-12T21:27:19.410985Z","shell.execute_reply":"2025-10-12T21:28:17.138023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# 3. VALIDACIÓN DE SPLITS Y ETIQUETAS\n# ============================================\ndef validate_dataset_splits(train_gen, val_gen):\n    print(\"\\n=== VALIDACIÓN DE DATASETS ===\")\n    total = train_gen.n + val_gen.n\n    print(f\"Entrenamiento: {train_gen.n} imágenes ({train_gen.n/total:.2%})\")\n    print(f\"Validación: {val_gen.n} imágenes ({val_gen.n/total:.2%})\")\nvalidate_dataset_splits(train_generator, val_generator)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T21:29:12.007368Z","iopub.execute_input":"2025-10-12T21:29:12.008059Z","iopub.status.idle":"2025-10-12T21:29:12.012577Z","shell.execute_reply.started":"2025-10-12T21:29:12.008034Z","shell.execute_reply":"2025-10-12T21:29:12.011854Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# 4. CARGA DEL MODELO BASE (CROPNET)\n# ============================================\ncropnet_path = \"/kaggle/input/cropnet/tensorflow1/classifier-cassava-disease-v1/1\"\nbase_model_layer = tf.keras.layers.TFSMLayer(cropnet_path, call_endpoint='default')\n\n# Inspeccionar salida\ninputs = tf.keras.Input(shape=(img_height, img_width, 3))\nbase_outputs = base_model_layer(inputs)\nx = base_outputs[list(base_outputs.keys())[0]] if isinstance(base_outputs, dict) else base_outputs\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T21:34:33.705860Z","iopub.execute_input":"2025-10-12T21:34:33.706139Z","iopub.status.idle":"2025-10-12T21:34:36.287800Z","shell.execute_reply.started":"2025-10-12T21:34:33.706120Z","shell.execute_reply":"2025-10-12T21:34:36.287187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# 5. MODELO FINAL Y ENTRENAMIENTO\n# ============================================\nx = layers.Dense(256, activation='relu')(x)\nx = layers.Dropout(0.5)(x)\noutputs = layers.Dense(num_classes, activation='softmax')(x)\nmodel = Model(inputs=inputs, outputs=outputs)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T21:35:55.887239Z","iopub.execute_input":"2025-10-12T21:35:55.887571Z","iopub.status.idle":"2025-10-12T21:35:57.058777Z","shell.execute_reply.started":"2025-10-12T21:35:55.887548Z","shell.execute_reply":"2025-10-12T21:35:57.058149Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Fase 1: entrenar solo la cabeza\nbase_model_layer.trainable = True\nmodel.compile(optimizer=optimizers.Adam(1e-4), loss='categorical_crossentropy', metrics=['accuracy'])\n\ncallbacks = [\n    ModelCheckpoint(os.path.join(PATHS['WEIGHTS'], 'best_model.keras'),\n                    monitor='val_accuracy', save_best_only=True, mode='max', verbose=1),\n    EarlyStopping(monitor='val_accuracy', patience=10, restore_best_weights=True, verbose=1),\n    ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=5, min_lr=1e-6, verbose=1)\n]\n\nsteps_per_epoch = train_generator.n // batch_size\nval_steps = val_generator.n // batch_size\n\nprint(\"Entrenando con modelo base congelado...\")\nhistory_frozen = model.fit(\n    train_generator, steps_per_epoch=steps_per_epoch,\n    validation_data=val_generator, validation_steps=val_steps,\n    epochs=10, callbacks=callbacks\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T21:36:21.127651Z","iopub.execute_input":"2025-10-12T21:36:21.128339Z","iopub.status.idle":"2025-10-12T21:59:54.298341Z","shell.execute_reply.started":"2025-10-12T21:36:21.128315Z","shell.execute_reply":"2025-10-12T21:59:54.297583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Fase 2: fine-tuning\nbase_model_layer.trainable = True\nmodel.compile(optimizer=optimizers.Adam(5e-6), loss='categorical_crossentropy', metrics=['accuracy'])\n\nprint(\"Entrenando con modelo base descongelado (fine-tuning)...\")\nhistory_unfrozen = model.fit(\n    train_generator, steps_per_epoch=steps_per_epoch,\n    validation_data=val_generator, validation_steps=val_steps,\n    epochs=1, initial_epoch=history_frozen.epoch[-1]+1, callbacks=callbacks\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T22:00:56.215716Z","iopub.execute_input":"2025-10-12T22:00:56.216335Z","iopub.status.idle":"2025-10-12T22:22:16.884218Z","shell.execute_reply.started":"2025-10-12T22:00:56.216309Z","shell.execute_reply":"2025-10-12T22:22:16.883652Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# 6. VISUALIZACIÓN DEL ENTRENAMIENTO\n# ============================================\ndef plot_training_history(history_frozen, history_unfrozen):\n    merged = {}\n    for m in history_frozen.history:\n        merged[m] = list(history_frozen.history[m])\n    for m in history_unfrozen.history:\n        merged[m].extend(history_unfrozen.history[m])\n    epochs = range(1, len(merged['accuracy']) + 1)\n\n    plt.figure(figsize=(14,5))\n    plt.subplot(1,2,1)\n    plt.plot(epochs, merged['accuracy'], 'b', label='Entrenamiento')\n    plt.plot(epochs, merged['val_accuracy'], 'r', label='Validación')\n    plt.axvline(x=len(history_frozen.history['accuracy']), color='g', linestyle='--')\n    plt.title('Precisión del modelo')\n    plt.legend()\n    \n    plt.subplot(1,2,2)\n    plt.plot(epochs, merged['loss'], 'b', label='Entrenamiento')\n    plt.plot(epochs, merged['val_loss'], 'r', label='Validación')\n    plt.axvline(x=len(history_frozen.history['accuracy']), color='g', linestyle='--')\n    plt.title('Pérdida del modelo')\n    plt.legend()\n    plt.tight_layout()\n    plt.show()\n\nplot_training_history(history_frozen, history_unfrozen)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T22:23:10.305894Z","iopub.execute_input":"2025-10-12T22:23:10.306521Z","iopub.status.idle":"2025-10-12T22:23:10.709360Z","shell.execute_reply.started":"2025-10-12T22:23:10.306469Z","shell.execute_reply":"2025-10-12T22:23:10.708727Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# 7. GUARDADO DEL MODELO Y PREDICCIONES\n# ============================================\n@tf.function(input_signature=[tf.TensorSpec(shape=[None, img_height, img_width, 3], dtype=tf.float32)])\ndef serving_fn(input_image):\n    return {'predictions': model(input_image, training=False)}\n\ntf.saved_model.save(model, PATHS['SAVED_MODEL'], signatures={'serving_default': serving_fn})\nprint(f\"Modelo guardado en: {PATHS['SAVED_MODEL']}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T22:23:19.348174Z","iopub.execute_input":"2025-10-12T22:23:19.348831Z","iopub.status.idle":"2025-10-12T22:23:20.847111Z","shell.execute_reply.started":"2025-10-12T22:23:19.348806Z","shell.execute_reply":"2025-10-12T22:23:20.846328Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# 8. MATRIZ DE CONFUSIÓN Y MÉTRICAS\n# ============================================\nprint(\"\\n=== EVALUACIÓN DEL MODELO EN VALIDACIÓN ===\")\nval_generator.reset()\npreds = model.predict(val_generator, steps=val_steps + 1)\ny_pred = np.argmax(preds, axis=1)\ny_true = val_generator.classes[:len(y_pred)]\n\n# Reporte de clasificación\nreport = classification_report(y_true, y_pred, target_names=[disease_map[i] for i in range(num_classes)])\nprint(report)\n\n# Matriz de confusión\ncm = confusion_matrix(y_true, y_pred)\nplt.figure(figsize=(8,6))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues',\n            xticklabels=[disease_map[i] for i in range(num_classes)],\n            yticklabels=[disease_map[i] for i in range(num_classes)])\nplt.xlabel(\"Predicho\")\nplt.ylabel(\"Real\")\nplt.title(\"Matriz de confusión - Validación\")\nplt.show()\n\nacc = accuracy_score(y_true, y_pred)\nprint(f\"Precisión global: {acc:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-12T22:23:24.369471Z","iopub.execute_input":"2025-10-12T22:23:24.370175Z","iopub.status.idle":"2025-10-12T22:23:47.242791Z","shell.execute_reply.started":"2025-10-12T22:23:24.370151Z","shell.execute_reply":"2025-10-12T22:23:47.242072Z"}},"outputs":[],"execution_count":null}]}