{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"},{"sourceId":6819899,"sourceType":"datasetVersion","datasetId":3922540},{"sourceId":7320027,"sourceType":"datasetVersion","datasetId":4247945},{"sourceId":7755213,"sourceType":"datasetVersion","datasetId":4534686},{"sourceId":9912034,"sourceType":"datasetVersion","datasetId":6090444},{"sourceId":12423607,"sourceType":"datasetVersion","datasetId":7836008},{"sourceId":12434794,"sourceType":"datasetVersion","datasetId":7843649},{"sourceId":12434906,"sourceType":"datasetVersion","datasetId":7843701},{"sourceId":12470620,"sourceType":"datasetVersion","datasetId":7867577},{"sourceId":12470628,"sourceType":"datasetVersion","datasetId":7867580}],"dockerImageVersionId":30747,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install transformers==4.28.0\n!pip install openpyxl\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport torch\nimport matplotlib.pyplot as plt\nimport torchvision.transforms.functional as F\nfrom datasets import load_dataset, DatasetDict, load_metric\nfrom transformers import BeitImageProcessor, BeitForImageClassification, TrainingArguments, Trainer\nfrom torchvision.transforms import Compose, Normalize, RandomHorizontalFlip, RandomVerticalFlip, RandomRotation, Resize, ToTensor\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom sklearn.model_selection import train_test_split\nimport openpyxl\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\n\n# Ruta del archivo con etiquetas\nexcel_path = \"/kaggle/input/labels-multilabel-diabetic-retinopathy-aptos/labels_multietiqueta_biomarcadores.csv\"\ndata = pd.read_csv(excel_path)\n\n# Etiquetas a usar\nlabel_columns = ['exudados duros', 'hemorragias', 'laser spot', 'microaneurismas']\ndata = data[['Fundus'] + label_columns]\ndata[label_columns] = data[label_columns].fillna(0).astype(float)\nlabels = data[label_columns].values\n\n# Directorio real de imágenes\nimage_dir = \"/kaggle/input/aptos2019-blindness-detection/train_images\"\nreal_files = os.listdir(image_dir)\n\n# Buscar con \"in\"\nvalid_image_filenames = []\nvalid_labels = []\n\nfor i, partial_name in enumerate(data['Fundus']):\n    partial_name = partial_name[0:len(partial_name)-4]\n    matched = [f for f in real_files if partial_name in f]\n    if matched:\n        full_path = os.path.join(image_dir, matched[0])\n        valid_image_filenames.append(full_path)\n        valid_labels.append(labels[i])\n\n# Convertir a arrays\nvalid_image_filenames = np.array(valid_image_filenames)\nvalid_labels = np.array(valid_labels)\n\nprint(f\"Total imágenes encontradas: {len(valid_image_filenames)}\")\n\n# División\ntrain_files, test_files, train_labels, test_labels = train_test_split(\n    valid_image_filenames, valid_labels, test_size=0.2, random_state=42\n)\ntrain_files, val_files, train_labels, val_labels = train_test_split(\n    train_files, train_labels, test_size=0.1, random_state=42\n)\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_files","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_labels","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"imagePreprocessing = BeitImageProcessor.from_pretrained(\"microsoft/beit-base-patch16-224-pt22k-ft22k\")\n\nnormalize = Normalize(mean=imagePreprocessing.image_mean, std=imagePreprocessing.image_std)\nsize = (imagePreprocessing.size[\"height\"], imagePreprocessing.size[\"width\"])\n\ntrainingTransforms = Compose([\n    Resize(size),\n    RandomRotation(180),\n    RandomHorizontalFlip(),\n    ToTensor(),\n    normalize\n])\n\ndef load_image(file):\n    image = Image.open(file).convert(\"RGB\")\n    return trainingTransforms(image)\n\ndef process_images(image_files, labels):\n    images = [load_image(file) for file in image_files]\n    return torch.stack(images), torch.tensor(labels)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CustomDataset(torch.utils.data.Dataset):\n    def __init__(self, image_files, labels):\n        self.image_files = image_files\n        self.labels = labels\n\n    def __len__(self):\n        return len(self.image_files)\n\n    def __getitem__(self, idx):\n        image = load_image(self.image_files[idx])\n        label = torch.tensor(self.labels[idx])\n        return {\"pixel_values\": image, \"labels\": label}\n\ntrain_dataset = CustomDataset(train_files, train_labels)\nval_dataset = CustomDataset(val_files, val_labels)\ntest_dataset = CustomDataset(test_files, test_labels)\n\ndef collate_fn(batch):\n    pixel_values = torch.stack([x[\"pixel_values\"] for x in batch])\n    labels = torch.stack([x[\"labels\"] for x in batch]).float()  # Asegurarse de que las etiquetas sean float\n    return {\"pixel_values\": pixel_values, \"labels\": labels}\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Asegúrate de tener importada la librería PIL\nfrom PIL import Image\n\ndef load_image(file):\n    image = Image.open(file).convert(\"RGB\")\n    return trainingTransforms(image)\n\nmodel = BeitForImageClassification.from_pretrained(\n    \"microsoft/beit-base-patch16-224-pt22k-ft22k\",\n    num_labels=4,  # Cambiado para multilabel\n    problem_type=\"multi_label_classification\",\n    ignore_mismatched_sizes=True\n)\n\n\nfrom sklearn.metrics import f1_score\n\ndef compute_metrics(p):\n    pred_logits = p.predictions\n    pred_labels = (pred_logits > 0.5).astype(int)\n    true_labels = p.label_ids\n\n    f1 = f1_score(true_labels, pred_labels, average='samples')  # Calcula el F1 score\n    return {\"f1\": f1}\n\n\ntraining_args = TrainingArguments(\n    output_dir=\"./Beit-for-ODIR\",\n    remove_unused_columns=False,\n    per_device_train_batch_size=32,\n    per_device_eval_batch_size=32,\n    fp16=True,\n    load_best_model_at_end=True,\n    evaluation_strategy=\"steps\",\n    save_steps=30,\n    save_total_limit=2,\n    logging_steps=10,\n    num_train_epochs=10,\n    report_to=\"none\",\n)\n\ntrainer = Trainer(\n    model=model,\n    args=training_args,\n    data_collator=collate_fn,\n    compute_metrics=compute_metrics,\n    train_dataset=train_dataset,\n    eval_dataset=val_dataset,\n    tokenizer=imagePreprocessing,\n)\n\ntrain_results = trainer.train()\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions = trainer.predict(test_dataset)\npred_logits = predictions.predictions\npred_labels = (pred_logits > 0.5).astype(int)\ntrue_labels = np.array([x['labels'] for x in test_dataset])\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Guarda el modelo entrenado\ntrainer.save_model(\"/kaggle/working/\")\n\n# Guarda el procesador de imágenes\nimagePreprocessing.save_pretrained(\"/kaggle/working/\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import f1_score, accuracy_score, precision_score, recall_score, confusion_matrix\n\nf1 = f1_score(true_labels, pred_labels, average='samples')\naccuracy = accuracy_score(true_labels, pred_labels)\nprecision = precision_score(true_labels, pred_labels, average='samples')\nrecall = recall_score(true_labels, pred_labels, average='samples')\n\nprint(f\"F1 Score: {f1}\")\nprint(f\"Accuracy: {accuracy}\")\nprint(f\"Precision: {precision}\")\nprint(f\"Recall: {recall}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\ndef plot_confusion_matrix(conf_matrix, class_names):\n    fig, ax = plt.subplots(figsize=(10, 10))\n    cax = ax.matshow(conf_matrix, cmap=plt.cm.Blues)\n    fig.colorbar(cax)\n\n    ax.set_xticks(np.arange(len(class_names)))\n    ax.set_yticks(np.arange(len(class_names)))\n\n    ax.set_xticklabels(class_names)\n    ax.set_yticklabels(class_names)\n\n    plt.xlabel('Predicted')\n    plt.ylabel('True')\n    plt.show()\n\n# Calcular la matriz de confusión\nconf_matrix = confusion_matrix(true_labels.argmax(axis=1), pred_labels.argmax(axis=1))\nclass_names = [ 'CNV ', 'Laqquer Cracks', 'atrofia parcheada', 'tessellated', 'atrofia peripapilar']\nplot_confusion_matrix(conf_matrix, class_names)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, auc\n\ndef plot_roc(true_labels, pred_probs, class_names):\n    plt.figure(figsize=(15, 10))\n    for i, class_name in enumerate(class_names):\n        fpr, tpr, _ = roc_curve(true_labels[:, i], pred_probs[:, i])\n        roc_auc = auc(fpr, tpr)\n        plt.plot(fpr, tpr, lw=2, label=f'{class_name} (AUC = {roc_auc:0.2f})')\n\n    plt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\n    plt.xlim([0.0, 1.0])\n    plt.ylim([0.0, 1.05])\n    plt.xlabel('False Positive Rate')\n    plt.ylabel('True Positive Rate')\n    plt.title('Receiver Operating Characteristic (ROC)')\n    plt.legend(loc=\"lower right\")\n    plt.show()\n\n# Graficar las curvas ROC\nplot_roc(true_labels, pred_logits, class_names)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generar predicciones en el conjunto de prueba\npredictions = trainer.predict(test_dataset)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.metrics import classification_report, confusion_matrix\n\n# Obtener los logits y etiquetas verdaderas\nlogits = predictions.predictions\ny_true = predictions.label_ids\n\n# Aplicar la función sigmoide para obtener probabilidades\nprobabilities = 1 / (1 + np.exp(-logits))\n\n# Convertir probabilidades a etiquetas binarias con un umbral de 0.5\ny_pred = (probabilities > 0.5).astype(int)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Suponiendo que label_columns es la lista de nombres de tus etiquetas\nlabel_columns = ['Normal ', 'Diabetic', 'Edema1',\"Edema2\"]\n\nfor i in range(len(label_columns)):\n    cm = confusion_matrix(y_true[:, i], y_pred[:, i])\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues')\n    plt.title(f\"Matriz de confusión para {label_columns[i]}\")\n    plt.xlabel('Predicción')\n    plt.ylabel('Valor Verdadero')\n    plt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generar el informe de clasificación\nreport = classification_report(y_true, y_pred, target_names=label_columns)\nprint(report)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Guardar predicciones y etiquetas verdaderas\nnp.save('y_pred.npy', y_pred)\nnp.save('y_true.npy', y_true)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_filenames = test_files  # Esto ya lo tienes; es el arreglo original con paths\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# Obtener logits y etiquetas verdaderas\nlogits = predictions.predictions\ny_true = predictions.label_ids\n\n# Aplicar sigmoide y umbral\nprobabilities = 1 / (1 + np.exp(-logits))\ny_pred = (probabilities > 0.5).astype(int)\n\n# Etiquetas\nlabel_columns = ['Normal', 'Diabetic', 'Edema1',\"Edema2\"]\n\n# Guardar nombres sin la ruta completa (solo el nombre del archivo)\nfilenames = [os.path.basename(path) for path in test_filenames]\n\n# Crear DataFrame\ndf = pd.DataFrame({\n    \"filename\": filenames,\n    **{f\"True_{label}\": y_true[:, i] for i, label in enumerate(label_columns)},\n    **{f\"Pred_{label}\": y_pred[:, i] for i, label in enumerate(label_columns)},\n})\n\n# Guardar como CSV\ndf.to_csv(\"resultados_predicciones3.csv\", index=False)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}