{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-06T03:36:22.459977Z","iopub.execute_input":"2025-08-06T03:36:22.460169Z","iopub.status.idle":"2025-08-06T03:36:33.385152Z","shell.execute_reply.started":"2025-08-06T03:36:22.460149Z","shell.execute_reply":"2025-08-06T03:36:33.384281Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install tensorflow==2.12.0  # Phiên bản ổn định với Kaggle","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-19T15:50:19.053805Z","iopub.execute_input":"2025-06-19T15:50:19.054440Z","iopub.status.idle":"2025-06-19T15:50:40.999392Z","shell.execute_reply.started":"2025-06-19T15:50:19.054412Z","shell.execute_reply":"2025-06-19T15:50:40.998637Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip uninstall tensorflow -y\n!pip install tensorflow-cpu==2.10.0  # Phiên bản ổn định","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-19T15:49:12.469472Z","iopub.execute_input":"2025-06-19T15:49:12.469750Z","iopub.status.idle":"2025-06-19T15:49:16.781383Z","shell.execute_reply.started":"2025-06-19T15:49:12.469725Z","shell.execute_reply":"2025-06-19T15:49:16.780646Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, classification_report\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Input\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom xgboost import XGBClassifier\nimport tensorflow as tf\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-06T03:36:54.849951Z","iopub.execute_input":"2025-08-06T03:36:54.850527Z","iopub.status.idle":"2025-08-06T03:37:10.309923Z","shell.execute_reply.started":"2025-08-06T03:36:54.850504Z","shell.execute_reply":"2025-08-06T03:37:10.309028Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Đọc dữ liệu từ Kaggle dataset\ndata_dir = '/kaggle/input/aptos2019-blindness-detection/train_images'\ncsv_file = '/kaggle/input/aptos2019-blindness-detection/train.csv'\ndf = pd.read_csv(csv_file)\n\n# Tiền xử lý ảnh với xử lý lỗi\ndef preprocess_image(image_path, target_size=(224, 224)):\n    try:\n        img = cv2.imread(image_path)\n        if img is None:\n            raise ValueError(f\"Không thể đọc ảnh từ {image_path}\")\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        img = cv2.resize(img, target_size)\n        img = img / 255.0\n        return img\n    except Exception as e:\n        print(f\"Lỗi khi xử lý ảnh {image_path}: {str(e)}\")\n        return None\n# Chuẩn bị dữ liệu với thanh tiến trình\nfrom tqdm import tqdm\n\nimages = []\nlabels = []\nfailed_images = []\n\nfor index, row in tqdm(df.iterrows(), total=len(df)):\n    image_path = os.path.join(data_dir, row['id_code'] + '.png')\n    img = preprocess_image(image_path)\n    if img is not None:\n        images.append(img)\n        labels.append(row['diagnosis'])\n    else:\n        failed_images.append(row['id_code'])\n\nprint(f\"\\nTổng số ảnh không thể đọc: {len(failed_images)}\")\n\nimages = np.array(images)\nlabels = np.array(labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-06T03:37:25.438858Z","iopub.execute_input":"2025-08-06T03:37:25.439459Z","iopub.status.idle":"2025-08-06T03:44:33.036301Z","shell.execute_reply.started":"2025-08-06T03:37:25.439428Z","shell.execute_reply":"2025-08-06T03:44:33.035691Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# split data\nX_train, X_test, y_train, y_test = train_test_split(\n    images, labels, test_size=0.2, random_state=42, stratify=labels\n)\n\n# Data Augmentation\ndatagen = ImageDataGenerator(\n    rotation_range=20,\n    zoom_range=0.15,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n# build CNN Feature Extractor\ndef build_cnn_feature_extractor(input_shape=(224, 224, 3)):\n    inputs = Input(shape=input_shape)\n    x = Conv2D(32, (3, 3), activation='relu')(inputs)\n    x = MaxPooling2D((2, 2))(x)\n    x = Conv2D(64, (3, 3), activation='relu')(x)\n    x = MaxPooling2D((2, 2))(x)\n    x = Conv2D(128, (3, 3), activation='relu')(x) \n    x = MaxPooling2D((2, 2))(x)\n    x = Flatten()(x)\n    x = Dense(256, activation='relu')(x)  \n    model = Model(inputs=inputs, outputs=x)\n    return model\n# trainding CNN\ncnn_model = build_cnn_feature_extractor()\ncnn_model.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\n# Trích xuất đặc trưng\nprint(\"\\nTrích xuất đặc trưng từ CNN...\")\ntrain_features = cnn_model.predict(X_train, batch_size=32)\ntest_features = cnn_model.predict(X_test, batch_size=32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-06T03:45:24.927786Z","iopub.execute_input":"2025-08-06T03:45:24.928106Z","iopub.status.idle":"2025-08-06T03:45:40.081067Z","shell.execute_reply.started":"2025-08-06T03:45:24.928081Z","shell.execute_reply":"2025-08-06T03:45:40.080305Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# trainding XGBoost\nprint(\"\\nHuấn luyện XGBoost...\")\nxgb_model = XGBClassifier(\n    n_estimators=150,\n    max_depth=7,\n    learning_rate=0.05,\n    objective='multi:softmax',\n    num_class=5,\n    tree_method='hist',  \n    device='cpu',       \n    random_state=42\n)\n\nxgb_model.fit(train_features, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-19T16:01:02.265005Z","iopub.execute_input":"2025-06-19T16:01:02.265771Z","iopub.status.idle":"2025-06-19T16:01:22.992471Z","shell.execute_reply.started":"2025-06-19T16:01:02.265747Z","shell.execute_reply":"2025-06-19T16:01:22.991814Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, classification_report, confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# forecast\ny_pred = xgb_model.predict(test_features)\n\n# Print report results\naccuracy = accuracy_score(y_test, y_pred)\nprint(f\"✅ Độ chính xác (Accuracy): {accuracy:.4f}\")\nprint(\"\\n📄 Classification Report:\")\nprint(classification_report(y_test, y_pred))\n\n# Confusion Matrix\ncm = confusion_matrix(y_test, y_pred)\nplt.figure(figsize=(8, 6))\nsns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\", xticklabels=range(5), yticklabels=range(5))\nplt.xlabel(\"Dự đoán\")\nplt.ylabel(\"Thực tế\")\nplt.title(\"🔍 Confusion Matrix - XGBoost\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-19T16:02:52.121380Z","iopub.execute_input":"2025-06-19T16:02:52.122007Z","iopub.status.idle":"2025-06-19T16:02:53.213170Z","shell.execute_reply.started":"2025-06-19T16:02:52.121985Z","shell.execute_reply":"2025-06-19T16:02:53.212393Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# save model\ncnn_model.save('/kaggle/working/cnn_feature_extractor.h5')\nxgb_model.save_model('/kaggle/working/xgboost_model.json')\n\nprint(\"\\nĐã lưu mô hình vào thư mục /kaggle/working/\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-19T16:03:05.042619Z","iopub.execute_input":"2025-06-19T16:03:05.043514Z","iopub.status.idle":"2025-06-19T16:03:05.227556Z","shell.execute_reply.started":"2025-06-19T16:03:05.043491Z","shell.execute_reply":"2025-06-19T16:03:05.226728Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Come see samples: https://colab.research.google.com/drive/1XM23__eWWgUbp0F8VxfFqeTwhovUacGS?usp=sharing\n# download 2 model về, up lên google drive\n# Connect google drive and google colab\n# copy path link model and paste \n# run GUI gradio","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}