{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":4104,"databundleVersionId":46661,"sourceType":"competition"},{"sourceId":7251,"sourceType":"datasetVersion","datasetId":2798},{"sourceId":7866129,"sourceType":"datasetVersion","datasetId":4614938},{"sourceId":7869237,"sourceType":"datasetVersion","datasetId":4617269},{"sourceId":11639089,"sourceType":"datasetVersion","datasetId":7303110},{"sourceId":11664461,"sourceType":"datasetVersion","datasetId":7320498}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Overview\nThe goal is to make a nice retinopathy model by using a pretrained inception v3 as a base and retraining some modified final layers with attention\n\nThis can be massively improved with \n* high-resolution images\n* better data sampling\n* ensuring there is no leaking between training and validation sets, ```sample(replace = True)``` is real dangerous\n* better target variable (age) normalization\n* pretrained models\n* attention/related techniques to focus on areas","metadata":{"_uuid":"34b6997bb115a11f47a7f54ce9d9052791b2e707","_cell_guid":"c67e806c-e0e6-415b-8cf0-ed92eba5ed37"}},{"cell_type":"code","source":"!pip install tensorflow==2.12.0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T10:39:50.371767Z","iopub.execute_input":"2025-05-03T10:39:50.372072Z","iopub.status.idle":"2025-05-03T10:39:54.784753Z","shell.execute_reply.started":"2025-05-03T10:39:50.372022Z","shell.execute_reply":"2025-05-03T10:39:54.784097Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt # showing and rendering figures\n# io related\nfrom skimage.io import imread\nimport os\nfrom glob import glob\n# not needed in Kaggle, but required in Jupyter\n%matplotlib inline \n\nimport tensorflow as tf\nprint(tf.__version__)\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"725d378daf5f836d4885d67240fc7955f113309d","_cell_guid":"c3cc4285-bfa4-4612-ac5f-13d10678c09a","trusted":true,"execution":{"iopub.status.busy":"2025-05-03T10:40:34.308524Z","iopub.execute_input":"2025-05-03T10:40:34.309134Z","iopub.status.idle":"2025-05-03T10:40:48.768008Z","shell.execute_reply.started":"2025-05-03T10:40:34.309112Z","shell.execute_reply":"2025-05-03T10:40:48.767376Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\n    Examine the distribution of eye and severity\n'''\nimport zipfile\nimport os\nimport pandas as pd\nfrom tensorflow.keras.utils import to_categorical\n\nzip_path = '/kaggle/input/diabetic-retinopathy-detection/trainLabels.csv.zip'\nextract_dir = '/kaggle/working'\n\nwith zipfile.ZipFile(zip_path, 'r') as zip_ref:\n    zip_ref.extractall(extract_dir)\n\n# B2: Đặt đường dẫn tới ảnh\nbase_image_dir = '/kaggle/input/diabetic-retinopathy-train-unzipped/train'\n\n# B3: Đọc file CSV đã giải nén\ncsv_path = os.path.join(extract_dir, 'trainLabels.csv')\nretina_df = pd.read_csv(csv_path)\n\n# B4: Xử lý dataframe\nretina_df['level'] = retina_df['level'].round().astype(int)\nretina_df['PatientId'] = retina_df['image'].map(lambda x: x.split('_')[0])\nretina_df['path'] = retina_df['image'].map(lambda x: os.path.join(base_image_dir, f'{x}.jpeg'))\nretina_df['exists'] = retina_df['path'].map(os.path.exists)\nprint(retina_df['exists'].sum(), 'images found of', retina_df.shape[0], 'total')\n\nretina_df['eye'] = retina_df['image'].map(lambda x: 1 if x.split('_')[-1] == 'left' else 0)\n\n# B5: Lọc ảnh tồn tại và loại bỏ dữ liệu thiếu\nretina_df.dropna(inplace=True)\nretina_df = retina_df[retina_df['exists']]","metadata":{"_uuid":"346da81db6ee7a34af8da8af245b42e681f2ba48","_cell_guid":"c4b38df6-ffa1-4847-b605-511e72b68231","trusted":true,"execution":{"iopub.status.busy":"2025-05-03T10:40:52.36971Z","iopub.execute_input":"2025-05-03T10:40:52.369988Z","iopub.status.idle":"2025-05-03T10:42:42.315003Z","shell.execute_reply.started":"2025-05-03T10:40:52.369968Z","shell.execute_reply":"2025-05-03T10:42:42.314341Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"retina_df[['level', 'eye']].hist(figsize = (10, 5))","metadata":{"_uuid":"60a8111c4093ca6f69d27a4499442ba7dd750839","_cell_guid":"5c8bd288-8261-4cbe-a954-e62ac795cc3e","trusted":true,"execution":{"iopub.status.busy":"2025-05-03T10:42:47.480725Z","iopub.execute_input":"2025-05-03T10:42:47.481368Z","iopub.status.idle":"2025-05-03T10:42:47.961814Z","shell.execute_reply.started":"2025-05-03T10:42:47.481334Z","shell.execute_reply":"2025-05-03T10:42:47.96112Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\n    Split Data into Training and Validation\n'''\n# SPLIT RANDOM\nfrom sklearn.model_selection import train_test_split\nrr_df = retina_df[['PatientId', 'level']].drop_duplicates()\ntrain_ids, valid_ids = train_test_split(rr_df['PatientId'], \n                                   test_size = 0.25, \n                                   random_state = 2018,\n                                   stratify = rr_df['level'])\ntrain_df = retina_df[retina_df['PatientId'].isin(train_ids)]\nvalid_df = retina_df[retina_df['PatientId'].isin(valid_ids)]\nprint('train', train_df.shape[0], 'validation', valid_df.shape[0])","metadata":{"_uuid":"a48b300ca4d37a6e8b39f82e3c172739635e4baa","_cell_guid":"1192c6b3-a940-4fa0-a498-d7e0d400a796","trusted":true,"execution":{"iopub.status.busy":"2025-05-03T10:42:51.25078Z","iopub.execute_input":"2025-05-03T10:42:51.251562Z","iopub.status.idle":"2025-05-03T10:42:51.697836Z","shell.execute_reply.started":"2025-05-03T10:42:51.251536Z","shell.execute_reply":"2025-05-03T10:42:51.697198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.utils import class_weight\nfrom tensorflow.keras.applications import EfficientNetB4\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, Dense, GlobalAveragePooling2D\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\n\n# ======= Image Parameters =======\nIMG_SIZE = 380  # EfficientNetB4 yêu cầu ảnh kích thước lớn\nBATCH_SIZE = 16\n\n# ======= Image Data Generators =======\ntrain_datagen = ImageDataGenerator(rescale=1./255)\n\n\nvalid_datagen = ImageDataGenerator(rescale=1./255)\n\ntrain_generator = train_datagen.flow_from_dataframe(\n    dataframe=train_df,\n    x_col='path',\n    y_col='level',\n    target_size=(IMG_SIZE, IMG_SIZE),\n    batch_size=BATCH_SIZE,\n    class_mode='raw'\n)\n\nvalid_generator = valid_datagen.flow_from_dataframe(\n    dataframe=valid_df,\n    x_col='path',\n    y_col='level',\n    target_size=(IMG_SIZE, IMG_SIZE),\n    batch_size=BATCH_SIZE,\n    class_mode='raw'\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T10:42:57.253994Z","iopub.execute_input":"2025-05-03T10:42:57.254731Z","iopub.status.idle":"2025-05-03T10:43:13.146833Z","shell.execute_reply.started":"2025-05-03T10:42:57.254706Z","shell.execute_reply":"2025-05-03T10:43:13.146247Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train_df = raw_train_df.groupby(['level', 'eye']).apply(lambda x: x.sample(min_count, replace = True)\n#                                                       ).reset_index(drop = True)\n# print('New Data Size:', train_df.shape[0], 'Old Size:', raw_train_df.shape[0])\n# train_df[['level', 'eye']].hist(figsize = (10, 5))","metadata":{"_uuid":"ba7befa238b8c11f9672e3539ac58f3da6955bd9","_cell_guid":"7a130199-fbf6-4c60-95f5-0797b2f3eaf1","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Lấy một batch từ generator\nimages, labels = next(valid_generator)\n\n# In 6 ảnh đầu tiên\nplt.figure(figsize=(15, 6))\nfor i in range(6):\n    ax = plt.subplot(2, 3, i + 1)\n    plt.imshow(images[i])\n    plt.title(f\"Label: {int(labels[i])}\")\n    plt.axis(\"off\")\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"_uuid":"6810407e25b887dd8b352f1e46fb3faceaa58ab7","execution":{"iopub.status.busy":"2025-05-03T10:43:50.847782Z","iopub.execute_input":"2025-05-03T10:43:50.848458Z","iopub.status.idle":"2025-05-03T10:43:52.544481Z","shell.execute_reply.started":"2025-05-03T10:43:50.848432Z","shell.execute_reply":"2025-05-03T10:43:52.543746Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ======= Xây dựng mô hình EfficientNetB3 =======\n# from tensorflow.keras.applications import EfficientNetB3\ninput_tensor = Input(shape=(IMG_SIZE, IMG_SIZE, 3))\nbase_model = EfficientNetB4(include_top=False, weights='imagenet', input_tensor=input_tensor)\n\nx = GlobalAveragePooling2D()(base_model.output)\noutput = Dense(5, activation='softmax')(x)\n\nmodel = Model(inputs=base_model.input, outputs=output)\n\nmodel.compile(\n    optimizer=Adam(learning_rate=1e-4),\n    loss='sparse_categorical_crossentropy',\n    metrics=['accuracy']\n)\n\ncallbacks = [\n    EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True, verbose=1),\n    ReduceLROnPlateau(monitor='val_loss', factor=0.5, patience=3, verbose=1)\n]\n\nhistory = model.fit(\n    train_generator,\n    validation_data=valid_generator,\n    epochs=30,\n    callbacks=callbacks\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T10:44:09.958601Z","iopub.execute_input":"2025-05-03T10:44:09.959288Z","iopub.status.idle":"2025-05-03T15:37:45.001216Z","shell.execute_reply.started":"2025-05-03T10:44:09.959265Z","shell.execute_reply":"2025-05-03T15:37:44.999142Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Lưu model\nmodel.save('/kaggle/working/my_model.h5')\n\n# Kiểm tra thư mục hiện tại\nimport os\nprint(\"Model saved to:\", os.path.join(os.getcwd(), 'my_model.h5'))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T16:23:16.327636Z","iopub.execute_input":"2025-05-03T16:23:16.328394Z","iopub.status.idle":"2025-05-03T16:23:17.832228Z","shell.execute_reply.started":"2025-05-03T16:23:16.328344Z","shell.execute_reply":"2025-05-03T16:23:17.831578Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.metrics import accuracy_score\n\n# Giả sử bạn đã load model từ trước\n# Nếu muốn thử trên một tập test khác không cần train thì mở comment cái dòng load model \n# from tensorflow.keras.models import load_model\n# model = load_model('/kaggle/input/my-answer/my_model.h5')\n\n# -------- Bước 1: Đọc file submission chứa nhãn thật --------\nsub_path = '/kaggle/input/solution/retinopathy_solution.csv'\nsubmission_df = pd.read_csv(sub_path)\n\n# -------- Bước 2: Tạo đường dẫn đầy đủ đến ảnh --------\nbase_image_dir = '/kaggle/input/diabetic-retinopathy-test-unzipped/test'\nsubmission_df['path'] = submission_df['image'].map(lambda x: os.path.join(base_image_dir, f'{x}.jpeg'))\n\n# -------- Bước 3: Kiểm tra ảnh tồn tại --------\nsubmission_df['exists'] = submission_df['path'].map(os.path.exists)\nprint(submission_df['exists'].sum(), 'images found of', submission_df.shape[0], 'total')\nsubmission_df = submission_df[submission_df['exists']]  # Lọc ảnh có tồn tại\n\n\n# -------- Bước 4: Đổi tên cột 'level' thành 'label' --------\nsubmission_df = submission_df.rename(columns={'level': 'label'})\n\n# -------- Bước 5: Tạo ImageDataGenerator (chuẩn hóa ảnh) --------\ntest_datagen = ImageDataGenerator(rescale=1./255)\n\nIMG_SIZE = 380\n\ntest_generator = test_datagen.flow_from_dataframe(\n    dataframe=submission_df,\n    x_col='path',\n    y_col='label',\n    target_size=(IMG_SIZE, IMG_SIZE),\n    batch_size=32,\n    class_mode='raw',\n    shuffle=False\n)\n\n# -------- Bước 6: Dự đoán --------\npredictions = model.predict(test_generator, verbose=1)\npredicted_classes = predictions.argmax(axis=1)\n\n# -------- Bước 7: Tính accuracy --------\ntrue_labels = submission_df['label'].values\naccuracy = accuracy_score(true_labels, predicted_classes)\nprint(f\"✅ Test Accuracy: {accuracy:.4f}\")\n\n# -------- Bước 8: Ghi kết quả ra file submission.csv --------\nsubmission_df['predicted'] = predicted_classes\nsubmission_df[['image', 'predicted']].to_csv('submission.csv', index=False)\nprint(\"📁 submission.csv đã được tạo.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T16:50:01.612715Z","iopub.execute_input":"2025-05-03T16:50:01.613004Z","iopub.status.idle":"2025-05-03T17:48:43.790335Z","shell.execute_reply.started":"2025-05-03T16:50:01.612982Z","shell.execute_reply":"2025-05-03T17:48:43.789677Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Xem xem level nào bị dự đoán sai nhiều nhất \nimport pandas as pd\nanswer_df = pd.read_csv('/kaggle/input/solution/retinopathy_solution.csv')\npredicted_df = pd.read_csv('/kaggle/input/my-answer/submission.csv')\ncompare_df = answer_df.merge(predicted_df, on='image')\n\ncompare_df['is_wrong'] = compare_df['predicted'] != compare_df['level']\n\nimages_by_class = compare_df.groupby('level').size()\nprint(images_by_class)\n\n# Tổng số ảnh đáng lẽ phải thuộc label này nhưng model lại gán cho nó label khác (VD: số ảnh thuộc lớp 0 bị nhầm thành thuộc cái khác)\nerrors_by_class = compare_df[compare_df['is_wrong']].groupby('level').size()\nprint(errors_by_class)\n\n# Tổng số lần model dự đoán mỗi label, nhưng bị sai (tức là nhãn đó là sai!) (VD: số ảnh bị nhầm là thuộc lớp 0 nhưng thật ra không phải)\nmistaken_as = compare_df[compare_df['is_wrong']].groupby('predicted').size()\nprint(mistaken_as)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T18:41:46.91498Z","iopub.execute_input":"2025-05-03T18:41:46.915693Z","iopub.status.idle":"2025-05-03T18:41:47.031986Z","shell.execute_reply.started":"2025-05-03T18:41:46.915668Z","shell.execute_reply":"2025-05-03T18:41:47.031389Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Đảm bảo lỗi và tổng có cùng index (nếu thiếu lớp nào thì điền 0)\nall_labels = sorted(compare_df['level'].unique())\nimages_by_class = compare_df.groupby('level').size().reindex(all_labels, fill_value=0)\n\n# Tạo thêm một cột để đếm số lượng ảnh bị nhầm thành label mà đáng lẽ phải thuộc class này (level - true answer)\nerrors_by_class = compare_df[compare_df['is_wrong']].groupby('level').size().reindex(all_labels, fill_value=0)\n\n# Tạo thêm một DataFrame để đếm số ảnh có thể đã thuộc class khác nhưng bị nhầm thành class hiện tại \nmistaken_from_label = compare_df[compare_df['is_wrong']].groupby('predicted').size().reindex(all_labels, fill_value=0)\n\n\nx = range(len(all_labels))\nbar_width = 0.25  # Tăng chiều rộng\n# Cập nhật phần biểu đồ với dữ liệu mới\nplt.bar([i - bar_width/2 for i in x], images_by_class, width=bar_width, label='Tổng ảnh', color='skyblue')\nplt.bar([i + bar_width/2 for i in x], errors_by_class, width=bar_width, label='Dự đoán sai', color='salmon')\nplt.bar([i + 1.5*bar_width for i in x], mistaken_from_label, width=bar_width, label='Dự đoán nhầm thành label này', color='orange')\n\n# Cấu hình hiển thị\nplt.xticks(x, all_labels)\nplt.xlabel('Label')\nplt.ylabel('Số lượng ảnh')\nplt.title('Số ảnh và số ảnh dự đoán sai theo từng lớp và bị nhầm thành label khác')\nplt.legend()\nplt.tight_layout()\nplt.show()\n\n# Vị trí các nhãn\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T18:43:23.212109Z","iopub.execute_input":"2025-05-03T18:43:23.212657Z","iopub.status.idle":"2025-05-03T18:43:23.410058Z","shell.execute_reply.started":"2025-05-03T18:43:23.212636Z","shell.execute_reply":"2025-05-03T18:43:23.40931Z"}},"outputs":[],"execution_count":null}]}