{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":4104,"databundleVersionId":46661,"sourceType":"competition"},{"sourceId":265751,"sourceType":"datasetVersion","datasetId":110097}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np  \nimport pandas as pd \nimport os \n\ninput_dir = '/kaggle/input'\nfile_paths = []\n\nfor dirname, _, filenames in os.walk(input_dir):\n    for filename in filenames:\n        file_path = os.path.join(dirname, filename)\n        file_paths.append(file_path)\n\nprint(\"Total files found:\", len(file_paths))\nprint(\"files:\", file_paths[:])  \n\nexpected_files = ['trainLabels.csv.zip', 'sampleSubmission.csv.zip']\nfor ef in expected_files:\n    if not any(ef in path for path in file_paths):\n        print(f\"Warning: Expected file {ef} not found!\")\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2025-01-06T07:00:45.701755Z","iopub.execute_input":"2025-01-06T07:00:45.702075Z","iopub.status.idle":"2025-01-06T07:00:46.696364Z","shell.execute_reply.started":"2025-01-06T07:00:45.702035Z","shell.execute_reply":"2025-01-06T07:00:46.695429Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np  # 用於數值計算\nimport matplotlib.pyplot as plt  # 用於資料視覺化\nimport seaborn as sns  # 用於強化視覺化的繪圖風格\nimport tensorflow as tf  # TensorFlow 深度學習框架\nfrom glob import glob  # 用於查找文件路徑\nfrom skimage.io import imread  # 用於圖像讀取\n\nfrom tensorflow.keras.applications import EfficientNetB0  # 預訓練模型\nfrom tensorflow.keras.optimizers import Adam  # 優化器\nfrom tensorflow.keras.losses import SparseCategoricalCrossentropy  # 損失函數\nfrom tensorflow.keras.layers import Dense, Flatten, Dropout  # 常用層\nfrom tensorflow.keras.models import Model  # 模型\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, EarlyStopping  # 訓練回調\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator  # 圖像生成工具\nfrom tensorflow.keras.utils import to_categorical  # 類別處理工具\nfrom sklearn.metrics import classification_report, confusion_matrix  # 評估工具\nfrom sklearn.model_selection import train_test_split  # 資料分割工具\n\nimport warnings\nwarnings.filterwarnings('ignore', category=FutureWarning)  # 只忽略FutureWarning\n\nprint(\"All necessary modules have been successfully imported!\")\n","metadata":{"execution":{"iopub.status.busy":"2025-01-06T07:00:46.698744Z","iopub.execute_input":"2025-01-06T07:00:46.699530Z","iopub.status.idle":"2025-01-06T07:00:59.219351Z","shell.execute_reply.started":"2025-01-06T07:00:46.699497Z","shell.execute_reply":"2025-01-06T07:00:59.218475Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import itertools\ndef plot_confusion_matrix(cm, classes,\n                          normalize=False,\n                          title='Confusion matrix',\n                          cmap=plt.cm.Blues):\n    \"\"\"\n    This function prints and plots the confusion matrix.\n    Normalization can be applied by setting `normalize=True`.\n    \"\"\"\n    plt.figure(figsize = (6,6))\n    plt.imshow(cm, interpolation='nearest', cmap=cmap)\n    plt.title(title)\n    plt.colorbar()\n    tick_marks = np.arange(len(classes))\n    plt.xticks(tick_marks, classes, rotation=90)\n    plt.yticks(tick_marks, classes)\n    if normalize:\n        cm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n\n    thresh = cm.max() / 2.\n    cm = np.round(cm,2)\n    for i, j in itertools.product(range(cm.shape[0]), range(cm.shape[1])):\n        plt.text(j, i, cm[i, j],\n                 horizontalalignment=\"center\",\n                 color=\"white\" if cm[i, j] > thresh else \"black\")\n    plt.tight_layout()\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2025-01-06T07:00:59.220368Z","iopub.execute_input":"2025-01-06T07:00:59.220826Z","iopub.status.idle":"2025-01-06T07:00:59.229324Z","shell.execute_reply.started":"2025-01-06T07:00:59.220803Z","shell.execute_reply":"2025-01-06T07:00:59.228317Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Loading data\ninfo = pd.read_csv(\"../input/prepossessed-arrays-of-binary-data/1000_Binary Dataframe\")\ninfo = info.drop('Unnamed: 0', axis=1)\nBinary_90 = np.load('../input/prepossessed-arrays-of-binary-data/1000_Binary_images_data_90.npz')\nX_90 = Binary_90['a']\nBinary_128 = np.load('../input/prepossessed-arrays-of-binary-data/1000_Binary_images_data_128.npz')\nX_128 = Binary_128['a']\nBinary_264 = np.load('../input/prepossessed-arrays-of-binary-data/1000_Binary_images_data_264.npz')\nX_264 = Binary_264['a']\ny = info['level'].values\n\n# Reshape images\nX_90 = X_90.reshape(1000, 90, 90, 3)\nX_128 = X_128.reshape(1000, 128, 128, 3)\nX_264 = X_264.reshape(1000, 264, 264, 3)\n\n# Display images\nplt.title(\"90*90*3 Image\")\nplt.imshow(X_90[1])\nplt.show()\n\nplt.title(\"128*128*3 Image\")\nplt.imshow(X_128[1])\nplt.show()\n\nplt.title(\"264*264*3 Image\")\nplt.imshow(X_264[1])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-01-06T07:00:59.230564Z","iopub.execute_input":"2025-01-06T07:00:59.230890Z","iopub.status.idle":"2025-01-06T07:01:14.401562Z","shell.execute_reply.started":"2025-01-06T07:00:59.230843Z","shell.execute_reply":"2025-01-06T07:01:14.400674Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prepare data for training\nX = np.array(X_264)\nY = np.array(y)\nY = to_categorical(Y, 5)\nx_train, x_test1, y_train, y_test1 = train_test_split(X, Y, test_size=0.4, random_state=42)\nx_val, x_test, y_val, y_test = train_test_split(x_test1, y_test1, test_size=0.5, random_state=42)\n\nprint(f\"Training set size: {len(x_train)}\")\nprint(f\"Validation set size: {len(x_val)}\")\nprint(f\"Test set size: {len(x_test)}\")","metadata":{"execution":{"iopub.status.busy":"2025-01-06T07:01:14.402713Z","iopub.execute_input":"2025-01-06T07:01:14.402991Z","iopub.status.idle":"2025-01-06T07:01:15.662330Z","shell.execute_reply.started":"2025-01-06T07:01:14.402969Z","shell.execute_reply":"2025-01-06T07:01:15.661445Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Data augmentation\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=20,\n    zoom_range=0.15,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.15,\n    horizontal_flip=True,\n    fill_mode=\"nearest\"\n)\n\nval_datagen = ImageDataGenerator(rescale=1./255)","metadata":{"execution":{"iopub.status.busy":"2025-01-06T07:01:15.663333Z","iopub.execute_input":"2025-01-06T07:01:15.663591Z","iopub.status.idle":"2025-01-06T07:01:15.668382Z","shell.execute_reply.started":"2025-01-06T07:01:15.663571Z","shell.execute_reply":"2025-01-06T07:01:15.667474Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Callbacks\nreduce_lr = ReduceLROnPlateau(\n    monitor=\"val_loss\",\n    factor=0.1,\n    patience=2,\n    mode=\"auto\",\n    min_delta=0.0001,\n    cooldown=0,\n    min_lr=0.001\n)\n\nearly_stopping = EarlyStopping(\n    monitor='val_loss',\n    patience=5,\n    restore_best_weights=True\n)\n\ncallbacks = [reduce_lr, early_stopping]","metadata":{"execution":{"iopub.status.busy":"2025-01-06T07:01:15.670969Z","iopub.execute_input":"2025-01-06T07:01:15.671638Z","iopub.status.idle":"2025-01-06T07:01:15.684586Z","shell.execute_reply.started":"2025-01-06T07:01:15.671615Z","shell.execute_reply":"2025-01-06T07:01:15.683803Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Model with transfer learning (VGG16)\nfrom tensorflow.keras.applications import VGG16\nfrom tensorflow.keras.layers import Flatten, Dense, Dropout, Input\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.regularizers import l2\nfrom sklearn import metrics  \n# 使用 Functional API 來構建模型\ninput_shape = (264, 264, 3)\ninputs = Input(shape=input_shape)\n\n# 加載預訓練的 VGG16 模型，不包括頂部的全連接層\nbase_model = VGG16(weights='imagenet', include_top=False, input_tensor=inputs)\n\n# 凍結預訓練模型的所有層\nfor layer in base_model.layers:\n    layer.trainable = False\n\n# 添加自定義的全連接層\nx = Flatten()(base_model.output)\nx = Dense(256, activation='relu', kernel_regularizer=l2(0.01))(x)\nx = Dropout(0.5)(x)\noutputs = Dense(5, activation='softmax')(x)\n\n# 定義模型\nmodel = Model(inputs, outputs)\n\n# 編譯模型\nmodel.compile(optimizer=Adam(learning_rate=0.0001), loss='categorical_crossentropy', metrics=['accuracy', 'AUC'])\n\n# 顯示模型結構\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2025-01-06T07:01:15.685648Z","iopub.execute_input":"2025-01-06T07:01:15.686174Z","iopub.status.idle":"2025-01-06T07:01:17.262044Z","shell.execute_reply.started":"2025-01-06T07:01:15.686147Z","shell.execute_reply":"2025-01-06T07:01:17.261205Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the model\n\nhistory = model.fit(\n    train_datagen.flow(x_train, y_train, batch_size=8),\n    validation_data=val_datagen.flow(x_val, y_val),\n    epochs=20,\n    callbacks=callbacks,\n)\n\n# Evaluate the model\nevaluation = model.evaluate(x_test, y_test)\nprint(f'Test Accuracy: {evaluation[1]*100:.2f}%')\n\n# Predictions and report\ny_test_labels = np.argmax(y_test, axis=1)\ny_pred_labels = np.argmax(model.predict(x_test), axis=-1)\n\nprint(\"Performance Report:\")\nprint('Accuracy score:', metrics.accuracy_score(y_test_labels, y_pred_labels))\nprint('Precision score:', metrics.precision_score(y_test_labels, y_pred_labels, average='weighted'))\nprint('Recall score:', metrics.recall_score(y_test_labels, y_pred_labels, average='weighted'))\nprint('F1 Score:', metrics.f1_score(y_test_labels, y_pred_labels, average='weighted'))\nprint('Cohen Kappa Score:', metrics.cohen_kappa_score(y_test_labels, y_pred_labels))\nprint('\\t\\tClassification Report:\\n', classification_report(y_test_labels, y_pred_labels))","metadata":{"execution":{"iopub.status.busy":"2025-01-06T07:01:17.263094Z","iopub.execute_input":"2025-01-06T07:01:17.263330Z","iopub.status.idle":"2025-01-06T07:03:53.997871Z","shell.execute_reply.started":"2025-01-06T07:01:17.263310Z","shell.execute_reply":"2025-01-06T07:03:53.997011Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Confusion Matrix visualization\nconf_matrix = confusion_matrix(y_test_labels, y_pred_labels)\nplt.figure(figsize=(4, 3))\nsns.heatmap(conf_matrix, annot=True, fmt='d', cmap='Blues', xticklabels=range(5), yticklabels=range(5))\nplt.title('Confusion Matrix')\nplt.ylabel('True Label')\nplt.xlabel('Predicted Label')\nplt.show()\n\n# Extracting the metrics using correct keys\nauc = history.history['AUC']\nval_auc = history.history['val_AUC']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\n# Plotting the metrics\nepochs = range(len(auc))\nplt.figure(figsize=(18, 4.8))\n\n# Plotting Training and Validation AUC\nplt.subplot(1, 3, 1)\nplt.plot(epochs, auc, 'r', label='Training AUC')\nplt.plot(epochs, val_auc, 'b', label='Validation AUC')\nplt.ylim(0, 1)\nplt.title('Training and Validation AUC')\nplt.legend(loc=0)\n\n# Plotting Training and Validation Loss\nplt.subplot(1, 3, 2)\nplt.plot(epochs, loss, 'y-.', label='Training Loss')\nplt.plot(epochs, val_loss, 'g-.', label='Validation Loss')\nplt.title('Training and Validation Loss')\nplt.ylim(0, 2)\nplt.legend(loc=0)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-01-06T07:03:53.999048Z","iopub.execute_input":"2025-01-06T07:03:53.999311Z","iopub.status.idle":"2025-01-06T07:03:54.642049Z","shell.execute_reply.started":"2025-01-06T07:03:53.999289Z","shell.execute_reply":"2025-01-06T07:03:54.641241Z"},"trusted":true},"outputs":[],"execution_count":null}]}