{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport tensorflow_addons as tfa\nimport os\nimport tensorflow as tf\nimport os, shutil, pandas as pd\nimport cv2\nfrom keras import backend as K\nfrom matplotlib import pyplot as plt\nimport numpy as np\nimport tensorflow as tf\nimport keras\nimport math, re\nimport tensorflow as tf\nimport numpy as np\nfrom tensorflow import keras\nfrom functools import partial\nfrom sklearn.model_selection import train_test_split\nimport tensorflow_hub as hub\nfrom keras.applications.resnet import ResNet50\nfrom sklearn.metrics import confusion_matrix\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport numpy as np\nimport pandas as pd\nfrom tensorflow.keras.layers import Dense, Dropout\nfrom tensorflow.keras.applications import EfficientNetB4\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.optimizers import Adam\nfrom keras.callbacks import EarlyStopping, ModelCheckpoint\nfrom keras.models import load_model\nimport tensorflow as tf\nfrom PIL import Image\nimport os\nimport matplotlib.pyplot as plt\n# detect and init the TPU\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-08T11:18:48.235719Z","iopub.execute_input":"2023-06-08T11:18:48.236109Z","iopub.status.idle":"2023-06-08T11:18:48.246865Z","shell.execute_reply.started":"2023-06-08T11:18:48.236080Z","shell.execute_reply":"2023-06-08T11:18:48.245725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# 讀取csv文件，這裡假設文件第一行為標題\ndf = pd.read_csv('../input/cassava-leaf-disease-classification/train.csv')\ngeneral_path = '../input/cassava-leaf-disease-classification/'\n\nOUTPUT_DIR = './'\nif not os.path.exists(OUTPUT_DIR):\n    os.makedirs(OUTPUT_DIR)\n    \n# 指定要保存圖片的目標資料夾\ntarget_folder = \"dataset\"\npath = \"../input/cassava-leaf-disease-classification/train_images\"\n\n# 創建目標資料夾，如果不存在\nif not os.path.exists(target_folder):\n    os.mkdir(target_folder)\n\n# 將圖像根據標籤分類\nfor label in df['label'].unique():\n    # 創建標籤對應的資料夾，如果不存在\n    label_folder = os.path.join(target_folder, str(label))\n    if not os.path.exists(label_folder):\n        os.mkdir(label_folder)\n\n    # 選擇所有標籤為label的行\n    mask = df['label'] == label\n    rows = df.loc[mask]\n    \n    # 遍歷這些行，將圖像移動到對應的資料夾中\n    for _, row in rows.iterrows():\n        image_name = row['image_id']\n        src_path = os.path.join(path, image_name)\n        dst_path = os.path.join(label_folder, image_name)\n        shutil.copy(src_path, dst_path)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-08T11:18:48.248977Z","iopub.execute_input":"2023-06-08T11:18:48.249355Z","iopub.status.idle":"2023-06-08T11:19:51.617057Z","shell.execute_reply.started":"2023-06-08T11:18:48.249315Z","shell.execute_reply":"2023-06-08T11:19:51.616116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def img_train_dataset(directory, batch_size=32, image_size=(224, 224), validation_split=0.1, seed=123):\n    '''\n    to build train dataset,  label_mode is categorical, color_mode is rgb\n    '''\n\n    train_dataset = tf.keras.utils.image_dataset_from_directory(\n        directory,\n        labels=\"inferred\",                  # {None, list/tuple of integer labels}\n        label_mode=\"categorical\",           # {int, binary, None}\n        class_names=None,                   # Only valid if \"labels\" is \"inferred\", list, to control order\n        color_mode=\"rgb\",                   # {\"grayscale\", \"rgba\"}\n        batch_size=batch_size,  \n        image_size=image_size,  \n        shuffle=True,   \n        seed=seed,                          # Optional random seed for shuffling and transformations.\n        validation_split=validation_split,\n        subset=\"training\",                # Only valid if validation_split is not None, One of \"training\" or \"validation\"\n        interpolation=\"bilinear\",           # {nearest, bicubic, area, lanczos3, lanczos5, gaussian, mitchellcubic}\n        crop_to_aspect_ratio=False          # 使圖像不失真\n    )\n    if validation_split != None:\n        valid_dataset = tf.keras.utils.image_dataset_from_directory(\n            directory,\n            labels=\"inferred\",                  # {None, list/tuple of integer labels}\n            label_mode=\"categorical\",           # {int, binary, None}\n            class_names=None,                   # Only valid if \"labels\" is \"inferred\", list, to control order\n            color_mode=\"rgb\",                   # {\"grayscale\", \"rgba\"}\n            batch_size=batch_size,  \n            image_size=image_size,  \n            shuffle=True,   \n            seed=seed,                          # Optional random seed for shuffling and transformations.\n            validation_split=validation_split,\n            subset=\"validation\",                # Only valid if validation_split is not None, One of \"training\" or \"validation\"\n            interpolation=\"bilinear\",           # {nearest, bicubic, area, lanczos3, lanczos5, gaussian, mitchellcubic}\n            crop_to_aspect_ratio=False          # 使圖像不失真\n        )\n        return train_dataset, valid_dataset\n    else:\n        return train_dataset","metadata":{"execution":{"iopub.status.busy":"2023-06-08T11:19:51.619201Z","iopub.execute_input":"2023-06-08T11:19:51.619931Z","iopub.status.idle":"2023-06-08T11:19:51.629069Z","shell.execute_reply.started":"2023-06-08T11:19:51.619894Z","shell.execute_reply":"2023-06-08T11:19:51.628134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = '/kaggle/working/dataset'\n\ntrain_dataset, test_dataset = img_train_dataset(path, 16)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T11:19:51.630618Z","iopub.execute_input":"2023-06-08T11:19:51.631279Z","iopub.status.idle":"2023-06-08T11:19:53.850778Z","shell.execute_reply.started":"2023-06-08T11:19:51.631244Z","shell.execute_reply":"2023-06-08T11:19:53.849512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def dataset_to_x_y(dataset, is_categorical=True):\n    y_true = []\n    x_true = []\n    for x, y in dataset:\n        y_true.append(y)\n        x_true.append(x)\n    y_true = np.concatenate(y_true, axis=0)\n    x_true = np.concatenate(x_true, axis=0)\n    if is_categorical:\n        y_true = np.argmax(y_true, axis=-1)\n\n    return x_true, y_true\n\ndef data_visualization(dataset, class_names=None):\n    x_true, y_true = dataset_to_x_y(dataset)\n    print(x_true.shape, y_true.shape)\n    data_y = pd.DataFrame({\"label\": y_true})\n    # print(y_true.shape, x_true.shape)\n    if class_names == None:\n        class_names = dataset.class_names\n    # print(class_names)\n\n    # 設定 視覺化風格\n    plt.style.use(\"tableau-colorblind10\")\n    # 以下程式碼從全域性設定字型為SimHei（黑體），解決顯示中文問題【Windows】\n    plt.rcParams[\"font.sans-serif\"] = [\"SimHei\"]\n    # 解決中文字型下座標軸負數的負號顯示問題\n    plt.rcParams[\"axes.unicode_minus\"] = False\n\n    # plot the num of every class\n    Label = data_y.groupby(['label']).size().to_list()\n    plt.grid(zorder=0)\n    plt.barh(class_names, Label)\n    # plt.show()\n    plt.clf()\n\n    # plot top 25 images in dataset\n    plt.figure(figsize=(10,10))\n    for i in range(25):\n        plt.subplot(5,5,i+1)\n        plt.xticks([])\n        plt.yticks([])\n        plt.imshow(x_true[i].astype(\"uint8\"))\n        plt.xlabel(class_names[y_true[i]])\n    # plt.show()\n    plt.clf()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T11:19:53.853380Z","iopub.execute_input":"2023-06-08T11:19:53.853756Z","iopub.status.idle":"2023-06-08T11:19:53.864669Z","shell.execute_reply.started":"2023-06-08T11:19:53.853724Z","shell.execute_reply":"2023-06-08T11:19:53.863326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_names = train_dataset.class_names # 類別名稱 \n#data_visualization(train_dataset) # 資料視覺化","metadata":{"execution":{"iopub.status.busy":"2023-06-08T11:19:53.865900Z","iopub.execute_input":"2023-06-08T11:19:53.866216Z","iopub.status.idle":"2023-06-08T11:19:53.884078Z","shell.execute_reply.started":"2023-06-08T11:19:53.866182Z","shell.execute_reply":"2023-06-08T11:19:53.882710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"class_names = train_dataset.class_names # 類別名稱\ndata_visualization(train_dataset) # 資料視覺化","metadata":{}},{"cell_type":"code","source":"num_samples = len(train_dataset)\nval_size = int(num_samples * 0.1)\ntrain_dataset = train_dataset.skip(val_size)\nvalid_dataset = train_dataset.take(val_size)\nprint(\"Train dataset size:\", train_dataset.cardinality().numpy())\nprint(\"Validation dataset size:\", valid_dataset.cardinality().numpy())","metadata":{"execution":{"iopub.status.busy":"2023-06-08T11:19:53.885799Z","iopub.execute_input":"2023-06-08T11:19:53.886152Z","iopub.status.idle":"2023-06-08T11:19:53.911608Z","shell.execute_reply.started":"2023-06-08T11:19:53.886116Z","shell.execute_reply":"2023-06-08T11:19:53.910684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"HEIGHT = 224\nWITH = 224\nSEED = 123\ndef data_augmentation():\n    data_aug = tf.keras.Sequential(\n        [\n            # tf.keras.layers.RandomCrop(height=HEIGHT, width=WITH, seed=SEED),\n            tf.keras.layers.RandomFlip(\n                mode=\"horizontal\",              # {\"vertical\", \"horizontal_and_vertical\"}\n                seed=SEED),\n            tf.keras.layers.RandomTranslation(\n                height_factor=0.2,  \n                width_factor=0.2, \n                fill_mode=\"reflect\",            # {\"constant\", \"wrap\", \"nearest\"}\n                interpolation=\"bilinear\",       # {\"nearest\"}\n                seed=SEED,\n                fill_value=0.0), \n            tf.keras.layers.RandomRotation(\n                factor=0.2,\n                fill_mode=\"reflect\",            # {\"constant\", \"wrap\", \"nearest\"}\n                interpolation=\"bilinear\",       # {\"nearest\"}\n                seed=SEED,\n                fill_value=0.0),\n            tf.keras.layers.RandomZoom(\n                height_factor=0.2,\n                width_factor=None,\n                fill_mode=\"reflect\",            # {\"constant\", \"wrap\", \"nearest\"}\n                interpolation=\"bilinear\",       # {\"nearest\"}\n                seed=SEED,\n                fill_value=0.0),\n            # tf.keras.layers.RandomHeight(\n            #     factor=0.2,\n            #     interpolation=\"bilinear\",       # {\"nearest\", \"bicubic\", \"area\", \"lanczos3\", \"lanczos5\", \"gaussian\", \"mitchellcubic\"}\n            #     seed=SEED),\n            # tf.keras.layers.RandomWidth(\n            #     factor=0.2,\n            #     interpolation=\"bilinear\",       # {\"nearest\", \"bicubic\", \"area\", \"lanczos3\", \"lanczos5\", \"gaussian\", \"mitchellcubic\"}\n            #     seed=SEED),\n            tf.keras.layers.RandomContrast(\n                factor=0.2,                     # (x - mean) * factor + mean\n                seed=SEED),\n        ]\n    )\n    return data_aug","metadata":{"execution":{"iopub.status.busy":"2023-06-08T11:19:53.912886Z","iopub.execute_input":"2023-06-08T11:19:53.913189Z","iopub.status.idle":"2023-06-08T11:19:53.923034Z","shell.execute_reply.started":"2023-06-08T11:19:53.913163Z","shell.execute_reply":"2023-06-08T11:19:53.921618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_aug = data_augmentation()\naugmented_dataset = train_dataset.repeat(2).map(lambda x, y: (data_aug(x), y))\ntrain_dataset = train_dataset.concatenate(augmented_dataset)\n# train_dataset = train_dataset.map(lambda x, y: (tf.squeeze(x, axis=0), tf.squeeze(y, axis=0)))\n# train_dataset = train_dataset.shuffle(1000).batch(32)\nprint(\"augmentation train dataset size:\", train_dataset.cardinality().numpy())","metadata":{"execution":{"iopub.status.busy":"2023-06-08T11:19:53.924549Z","iopub.execute_input":"2023-06-08T11:19:53.925095Z","iopub.status.idle":"2023-06-08T11:19:54.685270Z","shell.execute_reply.started":"2023-06-08T11:19:53.925047Z","shell.execute_reply":"2023-06-08T11:19:54.683820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"efficientweight = '/kaggle/input/tfkeras-efficientnetsv2/1k_notop/efficientnetv2-s_notop.h5'\ndef efficientnet_model(size=(224, 224, 3), classes=10):\n    model = tf.keras.applications.efficientnet_v2.EfficientNetV2S(weights=efficientweight, include_top=False, input_shape=size)\n    x = tf.keras.layers.GlobalAveragePooling2D()(model.output)\n    x = tf.keras.layers.Dense(256, activation='relu')(x)\n    x = tf.keras.layers.Dropout(0.5)(x)\n    output = tf.keras.layers.Dense(classes, activation='softmax')(x)\n    model = tf.keras.models.Model(inputs=model.input, outputs=output)\n    return model\nmodel = efficientnet_model(classes=5)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T11:19:54.686611Z","iopub.execute_input":"2023-06-08T11:19:54.687054Z","iopub.status.idle":"2023-06-08T11:20:03.116191Z","shell.execute_reply.started":"2023-06-08T11:19:54.687008Z","shell.execute_reply":"2023-06-08T11:20:03.114916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_model(model, train_dataset, valid_dataset, epochs=10):\n    try:\n        os.mkdir(save_path)\n    except:\n        pass\n    reduce_lr = tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.5,\n                            patience=1, mode='auto', verbose=0, cooldown=0,\n                            min_lr=1e-7)\n\n    model.compile(\n        optimizer=tf.optimizers.Adam(learning_rate=0.001),\n        loss=tfa.losses.SigmoidFocalCrossEntropy(alpha=0.25, gamma=2),\n        metrics=[\n            tf.keras.metrics.CategoricalAccuracy(name=\"accuracy\"),]\n    )\n    # model.load_weights(save_path + \"/weights.01-1.53.hdf5\")\n    history = model.fit(\n            train_dataset,\n            epochs=epochs,\n            callbacks=[\n                reduce_lr\n                ],\n            validation_data=valid_dataset\n    )\n\n    plt.plot(history.history['accuracy'])\n    plt.plot(history.history['val_accuracy'])\n    plt.title('Model accuracy')\n    plt.ylabel('Accuracy')\n    plt.xlabel('Epoch')\n    plt.legend(['Train', 'Test'], loc='upper left')\n    plt.show()\n    plt.clf()\n\n    # 绘制训练 & 验证的损失值\n    plt.plot(history.history['loss'])\n    plt.plot(history.history['val_loss'])\n    plt.title('Model loss')\n    plt.ylabel('Loss')\n    plt.xlabel('Epoch')\n    plt.legend(['Train', 'Test'], loc='upper left')\n    plt.show()\n    plt.clf()\n\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-06-08T11:20:03.120882Z","iopub.execute_input":"2023-06-08T11:20:03.121499Z","iopub.status.idle":"2023-06-08T11:20:03.134183Z","shell.execute_reply.started":"2023-06-08T11:20:03.121455Z","shell.execute_reply":"2023-06-08T11:20:03.132786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = train_model(model, train_dataset, valid_dataset, epochs=10)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T11:20:03.135569Z","iopub.execute_input":"2023-06-08T11:20:03.136579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(columns=['image_id','label'])\nfor image_name in os.listdir(general_path + 'test_images'):\n    image_path = os.path.join(general_path + 'test_images', image_name)\n    image = tf.keras.preprocessing.image.load_img(image_path)\n    resized_image = image.resize((WITH, HEIGHT))\n    numpied_image = np.expand_dims(resized_image, 0)\n    tensored_image = tf.cast(numpied_image, tf.float32)\n    y_pred = model.predict(tensored_image)\n    y_pred = np.argmax(y_pred, axis=-1)\n    submission = submission.append(pd.DataFrame({'image_id': image_name,\n                                                 'label': y_pred }))\nsubmission.to_csv('submission.csv', index=False)\nfolder = '/kaggle/working/dataset'\ntry:\n    shutil.rmtree(folder)\n    print(\"資料夾已刪除\")\nexcept OSError as e:\n    print(f\"刪除失敗: {e}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-06-08T15:07:14.465556Z","iopub.execute_input":"2023-06-08T15:07:14.466094Z","iopub.status.idle":"2023-06-08T15:07:14.925560Z","shell.execute_reply.started":"2023-06-08T15:07:14.466063Z","shell.execute_reply":"2023-06-08T15:07:14.923168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}