{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%matplotlib inline\n#這是juoyter notebook的magic word˙\n\nimport matplotlib\nimport matplotlib.pyplot as plt\nfrom IPython import display","metadata":{"execution":{"iopub.status.busy":"2021-06-25T10:14:40.134370Z","iopub.execute_input":"2021-06-25T10:14:40.134678Z","iopub.status.idle":"2021-06-25T10:14:40.145054Z","shell.execute_reply.started":"2021-06-25T10:14:40.134591Z","shell.execute_reply":"2021-06-25T10:14:40.143849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install ipyplot\nimport os\n#判斷是否在jupyter notebook上\ndef is_in_ipython():\n    \"Is the code running in the ipython environment (jupyter including)\"\n    program_name = os.path.basename(os.getenv('_', ''))\n\n    if ('jupyter-notebook' in program_name or # jupyter-notebook\n        'ipython'          in program_name or # ipython\n        'jupyter' in program_name or  # jupyter\n        'JPY_PARENT_PID'   in os.environ):    # ipython-notebook\n        return True\n    else:\n        return False\n\n\n#判斷是否在colab上\ndef is_in_colab():\n    if not is_in_ipython(): return False\n    try:\n        from google import colab\n        return True\n    except: return False\n\n#判斷是否在kaggke_kernal上\ndef is_in_kaggle_kernal():\n    if 'kaggle' in os.environ['PYTHONPATH']:\n        return True\n    else:\n        return False\n\nif is_in_colab():\n    from google.colab import drive\n    drive.mount('/content/gdrive')","metadata":{"execution":{"iopub.status.busy":"2021-06-25T10:14:50.023332Z","iopub.execute_input":"2021-06-25T10:14:50.023760Z","iopub.status.idle":"2021-06-25T10:14:57.935710Z","shell.execute_reply.started":"2021-06-25T10:14:50.023632Z","shell.execute_reply":"2021-06-25T10:14:57.934779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.environ['TRIDENT_BACKEND'] = 'pytorch'\n\nif is_in_kaggle_kernal():\n    os.environ['TRIDENT_HOME'] = './trident'\n    \nelif is_in_colab():\n    os.environ['TRIDENT_HOME'] = '/content/gdrive/My Drive/trident'","metadata":{"execution":{"iopub.status.busy":"2021-06-25T10:14:57.938325Z","iopub.execute_input":"2021-06-25T10:14:57.938707Z","iopub.status.idle":"2021-06-25T10:14:57.944722Z","shell.execute_reply.started":"2021-06-25T10:14:57.938659Z","shell.execute_reply":"2021-06-25T10:14:57.942346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#為確保安裝最新版 \n!pip uninstall tridentx -y\n!pip install tridentx --upgrade","metadata":{"execution":{"iopub.status.busy":"2021-06-25T10:14:58.063530Z","iopub.execute_input":"2021-06-25T10:14:58.063831Z","iopub.status.idle":"2021-06-25T10:15:06.747906Z","shell.execute_reply.started":"2021-06-25T10:14:58.063806Z","shell.execute_reply":"2021-06-25T10:15:06.746941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import copy\nimport numpy as np\n#調用trident api\nimport trident as T\nfrom trident import *\nfrom trident.models import resnet,efficientnet","metadata":{"execution":{"iopub.status.busy":"2021-06-25T10:15:06.749670Z","iopub.execute_input":"2021-06-25T10:15:06.750043Z","iopub.status.idle":"2021-06-25T10:15:10.353613Z","shell.execute_reply.started":"2021-06-25T10:15:06.750003Z","shell.execute_reply":"2021-06-25T10:15:10.352531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport os\nimport PIL\nimport PIL.Image\nimport tensorflow as tf\nimport tensorflow_datasets as tfds\nimport pathlib\nimport cv2\nimport random\nfrom shutil import copyfile\nimport shutil\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2021-06-25T10:15:12.692306Z","iopub.execute_input":"2021-06-25T10:15:12.692668Z","iopub.status.idle":"2021-06-25T10:15:17.546188Z","shell.execute_reply.started":"2021-06-25T10:15:12.692615Z","shell.execute_reply":"2021-06-25T10:15:17.545359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"src_dataset = '../input/histopathologic-cancer-detection'\ntrain_data_dir = 'dataset/train'\ntest_data_dir = 'dataset/test'\nvalidation_split = 0.2\n\nbatch_size = 32\nimg_height = 96 #96 original size\nimg_width = 96","metadata":{"execution":{"iopub.status.busy":"2021-06-25T10:15:19.755415Z","iopub.execute_input":"2021-06-25T10:15:19.755776Z","iopub.status.idle":"2021-06-25T10:15:19.760425Z","shell.execute_reply.started":"2021-06-25T10:15:19.755741Z","shell.execute_reply":"2021-06-25T10:15:19.759607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess_dataset():\n    print('dataset re-structure...')\n    if not os.path.exists(train_data_dir):\n        os.makedirs(train_data_dir)\n            \n    for file in tqdm(os.listdir( os.path.join(src_dataset, 'train'))):\n        filename, file_extension = os.path.splitext(file)\n        if file_extension == '.tif':\n            file_path = os.path.join(src_dataset, 'train', file)\n            new_img_path = os.path.join(train_data_dir, filename+'.jpg')\n            cv2.imwrite(new_img_path, cv2.imread(file_path)  )\n\ndef get_combined(img, ccrop=(16,32,64,96)):\n    width = img.shape[1]\n    height = img.shape[0]\n    \n    cimgs = []\n    for c in ccrop:\n        x1 = int(width/2 - c/2)\n        y1 = int(height/2 - c/2)\n        x2 = x1 + c\n        y2 = y1 + c\n        center_img = img[y1:y2, x1:x2]\n        center_img = cv2.resize(center_img, (48,48))\n        cimgs.append(center_img)\n        \n    vis1 = np.concatenate((cimgs[0], cimgs[1]), axis=1)\n    vis2 = np.concatenate((cimgs[2], cimgs[3]), axis=1)\n    vis = np.concatenate((vis1, vis2), axis=0)\n    \n    return vis\n\ndef get_combined_B(img, ccrop=(32,32)):\n    width = img.shape[1]\n    height = img.shape[0]\n    x1 = int(width/2 - ccrop[0]/2)\n    y1 = int(height/2 - ccrop[1]/2)\n    x2 = x1 + ccrop[0]\n    y2 = y1 + ccrop[1]\n    center_img = cv2.resize(img[y1:y2, x1:x2], (48,96))\n    img = cv2.resize(img, (48,96))\n    vis = np.concatenate((center_img, img), axis=1)\n    return vis\n\ndef preprocess_image():\n    for classname in ['0', '1']:\n        print('image pre-processt, class' ,classname)\n        folder_path = os.path.join('dataset/train/', classname)\n        for file in tqdm(os.listdir(folder_path)):\n            filename, file_extension = os.path.splitext(file)\n            if file_extension == '.jpg':\n                file_path = os.path.join(folder_path, file)\n                img = get_combined( cv2.imread(file_path) )\n                cv2.imwrite( file_path, img )\n\ndef convert_test_ds():\n    src_folder = os.path.join(src_dataset, 'test')\n    folder = 'dataset/test'\n    for file in tqdm(os.listdir(src_folder)):\n            filename, file_extension = os.path.splitext(file)\n            if file_extension == '.tif':\n                src_file_path = os.path.join(src_folder, file)\n                file_path = os.path.join(folder, file)\n                img = cv2.imread(src_file_path)\n                img = get_combined(img, ccrop=(16,32,64,96))\n                cv2.imwrite( os.path.join(folder, filename+'.jpg'), img )\n                #os.remove(file_path)","metadata":{"execution":{"iopub.status.busy":"2021-06-25T10:16:16.669880Z","iopub.execute_input":"2021-06-25T10:16:16.670266Z","iopub.status.idle":"2021-06-25T10:16:16.696686Z","shell.execute_reply.started":"2021-06-25T10:16:16.670230Z","shell.execute_reply":"2021-06-25T10:16:16.694516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#for first time only\npreprocess_dataset()\npreprocess_image()\nconvert_test_ds()","metadata":{"execution":{"iopub.status.busy":"2021-06-25T10:16:20.265864Z","iopub.execute_input":"2021-06-25T10:16:20.266212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_dir = pathlib.Path(train_data_dir)\ntest_data_dir = pathlib.Path(test_data_dir)\n\nimage_count = len(list(train_data_dir.glob('0/*.jpg')))\nprint(\"0 class count of training dataset:\", image_count)\nimage_count = len(list(train_data_dir.glob('1/*.jpg')))\nprint(\"1 class count of training dataset:\", image_count)\n\ntrain_img_list = list(train_data_dir.glob('0/*.jpg'))\nprint(train_img_list[1])\nimg = cv2.imread( str(train_img_list[1]) )\nprint('images shape is', img.shape)","metadata":{"execution":{"iopub.status.busy":"2021-06-25T09:00:39.483394Z","iopub.execute_input":"2021-06-25T09:00:39.483715Z","iopub.status.idle":"2021-06-25T09:00:41.799537Z","shell.execute_reply.started":"2021-06-25T09:00:39.483683Z","shell.execute_reply":"2021-06-25T09:00:41.798694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\n#透過glob所全部train資料夾中所有可用圖片\nimgs=glob.glob( os.path.join(train_data_dir, '*.jpg') )\nprint(len(imgs))\n\n#ImageDatset(imgs,symbol='image')\nf=open( os.path.join(src_dataset, 'train_labels.csv'),'r',encoding='utf-8-sig')\nrows=f.readlines()\nprint(rows[:3])\nimage_path=[]\nlabels=[]\ntest_image_path=[]\ntest_labels=[]\n\nrows=rows[1:] #拿掉第一筆標頭\nrandom.shuffle(rows)#隨機洗牌\nfor row in rows:\n    cols=row.strip().split(',') #移除\\n然後逗號分割\n    if random.random()<=0.3:\n        test_image_path.append( os.path.join(train_data_dir,'.jpg'.format(cols[0])))\n        test_labels.append(int(cols[1]))\n    else:\n        image_path.append(os.path.join(train_data_dir,'.jpg'.format(cols[0])))\n        labels.append(int(cols[1]))\nprint(len(image_path))\nprint(len(labels))\nprint(len(test_image_path))\nprint(len(test_labels))","metadata":{"execution":{"iopub.status.busy":"2021-06-25T09:21:22.506747Z","iopub.execute_input":"2021-06-25T09:21:22.507125Z","iopub.status.idle":"2021-06-25T09:21:23.807415Z","shell.execute_reply.started":"2021-06-25T09:21:22.507081Z","shell.execute_reply":"2021-06-25T09:21:23.806033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#資料集\nds1=ImageDataset(image_path,symbol='image')\nds2=LabelDataset(labels,symbol='label')\n\nds1_t=ImageDataset(test_image_path,symbol='image')\nds2_t=LabelDataset(test_labels,symbol='label')\n\n#與Iterator構成data provider\ndata_provider=DataProvider(traindata=Iterator(data=ds1,label=ds2),testdata=Iterator(data=ds1_t,label=ds2_t))\n\n#設定DataProvider的預處理流程\ndata_provider.image_transform_funcs=[Normalize(127.5,127.5)]\n\n#即可完成設定，可以透過next()來確認數據是否正常拋出，以及是否有正確產生輸出數據的signature\nimg_data,label_data=data_provider.next()\nprint(data_provider.signature)\nprint(img_data.shape)\nprint(label_data.shape)\n\ndata_provider.image_transform_funcs=[\n    auto_level(),\n    RandomAdjustGamma(scale=(0.6,1.4)),#調整明暗\n    RandomAdjustHue(scale=(-0.5,0.5)),#調整色相\n    RandomAdjustSaturation(scale=(0.6,1.4)),#調整飽和度\n    SaltPepperNoise(0.05),#加入胡椒鹽噪音\n    RandomErasing(), #加入隨機擦去\n    Normalize(127.5,127.5)] #標準化\n\n#測試集數據不需要做數據增強\ndata_provider.testdata.data.image_transform_funcs=[\n    Normalize(127.5,127.5)] #標準化\n\nimg_data,label_data=data_provider.next()\nprint(img_data.shape)\nprint(label_data.shape)\ntest_img_data,test_label_data=data_provider.next_test()\nprint(test_img_data.shape)\nprint(test_label_data.shape)\ndata_provider.preview_images()","metadata":{"execution":{"iopub.status.busy":"2021-06-25T09:22:10.220007Z","iopub.execute_input":"2021-06-25T09:22:10.220352Z","iopub.status.idle":"2021-06-25T09:22:10.268874Z","shell.execute_reply.started":"2021-06-25T09:22:10.220321Z","shell.execute_reply":"2021-06-25T09:22:10.266971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(val_batches)","metadata":{"execution":{"iopub.status.busy":"2021-06-25T09:02:02.294526Z","iopub.execute_input":"2021-06-25T09:02:02.294859Z","iopub.status.idle":"2021-06-25T09:02:02.299836Z","shell.execute_reply.started":"2021-06-25T09:02:02.294830Z","shell.execute_reply":"2021-06-25T09:02:02.298833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTOTUNE = tf.data.AUTOTUNE\n\ntrain_ds = train_ds.prefetch(buffer_size=AUTOTUNE)\nval_ds = val_ds.prefetch(buffer_size=AUTOTUNE)\ntest_ds = test_ds.prefetch(buffer_size=AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2021-06-25T09:02:14.500790Z","iopub.execute_input":"2021-06-25T09:02:14.501146Z","iopub.status.idle":"2021-06-25T09:02:14.507612Z","shell.execute_reply.started":"2021-06-25T09:02:14.501116Z","shell.execute_reply":"2021-06-25T09:02:14.506424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Augmentation\n\ndata_augmentation = tf.keras.Sequential([\n  tf.keras.layers.experimental.preprocessing.RandomFlip('horizontal'),\n  tf.keras.layers.experimental.preprocessing.RandomFlip('vertical'),\n  tf.keras.layers.experimental.preprocessing.RandomRotation(1.0),\n  tf.keras.layers.experimental.preprocessing.RandomContrast(factor=0.2)\n])\n'''\ndata_augmentation = tf.keras.Sequential([\n  tf.keras.layers.experimental.preprocessing.RandomContrast(factor=0.5)\n])\n'''","metadata":{"execution":{"iopub.status.busy":"2021-06-25T09:02:27.558892Z","iopub.execute_input":"2021-06-25T09:02:27.559270Z","iopub.status.idle":"2021-06-25T09:02:27.591926Z","shell.execute_reply.started":"2021-06-25T09:02:27.559239Z","shell.execute_reply":"2021-06-25T09:02:27.591213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#preview for augmentations\nfor image, _ in train_ds.take(1):\n  plt.figure(figsize=(10, 10))\n  first_image = image[0]\n  for i in range(9):\n    ax = plt.subplot(3, 3, i + 1)\n    augmented_image = data_augmentation(tf.expand_dims(first_image, 0))\n    plt.imshow(augmented_image[0] / 255)\n    plt.axis('off')","metadata":{"execution":{"iopub.status.busy":"2021-06-25T09:03:23.261819Z","iopub.execute_input":"2021-06-25T09:03:23.262187Z","iopub.status.idle":"2021-06-25T09:03:24.292497Z","shell.execute_reply.started":"2021-06-25T09:03:23.262156Z","shell.execute_reply":"2021-06-25T09:03:24.291689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preprocess_input = tf.keras.applications.resnet50.preprocess_input\nrescale = tf.keras.layers.experimental.preprocessing.Rescaling(1./127.5, offset= -1)","metadata":{"execution":{"iopub.status.busy":"2021-06-25T09:03:43.976632Z","iopub.execute_input":"2021-06-25T09:03:43.976996Z","iopub.status.idle":"2021-06-25T09:03:43.983444Z","shell.execute_reply.started":"2021-06-25T09:03:43.976946Z","shell.execute_reply":"2021-06-25T09:03:43.982286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create the base model from the pre-trained model MobileNet V2\nIMG_SHAPE = (img_height,img_width) + (3,)\n\nbasemodel = tf.keras.applications.ResNet50(input_shape=IMG_SHAPE, include_top=False, weights='imagenet')","metadata":{"execution":{"iopub.status.busy":"2021-06-25T09:04:02.783337Z","iopub.execute_input":"2021-06-25T09:04:02.783658Z","iopub.status.idle":"2021-06-25T09:04:04.941297Z","shell.execute_reply.started":"2021-06-25T09:04:02.783631Z","shell.execute_reply":"2021-06-25T09:04:04.940436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_batch, label_batch = next(iter(train_ds))\nfeature_batch = basemodel(image_batch)\nprint(feature_batch.shape)","metadata":{"execution":{"iopub.status.busy":"2021-06-25T09:04:19.153879Z","iopub.execute_input":"2021-06-25T09:04:19.154261Z","iopub.status.idle":"2021-06-25T09:04:23.410660Z","shell.execute_reply.started":"2021-06-25T09:04:19.154230Z","shell.execute_reply":"2021-06-25T09:04:23.409723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"basemodel.trainable = False\nbasemodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-06-25T09:04:33.404014Z","iopub.execute_input":"2021-06-25T09:04:33.404381Z","iopub.status.idle":"2021-06-25T09:04:33.494376Z","shell.execute_reply.started":"2021-06-25T09:04:33.404348Z","shell.execute_reply":"2021-06-25T09:04:33.493533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"global_average_layer = tf.keras.layers.GlobalAveragePooling2D()\nfeature_batch_average = global_average_layer(feature_batch)\nprint(feature_batch_average.shape)","metadata":{"execution":{"iopub.status.busy":"2021-06-25T09:04:58.138638Z","iopub.execute_input":"2021-06-25T09:04:58.138980Z","iopub.status.idle":"2021-06-25T09:04:58.155452Z","shell.execute_reply.started":"2021-06-25T09:04:58.138938Z","shell.execute_reply":"2021-06-25T09:04:58.154509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_layer = tf.keras.layers.Dense(1)\nprediction_batch = prediction_layer(feature_batch_average)\nprint(prediction_batch.shape)","metadata":{"execution":{"iopub.status.busy":"2021-06-25T09:05:07.900495Z","iopub.execute_input":"2021-06-25T09:05:07.900915Z","iopub.status.idle":"2021-06-25T09:05:07.927582Z","shell.execute_reply.started":"2021-06-25T09:05:07.900866Z","shell.execute_reply":"2021-06-25T09:05:07.926525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs = tf.keras.Input(shape=IMG_SHAPE)\nx = data_augmentation(inputs)\nx = rescale(x)\nx = preprocess_input(x)\nx = basemodel(x, training=False)\nx = global_average_layer(x)\nx = tf.keras.layers.Dropout(0.2)(x)\noutputs = prediction_layer(x)\nmodel = tf.keras.Model(inputs, outputs)","metadata":{"execution":{"iopub.status.busy":"2021-06-25T09:05:32.340630Z","iopub.execute_input":"2021-06-25T09:05:32.340990Z","iopub.status.idle":"2021-06-25T09:05:32.876786Z","shell.execute_reply.started":"2021-06-25T09:05:32.340942Z","shell.execute_reply":"2021-06-25T09:05:32.875879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_learning_rate = 0.0005\nmodel.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=base_learning_rate),\n              loss=tf.keras.losses.BinaryCrossentropy(from_logits=True),\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2021-06-25T09:05:43.540921Z","iopub.execute_input":"2021-06-25T09:05:43.541296Z","iopub.status.idle":"2021-06-25T09:05:43.562138Z","shell.execute_reply.started":"2021-06-25T09:05:43.541266Z","shell.execute_reply":"2021-06-25T09:05:43.561227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2021-06-25T09:05:54.587621Z","iopub.execute_input":"2021-06-25T09:05:54.587960Z","iopub.status.idle":"2021-06-25T09:05:54.609464Z","shell.execute_reply.started":"2021-06-25T09:05:54.587926Z","shell.execute_reply":"2021-06-25T09:05:54.608405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss0, accuracy0 = model.evaluate(val_ds)","metadata":{"execution":{"iopub.status.busy":"2021-06-25T09:06:04.314430Z","iopub.execute_input":"2021-06-25T09:06:04.314758Z","iopub.status.idle":"2021-06-25T09:06:30.577936Z","shell.execute_reply.started":"2021-06-25T09:06:04.314729Z","shell.execute_reply":"2021-06-25T09:06:30.576835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"initial loss: {:.2f}\".format(loss0))\nprint(\"initial accuracy: {:.2f}\".format(accuracy0))","metadata":{"execution":{"iopub.status.busy":"2021-06-25T09:06:30.582226Z","iopub.execute_input":"2021-06-25T09:06:30.582579Z","iopub.status.idle":"2021-06-25T09:06:30.594254Z","shell.execute_reply.started":"2021-06-25T09:06:30.582545Z","shell.execute_reply":"2021-06-25T09:06:30.593031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"initial_epochs = 30\nhistory = model.fit(train_ds,\n                    epochs=initial_epochs,\n                    validation_data=val_ds)","metadata":{"execution":{"iopub.status.busy":"2021-06-25T09:16:27.599833Z","iopub.execute_input":"2021-06-25T09:16:27.600190Z","iopub.status.idle":"2021-06-25T09:16:29.044757Z","shell.execute_reply.started":"2021-06-25T09:16:27.600160Z","shell.execute_reply":"2021-06-25T09:16:29.042413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"initial_epochs = history.epoch[-1]  #continued from last\nadd_epochs = 20\ntotal_epochs =  initial_epochs + add_epochs + 1\n\nacc += history.history['accuracy']\nval_acc += history.history['val_accuracy']\n\nloss += history.history['loss']\nval_loss += history.history['val_loss']\n\nhistory = model.fit(train_ds,\n                         epochs=total_epochs,\n                         initial_epoch=initial_epochs,\n                         validation_data=val_ds)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Load images list, and load labels\nimport glob\n#透過glob所全部train資料夾中所有可用圖片\nimgs=glob.glob('../input/histopathologic-cancer-detection/train/*.tif')\nprint(len(imgs))\n\n#ImageDatset(imgs,symbol='image')\nf=open('../input/histopathologic-cancer-detection/train_labels.csv','r',encoding='utf-8-sig')\nrows=f.readlines()\nprint(rows[:3])\n\nrows=rows[1:] #拿掉第一筆標頭\nrandom.shuffle(rows)#隨機洗牌","metadata":{"execution":{"iopub.status.busy":"2021-06-23T03:25:07.130185Z","iopub.execute_input":"2021-06-23T03:25:07.130519Z","iopub.status.idle":"2021-06-23T03:25:11.817367Z","shell.execute_reply.started":"2021-06-23T03:25:07.130488Z","shell.execute_reply":"2021-06-23T03:25:11.816489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Split to training & testing dataset\n\nimage_path=[]\nlabels=[]\ntest_image_path=[]\ntest_labels=[]\n\nfor row in rows:\n    cols=row.strip().split(',') #移除\\n然後逗號分割\n    if random.random()<=0.3:\n        test_image_path.append('../input/histopathologic-cancer-detection/train/{0}.tif'.format(cols[0]))\n        test_labels.append(int(cols[1]))\n    else:\n        image_path.append('../input/histopathologic-cancer-detection/train/{0}.tif'.format(cols[0]))\n        labels.append(int(cols[1]))\nprint('training dataset imgages / labels', len(image_path), '/', len(labels))\n\nprint('testing dataset images / labels', len(test_image_path), '/', len(test_labels))\n","metadata":{"execution":{"iopub.status.busy":"2021-06-23T03:25:14.43244Z","iopub.execute_input":"2021-06-23T03:25:14.432808Z","iopub.status.idle":"2021-06-23T03:25:14.795027Z","shell.execute_reply.started":"2021-06-23T03:25:14.432758Z","shell.execute_reply":"2021-06-23T03:25:14.793972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#將一個個放置於image_path, labels及test_image_path, test_labels的列表，透過Trident提供的ImageDataset和LabelDataset載入。\nds1_tmp=ImageDataset(image_path,symbol='image')\nds2=LabelDataset(labels,symbol='label')\n\nds1_t=ImageDataset(test_image_path,symbol='image')\nds2_t=LabelDataset(test_labels,symbol='label')\n","metadata":{"execution":{"iopub.status.busy":"2021-06-23T03:25:16.532568Z","iopub.execute_input":"2021-06-23T03:25:16.532911Z","iopub.status.idle":"2021-06-23T03:25:16.538149Z","shell.execute_reply.started":"2021-06-23T03:25:16.532877Z","shell.execute_reply":"2021-06-23T03:25:16.537306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#let's preview some images\n#!pip install ipyplot\nimport ipyplot\nimport cv2\nfrom tqdm import tqdm\n\ndef get_combined(img, ccrop=(32,32)):\n    width = img.shape[1]\n    height = img.shape[0]\n    x1 = int(width/2 - ccrop[0]/2)\n    y1 = int(height/2 - ccrop[1]/2)\n    x2 = x1 + ccrop[0]\n    y2 = y1 + ccrop[1]\n    center_img = cv2.resize(img[y1:y2, x1:x2], (48,96))\n    img = cv2.resize(img, (48,96))\n    vis = np.concatenate((center_img, img), axis=1)\n    return vis\n\n\nds1 = []\nfor img in tqdm(ds1_tmp):\n    img_c = get_combined(img, (32,32))\n    ds1.append(img_c.copy())\n\nprint(ds1[3].shape)\nipyplot.plot_images([ds1[3]], img_width=250)\n","metadata":{"execution":{"iopub.status.busy":"2021-06-23T03:25:44.227463Z","iopub.execute_input":"2021-06-23T03:25:44.227866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torchvision, torch, torchaudio, matplotlib\nprint(torchvision.__version__)\nprint(torch.__version__)\nprint(torchaudio.__version__)\nprint(matplotlib.__version__)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T03:57:34.769975Z","iopub.execute_input":"2021-06-23T03:57:34.770315Z","iopub.status.idle":"2021-06-23T03:57:34.777599Z","shell.execute_reply.started":"2021-06-23T03:57:34.770283Z","shell.execute_reply":"2021-06-23T03:57:34.776488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#與Iterator構成data provider\n#設定Trident Data provider，很類似Keras的Data generator，但使用起來更為方便。\ndata_provider=DataProvider(traindata=Iterator(data=ds1,label=ds2),testdata=Iterator(data=ds1_t,label=ds2_t))\n\n#即可完成設定，可以透過next()來確認數據是否正常拋出，以及是否有正確產生輸出數據的signature\nimg_data,label_data=data_provider.next()\nprint(data_provider.signature)\nprint(img_data.shape)\nprint(label_data.shape)\n\n#設定DataProvider的預處理流程\ndata_provider.image_transform_funcs=[\n    auto_level(),\n    RandomAdjustGamma(scale=(0.6,1.4)),#調整明暗\n    RandomAdjustHue(scale=(-0.5,0.5)),#調整色相\n    RandomAdjustSaturation(scale=(0.6,1.4)),#調整飽和度\n    SaltPepperNoise(0.05),#加入胡椒鹽噪音\n    RandomErasing(), #加入隨機擦去\n    Normalize(127.5,127.5)] #標準化\n\nimg_data,label_data=data_provider.next()\nprint(img_data.shape)\nprint(label_data.shape)\n\ndata_provider.preview_images()","metadata":{"execution":{"iopub.status.busy":"2021-06-21T09:01:18.992267Z","iopub.execute_input":"2021-06-21T09:01:18.992624Z","iopub.status.idle":"2021-06-21T09:01:19.469021Z","shell.execute_reply.started":"2021-06-21T09:01:18.992593Z","shell.execute_reply":"2021-06-21T09:01:19.468105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#測試集數據不需要做數據增強\ndata_provider.testdata.data.image_transform_funcs=[\n    Normalize(127.5,127.5)] #標準化\n\ntest_img_data,test_label_data=data_provider.next_test()\nprint(test_img_data.shape)\nprint(test_label_data.shape)\n\ndata_provider.preview_images()","metadata":{"execution":{"iopub.status.busy":"2021-06-21T09:01:25.960042Z","iopub.execute_input":"2021-06-21T09:01:25.960384Z","iopub.status.idle":"2021-06-21T09:01:26.126509Z","shell.execute_reply.started":"2021-06-21T09:01:25.960352Z","shell.execute_reply":"2021-06-21T09:01:26.121718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from trident.models import efficientnet,densenet,resnet\nnet1=efficientnet.EfficientNetB0(pretrained=True,include_top=True,freeze_features=True,input_shape=(3,96,96),classes=2)\nnet1.model[-1].add_noise=True\nnet1.model[-1].noise_intensity=0.12","metadata":{"execution":{"iopub.status.busy":"2021-06-21T09:02:10.315401Z","iopub.execute_input":"2021-06-21T09:02:10.315764Z","iopub.status.idle":"2021-06-21T09:02:16.613896Z","shell.execute_reply.started":"2021-06-21T09:02:10.315734Z","shell.execute_reply":"2021-06-21T09:02:16.612851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CustomCropFlatten(Layer):\n    \"\"\"\n  \n    \"\"\"\n    def __init__(self,name='CustomCropFlatten'):\n        super(CustomCropFlatten, self).__init__()\n        self.name = name\n\n    def forward(self, x, **kwargs):\n        #(None, 112, 6, 6)\n        #target_area=2x2\n        target=x[:,:,2:4,2:4]\n        B,C,H,W=int_shape(target)\n        outside1=reduce_max(x[:,:,1,1:5],axis=-1)\n        outside2=reduce_max(x[:,:,4,1:5],axis=-1)\n        outside3=reduce_max(x[:,:,1:5,1],axis=-1)\n        outside4=reduce_max(x[:,:,1:5,4],axis=-1)\n        outside=stack([outside1,outside2,outside3,outside4],axis=-1)\n        target=reshape(target,(B,C,-1))\n        return reshape(concate([target,outside],axis=-1),(B,-1))\n\nnet2=efficientnet.EfficientNetB0(pretrained=True,include_top=False,freeze_features=True,input_shape=(3,96,96),classes=2)\nnet2.model.remove_at(-1)\nnet2.model.remove_at(-1)\nnet2.model.remove_at(-1)\nnet2.model.remove_at(-1)\nnet2.model.remove_at(-1)\nnet2.model.remove_at(-1)\nnet2.model.add_module('custom',CustomCropFlatten())\nnet2.model.add_module('fc',Dense(2,activation=None))\nnet2.model.add_module('softmax',SoftMax(-1,add_noise=True,noise_intensity=0.12))\nnet2.summary()","metadata":{"execution":{"iopub.status.busy":"2021-06-21T09:02:16.615546Z","iopub.execute_input":"2021-06-21T09:02:16.615887Z","iopub.status.idle":"2021-06-21T09:02:17.236981Z","shell.execute_reply.started":"2021-06-21T09:02:16.61585Z","shell.execute_reply":"2021-06-21T09:02:17.236029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"net3=resnet.ResNet50(pretrained=True,include_top=False,freeze_features=True,input_shape=(3,96,96),classes=2)\nnet3.model.remove_at(-1)\nnet3.model.add_module('custom',CustomCropFlatten())\nnet3.model.add_module('fc',Dense(2,activation=None))\nnet3.model.add_module('softmax',SoftMax(axis=-1,add_noise=True,noise_intensity=0.12))\nnet3.summary()","metadata":{"execution":{"iopub.status.busy":"2021-06-21T09:02:17.240547Z","iopub.execute_input":"2021-06-21T09:02:17.240861Z","iopub.status.idle":"2021-06-21T09:02:19.582874Z","shell.execute_reply.started":"2021-06-21T09:02:17.240822Z","shell.execute_reply":"2021-06-21T09:02:19.582031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, auc, roc_auc_score\n# make a prediction\n\ndef auc(output,target):\n    \n    output_np=to_numpy(exp(output))[:,1]\n    target_np=to_numpy(target)\n    return roc_auc_score(target_np, output_np)\n\ndef punishment(output,target):\n    mask=target==1\n    mask_neg=target==0\n    masked_positive=exp(output)[:,1][mask]\n    masked_negative=exp(output)[:,1][mask_neg]\n    return 1-(masked_positive.mean()-masked_negative.mean())\n\n#challenger1 使用DiffGrad優化器、累積梯度\nnet1.with_optimizer(DiffGrad,lr=1e-3,gradient_centralization='all')\\\n.with_loss(CrossEntropyLoss(auto_balance=True,label_smooth=True))\\\n.with_loss(F1ScoreLoss,loss_weight=0.5)\\\n.with_loss(punishment,loss_weight=0.2)\\\n.with_metric(accuracy,ignore_index=0)\\\n.with_metric(recall,ignore_index=0)\\\n.with_metric(auc)\\\n.with_regularizer('l2',1e-5)\\\n.with_model_save_path('./Models/eff_1.pth')\\\n.with_accumulate_grads(5)\\\n.with_callbacks(CosineLR(max_lr=1e-3, min_lr=1e-8,period=5000))\\\n.unfreeze_model_scheduling(300,unit='batch',module_name='block7a')\\\n.with_automatic_mixed_precision_training()\n\n#challenger1 使用DiffGrad優化器、累積梯度\nnet2.with_optimizer(DiffGrad,lr=1e-3,gradient_centralization='all')\\\n.with_loss(CrossEntropyLoss(auto_balance=True,label_smooth=True))\\\n.with_loss(F1ScoreLoss,loss_weight=0.5)\\\n.with_loss(punishment,loss_weight=0.2)\\\n.with_metric(accuracy,ignore_index=0)\\\n.with_metric(recall,ignore_index=0)\\\n.with_metric(auc)\\\n.with_regularizer('l2',1e-5)\\\n.with_model_save_path('./Models/eff_2.pth')\\\n.with_accumulate_grads(10)\\\n.with_callbacks(CosineLR(max_lr=1e-3, min_lr=1e-8,period=5000))\\\n.unfreeze_model_scheduling(300,unit='batch',module_name='block5c')\\\n.with_automatic_mixed_precision_training()\n\n\n\n#challenger1 使用DiffGrad優化器、累積梯度\nnet3.with_optimizer(DiffGrad,lr=1e-3,gradient_centralization='all')\\\n.with_loss(CrossEntropyLoss(auto_balance=True,label_smooth=True))\\\n.with_loss(F1ScoreLoss,loss_weight=0.5)\\\n.with_loss(punishment,loss_weight=0.2)\\\n.with_metric(accuracy,ignore_index=0)\\\n.with_metric(recall,ignore_index=0)\\\n.with_metric(auc)\\\n.with_regularizer('l2',1e-5)\\\n.with_model_save_path('./Models/resnet_1.pth')\\\n.with_accumulate_grads(10)\\\n.with_callbacks(CosineLR(max_lr=1e-3, min_lr=1e-8,period=5000))\\\n.unfreeze_model_scheduling(300,unit='batch',module_name='layer3.5')\\\n.with_automatic_mixed_precision_training()","metadata":{"execution":{"iopub.status.busy":"2021-06-21T09:02:26.148052Z","iopub.execute_input":"2021-06-21T09:02:26.148559Z","iopub.status.idle":"2021-06-21T09:02:26.460852Z","shell.execute_reply.started":"2021-06-21T09:02:26.148527Z","shell.execute_reply":"2021-06-21T09:02:26.460078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#使用outsample驗證\ndef draw_roc(training_context): #建模訓練階段都靠training_context通訊，傳遞所有需要的訊息\n    if training_context['steps']==10 or (training_context['steps']+1)%100==0:\n        model=copy.deepcopy(training_context['current_model'])#複製一份模型\n        model.eval()  #切換為推論模式\n        traindata=training_context['train_data'] #取出測試數據\n        input_data=traindata['image']\n        target_data=traindata['label']\n        \n        output=model(input_data)\n        target_np=to_numpy(target_data)\n        output_np=to_numpy(output) #推論階段softmax就會是一般softmax(訓練階段是log_softmax)，所以不必再做處理，而且所有noise, dropout都會消失\n        fpr, tpr,_=roc_curve(target_np, output_np[:,1])\n        plt.figure(1)\n        plt.plot([0, 1], [0, 1], 'k--')\n        plt.plot(fpr, tpr, label='area = {:.3f}'.format(roc_auc_score(target_np, output_np[:,1])))\n        plt.xlabel('False positive rate')\n        plt.ylabel('True positive rate')\n        plt.title('ROC curve')\n        plt.legend(loc='best')\n        plt.show()\n        \nnet1.trigger_when('on_batch_end', frequency=1, action=draw_roc)\nnet2.trigger_when('on_batch_end', frequency=1, action=draw_roc)\nnet3.trigger_when('on_batch_end', frequency=1, action=draw_roc)","metadata":{"execution":{"iopub.status.busy":"2021-06-21T09:02:31.130047Z","iopub.execute_input":"2021-06-21T09:02:31.130413Z","iopub.status.idle":"2021-06-21T09:02:31.145659Z","shell.execute_reply.started":"2021-06-21T09:02:31.130378Z","shell.execute_reply":"2021-06-21T09:02:31.144495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tensorboard","metadata":{"execution":{"iopub.status.busy":"2021-06-21T07:52:22.922737Z","iopub.execute_input":"2021-06-21T07:52:22.923068Z","iopub.status.idle":"2021-06-21T07:52:28.793338Z","shell.execute_reply.started":"2021-06-21T07:52:22.923037Z","shell.execute_reply":"2021-06-21T07:52:28.792261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plan=TrainingPlan()\\\n    .add_training_item(net1,name='net1')\\\n    .add_training_item(net2,name='net2')\\\n    .add_training_item(net3,name='net3')\\\n    .with_data_loader(data_provider)\\\n    .with_batch_size(128)\\\n    .repeat_epochs(5)\\\n    .out_sample_evaluation_scheduling(100)\\\n    .print_gradients_scheduling(100,unit='batch')\\\n    .print_progress_scheduling(10,unit='batch')\\\n    .display_loss_metric_curve_scheduling(200)\\\n    .save_model_scheduling(20,unit='batch')\n    #.with_tensorboard()\n\nplan.start_now(collect_data_inteval=10)","metadata":{"execution":{"iopub.status.busy":"2021-06-21T09:02:34.999115Z","iopub.execute_input":"2021-06-21T09:02:34.999473Z","iopub.status.idle":"2021-06-21T11:11:15.432996Z","shell.execute_reply.started":"2021-06-21T09:02:34.999443Z","shell.execute_reply":"2021-06-21T11:11:15.432017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}