{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"感谢大佬[h053473666](https://www.kaggle.com/h053473666)的数据集[https://www.kaggle.com/h053473666/siimcovid19-512-img-png-600-study-png](https://www.kaggle.com/h053473666/siimcovid19-512-img-png-600-study-png)","metadata":{}},{"cell_type":"code","source":"!pip install efficientnet -q","metadata":{"execution":{"iopub.status.busy":"2021-06-25T01:50:56.651737Z","iopub.execute_input":"2021-06-25T01:50:56.652321Z","iopub.status.idle":"2021-06-25T01:51:05.448226Z","shell.execute_reply.started":"2021-06-25T01:50:56.652214Z","shell.execute_reply":"2021-06-25T01:51:05.44711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\nimport numpy as np\nimport pandas as pd\nfrom kaggle_datasets import KaggleDatasets\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom sklearn.model_selection import KFold","metadata":{"execution":{"iopub.status.busy":"2021-06-25T01:51:05.450328Z","iopub.execute_input":"2021-06-25T01:51:05.450851Z","iopub.status.idle":"2021-06-25T01:51:12.445499Z","shell.execute_reply.started":"2021-06-25T01:51:05.450778Z","shell.execute_reply":"2021-06-25T01:51:12.444504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SEED = 0        #随机数种子，用来KFold分数据集\nFOLDS = 5        #交叉验证次数\nBATCH_SIZES = 18\nEPOCHS = 40\nls = 0.015       # 标签平滑，可以尝试0.015，不用请写0，可抗过拟合\nIMAGE_SIZE = (224, 240, 260, 300, 380, 456, 528, 600, 512)\n\nSIIM_para = {}\nSIIM_para['SEED'] = SEED\nSIIM_para['FOLDS'] = FOLDS\nSIIM_para['BATCH_SIZES'] = BATCH_SIZES\nSIIM_para['EPOCHS'] = EPOCHS\nSIIM_para['IMAGE_SIZE'] = IMAGE_SIZE[8]\nprint('SIIM_parameters: {}'.format(SIIM_para))","metadata":{"execution":{"iopub.status.busy":"2021-06-25T01:51:12.447666Z","iopub.execute_input":"2021-06-25T01:51:12.448033Z","iopub.status.idle":"2021-06-25T01:51:12.455735Z","shell.execute_reply.started":"2021-06-25T01:51:12.448002Z","shell.execute_reply":"2021-06-25T01:51:12.454671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_lr_callback():\n    lr_start   = 1e-4 # 初始学习率\n    lr_max   =  2e-4# 最大学习率\n    lr_min     = 1e-7 #最小学习率\n    lr_ramp_ep =  3 # 用几个epoch达到最大学习率\n    lr_sus_ep  =  3# 用最大的学习率跑几个epoch\n    lr_decay   = .4 # 退火，常用方法\n   \n    def lrfn(epoch):\n        if epoch < lr_ramp_ep:\n            lr = (lr_max - lr_start) / lr_ramp_ep * epoch + lr_start\n            \n        elif epoch < lr_ramp_ep + lr_sus_ep:\n            lr = lr_max\n            \n        else:\n            lr = (lr_max - lr_min) * lr_decay**(epoch - lr_ramp_ep - lr_sus_ep) + lr_min\n            \n        return lr\n\n    lr_callback = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose=False)\n    return lr_callback","metadata":{"execution":{"iopub.status.busy":"2021-06-25T01:51:12.457539Z","iopub.execute_input":"2021-06-25T01:51:12.457845Z","iopub.status.idle":"2021-06-25T01:51:12.467222Z","shell.execute_reply.started":"2021-06-25T01:51:12.457808Z","shell.execute_reply":"2021-06-25T01:51:12.466183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def auto_select_strategy():\n    try:\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n        print(\"Running on TPU:\", tpu.master())\n    except ValueError:\n        strategy = tf.distribute.get_strategy()\n    print(f\"Running on {strategy.num_replicas_in_sync} replicas\")\n    \n    return strategy","metadata":{"execution":{"iopub.status.busy":"2021-06-25T01:51:12.46856Z","iopub.execute_input":"2021-06-25T01:51:12.468907Z","iopub.status.idle":"2021-06-25T01:51:12.478906Z","shell.execute_reply.started":"2021-06-25T01:51:12.468878Z","shell.execute_reply":"2021-06-25T01:51:12.478037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_decoder(with_labels=True, target_size=(256, 256), ext='jpg'):\n    def decode(path):\n        file_bytes = tf.io.read_file(path)\n\n        if ext == 'png':\n            img = tf.image.decode_png(file_bytes, channels=3)\n        elif ext in ['jpg', 'jpeg']:\n            img = tf.image.decode_jpeg(file_bytes, channels=3)\n        else:\n            raise ValueError(\"Image extension not supported\")\n        img = tf.cast(img, tf.float32) / 255.0\n        img = tf.image.resize(img, target_size)\n\n        return img\n    \n    def decode_with_labels(path, label):\n        return decode(path), label\n    \n    return decode_with_labels if with_labels else decode\n\n\ndef build_augmenter(with_labels=True):\n    def augment(img):\n        if tf.random.uniform(()) > 0.5:\n            img = tf.image.flip_left_right(img)\n    \n        if tf.random.uniform(()) > 0.4:\n            img = tf.image.flip_up_down(img)\n\n        if tf.random.uniform(()) > 0.5:\n            img = tf.image.rot90(img, k=1)\n\n#        if tf.random.uniform(()) > 0.45:\n#            img = tf.image.random_saturation(img, 0.7, 1.3)\n\n#        if tf.random.uniform(()) > 0.45:\n#            img = tf.image.random_contrast(img, 0.8, 1.2)\n#            \n#        if tf.random.uniform(()) > 0.45:\n#            img = tf.image.random_brightness(img, 0.1)\n        return img\n    \n    def augment_with_labels(img, label):\n        return augment(img), label\n    \n    return augment_with_labels if with_labels else augment\n\n\ndef build_dataset(paths, labels=None, bsize=128, cache=True,\n                  decode_fn=None, augment_fn=None,\n                  augment=True, repeat=True, shuffle=1024, \n                  cache_dir=\"\"):\n    if cache_dir != \"\" and cache is True:\n        os.makedirs(cache_dir, exist_ok=True)\n    \n    if decode_fn is None:\n        decode_fn = build_decoder(labels is not None)\n    \n    if augment_fn is None:\n        augment_fn = build_augmenter(labels is not None)\n    \n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = paths if labels is None else (paths, labels)\n    \n    dset = tf.data.Dataset.from_tensor_slices(slices)\n    dset = dset.map(decode_fn, num_parallel_calls=AUTO)\n    dset = dset.cache(cache_dir) if cache else dset\n    dset = dset.map(augment_fn, num_parallel_calls=AUTO) if augment else dset\n    dset = dset.repeat() if repeat else dset\n    dset = dset.shuffle(shuffle) if shuffle else dset\n    dset = dset.batch(bsize).prefetch(AUTO)\n    \n    return dset","metadata":{"execution":{"iopub.status.busy":"2021-06-25T01:51:12.480355Z","iopub.execute_input":"2021-06-25T01:51:12.480756Z","iopub.status.idle":"2021-06-25T01:51:12.497502Z","shell.execute_reply.started":"2021-06-25T01:51:12.480716Z","shell.execute_reply":"2021-06-25T01:51:12.496387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"COMPETITION_NAME = \"siimcovid19-512-img-png-600-study-png\"\nstrategy = auto_select_strategy()\nREPLICAS = strategy.num_replicas_in_sync * BATCH_SIZES\nGCS_DS_PATH = KaggleDatasets().get_gcs_path(COMPETITION_NAME)","metadata":{"execution":{"iopub.status.busy":"2021-06-25T01:51:12.499184Z","iopub.execute_input":"2021-06-25T01:51:12.499585Z","iopub.status.idle":"2021-06-25T01:51:18.780494Z","shell.execute_reply.started":"2021-06-25T01:51:12.499541Z","shell.execute_reply":"2021-06-25T01:51:18.779246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"load_dir = f\"/kaggle/input/{COMPETITION_NAME}/\"\ndf = pd.read_csv('../input/siimcov19csv/train.csv')\nlabel_cols = df.columns[4]","metadata":{"execution":{"iopub.status.busy":"2021-06-25T01:51:18.78504Z","iopub.execute_input":"2021-06-25T01:51:18.785395Z","iopub.status.idle":"2021-06-25T01:51:18.864571Z","shell.execute_reply.started":"2021-06-25T01:51:18.785363Z","shell.execute_reply":"2021-06-25T01:51:18.863325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model(dim=512):\n    \n    inp = tf.keras.layers.Input(shape=(dim,dim,3))\n    base = tf.keras.applications.ResNet101(input_shape=(dim,dim,3),weights='imagenet',include_top=False)\n\n    x = base(inp)\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n\n    x = tf.keras.layers.Dense(1024, activation='relu')(x)\n    x = tf.keras.layers.Dropout(0.2)(x)\n\n    x = tf.keras.layers.Dense(1, activation='sigmoid')(x)\n    \n    model = tf.keras.Model(inputs=inp,outputs=x)\n    opt = tf.keras.optimizers.Adam(learning_rate=0.001)\n    loss = tf.keras.losses.BinaryCrossentropy(label_smoothing=ls) \n    model.compile(optimizer=opt,loss=loss,metrics=['AUC'])\n    model.summary()\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2021-06-25T01:51:18.866348Z","iopub.execute_input":"2021-06-25T01:51:18.866732Z","iopub.status.idle":"2021-06-25T01:51:18.878536Z","shell.execute_reply.started":"2021-06-25T01:51:18.86669Z","shell.execute_reply":"2021-06-25T01:51:18.877023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skf = KFold(n_splits=FOLDS,shuffle=True,random_state=SEED)\ndf['fold'] = -1\nfor fold, (train_idx, val_idx) in enumerate(skf.split(df, groups = df.StudyInstanceUID.tolist())):\n    df.loc[val_idx, 'fold'] = fold","metadata":{"execution":{"iopub.status.busy":"2021-06-25T01:51:18.880253Z","iopub.execute_input":"2021-06-25T01:51:18.880684Z","iopub.status.idle":"2021-06-25T01:51:18.906447Z","shell.execute_reply.started":"2021-06-25T01:51:18.880637Z","shell.execute_reply":"2021-06-25T01:51:18.905228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(5):\n    \n    valid_paths = GCS_DS_PATH + '/image/' + df[df['fold'] == i]['id'] + '.png' #\"/train/\"\n    train_paths = GCS_DS_PATH + '/image/' + df[df['fold'] != i]['id'] + '.png' #\"/train/\" \n    valid_labels = df[df['fold'] == i][label_cols].values\n    train_labels = df[df['fold'] != i][label_cols].values\n\n\n\n    decoder = build_decoder(with_labels=True, target_size=(IMAGE_SIZE[8], IMAGE_SIZE[8]), ext='png')\n    test_decoder = build_decoder(with_labels=False, target_size=(IMAGE_SIZE[8], IMAGE_SIZE[8]),ext='png')\n\n    train_dataset = build_dataset(\n        train_paths, train_labels, bsize=REPLICAS, decode_fn=decoder\n    )\n\n    valid_dataset = build_dataset(\n        valid_paths, valid_labels, bsize=REPLICAS, decode_fn=decoder,\n        repeat=False, shuffle=False, augment=False\n    )\n\n    try:\n        n_labels = train_labels.shape[1]\n    except:\n        n_labels = 1\n\n    with strategy.scope():\n        model = build_model(dim=IMAGE_SIZE[8])\n\n    steps_per_epoch = train_paths.shape[0] // REPLICAS\n    checkpoint = tf.keras.callbacks.ModelCheckpoint(\n        f'model{i}.h5', save_best_only=True, monitor='val_loss', mode='min')\n\n\n    history = model.fit(\n        train_dataset, \n        epochs=EPOCHS,\n        verbose=1,\n        callbacks=[checkpoint, get_lr_callback()],\n        steps_per_epoch=steps_per_epoch,\n        validation_data=valid_dataset)\n\n    hist_df = pd.DataFrame(history.history)\n    hist_df.to_csv(f'history{i}.csv')","metadata":{"execution":{"iopub.status.busy":"2021-06-25T01:51:18.908096Z","iopub.execute_input":"2021-06-25T01:51:18.908664Z"},"trusted":true},"execution_count":null,"outputs":[]}]}