{"cells":[{"metadata":{"papermill":{"duration":0.012261,"end_time":"2020-08-20T13:10:33.841122","exception":false,"start_time":"2020-08-20T13:10:33.828861","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Overview & Remarks\n\nThis notebook contains the configurations required to train an efficientnet model for K-folds.\n\nIt is possible to hit -0.6910 LB by tweaking parameters in this notebook!"},{"metadata":{"_kg_hide-output":true,"execution":{"iopub.execute_input":"2020-08-20T13:10:33.870556Z","iopub.status.busy":"2020-08-20T13:10:33.869709Z","iopub.status.idle":"2020-08-20T13:10:54.378664Z","shell.execute_reply":"2020-08-20T13:10:54.377677Z"},"papermill":{"duration":20.529455,"end_time":"2020-08-20T13:10:54.378916","exception":false,"start_time":"2020-08-20T13:10:33.849461","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"!pip install ../input/kerasapplications/keras-team-keras-applications-3b180cb -f ./ --no-index\n!pip install ../input/efficientnet/efficientnet-1.1.0/ -f ./ --no-index","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.execute_input":"2020-08-20T13:10:54.404282Z","iopub.status.busy":"2020-08-20T13:10:54.403401Z","iopub.status.idle":"2020-08-20T13:11:01.208783Z","shell.execute_reply":"2020-08-20T13:11:01.209418Z"},"papermill":{"duration":6.82276,"end_time":"2020-08-20T13:11:01.20963","exception":false,"start_time":"2020-08-20T13:10:54.38687","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"import os\nimport cv2\nimport pydicom\nimport pandas as pd\nimport numpy as np \nimport tensorflow as tf \nimport matplotlib.pyplot as plt \nfrom tqdm.notebook import tqdm \nfrom tensorflow.keras.layers import (\n    Dense, Dropout, Activation, Flatten, Input, BatchNormalization, GlobalAveragePooling2D, Add, Conv2D, AveragePooling2D, \n    LeakyReLU, Concatenate \n)\nfrom tensorflow.keras import Model\nfrom tensorflow.keras.utils import Sequence\nimport tensorflow.keras.backend as K\nimport tensorflow.keras.applications as tfa\nimport efficientnet.tfkeras as efn\nfrom sklearn.model_selection import train_test_split, KFold\nimport seaborn as sns","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T13:11:01.237312Z","iopub.status.busy":"2020-08-20T13:11:01.236532Z","iopub.status.idle":"2020-08-20T13:11:03.734991Z","shell.execute_reply":"2020-08-20T13:11:03.734382Z"},"papermill":{"duration":2.514265,"end_time":"2020-08-20T13:11:03.73512","exception":false,"start_time":"2020-08-20T13:11:01.220855","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"config = tf.compat.v1.ConfigProto()\nconfig.gpu_options.allow_growth = True\nsession = tf.compat.v1.Session(config=config)","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.008417,"end_time":"2020-08-20T13:11:03.751195","exception":false,"start_time":"2020-08-20T13:11:03.742778","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Training Parameters\n\n- `EPOCHS`: number of epochs to train for in each fold\n- `BATCH_SIZE`: batch size of images during training\n- `NFOLD`: number of folds in K-fold cross-validation (CV)\n- `LR`: learning rate\n- `SAVE_BEST`: default is True to save best weights on validation loss\n- `MODEL_CLASS`: the class of model. E.g. \"b1\" for EfficientNet-B1"},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T13:11:03.770223Z","iopub.status.busy":"2020-08-20T13:11:03.769612Z","iopub.status.idle":"2020-08-20T13:11:03.773847Z","shell.execute_reply":"2020-08-20T13:11:03.773374Z"},"papermill":{"duration":0.015392,"end_time":"2020-08-20T13:11:03.773943","exception":false,"start_time":"2020-08-20T13:11:03.758551","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"EPOCHS = 2\nBATCH_SIZE = 8\nNFOLD = 5\nLR = 0.003\nSAVE_BEST = True\nMODEL_CLASS = 'b1'","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T13:11:03.797636Z","iopub.status.busy":"2020-08-20T13:11:03.796851Z","iopub.status.idle":"2020-08-20T13:11:03.809048Z","shell.execute_reply":"2020-08-20T13:11:03.80851Z"},"papermill":{"duration":0.027299,"end_time":"2020-08-20T13:11:03.809146","exception":false,"start_time":"2020-08-20T13:11:03.781847","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/train.csv') ","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T13:11:03.839331Z","iopub.status.busy":"2020-08-20T13:11:03.838573Z","iopub.status.idle":"2020-08-20T13:11:03.851608Z","shell.execute_reply":"2020-08-20T13:11:03.850982Z"},"papermill":{"duration":0.035288,"end_time":"2020-08-20T13:11:03.851713","exception":false,"start_time":"2020-08-20T13:11:03.816425","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T13:11:03.873351Z","iopub.status.busy":"2020-08-20T13:11:03.872738Z","iopub.status.idle":"2020-08-20T13:11:03.879439Z","shell.execute_reply":"2020-08-20T13:11:03.878952Z"},"papermill":{"duration":0.020138,"end_time":"2020-08-20T13:11:03.879533","exception":false,"start_time":"2020-08-20T13:11:03.859395","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"train.SmokingStatus.unique()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T13:11:03.905438Z","iopub.status.busy":"2020-08-20T13:11:03.903861Z","iopub.status.idle":"2020-08-20T13:11:03.906306Z","shell.execute_reply":"2020-08-20T13:11:03.906759Z"},"papermill":{"duration":0.01946,"end_time":"2020-08-20T13:11:03.906908","exception":false,"start_time":"2020-08-20T13:11:03.887448","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"def get_tab(df):\n    vector = [(df.Age.values[0] - 30) / 30] \n    \n    if df.Sex.values[0].lower() == 'male':\n       vector.append(0)\n    else:\n       vector.append(1)\n    \n    if df.SmokingStatus.values[0] == 'Never smoked':\n        vector.extend([0,0])\n    elif df.SmokingStatus.values[0] == 'Ex-smoker':\n        vector.extend([1,1])\n    elif df.SmokingStatus.values[0] == 'Currently smokes':\n        vector.extend([0,1])\n    else:\n        vector.extend([1,0])\n    return np.array(vector) ","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T13:11:03.9316Z","iopub.status.busy":"2020-08-20T13:11:03.930755Z","iopub.status.idle":"2020-08-20T13:11:04.243675Z","shell.execute_reply":"2020-08-20T13:11:04.243093Z"},"papermill":{"duration":0.329148,"end_time":"2020-08-20T13:11:04.243842","exception":false,"start_time":"2020-08-20T13:11:03.914694","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"A = {} \nTAB = {} \nP = [] \nfor i, p in tqdm(enumerate(train.Patient.unique())):\n    sub = train.loc[train.Patient == p, :] \n    fvc = sub.FVC.values\n    weeks = sub.Weeks.values\n    c = np.vstack([weeks, np.ones(len(weeks))]).T\n    a, b = np.linalg.lstsq(c, fvc)[0]\n    \n    A[p] = a\n    TAB[p] = get_tab(sub)\n    P.append(p)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T13:11:04.266426Z","iopub.status.busy":"2020-08-20T13:11:04.265843Z","iopub.status.idle":"2020-08-20T13:11:04.269515Z","shell.execute_reply":"2020-08-20T13:11:04.26999Z"},"papermill":{"duration":0.017061,"end_time":"2020-08-20T13:11:04.270109","exception":false,"start_time":"2020-08-20T13:11:04.253048","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"def get_img(path):\n    d = pydicom.dcmread(path)\n    return cv2.resize((d.pixel_array - d.RescaleIntercept) / (d.RescaleSlope * 1000), (512, 512))","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T13:11:04.312319Z","iopub.status.busy":"2020-08-20T13:11:04.292846Z","iopub.status.idle":"2020-08-20T13:15:13.311411Z","shell.execute_reply":"2020-08-20T13:15:13.308024Z"},"papermill":{"duration":249.033369,"end_time":"2020-08-20T13:15:13.311573","exception":false,"start_time":"2020-08-20T13:11:04.278204","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"x, y = [], []\nfor p in tqdm(train.Patient.unique()):\n    try:\n        ldir = os.listdir(f'../input/osic-pulmonary-fibrosis-progression-lungs-mask/mask_noise/mask_noise/{p}/')\n        numb = [float(i[:-4]) for i in ldir]\n        for i in ldir:\n            x.append(cv2.imread(f'../input/osic-pulmonary-fibrosis-progression-lungs-mask/mask_noise/mask_noise/{p}/{i}', 0).mean())\n            y.append(float(i[:-4]) / max(numb))\n    except:\n        pass","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T13:15:13.345343Z","iopub.status.busy":"2020-08-20T13:15:13.341014Z","iopub.status.idle":"2020-08-20T13:15:13.347919Z","shell.execute_reply":"2020-08-20T13:15:13.348389Z"},"papermill":{"duration":0.028408,"end_time":"2020-08-20T13:15:13.348505","exception":false,"start_time":"2020-08-20T13:15:13.320097","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"class IGenerator(Sequence):\n    BAD_ID = ['ID00011637202177653955184', 'ID00052637202186188008618']\n    def __init__(self, keys, a, tab, batch_size=BATCH_SIZE):\n        self.keys = [k for k in keys if k not in self.BAD_ID]\n        self.a = a\n        self.tab = tab\n        self.batch_size = batch_size\n        \n        self.train_data = {}\n        for p in train.Patient.values:\n            self.train_data[p] = os.listdir(f'../input/osic-pulmonary-fibrosis-progression/train/{p}/')\n    \n    def __len__(self):\n        return 1000\n    \n    def __getitem__(self, idx):\n        x = []\n        a, tab = [], [] \n        keys = np.random.choice(self.keys, size = self.batch_size)\n        for k in keys:\n            try:\n                i = np.random.choice(self.train_data[k], size=1)[0]\n                img = get_img(f'../input/osic-pulmonary-fibrosis-progression/train/{k}/{i}')\n                x.append(img)\n                a.append(self.a[k])\n                tab.append(self.tab[k])\n            except:\n                print(k, i)\n       \n        x,a,tab = np.array(x), np.array(a), np.array(tab)\n        x = np.expand_dims(x, axis=-1)\n        return [x, tab] , a","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img = get_img(f'../input/osic-pulmonary-fibrosis-progression/train/ID00007637202177411956430/10.dcm')\nimg.shape\n a.append(self.a[k])\n                tab.append(self.tab[k])\n            except:\n                print(k, i)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"A","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"IGenerator(keys=P[0], a = A, tab = TAB)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T13:15:13.374062Z","iopub.status.busy":"2020-08-20T13:15:13.373386Z","iopub.status.idle":"2020-08-20T13:15:13.377306Z","shell.execute_reply":"2020-08-20T13:15:13.376812Z"},"papermill":{"duration":0.020256,"end_time":"2020-08-20T13:15:13.377421","exception":false,"start_time":"2020-08-20T13:15:13.357165","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"def get_efficientnet(model, shape):\n    models_dict = {\n        'b0': efn.EfficientNetB0(input_shape=shape,weights='imagenet',include_top=False),\n        'b1': efn.EfficientNetB1(input_shape=shape,weights='imagenet',include_top=False),\n        'b2': efn.EfficientNetB2(input_shape=shape,weights='imagenet',include_top=False),\n        'b3': efn.EfficientNetB3(input_shape=shape,weights='imagenet',include_top=False),\n        'b4': efn.EfficientNetB4(input_shape=shape,weights='imagenet',include_top=False),\n        'b5': efn.EfficientNetB5(input_shape=shape,weights='imagenet',include_top=False),\n        'b6': efn.EfficientNetB6(input_shape=shape,weights='imagenet',include_top=False),\n        'b7': efn.EfficientNetB7(input_shape=shape,weights='imagenet',include_top=False)\n    }\n    return models_dict[model]\n\ndef build_model(shape=(256, 256,3), model_class=None):\n    inp = Input(shape=shape)\n    base = get_efficientnet(model_class, shape)\n    x = base(inp)\n    x = GlobalAveragePooling2D()(x)\n    inp2 = Input(shape=(4,))\n    x2 = tf.keras.layers.GaussianNoise(0.2)(inp2)\n    x = Concatenate()([x, x2]) \n    x = Dropout(0.5)(x) \n    x = Dense(1)(x)\n    model = Model([inp, inp2] , x)\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install efficientnet-pytorch","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from efficientnet_pytorch import EfficientNet\nmodel = EfficientNet.from_pretrained('efficientnet-b0')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Training"},{"metadata":{"execution":{"iopub.execute_input":"2020-08-20T13:15:13.408437Z","iopub.status.busy":"2020-08-20T13:15:13.407625Z","iopub.status.idle":"2020-08-20T14:38:47.301073Z","shell.execute_reply":"2020-08-20T14:38:47.300505Z"},"papermill":{"duration":5013.916312,"end_time":"2020-08-20T14:38:47.302146","exception":false,"start_time":"2020-08-20T13:15:13.385834","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"kf = KFold(n_splits=NFOLD, random_state=42,shuffle=False)\nP = np.array(P)\nsubs = []\nfolds_history = []\nfor fold, (tr_idx, val_idx) in enumerate(kf.split(P)):\n    print('#####################')\n    print('####### Fold %i ######'%fold)\n    print('#####################')\n    print('Training...')\n    \n    er = tf.keras.callbacks.EarlyStopping(\n        monitor=\"val_loss\",\n        min_delta=1e-3,\n        patience=10,\n        verbose=1,\n        mode=\"auto\",\n        baseline=None,\n        restore_best_weights=True,\n    )\n\n    cpt = tf.keras.callbacks.ModelCheckpoint(\n        filepath='fold-%i.h5'%fold,\n        monitor='val_loss', \n        verbose=1, \n        save_best_only=SAVE_BEST,\n        mode='auto'\n    )\n\n    rlp = tf.keras.callbacks.ReduceLROnPlateau(\n        monitor='val_loss', \n        factor=0.5,\n        patience=5, \n        verbose=1, \n        min_lr=1e-8\n    )\n    model = build_model(model_class=MODEL_CLASS)\n    model.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=LR), loss=\"mae\") \n    history = model.fit_generator(IGenerator(keys=P[tr_idx], \n                                   a = A, \n                                   tab = TAB), \n                        steps_per_epoch = 32,\n                        validation_data=IGenerator(keys=P[val_idx], \n                                   a = A, \n                                   tab = TAB),\n                        validation_steps = 16, \n                        callbacks = [cpt, rlp], \n                        epochs=EPOCHS)\n    folds_history.append(history.history)\n    print('Training done!')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# CV Evaluation"},{"metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-08-20T14:38:48.311214Z","iopub.status.busy":"2020-08-20T14:38:48.310434Z","iopub.status.idle":"2020-08-20T14:38:48.315566Z","shell.execute_reply":"2020-08-20T14:38:48.315096Z"},"papermill":{"duration":0.548975,"end_time":"2020-08-20T14:38:48.315673","exception":false,"start_time":"2020-08-20T14:38:47.766698","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"if SAVE_BEST:\n    mean_val_loss = np.mean([np.min(h['val_loss']) for h in folds_history])\nelse:\n    mean_val_loss = np.mean([h['val_loss'][-1] for h in folds_history])\nprint('Our mean CV MAE is: ' + str(mean_val_loss))","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}