{"cells":[{"metadata":{},"cell_type":"markdown","source":"V2: 无数据泄露的分层KFOLD划分         \nV3: mask_contour数据增强的debug         \nV4: 输出训练集和public测试集的mask_contour,但没有划分CV"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nfrom sklearn.cluster import KMeans\nfrom tqdm import tqdm\nfrom sklearn.preprocessing import LabelEncoder\nfrom collections import Counter, defaultdict\nfrom sklearn.utils import check_random_state","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Params"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"seed = 35#32\nnfold = 5","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Augmentation of lung mask"},{"metadata":{"trusted":true},"cell_type":"code","source":"# First, installing the dependencies\n!pip install -U ../input/kerasapplications/Keras_Applications-1.0.8-py3-none-any.whl\n!pip install ../input/qubvel/efficientnet-1.0.0-py3-none-any.whl\n!pip install ../input/qubvel/image_classifiers-1.0.0-py3-none-any.whl\n\n# Now, installing segmentation_models (short for 'sm')\n!pip install ../input/qubvel-segmentation-model-keras-v101/segmentation_models-master\n\n# sm can work with both Keras and Tensorflow.\n# By default, it look for keras.\n# But, with Keras, it's giving error during the import. \n# So, we will be using Tensorflow as the backend for sm.\n%env SM_FRAMEWORK=tf.keras","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport cv2\nimport os\nimport matplotlib.pyplot as plt\n\nfrom keras import backend as K\nfrom keras.callbacks import CSVLogger, ModelCheckpoint, LearningRateScheduler, EarlyStopping, ReduceLROnPlateau\n\nimport keras\nimport json\nfrom tqdm import tqdm\nfrom segmentation_models.losses import bce_jaccard_loss\nfrom segmentation_models.metrics import iou_score\nimport gc\nfrom segmentation_models import Unet\nimport segmentation_models  as sm\nfrom sklearn.model_selection import train_test_split\nfrom keras.utils import Sequence\nfrom keras.optimizers import Adam\nimport warnings\nwarnings.filterwarnings('ignore')\nprint(os.listdir('../input'))\n\nOUTPUT_DIR = './'\nif not os.path.exists(OUTPUT_DIR):\n    os.makedirs(OUTPUT_DIR)\n    \nfrom pathlib import Path\nimport ast\nfrom PIL import Image  ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ctr = pd.read_csv('../input/ranzcr-clip-lung-contours/RANZCR_CLiP_lung_contours.csv')\ntrain = pd.read_csv('../input/ranzcr-clip-catheter-line-classification/train.csv')\ntest = pd.read_csv('../input/ranzcr-clip-catheter-line-classification/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"DIMENSION =  (512, 512) #(128, 128)  #\nIMG_HEIGHT, IMG_WIDTH = DIMENSION\n\nBACKBONE = 'seresnet34'\nTRAIN_PATH = '../input/ranzcr-clip-catheter-line-classification/train/'\nTEST_PATH = '../input/ranzcr-clip-catheter-line-classification/test/'\n# all_images = os.listdir(TRAIN_PATH)\n# all_images = [Path(e).stem for e in all_images]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_mask(StudyInstanceUID):\n    img = cv2.imread(TRAIN_PATH+StudyInstanceUID+'.jpg',1)\n    ctr_left = ast.literal_eval(ctr.loc[ctr.StudyInstanceUID==StudyInstanceUID,'left_lung_contour'].values[0])\n    ctr_right = ast.literal_eval(ctr.loc[ctr.StudyInstanceUID==StudyInstanceUID,'right_lung_contour'].values[0])\n    img = cv2.drawContours(img, np.array([[np.array(x) for x in ctr_left]]), 0, (255,255,255), 1)\n    img = cv2.drawContours(img, np.array([[np.array(x) for x in ctr_right]]), 0, (255,255,255), 1)\n#     img = np.where(img>=255, 1.0, 0.0)\n    return img","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"StudyInstanceUID = '1.2.826.0.1.3680043.8.498.10000428974990117276582711948006105617'# train.loc[0,'StudyInstanceUID']\nimg = cv2.imread(TRAIN_PATH+StudyInstanceUID+'.jpg',1)\nmask_img = load_mask(StudyInstanceUID)\nfix, ax = plt.subplots(2,2, figsize=(10,10))\nfor i in range(2):\n    ax[i,0].imshow(img)\n    ax[i,1].imshow(mask_img)\n    ax[i,0].axis('off')\n    ax[i,1].axis('off')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.imshow(cv2.imread('../input/ranzcr-clip-lung-contours/train_lung_masks/train_lung_masks/1.2.826.0.1.3680043.8.498.10000428974990117276582711948006105617.jpg'))\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from time import time","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"start = time()\nfor tra_idx in tqdm(range(len(train))):\n    StudyInstanceUID = train.loc[tra_idx,'StudyInstanceUID']\n    #img = cv2.imread(TRAIN_PATH+StudyInstanceUID+'.jpg',1)\n    mask_img = load_mask(StudyInstanceUID)\n    cv2.imwrite('./train_ctrlung/'+StudyInstanceUID+'.jpg', mask_img)\nprint('draw contours time =',time()-start)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = Unet(backbone_name=BACKBONE, encoder_weights='imagenet', activation='sigmoid', classes=1, input_shape=(IMG_HEIGHT, IMG_WIDTH, 3))\n\nmodel.load_weights('../input/unet-lung-mask-512/unet_lungseg_512_12-0.075.hdf5')\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def mask2ctr(img,im):\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    img = 1 - img.astype(np.float32)\n    img = np.where(img>=0.5, 1, 0) \n    img = img.astype(np.uint8)\n    img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n    #腐蚀\n    kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (2, 2))\n    ctr = cv2.erode(img, kernel)\n    \n    ctr = img - ctr\n    ctr=Image.fromarray(ctr) \n    ctr = 255 * np.array(ctr).astype('uint8')\n    ctr = cv2.cvtColor(ctr, cv2.COLOR_GRAY2RGB)\n    \n    im = im.astype(np.float32) * 255\n    ctr_img = ctr + im[0,:,:,:]\n    ctr_img = ctr_img.astype(np.float32) / 255\n    \n    return ctr,ctr_img\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from PIL import Image  \n\nSIUID = test.loc[1234,'StudyInstanceUID']\nim = cv2.imread(TEST_PATH+SIUID+'.jpg') \nim = cv2.cvtColor(im, cv2.COLOR_BGR2RGB)\nim = im.astype(np.float32) # / 255.\nim = cv2.resize(im, dsize=(IMG_WIDTH, IMG_HEIGHT), interpolation=cv2.INTER_LANCZOS4)\nim1 = (im - np.min(im)) / (np.max(im) - np.min(im))\nim = im1.reshape(1,IMG_WIDTH, IMG_HEIGHT, -1)\n\ny_hat = model.predict(im)\ny_hat = y_hat.squeeze(0)\n_, ctr_img = mask2ctr(y_hat,im)\n\n# fix, ax = plt.subplots(1,2, figsize=(20,10))\n# ax[0,0].imshow(im)\n# ax[0,1].imshow(ctr)\n# plt.show()    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(im.mean())\nplt.imshow(im[0,:,:,:])\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# im = im.astype(np.float32) * 255\n# ctr_img = ctr + im[0,:,:,:]\n# ctr_img = ctr_img.astype(np.float32) / 255\n\n# ctr =  ctr.astype(np.float32)  / 255.\n# ctr_img = ctr + im[0,:,:,:]\n# ctr_img = (ctr_img - np.min(ctr_img)) / (np.max(ctr_img) - np.min(ctr_img))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(ctr_img.mean())\nplt.imshow(ctr_img)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for idx in tqdm(range(len(test))):\n    SIUID = test.loc[idx,'StudyInstanceUID']\n    im = cv2.imread(TEST_PATH+SIUID+'.jpg',-1)\n    im = cv2.cvtColor(im, cv2.COLOR_BGR2RGB)\n    im = im.astype(np.float32) # / 255.\n    im = cv2.resize(im, dsize=(IMG_WIDTH, IMG_HEIGHT), interpolation=cv2.INTER_LANCZOS4)\n    im = (im - np.min(im)) / (np.max(im) - np.min(im))\n    im = im.reshape(1,IMG_WIDTH, IMG_HEIGHT, -1)\n    y_hat = model.predict(im)\n    y_hat = y_hat.squeeze(0)\n    _, y_ctr = mask2ctr(y_hat,im)\n    cv2.imwrite('./test_ctrlung/'+SIUID+'.jpg', y_ctr)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Split folds\nIdeas:\n1. make sure that the labels are stratified\n2. one patient's images are grouped in one fold"},{"metadata":{"trusted":true},"cell_type":"code","source":"class RepeatedStratifiedGroupKFold():\n\n    def __init__(self, n_splits=5, n_repeats=1, random_state=None):\n        self.n_splits = n_splits\n        self.n_repeats = n_repeats\n        self.random_state = random_state\n        \n    def split(self, X, y=None, groups=None):\n        k = self.n_splits\n        def eval_y_counts_per_fold(y_counts, fold):\n            y_counts_per_fold[fold] += y_counts\n            std_per_label = []\n            for label in range(labels_num):\n                label_std = np.std(\n                    [y_counts_per_fold[i][label] / y_distr[label] for i in range(k)]\n                )\n                std_per_label.append(label_std)\n            y_counts_per_fold[fold] -= y_counts\n            return np.mean(std_per_label)\n            \n        rnd = check_random_state(self.random_state)\n        for repeat in range(self.n_repeats):\n            labels_num = np.max(y) + 1\n            y_counts_per_group = defaultdict(lambda: np.zeros(labels_num))\n            y_distr = Counter()\n            for label, g in zip(y, groups):\n                y_counts_per_group[g][label] += 1\n                y_distr[label] += 1\n\n            y_counts_per_fold = defaultdict(lambda: np.zeros(labels_num))\n            groups_per_fold = defaultdict(set)\n        \n            groups_and_y_counts = list(y_counts_per_group.items())\n            rnd.shuffle(groups_and_y_counts)\n\n            for g, y_counts in sorted(groups_and_y_counts, key=lambda x: -np.std(x[1])):\n                best_fold = None\n                min_eval = None\n                for i in range(k):\n                    fold_eval = eval_y_counts_per_fold(y_counts, i)\n                    if min_eval is None or fold_eval < min_eval:\n                        min_eval = fold_eval\n                        best_fold = i\n                y_counts_per_fold[best_fold] += y_counts\n                groups_per_fold[best_fold].add(g)\n            \n            all_groups = set(groups)\n            for i in range(k):\n                train_groups = all_groups - groups_per_fold[i]\n                test_groups = groups_per_fold[i]\n\n                train_indices = [i for i, g in enumerate(groups) if g in train_groups]\n                test_indices = [i for i, g in enumerate(groups) if g in test_groups]\n\n                yield train_indices, test_indices","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# let's first concat all the labels \n# e.g 00000000010   \ntarget_cols = train.drop(['StudyInstanceUID', 'PatientID'],axis=1).columns.values.tolist()\ntargets = train[target_cols].astype(str)\n# create a new col to store the label\ntrain['combined_tar'] = ''\nfor i in tqdm(range(targets.shape[1])):\n    train['combined_tar'] += targets.iloc[:,i]\n# take a look at it\ntrain.combined_tar.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['combined_tar'] = LabelEncoder().fit_transform(train['combined_tar'])\nlen(train['combined_tar'].unique())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"target_cols","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"targets","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X=train.values\ny=targets.values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['fold'] = -1\nrskf = RepeatedStratifiedGroupKFold(n_splits=nfold, random_state=seed)\nfor i, (train_idx, valid_idx) in enumerate(rskf.split(train, train.combined_tar, train.PatientID)): #(df, targets, group)\n    train.loc[valid_idx, 'fold'] = int(i)\n\n# !pip install /kaggle/input/iterative-stratification/iterative-stratification-master/\n# from iterstrat.ml_stratifiers import MultilabelStratifiedKFold\n# mskf = MultilabelStratifiedKFold(n_splits=nfold, random_state=seed)\n# for i, (train_idx, valid_idx) in enumerate(mskf.split(X,y)): #(df, targets, group)\n#     train.loc[valid_idx, 'fold'] = int(i)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Sanity Check\nYou wanna make sure this split makes sense. We can do that by checking the stratification and groups."},{"metadata":{"trusted":true},"cell_type":"code","source":"train.query('fold==0').combined_tar.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.query('fold==1').combined_tar.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.query('fold==2').combined_tar.value_counts()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"It seems that the label is very nicely stratified. Now let's check groups."},{"metadata":{"trusted":true},"cell_type":"code","source":"np.intersect1d(train.query('fold==0').PatientID.unique(), train.query('fold==1').PatientID.unique())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"np.intersect1d(train.query('fold==1').PatientID.unique(), train.query('fold==2').PatientID.unique())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"np.intersect1d(train.query('fold==2').PatientID.unique(), train.query('fold==3').PatientID.unique())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"No patient has appeared in two folds."},{"metadata":{},"cell_type":"markdown","source":"## Save final CSV"},{"metadata":{"trusted":true},"cell_type":"code","source":"train.drop('combined_tar', axis=1)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.drop('combined_tar', axis=1).to_csv('train_ctrlung_rskfolds.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}