{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Introduction\nensembles models are multiple models grouped together with their result generalised at the end. RandomForest is one of the most popular example of ensemble. the general architecture of the ensemble is shown in the following figure\n\n![general architecture ensemble model](https://www.researchgate.net/publication/324552457/figure/fig3/AS:616245728645121@1523935839872/An-example-scheme-of-stacking-ensemble-learning.png)\n\nnow you can see their are two levels in the ensemble model\n1. Level 0 -- this level is a stack of all the models which have been trained on the training                   data. \n2. Level 1 -- at this the prediction of each model are taken and then generalized. there are two               ways of generalizing.<br>\na) by aggregating all the prediction like average , max etc.<br>\nb) by training another linear model on these predictions with respect to actual labels <br>\nwe will be using approach b. in this notebook\n              \nbased on the stacked models there are two categories on ensemble models\n1. homogenous models -- in these models all the modesl in level 0 have the same architecture and                         they will of same type but they will be trained on different samples of                           training data.\n2. heterogenous models -- in these models all the models in level 0 are of types or have a                                 different architecture as shown in the figure below\n\n![hetrogenous model](https://blogs.sas.com/content/subconsciousmusings/files/2017/05/modelstacking.png)\n\nHowever in this notebook we will only look at the homogenous model. if there's time I will publish a seperate notebook explaining the heterogenous models","execution_count":null},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import tensorflow as tf\nfrom kaggle_datasets import KaggleDatasets\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow_addons as tfa\nfrom tensorflow.keras.layers import Dense , Dropout , BatchNormalization , concatenate\nfrom tensorflow.keras.layers import Activation , Input , GlobalAveragePooling2D  \nfrom tensorflow.keras.models import Model , Sequential , load_model\nfrom tensorflow.keras.applications import InceptionV3 , MobileNetV2\nfrom sklearn.model_selection import KFold\nfrom sklearn.linear_model import LogisticRegression\nimport re\nfrom PIL import Image","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# utility Functions\nagain first lets have some of the utility functions","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"IMG_DIMS = 64 # inception net has a minimum input size of 75x75 thus 150 is good\nCHANNELS = 3\nBATCH_SIZE = 32\nSEED = 42\nSPLITS = 5\n\nAUTO  = tf.data.experimental.AUTOTUNE","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# getting training data gcs path\nGCS_PATH = KaggleDatasets().get_gcs_path('melanoma-256x256')\ntrain_datasets = tf.io.gfile.glob(GCS_PATH + '/train*.tfrec')\n\nprint('number of TFRecords in train : ',len(train_datasets))\n\n# getting testing data\nGCS_PATH = KaggleDatasets().get_gcs_path('siim-isic-melanoma-classification')\ntest_datasets = tf.io.gfile.glob(GCS_PATH + '/tfrecords/test*.tfrec')\n\nprint('number of TFRecords in test : ' ,len(test_datasets))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# parse data from TF Records\ndef parse_TFR_data_labelled(sample):\n    features = {\n      'image': tf.io.FixedLenFeature([] , tf.string , default_value = ''),\n      'image_name': tf.io.FixedLenFeature([] , tf.string , default_value=''),\n      'patient_id': tf.io.FixedLenFeature([] , tf.int64 , default_value=0),\n      'sex': tf.io.FixedLenFeature([] , tf.int64 , default_value=0),\n      'age_approx': tf.io.FixedLenFeature([] , tf.int64 , default_value=0),\n      'anatom_site_general_challenge':tf.io.FixedLenFeature([] ,tf.int64 , default_value=0 ),\n      'diagnosis': tf.io.FixedLenFeature([] ,tf.int64 , default_value=0 ),\n      'target': tf.io.FixedLenFeature([] ,tf.int64 , default_value=0 ),\n      'width': tf.io.FixedLenFeature([] ,tf.int64 , default_value=0 ),\n      'height': tf.io.FixedLenFeature([] ,tf.int64 , default_value=0 )\n    }\n    \n    p = tf.io.parse_single_example(sample , features)\n    \n    img = p['image']\n    target = p['target']\n    \n    return img , target","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# decode img\ndef decode_image(img , IMG_DIMS):\n    img = tf.image.decode_jpeg(img, channels=3)\n    img = tf.image.resize(img , [IMG_DIMS , IMG_DIMS])\n    return img","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# load data set for training and validation\ndef _get_ds(files , train=True , repeat=True , img_dims=64 , batch_size=32):\n    ds = tf.data.TFRecordDataset(files , num_parallel_reads=AUTO)\n    ds = ds.cache()\n    \n    if repeat:\n        ds = ds.repeat()\n    ds = ds.map(parse_TFR_data_labelled , num_parallel_calls=AUTO)     \n    ds = ds.map(lambda img ,label: (decode_image(img, img_dims),label) , num_parallel_calls=AUTO)\n    if train:\n        ds = ds.shuffle(buffer_size=1000)\n       \n    ds = ds.batch(batch_size*REPLICAS)\n    ds = ds.prefetch(AUTO)\n    return ds","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def count_data_items(filenames):\n    n = [int(re.compile(r\"-([0-9]*)\\.\").search(filename).group(1)) \n         for filename in filenames]\n    return np.sum(n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# preparing the testing data\ndef parsed_TFR_unlabelled_2(sample):\n    feature_description = {\n        'image': tf.io.FixedLenFeature([], tf.string, default_value=''),\n        'image_name': tf.io.FixedLenFeature([], tf.string, default_value=''),\n        'target': tf.io.FixedLenFeature([], tf.int64, default_value=0),\n    }\n    p = tf.io.parse_single_example(sample , feature_description)\n    img = p['image']\n    name = p['image_name']\n    return name , img\n\ntest_data= tf.data.TFRecordDataset(test_datasets)\ntest_data = test_data.map(parsed_TFR_unlabelled_2 , num_parallel_calls=AUTO)\ntest_data = test_data.map(lambda name , img: (name , decode_image(img , 64)))\n\nsub_df = pd.read_csv('../input/siim-isic-melanoma-classification/sample_submission.csv')\n\nx_dict = {}\nfor p in test_data:\n    temp = {p[0].numpy().decode() : p[1].numpy()}\n    x_dict.update(temp)\n    \nprint(f'number of samples in testing data : {len(x_dict)}')\n\ntest = []\nfor i in sub_df['image_name']:\n    test.append(x_dict[i])\n    del(x_dict[i])\n    \ntest = np.array(test)\nprint(f'sahpe of testing set sorted according to the submission file : {test.shape}')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Level 0\n\nnow we will use MobileNetV2 as our base model and create an inference model over it that create different instances of this model and save those modesls.. we will be using KFold cross validation technique\n\nso first lets create our model","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def create_Inf_model(IMG_DIMS , CHANNELS):\n    in_put = Input(shape=(IMG_DIMS , IMG_DIMS , 3))\n    # applying auggumentations\n    #pre = aug(in_put)\n    # pre process layer\n    pre_process_layer = tf.keras.applications.mobilenet_v2.preprocess_input\n    pre = pre_process_layer(in_put)\n    # base model non trainable\n    base_model = MobileNetV2(input_shape=(IMG_DIMS , IMG_DIMS , 3) , include_top=False , weights='imagenet')\n    x = base_model(pre , training = False)\n    # top trainable model layers\n    x = GlobalAveragePooling2D()(x)\n    x = Dense(128 , activation = 'relu')(x)\n    x = Dropout(0.3)(x)\n    x = Dense(1 , activation = 'sigmoid')(x)\n    model = Model(inputs=in_put , outputs=x)\n    # optimizer\n    opt = tf.keras.optimizers.Adam(0.0001)\n    model.compile(optimizer=opt , loss='binary_crossentropy' , metrics=['accuracy' , 'AUC'])\n    return model","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"now lets initialize our TPU","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"TPU = tf.distribute.cluster_resolver.TPUClusterResolver()\ntf.config.experimental_connect_to_cluster(TPU)\ntf.tpu.experimental.initialize_tpu_system(TPU)\nstrategy = tf.distribute.experimental.TPUStrategy(TPU)\n\nREPLICAS = strategy.num_replicas_in_sync","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"to train these ensemble model I will be using the same KFold technique that I used in KFold with efficientnet notebook and transfer learning notebook","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"kf = KFold(n_splits=SPLITS)\noof_hist = []\noof_val = []\nfor f , (idxT , idxV) in enumerate(kf.split(train_datasets)):\n    train = []\n    val =[]\n    for idx in idxT:\n        train.append(train_datasets[idx])\n    for idx in idxV:\n        val.append(train_datasets[idx])\n\n    # instantiate model\n    with strategy.scope():\n         # cretae model check points\n        cp_callback = tf.keras.callbacks.ModelCheckpoint(filepath='model_fold_'+str(f)+'_weights.hdf5' , \n                                                         monitor='val_auc',\n                                                         mode='max',\n                                                         save_best_only =True,\n                                                         verbose = 1 )\n        # early stopping\n        es_callback = tf.keras.callbacks.EarlyStopping(monitor='val_auc' , patience=5 , mode='max' )\n        model = create_Inf_model(IMG_DIMS , CHANNELS)\n        \n        history = model.fit(_get_ds(train) ,\n                        epochs = 20 ,\n                        steps_per_epoch = count_data_items(train)/BATCH_SIZE//REPLICAS,\n                        validation_data=_get_ds(val , train=False , repeat=False),\n                        callbacks = [cp_callback , es_callback],\n                        verbose = 0)\n        model.save('model_'+str(f)+'_.hdf5')\n    oof_hist.append(history)\n    oof_val.append(_get_ds(val, train=False))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Level 1 -- Generalization\nnow what we are going to do is to first prepare the training set for these models to predict on it so than we will create the following function \n* level0_predict\n* create_stacked_dataset\n* ensemsble_predict\n\n\nwe will use the stacked dataset to create a dataset which will be use to by the aggregator to fit and generalize and to make a prediction\n\nnow prepare the training data for prediction","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"\ntrain_data= tf.data.TFRecordDataset(train_datasets)\ntrain_data = train_data.map(parse_TFR_data_labelled , num_parallel_calls=AUTO)\ntrain_data = train_data.map(lambda img , target: (decode_image(img , 64) , target))\n\nimg = []\nlabel = []\nfor sample in train_data:\n    i = sample[0].numpy()\n    t = sample[1].numpy()\n    img.append(i)\n    label.append(t)\n    \nimg = np.array(img)\nlabel = np.array(label)\n\nprint(f'shape of images : {img.shape}')\nprint(f'shape of labels : {label.shape}')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"now lets create level0_predict function<br>\nbut first lets create a list of all our memeber models","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"file_names = ['model_'+str(f)+'_.hdf5' for f in range(5)]\nmember_models = [load_model(m) for m in file_names]\nmember_models","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# level0_predict function\ndef level0_predict(m_models , train_x):\n    predictions = []\n    for model in m_models:\n        p = model.predict(train_x)\n        predictions.append(p)\n    return predictions","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"now lets created the stacked dataset function","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def stacked_set(predictions):\n    X = None\n    for p in predictions:\n        if X is None:\n            X = p\n        else:\n            X = np.dstack((X , p))\n    X = X.reshape(X.shape[0] , X.shape[1]*X.shape[2])\n    return X","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"the fit_aggregator function","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def fit_agg(X , Y):\n    model = Sequential([\n        Dense(3 , activation='relu'),\n        Dense(1 , activation ='sigmoid')\n    ])\n    model.compile(optimizer='adam' ,loss='mean_squared_error' , metrics=['accuracy'])\n    model.fit(X,Y , epochs = 10)\n    return model","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"the final ensemble predict function","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def ensemble_fit(X_train , Y , X_test , member_models):\n    # get prediction of each sub model\n    preds = level0_predict(member_models , X_train)\n    # prepare the stacked set\n    X = stacked_set(preds)\n    # fit aggregator\n    model = fit_agg(X,Y)\n    return model\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def ensemble_predict(agg, member_models , data ):\n    # get the sub model predictions\n    preds = level0_predict(member_models , data)\n    # prepare the stacked set\n    X = stacked_set(preds)\n    # final prediction\n    result = agg.predict(X)\n    return result","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"now our Level 1 is ready lets use it\n\nensemble fit","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"final_model = ensemble_fit(img , label , test , member_models)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"final prediction","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"result = ensemble_predict(final_model , member_models , test)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"now lets make a submission","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"result","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_df['target'] = result\nsub_df.set_index('image_name' , inplace=True)\nsub_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_df.to_csv('submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}