{"cells":[{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5"},"cell_type":"markdown","source":"# Introduction\nFor this sort of a (image) classification task, it is worthwhile knowing whether the distribution of the train data is somewhat similar to that of the test data. If they are very different, a model fitted to the train data cannot do much job for classifying the test data. To see whether distributions of the train and test is similar, we can do so-called **adversarial validation**. \n\nThis kernel is largely based on this kernel (https://www.kaggle.com/cdeotte/steel-adversarial-validation) which did the adversarial classification for the steel competition."},{"metadata":{},"cell_type":"markdown","source":"# Libraries"},{"metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"cell_type":"code","source":"import os, glob\nimport random\nfrom PIL import Image \nimport numpy as np\nimport pandas as pd\nimport multiprocessing\nimport keras\nimport keras.backend as K\nfrom keras.optimizers import Adam\nfrom keras.callbacks import Callback\nfrom keras.applications.densenet import DenseNet169\nfrom keras.layers import Dense, Flatten\nfrom keras.models import Model, load_model\nfrom keras.utils import Sequence\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras import layers\nfrom keras.callbacks import LearningRateScheduler\nimport matplotlib.pyplot as plt, time\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm_notebook as tqdm\nfrom numpy.random import seed\nseed(1220)\nfrom tensorflow import set_random_seed\nset_random_seed(1220)\n%matplotlib inline\nprint(\"libraries imported!\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true,"_kg_hide-output":true},"cell_type":"code","source":"test_imgs_folder = '../input/understanding_cloud_organization/test_images/'\ntrain_imgs_folder = '../input/understanding_cloud_organization/train_images/'\nnum_cores = multiprocessing.cpu_count()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# train : test = 1: 1"},{"metadata":{"trusted":true},"cell_type":"code","source":"# based on https://www.kaggle.com/cdeotte/steel-adversarial-validation/data\nTRAIN_IMG = os.listdir('../input/understanding_cloud_organization/train_images')\nTEST_IMG = os.listdir('../input/understanding_cloud_organization/test_images')\nprint('Original train count =',len(TRAIN_IMG),', Original test count =',len(TEST_IMG))\nos.mkdir('../tmp/')\nos.mkdir('../tmp/train_images/')\nr = np.random.choice(TRAIN_IMG,len(TEST_IMG),replace=False)\nfor i,f in enumerate(r):\n    img = Image.open('../input/understanding_cloud_organization/train_images/'+f)\n    img.save('../tmp/train_images/'+f)\nos.mkdir('../tmp/test_images/')\nfor i,f in enumerate(TEST_IMG):\n    img = Image.open('../input/understanding_cloud_organization/test_images/'+f)\n    img.save('../tmp/test_images/'+f)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"TRAIN_IMG_AV = os.listdir('../tmp/train_images')\nTEST_IMG_AV = os.listdir('../tmp/test_images')\nprint('New train count =',len(TRAIN_IMG_AV),', New test count =',len(TEST_IMG_AV))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"\n# Build an Adversarial Classifier\nI use a simple DenseNet169 as an adversarial classifier."},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_model():\n    K.clear_session()\n    base_model = DenseNet169(weights='imagenet', include_top=False, pooling='avg', input_shape=(224, 224, 3))\n    x = base_model.output\n    y_pred = Dense(1, activation='sigmoid')(x)\n    return Model(inputs=base_model.input, outputs=y_pred)\n\nmodel = get_model()\nmodel.compile(optimizer=Adam(lr=1e-4), loss='binary_crossentropy', metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img_dir = '../tmp/'\nimg_height = 224; img_width = 224\nbatch_size = 32; nb_epochs = 8\n\ntrain_datagen = ImageDataGenerator(rescale=1./255,\n    horizontal_flip=True,\n    vertical_flip=True,\n    validation_split=0.2) # set validation split\n\ntrain_generator = train_datagen.flow_from_directory(\n    img_dir,\n    target_size=(img_height, img_width),\n    batch_size=batch_size,\n    class_mode='binary',\n    subset='training') # set as training data\n\nvalidation_generator = train_datagen.flow_from_directory(\n    img_dir, # same directory as training data\n    target_size=(img_height, img_width),\n    batch_size=batch_size,\n    class_mode='binary',\n    subset='validation') # set as validation data\n\nannealer = LearningRateScheduler(lambda x: 0.0001 * 0.95 ** x)\n\n# fit!\nh = model.fit_generator(\n    train_generator,\n    steps_per_epoch = train_generator.samples // batch_size,\n    validation_data = validation_generator, \n    validation_steps = validation_generator.samples // batch_size,\n    epochs = nb_epochs,\n    callbacks = [annealer],\n    verbose=2)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Accuracy\nIf distributions of the train and test are similar to one another, the classification accuracy for validation should be around 0.5."},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(15,5))\nplt.plot(h.history['acc'],label='Train ACC')\nplt.plot(h.history['val_acc'],label='Val ACC')\nplt.title('TRAIN COMPARED WITH TEST. Training History')\nplt.legend()\nplt.show()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.7"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"22d16de5dc5140928a6c0d581b8e3a3b":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"DescriptionStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"417ad312018e44dfad8151c6be6e9331":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"6ef1f8163f6040268ece907cf7b15b03":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"b3fa440860214293b4d3c9c04fadc9ac":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"b93962a03ebf4c868a8b9f8e8a804dc0":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_6ef1f8163f6040268ece907cf7b15b03","placeholder":"​","style":"IPY_MODEL_22d16de5dc5140928a6c0d581b8e3a3b","value":"0/|/| 0/? [00:00&lt;?, ?it/s]"}},"bd48f48b47df48af8d3ef36314ecdf91":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_e67d77e01ab64050a9016af3b1502206","IPY_MODEL_b93962a03ebf4c868a8b9f8e8a804dc0"],"layout":"IPY_MODEL_417ad312018e44dfad8151c6be6e9331"}},"e0b948daaa3e4fcf96acfb58c17fc67d":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"e67d77e01ab64050a9016af3b1502206":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"IntProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"IntProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"ProgressView","bar_style":"danger","description":"","description_tooltip":null,"layout":"IPY_MODEL_b3fa440860214293b4d3c9c04fadc9ac","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_e0b948daaa3e4fcf96acfb58c17fc67d","value":1}}},"version_major":2,"version_minor":0}}},"nbformat":4,"nbformat_minor":1}