{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q efficientnet","metadata":{"_uuid":"ac7d3b8f-556d-4f70-80c8-f57cd7a1abe3","_cell_guid":"69e493a7-666d-441f-9caf-20dbaaa45b77","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport supporting_functions as sf\nimport tensorflow.keras.layers as L\nimport efficientnet.tfkeras as efn\nimport numpy as np\nfrom sklearn.model_selection import KFold\nfrom kaggle_datasets import KaggleDatasets\n\nGCS_PATH = KaggleDatasets().get_gcs_path('siim-isic-melanoma-classification')\nTRAINING_FILENAMES = np.sort(np.array(tf.io.gfile.glob(GCS_PATH + '/tfrecords/train*.tfrec')))\n\n\nIMAGE_SIZE, BATCH_SIZE = sf.get_sizes()\nEPOCHS = 3\nFOLDS = 4\n\nprint(IMAGE_SIZE)\nprint(BATCH_SIZE)\nprint(EPOCHS)\n\ndef efiNet_model():\n    model = efn.EfficientNetB5(input_shape=(*IMAGE_SIZE, 3), weights='imagenet', include_top=False)\n    return model","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is\n    # set: this is always the case on Kaggle.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"_uuid":"f054404d-049d-42ef-9cc5-43feb8f35cff","_cell_guid":"5f9c740e-3d4e-41ee-bdb8-e8075b2aa456","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Defining the optimizer and compiling the model\nwith strategy.scope():\n    model = tf.keras.Sequential([\n        efiNet_model(),\n        L.GlobalAveragePooling2D(),\n        #L.Dense(1024, activation = 'relu'),\n        #L.Dropout(0.3) , \n        L.Dense(512, activation= 'relu'), \n        L.Dropout(0.25), \n        #L.Dense(256, activation='relu'), \n        #L.Dropout(0.2), \n        #L.Dense(128, activation='relu'), \n        #L.Dropout(0.15), \n        L.Dense(1, activation='sigmoid')\n    ])\n\n    model.compile(\n        optimizer='adam',\n        loss = 'binary_crossentropy',\n        metrics=[tf.keras.metrics.AUC()]\n    )","metadata":{"_uuid":"c5831631-894c-4d95-a05a-c2d79ef21d2a","_cell_guid":"7b73d475-7dbe-4d5b-b68f-cc81d1c9cd13","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#f = open(\"./model_summary.txt\", \"w\")\n\n#model.summary(print_fn=lambda x: f.write(x + '\\n'))\n#f.close()\n\n#model.get_layer(index=0).summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skf = KFold(n_splits=FOLDS,shuffle=True,random_state=42)\nfor fold,(idxT,idxV) in enumerate(skf.split(np.arange(16))):\n    \n    # CREATE TRAIN AND VALIDATION SUBSETS\n    files_train = tf.io.gfile.glob([GCS_PATH + '/tfrecords/train%.2i*.tfrec'%x for x in idxT])\n    \n    files_valid = tf.io.gfile.glob([GCS_PATH + '/tfrecords/train%.2i*.tfrec'%x for x in idxV])\n    \n\n# hardcoded cuz I'm a hack\nNUM_VAL_IMAGES = 33126 * (1/FOLDS)\nNUM_TRAINING_IMAGES = 33126 - NUM_VAL_IMAGES\n\n\nTRAIN_STEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE\nprint('Dataset: {} training images'.format(NUM_TRAINING_IMAGES))","metadata":{"_kg_hide-output":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fitting the model\nhistory = model.fit(\n    sf.get_training_dataset(files_train),\n    steps_per_epoch=TRAIN_STEPS_PER_EPOCH,\n    epochs=EPOCHS,\n    use_multiprocessing=False\n    #validation_data=get_validation_dataset(),\n    #validation_steps=VAL_STEPS_PER_EPOCH\n)","metadata":{"_uuid":"8aba4f30-c5d8-4585-9af2-5db82d11b8a3","_cell_guid":"4fac61a9-34ff-4492-8180-edaa0bfde2b3","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('model.h5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test_predictions = model.predict(sf.get_testing_dataset(), steps=TEST_STEPS, use_multiprocessing=True)\n#print(test_predictions)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#GCS_PATH = KaggleDatasets().get_gcs_path('siim-isic-melanoma-classification')\n#TEST_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/tfrecords/test.tfrec')\n\n#for example in tf.python_io.tf_record_iterator(\"../input/siim-isic-melanoma-classification/tfrecords/train00-2071.tfrec\"):\n#    print(tf.train.Example.FromString(example))","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}