{"cells":[{"metadata":{},"cell_type":"markdown","source":"## Preprocessing SIIM-ISIC Melanoma Classification Data","execution_count":null},{"metadata":{"_uuid":"a793c902-66f5-4052-860b-8744f4349792","_cell_guid":"636d29ef-4676-43aa-93da-69d50ef19fc6","trusted":true},"cell_type":"code","source":"import tensorflow as tf","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"DIR_PATH = \"../input/siim-isic-melanoma-classification\"\nJPEG_PATH = f\"{DIR_PATH}/jpeg/train\"\nIMAGE_SIZE = 256\nN_FILES = 25\nDATA_CARD = 33126","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def _serialize_example(image, target):\n    \"\"\"Writes the image and target to a protobuf.\"\"\"\n    feature = {\n        \"image\": tf.train.Feature(\n            bytes_list=tf.train.BytesList(value=[image.numpy()])),\n        \"target\": tf.train.Feature(int64_list=tf.train.Int64List(value=[target]))\n    }\n    \n    example_proto = tf.train.Example(features=tf.train.Features(feature=feature))\n    return example_proto.SerializeToString()\n                                   \ndef load_resize_and_serialize(image_name, target):\n    image_file = tf.strings.join([image_name, \"jpg\"], separator=\".\")\n    \n    read_image = tf.io.read_file(\n        tf.strings.join([JPEG_PATH, image_file], separator=\"/\"))\n    image = tf.io.decode_jpeg(read_image, channels=3)\n                                                # h         # w\n    image = tf.image.resize_with_pad(image, IMAGE_SIZE, IMAGE_SIZE,\n                                     method=\"lanczos5\", antialias=True)\n    serialized_image = tf.io.serialize_tensor(tf.cast(image, tf.uint8))\n    \n    return tf.py_function(_serialize_example, (serialized_image, target), tf.string)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_pipe = tf.data.experimental.CsvDataset(\n    f\"{DIR_PATH}/train.csv\",\n    [tf.constant([], dtype=tf.string), tf.constant([], dtype=tf.int64)],\n    header=True, select_cols=[0, 7]\n).shuffle(\n    10000, seed=1236, reshuffle_each_iteration=False\n).map(\n    load_resize_and_serialize,\n    num_parallel_calls=tf.data.experimental.AUTOTUNE,\n    deterministic=False\n).window(DATA_CARD // N_FILES, drop_remainder=True\n).enumerate(\n).prefetch(tf.data.experimental.AUTOTUNE)\n\nfor n, data in data_pipe:\n    if n < 20:\n        file_name = tf.strings.format(\"train_{}_{}.tfrecord\", (n % 5, n % 4))\n    else:\n        file_name = tf.strings.format(\"valid_{}.tfrecord\", n % 5)\n    \n    tf.data.experimental.TFRecordWriter(\n        file_name, compression_type=\"GZIP\"\n    ).write(data)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}