{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport tensorflow as tf\nimport os\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"trainrecords = []\ntestrecords = []\nfor dirname, _, filenames in os.walk(\"../input/siim-isic-melanoma-classification/tfrecords\"):\n    for filename in filenames:\n        if(filename.startswith(\"train\")):\n            trainrecords.append(os.path.join(dirname, filename))\n        else:\n            testrecords.append(os.path.join(dirname, filename))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(trainrecords)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def _parse_function(record):\n    features = {\n        \"image\": tf.io.FixedLenFeature(shape=[], dtype=tf.string),\n        \"image_name\": tf.io.FixedLenFeature(shape=[], dtype=tf.string),\n        \"target\": tf.io.FixedLenFeature(shape=[], dtype=tf.int64)\n    }\n    \n    parsed_features = tf.io.parse_single_example(record, features)\n    \n    image_shape = tf.stack([1024, 1024, 3])\n    \n    image = tf.image.decode_jpeg(parsed_features[\"image\"])\n    image = tf.cast(image, tf.float32)\n    image = tf.reshape(image, image_shape)\n    image = tf.image.resize(image, (96, 96))\n    image = tf.image.rgb_to_grayscale(image)\n    mean = tf.math.reduce_mean(image)\n    image = image - mean\n    image = image/(tf.math.reduce_max(image) - tf.math.reduce_min(image))\n    \n    \n    target = parsed_features[\"target\"]\n    target = tf.cast(target, tf.float32)\n    \n    image_name = parsed_features[\"image_name\"]\n    return image, target","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"rawtrainds = tf.data.TFRecordDataset(trainrecords)\nrawtestds = tf.data.TFRecordDataset(testrecords)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"trainds = rawtrainds.map(_parse_function)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x_all = []\ny_all = []\nfor x_val, y_val in trainds.as_numpy_iterator():\n    x_all.append(x_val)\n    y_all.append(y_val)\nx_all = np.array(x_all)\ny_all = np.array(y_all)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"total = 30000\nx_train = x_all[:total]\nx_test = x_all[total:]\ny_train = y_all[:total]\ny_test = y_all[total:]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"weight_for_0 = (1 / (total - sum(y_train)))*(total)/2.0 \nweight_for_1 = (1 / sum(y_train))*(total)/2.0\n\nclass_weight = {0: weight_for_0, 1: weight_for_1}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = keras.Sequential()\nmodel.add(layers.Conv2D(32, kernel_size=5, activation=\"relu\", input_shape=(96,96,1)))\nmodel.add(layers.Flatten())\nmodel.add(layers.Dense(16, activation=\"relu\"))\nmodel.add(layers.Dense(1, activation=\"sigmoid\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.compile(optimizer='Adam', loss='binary_crossentropy', metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.fit(x_train, y_train, epochs=21, batch_size=100, class_weight=class_weight)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.evaluate(x_test, y_test, batch_size=100)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def _parse_test_function(record):\n    features = {\n        \"image\": tf.io.FixedLenFeature(shape=[], dtype=tf.string),\n        \"image_name\": tf.io.FixedLenFeature(shape=[], dtype=tf.string)\n    }\n    \n    parsed_features = tf.io.parse_single_example(record, features)\n    \n    image_shape = tf.stack([1024, 1024, 3])\n    \n    image = tf.image.decode_jpeg(parsed_features[\"image\"])\n    image = tf.cast(image, tf.float32)\n    image = tf.reshape(image, image_shape)\n    image = tf.image.resize(image, (96, 96))\n    image = tf.image.rgb_to_grayscale(image)\n    mean = tf.math.reduce_mean(image)\n    image = image - mean\n    image = image/(tf.math.reduce_max(image) - tf.math.reduce_min(image))\n    \n    image_name = parsed_features[\"image_name\"]\n    return image_name, image","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"testds = rawtrainds.map(_parse_test_function)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_test = []\nsub_image_name = []\nfor img_name, img in testds.as_numpy_iterator():\n        sub_image_name.append(img_name)\n        sub_test.append(img)\nsub_test = np.array(sub_test)\nsub_image_name = np.array(sub_image_name)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_target = model.predict(sub_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"my_submission = pd.DataFrame({'image_name': sub_image_name, 'target': sub_target.flatten()})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"my_submission.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}