{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import re\nimport os\nimport random\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\n\n#try:\n#    tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect()\n#    print(\"Device:\", tpu.master())\n#    strategy = tf.distribute.TPUStrategy(tpu)\n#except:\n#    strategy = tf.distribute.get_strategy()\n#print(\"Number of replicas:\", strategy.num_replicas_in_sync)","metadata":{"id":"XQe_NTuSCABB"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"strategy = tf.test.gpu_device_name()\nif len(strategy) > 0:\n    print(\"Found GPU at: {}\".format(strategy))\nelse:\n    strategy = \"/device:CPU:0\"\n    print(\"No GPU, using {}.\".format(strategy))","metadata":{"id":"_sCriHvfiGOf","outputId":"6bab605a-ba31-4f7e-cabb-0f250b4f6558"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gpu_info = !nvidia-smi\ngpu_info = '\\n'.join(gpu_info)\nif gpu_info.find('failed') >= 0:\n  print('Not connected to a GPU')\nelse:\n  print(gpu_info)","metadata":{"id":"OsFruE04KfBK","outputId":"8e1f7f0b-99d5-418b-cb16-5e5775a86776"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 32\nimg_height = 180\nimg_width = 180","metadata":{"id":"jR6VZEEqCPNZ"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from google.colab import drive\ndrive.mount('/content/gdrive')","metadata":{"id":"HIdhmXhdDFtD","outputId":"4042cb2b-e663-4824-eb8d-874baf1ce3c6"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds = tf.keras.utils.image_dataset_from_directory(\"/content/gdrive/MyDrive/xray\",\n                                                      validation_split=0.2,\n                                                      subset=\"training\",   \n                                                      seed=123,\n                                                      image_size=(img_height, img_width),\n                                                      batch_size=batch_size\n                                                        )\n\nval_ds = tf.keras.utils.image_dataset_from_directory(\"/content/gdrive/MyDrive/xray\",\n                                                      validation_split=0.2,\n                                                      subset=\"validation\",     \n                                                      seed=123,\n                                                      image_size=(img_height, img_width),\n                                                      batch_size=batch_size\n                                                        )\n","metadata":{"id":"erkLkucpCPPk","outputId":"fbc8794d-2f72-455b-a4ca-4bebbf890e6f"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"gRvOT9l3X4YT"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTOTUNE = tf.data.AUTOTUNE\n\ntest_ds = val_ds.take(10) \nval_ds = val_ds.skip(10)\n\ntrain_ds = train_ds.cache().prefetch(buffer_size=AUTOTUNE)\nval_ds = val_ds.cache().prefetch(buffer_size=AUTOTUNE)\ntest_ds = test_ds.cache().prefetch(buffer_size=AUTOTUNE)\n\n","metadata":{"id":"xKAaXno7CPao"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Gender Classification on TPU","metadata":{"id":"Ac7g5r0ICPeK"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTOTUNE = tf.data.AUTOTUNE\nBATCH_SIZE = 32\nIMAGE_SIZE = [180, 180]\n#CLASS_NAMES = [\"NORMAL\", \"PNEUMONIA\"]","metadata":{"id":"DOI-k0PxChp1"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Build the CNN","metadata":{"id":"IwYUJSQEChsF"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow import keras\nfrom tensorflow.keras import layers\n\n\ndef conv_block(filters, inputs):\n    x = layers.SeparableConv2D(filters, 3, activation=\"relu\", padding=\"same\")(inputs)\n    x = layers.SeparableConv2D(filters, 3, activation=\"relu\", padding=\"same\")(x)\n    x = layers.BatchNormalization()(x)\n    outputs = layers.MaxPool2D()(x)\n\n    return outputs\n\n\ndef dense_block(units, dropout_rate, inputs):\n    x = layers.Dense(units, activation=\"relu\")(inputs)\n    x = layers.BatchNormalization()(x)\n    outputs = layers.Dropout(dropout_rate)(x)\n\n    return outputs","metadata":{"id":"OJ0tfsyVChuZ"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model():\n    inputs = keras.Input(shape=(IMAGE_SIZE[0], IMAGE_SIZE[1], 3))\n    x = layers.Rescaling(1.0 / 255)(inputs)\n    x = layers.Conv2D(16, 3, activation=\"relu\", padding=\"same\")(x)\n    x = layers.Conv2D(16, 3, activation=\"relu\", padding=\"same\")(x)\n    x = layers.MaxPool2D()(x)\n\n    x = conv_block(32, x)\n    x = conv_block(64, x)\n\n    x = conv_block(128, x)\n    x = layers.Dropout(0.2)(x)\n\n    x = conv_block(256, x)\n    x = layers.Dropout(0.2)(x)\n\n    x = layers.Flatten()(x)\n    x = dense_block(512, 0.7, x)\n    x = dense_block(128, 0.5, x)\n    x = dense_block(64, 0.3, x)\n\n    outputs = layers.Dense(1, activation=\"sigmoid\")(x)\n\n    model = keras.Model(inputs=inputs, outputs=outputs)\n    return model","metadata":{"id":"ynquay7UChwe"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Train the model","metadata":{"id":"DUtZBY3NChyy"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Defining callbacks","metadata":{"id":"UCLb92wNCh08"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"checkpoint_cb = tf.keras.callbacks.ModelCheckpoint(\"xray_model.h5\", save_best_only=True)\n\nearly_stopping_cb = tf.keras.callbacks.EarlyStopping(patience=35, restore_best_weights=True)","metadata":{"id":"y977PfGVCh3P"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"initial_learning_rate = 0.015\nlr_schedule = tf.keras.optimizers.schedules.ExponentialDecay(initial_learning_rate, decay_steps=100000, decay_rate=0.96, staircase=True)","metadata":{"id":"3N5zZswBCh61"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Fit the model","metadata":{"id":"RqZELY63CGvI"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with tf.device(strategy):\n    model = build_model()\n\n    METRICS = [\n        tf.keras.metrics.AUC(name='AUC'),\n    ]\n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(learning_rate=lr_schedule),\n        loss=\"binary_crossentropy\",\n        metrics=METRICS,\n    )\n\nhistory = model.fit(\n    train_ds,\n    epochs=100,\n    validation_data=val_ds,\n    #class_weight=class_weight,\n    callbacks=[checkpoint_cb, early_stopping_cb],\n  )","metadata":{"id":"TwDoNXC0C21z","outputId":"6ebaada0-07db-45e8-80a3-6de124a46b99"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Visualizing model performance","metadata":{"id":"MXi5dC4wC238"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n","metadata":{"id":"muDkybZBZT26"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(20, 3))\nax = ax.ravel()\n\nfor i, met in enumerate([\"AUC\", \"loss\"]):\n    ax[i].plot(history.history[met])\n    ax[i].plot(history.history[met])\n    ax[i].set_title(\"Model {}\".format(met))\n    ax[i].set_xlabel(\"epochs\")\n    ax[i].set_ylabel(met)\n    ax[i].legend([\"train\", \"val\"])","metadata":{"id":"H8dhDqIFC26T","outputId":"b6b119b6-1bde-496b-cc63-f8a1e97d9d7f"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Predict and evaluate results","metadata":{"id":"3K5ykcQeC28j"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.evaluate(test_ds, return_dict=True)","metadata":{"id":"YdkbGWDbC2-8","outputId":"7e754c12-a9b7-4726-cefc-1edfc059f4cb"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#for image, label in test_ds.take(1):\n#  print(\"Image shape: \", image.numpy().shape)\n#  print(\"Label: \", label.numpy())","metadata":{"id":"7BstWSZkVt5u"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# imageId - same as the PNG filename\n# gender - 0 for female and 1 for male\nCLASS_NAMES = [\"FEMALE\", \"MALE\"]","metadata":{"id":"DUHPD2L-WCeH"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = keras.preprocessing.image_dataset_from_directory(\n    \"/content/gdrive/MyDrive/xraytest\", \n    batch_size=32, \n    image_size=(img_height, img_width), shuffle=False\n)\nfile_paths = test.file_paths\ntest = test.cache().prefetch(buffer_size=AUTOTUNE)","metadata":{"id":"szyA4-362BhQ","outputId":"3bcb25aa-b861-4722-e7bd-9dee54e0ac26"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dict = {'imageId': file_paths}\ndf = pd.DataFrame(dict)\ndf['imageId'] = df.imageId.str.split('/').str[6]\ndf['imageId'] = df.imageId.str.split('.').str[0]\ndf['imageId'] = df['imageId'].astype(int)\ndf.head()","metadata":{"id":"XKLpn08_3TiD","outputId":"78f60b58-914f-4c5a-c6f4-147ecf43d421"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = model.predict(test, batch_size=batch_size)\n#preds = preds.round(2)\npreds = np.squeeze(preds)","metadata":{"id":"cTZaCmKZey4v","outputId":"cdf35ed6-17c5-4c4f-ee7f-ec3ec2e30c49"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['gender'] = preds","metadata":{"id":"BSZXW-Y4_Rfz"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"id":"b8nsoSxBltdW","outputId":"dee0bcad-9dbf-419c-eb7a-852b1169c5c1"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def label_classes(pred_probs, threshold=0.5):\n    \"\"\"\n    Takes in a list of predicted probabilities and returns a list of class labels.\n    The class labels are determined by comparing each probability to a threshold value.\n    If the probability is greater than or equal to the threshold, the label is 1. Otherwise, it is 0.\n    Args:\n        pred_probs (list): A list of predicted probabilities.\n        threshold (float): The probability threshold for determining the class label. Default is 0.5.\n    Returns:\n        A list of class labels (0 or 1) corresponding to each predicted probability.\n    \"\"\"\n    class_labels = []\n    for prob in pred_probs:\n        if prob >= threshold:\n            class_labels.append(1)\n        else:\n            class_labels.append(0)\n    return class_labels\n\ndf['gender'] = label_classes(df['gender'], threshold=0.5)","metadata":{"id":"VtEZzTjnBJAL"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('/content/gdrive/MyDrive/submission_gender.csv', index=False)","metadata":{"id":"AEN37B4NAori"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"id":"D0yKdmhLZsb3","outputId":"f633559d-a9ed-40e1-d431-3de058b37d9d"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for image, label in test.take(1):\n  for j in range(0,30):\n    prediction = model.predict(test_ds.take(1))[j]\n    scores = [1 - prediction, prediction]\n    for score, name in zip(scores, CLASS_NAMES):\n        print(\"This image is %.2f percent %s\" % ((100 * score), name))\n    plt.imshow(image[j] / 255.0)\n    plt.title(CLASS_NAMES[label[j].numpy()])\n    plt.show()","metadata":{"id":"NyGMVDy9C3Ax","outputId":"c767c9ca-75cb-4735-c42a-d4bbfd72fb9c"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"roXBSlQiC3Ea"},"execution_count":null,"outputs":[]}]}