{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom kaggle_datasets import KaggleDatasets\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom sklearn.model_selection import GroupKFold","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"def auto_select_accelerator():\n    try:\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n        print(\"Running on TPU:\", tpu.master())\n    except ValueError:\n        strategy = tf.distribute.get_strategy()\n    print(f\"Running on {strategy.num_replicas_in_sync} replicas\")\n    \n    return strategy\n\n\ndef build_decoder(with_labels=True, target_size=(256, 256), ext='jpg'):\n    def decode(path):\n        file_bytes = tf.io.read_file(path)\n\n        if ext == 'png':\n            img = tf.image.decode_png(file_bytes, channels=3)\n        elif ext in ['jpg', 'jpeg']:\n            img = tf.image.decode_jpeg(file_bytes, channels=3)\n        else:\n            raise ValueError(\"Image extension not supported\")\n        img = tf.cast(img, tf.float32) / 255.0\n        img = tf.image.resize(img, target_size)\n\n        return img\n    \n    def decode_with_labels(path, label):\n        return decode(path), label\n    \n    return decode_with_labels if with_labels else decode\n\n\ndef build_augmenter(with_labels=True):\n    def augment(img):\n        img = tf.image.random_flip_left_right(img)\n        img = tf.image.random_flip_up_down(img)\n        return img\n    \n    def augment_with_labels(img, label):\n        return augment(img), label\n    \n    return augment_with_labels if with_labels else augment\n\n\ndef build_dataset(paths, labels=None, bsize=128, cache=True,\n                  decode_fn=None, augment_fn=None,\n                  augment=True, repeat=True, shuffle=1024, \n                  cache_dir=\"\"):\n    if cache_dir != \"\" and cache is True:\n        os.makedirs(cache_dir, exist_ok=True)\n    \n    if decode_fn is None:\n        decode_fn = build_decoder(labels is not None)\n    \n    if augment_fn is None:\n        augment_fn = build_augmenter(labels is not None)\n    \n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = paths if labels is None else (paths, labels)\n    \n    dset = tf.data.Dataset.from_tensor_slices(slices)\n    dset = dset.map(decode_fn, num_parallel_calls=AUTO)\n    dset = dset.cache(cache_dir) if cache else dset\n    dset = dset.map(augment_fn, num_parallel_calls=AUTO) if augment else dset\n    dset = dset.repeat() if repeat else dset\n    dset = dset.shuffle(shuffle) if shuffle else dset\n    dset = dset.batch(bsize).prefetch(AUTO)\n    \n    return dset","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"COMPETITION_NAME = \"hpa-768768\"\nstrategy = auto_select_accelerator()\nBATCH_SIZE = strategy.num_replicas_in_sync * 16\nGCS_DS_PATH = KaggleDatasets().get_gcs_path(COMPETITION_NAME)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GCS_DS_PATH","metadata":{"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#green\nload_dir = f\"/kaggle/input/{COMPETITION_NAME}/\"\ndf = pd.read_csv('../input/classification-label-csv-green/df_green.csv')\nlabel_cols = df.columns[2:21]\npaths = GCS_DS_PATH + '/' + df['ID'] + '.png'\nlabels = df[label_cols].values","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from kaggle_datasets import KaggleDatasets\nimport tensorflow as tf\n\nlabel2id_dict = {\n 'Nucleoplasm': 0,\n 'Nuclear Membrane': 1,\n 'Nucleoli': 2,\n 'Nucleoli Fibrillar Center': 3,\n 'Nuclear Speckles': 4,\n 'Nuclear Bodies': 5,\n 'Endoplasmic Reticulum': 6,\n 'Golgi Apparatus': 7,\n 'Intermediate Filaments': 8,\n 'Actin Filaments': 9,\n 'Microtubules': 10,\n 'Mitotic Spindle': 11,\n 'Centrosome': 12,\n 'Plasma Membrane': 13,\n 'Mitochondria': 14,\n 'Aggresome': 15,\n 'Cytosol': 16,\n 'Vesicles': 17,\n 'Negative': 18\n}\n\n# GCS_DS_PATH = KaggleDatasets().get_gcs_path(COMPETITION_NAME)\n# load_dir = GCS_DS_PATH\n\nCOMPETITION_NAME = \"hpa-single-cell-image-classification\"\nload_dir = f\"../input/{COMPETITION_NAME}/train/\"\nload_dir = \"gs://green_channels/\"\n\n#Preprocessing Dataset \ndf = pd.read_csv('../input/classification-label-csv-green/df_green.csv')\n# df['label_count'] = df.Label.str.split(\"|\").str.len()\n# df = df[df.label_count == 1]\n# df['label_name'] = df[\"Label\"].apply(lambda x: l_dict[int(x)])\ndf['path'] = df[\"ID\"].apply(lambda x: load_dir + x + \".png\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_cols = df.columns[2:21]\n\nfile_format = \"\"\n\ntraining_df = pd.read_csv(\"../input/human-cell-atlas-training/training_df.csv\")\ntrain_paths = load_dir + training_df['ID'] + file_format\ntraining_df[\"path\"] = train_paths\ntrain_labels = training_df[label_cols].values\n\nvalid_df = pd.read_csv(\"../input/human-cell-atlas-training/valid_df.csv\")\nvalid_paths = load_dir + valid_df['ID'] + file_format\nvalid_df[\"path\"] = valid_paths\nvalid_labels = valid_df[label_cols].values\n\ntest_df = pd.read_csv(\"../input/human-cell-atlas-training/test_df.csv\")\ntest_paths = load_dir + test_df['ID'] + file_format\ntest_df[\"path\"] = test_paths\ntest_labels = test_df[label_cols].values\n\nsample_df = df.sample(frac=0.01)\nsample_paths = load_dir + sample_df['ID'] + file_format\nsample_labels = sample_df[label_cols].values\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMSIZE = (224, 240, 260, 300, 380, 456, 528, 600, 675)\nIMS = 7\n\ndecoder = build_decoder(with_labels=True, target_size=(IMSIZE[IMS], IMSIZE[IMS]))\ntest_decoder = build_decoder(with_labels=False, target_size=(IMSIZE[IMS], IMSIZE[IMS]))\n\ntrain_dataset = build_dataset(\n    train_paths, train_labels, bsize=BATCH_SIZE, decode_fn=decoder\n)\n\nvalid_dataset = build_dataset(\n    valid_paths, valid_labels, bsize=BATCH_SIZE, decode_fn=decoder,\n    repeat=False, shuffle=False, augment=False\n)\n\ntest_dataset = build_dataset(\n    test_paths, cache=False, bsize=BATCH_SIZE, decode_fn=test_decoder,\n    repeat=False, shuffle=False, augment=False\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    n_labels = train_labels.shape[1]\nexcept:\n    n_labels = 1\n    \nwith strategy.scope():\n    model = tf.keras.Sequential([\n        tf.keras.applications.ResNet152V2(\n            input_shape =(IMSIZE[IMS], IMSIZE[IMS], 3),\n            weights='imagenet',\n            include_top=False),\n        tf.keras.layers.GlobalAveragePooling2D(),\n        tf.keras.layers.Dense(n_labels, activation='sigmoid')\n    ])\n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(),\n        loss='binary_crossentropy',\n        metrics=[tf.keras.metrics.AUC(multi_label=True)])\n        \n    model.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"colour = '_green'\nsteps_per_epoch = train_paths.shape[0] // BATCH_SIZE\ncheckpoint = tf.keras.callbacks.ModelCheckpoint(\n    f'ResNet152V2_model{colour}.h5', save_best_only=True, monitor='val_loss', mode='min')\nlr_reducer = tf.keras.callbacks.ReduceLROnPlateau(\n    monitor=\"val_loss\", patience=3, min_lr=1e-6, mode='min')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_dataset, \n    epochs=20,\n    verbose=1,\n    callbacks=[checkpoint, lr_reducer],\n    steps_per_epoch=steps_per_epoch,\n    validation_data=valid_dataset)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hist_dfResNet152V2 = pd.DataFrame(history.history)\nhist_dfResNet152V2.to_csv(f'history_ResNet152V2{colour}.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport matplotlib.pyplot as plt\ndef plot_hist(hist):\n    columns = list(hist.columns) \n    plt.plot(hist[columns[1]])\n    plt.plot(hist[columns[3]])\n    plt.title(\"Model Accuracy\")\n    plt.ylabel(\"accuracy\")\n    plt.xlabel(\"epouch\")\n    plt.legend([\"train\", \"validation\"], loc = \"upper left\")\n\n\ndef plot_loss(hist):\n    columns = list(hist.columns) \n    plt.plot(hist[columns[0]])\n    plt.plot(hist[columns[2]])\n    plt.title(\"Model Loss\")\n    plt.ylabel(\"Loss\")\n    plt.xlabel(\"epouch\")\n    plt.legend([\"train\", \"validation\"], loc = \"upper right\")\n\nnew_data_frame = history.history\nnew_data_frame = pd.DataFrame(new_data_frame)\nplot_loss(new_data_frame)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_hist(new_data_frame)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sn\nfrom sklearn.metrics import classification_report, confusion_matrix\n\n\nprediction_probs = model.predict(test_dataset, verbose=1)\nprediction_classes = np.argmax(prediction_probs, axis=-1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true = list(test_df[\"Label\"])\ny_true = list(map(lambda x: int(x), y_true))\n\ncmat = confusion_matrix(y_true, prediction_classes)\nfigure = plt.figure(figsize=(12,12))\nsn.heatmap(cmat,annot=True, fmt='')\n\n\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(y_true, prediction_classes))","metadata":{},"execution_count":null,"outputs":[]}]}