{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Best model for covid","metadata":{}},{"cell_type":"code","source":"import os\n\nimport numpy as np\nimport pandas as pd\nfrom kaggle_datasets import KaggleDatasets\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom sklearn.model_selection import GroupKFold","metadata":{"papermill":{"duration":7.602011,"end_time":"2021-05-21T02:30:23.132009","exception":false,"start_time":"2021-05-21T02:30:15.529998","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-10T08:34:46.634629Z","iopub.execute_input":"2021-07-10T08:34:46.63507Z","iopub.status.idle":"2021-07-10T08:34:52.442027Z","shell.execute_reply.started":"2021-07-10T08:34:46.634974Z","shell.execute_reply":"2021-07-10T08:34:52.441275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def auto_select_accelerator():\n    try:\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n        print(\"Running on TPU:\", tpu.master())\n    except ValueError:\n        strategy = tf.distribute.get_strategy()\n    print(f\"Running on {strategy.num_replicas_in_sync} replicas\")\n    \n    return strategy\n\n\ndef build_decoder(with_labels=True, target_size=(256, 256), ext='jpg'):\n    def decode(path):\n        file_bytes = tf.io.read_file(path)\n\n        if ext == 'png':\n            img = tf.image.decode_png(file_bytes, channels=3)\n        elif ext in ['jpg', 'jpeg']:\n            img = tf.image.decode_jpeg(file_bytes, channels=3)\n        else:\n            raise ValueError(\"Image extension not supported\")\n        img = tf.cast(img, tf.float32) / 255.0\n        img = tf.image.resize(img, target_size)\n\n        return img\n    \n    def decode_with_labels(path, label):\n        return decode(path), label\n    \n    return decode_with_labels if with_labels else decode\n\n\ndef build_augmenter(with_labels=True):\n    def augment(img):\n        img = tf.image.random_flip_left_right(img)\n        img = tf.image.random_flip_up_down(img)\n        return img\n    \n    def augment_with_labels(img, label):\n        return augment(img), label\n    \n    return augment_with_labels if with_labels else augment\n\n\ndef build_dataset(paths, labels=None, bsize=128, cache=True,\n                  decode_fn=None, augment_fn=None,\n                  augment=True, repeat=True, shuffle=1024, \n                  cache_dir=\"\"):\n    if cache_dir != \"\" and cache is True:\n        os.makedirs(cache_dir, exist_ok=True)\n    \n    if decode_fn is None:\n        decode_fn = build_decoder(labels is not None)\n    \n    if augment_fn is None:\n        augment_fn = build_augmenter(labels is not None)\n    \n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = paths if labels is None else (paths, labels)\n    \n    dset = tf.data.Dataset.from_tensor_slices(slices)\n    dset = dset.map(decode_fn, num_parallel_calls=AUTO)\n    dset = dset.cache(cache_dir) if cache else dset\n    dset = dset.map(augment_fn, num_parallel_calls=AUTO) if augment else dset\n    dset = dset.repeat() if repeat else dset\n    dset = dset.shuffle(shuffle) if shuffle else dset\n    dset = dset.batch(bsize).prefetch(AUTO)\n    \n    return dset","metadata":{"papermill":{"duration":0.031498,"end_time":"2021-05-21T02:30:23.171226","exception":false,"start_time":"2021-05-21T02:30:23.139728","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-10T08:34:52.443301Z","iopub.execute_input":"2021-07-10T08:34:52.443553Z","iopub.status.idle":"2021-07-10T08:34:52.45793Z","shell.execute_reply.started":"2021-07-10T08:34:52.44353Z","shell.execute_reply":"2021-07-10T08:34:52.457106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"COMPETITION_NAME = \"siimcovid19-512-img-png-600-study-png\"\nstrategy = auto_select_accelerator()\nBATCH_SIZE = strategy.num_replicas_in_sync * 16\nGCS_DS_PATH = KaggleDatasets().get_gcs_path(COMPETITION_NAME)","metadata":{"papermill":{"duration":5.980997,"end_time":"2021-05-21T02:30:29.160082","exception":false,"start_time":"2021-05-21T02:30:23.179085","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-10T08:34:52.459273Z","iopub.execute_input":"2021-07-10T08:34:52.45956Z","iopub.status.idle":"2021-07-10T08:34:58.341564Z","shell.execute_reply.started":"2021-07-10T08:34:52.459538Z","shell.execute_reply":"2021-07-10T08:34:58.340699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"load_dir = f\"/kaggle/input/{COMPETITION_NAME}/\"\ndf = pd.read_csv('../input/siim-covid19-detection/train_study_level.csv')\nlabel_cols = df.columns[1:5]\n","metadata":{"papermill":{"duration":0.049916,"end_time":"2021-05-21T02:30:29.218772","exception":false,"start_time":"2021-05-21T02:30:29.168856","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gkf  = GroupKFold(n_splits = 2)\ndf['fold'] = -1\nfor fold, (train_idx, val_idx) in enumerate(gkf.split(df, groups = df.id.tolist())):\n    df.loc[val_idx, 'fold'] = fold","metadata":{"papermill":{"duration":0.067287,"end_time":"2021-05-21T02:30:29.294557","exception":false,"start_time":"2021-05-21T02:30:29.22727","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(1):\n    \n    valid_paths = GCS_DS_PATH + '/study/' + df[df['fold'] == i]['id'] + '.png' #\"/train/\"\n    train_paths = GCS_DS_PATH + '/study/' + df[df['fold'] != i]['id'] + '.png' #\"/train/\" \n    valid_labels = df[df['fold'] == i][label_cols].values\n    train_labels = df[df['fold'] != i][label_cols].values\n    print(len(valid_labels))\n    \n    IMSIZE = (224, 240, 260, 300, 380, 456, 528, 331)\n    IMS = 0\n\n    decoder = build_decoder(with_labels=True, target_size=(IMSIZE[IMS], IMSIZE[IMS]), ext='png')\n    test_decoder = build_decoder(with_labels=False, target_size=(IMSIZE[IMS], IMSIZE[IMS]),ext='png')\n\n    train_dataset = build_dataset(\n        train_paths, train_labels, bsize=BATCH_SIZE, decode_fn=decoder\n    )\n\n    valid_dataset = build_dataset(\n        valid_paths, valid_labels, bsize=BATCH_SIZE, decode_fn=decoder,\n        repeat=False, shuffle=False, augment=False\n    )\n    \n    IMS_2 = 7\n\n    decoder_2 = build_decoder(with_labels=True, target_size=(IMSIZE[IMS_2], IMSIZE[IMS_2]), ext='png')\n    test_decoder_2 = build_decoder(with_labels=False, target_size=(IMSIZE[IMS], IMSIZE[IMS]),ext='png')\n\n    train_dataset_2 = build_dataset(\n        train_paths, train_labels, bsize=BATCH_SIZE, decode_fn=decoder_2\n    )\n\n    valid_dataset_2 = build_dataset(\n        valid_paths, valid_labels, bsize=BATCH_SIZE, decode_fn=decoder_2,\n        repeat=False, shuffle=False, augment=False\n    )\n\n    try:\n        n_labels = train_labels.shape[1]\n    except:\n        n_labels = 1","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Selecting model part","metadata":{}},{"cell_type":"code","source":"import inspect\n# List all available models\nmodel_dictionary = {m[0]:m[1] for m in inspect.getmembers(tf.keras.applications, inspect.isfunction)}\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model_dictionary","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install h5py","metadata":{"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\n# Loop over each model available in Keras\nmodel_benchmarks = {'model_name': [], 'num_model_params': [], 'validation_loss': []}\nnum_iterations = int(len(train_labels)//BATCH_SIZE)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n\n    for model_name, model in tqdm(model_dictionary.items()):\n        # Special handling for \"NASNetLarge\" since it requires input images with size (331,331)\n        if 'NASNetLarge' in model_name:\n            input_shape=(331,331,3)\n            train_processed = train_dataset_2\n            validation_processed = valid_dataset_2\n        else:\n            input_shape=(224,224,3)\n            train_processed = train_dataset\n            validation_processed = valid_dataset\n\n        # load the pre-trained model with global average pooling as the last layer and freeze the model weights\n        pre_trained_model = model(include_top=False, pooling='avg', input_shape=input_shape)\n        pre_trained_model.trainable = False\n\n        # custom modifications on top of pre-trained model\n        clf_model = tf.keras.models.Sequential()\n        clf_model.add(pre_trained_model)\n        clf_model.add(tf.keras.layers.Dense(n_labels, activation='softmax'))\n        clf_model.compile(optimizer=tf.keras.optimizers.Adam(),\n                          loss='categorical_crossentropy', \n                          metrics=[tf.keras.metrics.AUC(multi_label=True)])\n        history = clf_model.fit(train_processed, \n                                epochs=3, \n                                verbose = 1,\n                                validation_data=validation_processed, \n                                steps_per_epoch = num_iterations)\n\n        # Calculate all relevant metrics\n        model_benchmarks['model_name'].append(model_name)\n        model_benchmarks['num_model_params'].append(pre_trained_model.count_params())\n        model_benchmarks['validation_loss'].append(history.history['val_loss'][-1])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n# Convert Results to DataFrame for easy viewing\nbenchmark_df = pd.DataFrame(model_benchmarks)\nbenchmark_df.sort_values('validation_loss', inplace=True) # sort in ascending order of num_model_params column\nbenchmark_df.to_csv('benchmark_df.csv', index=False) # write results to csv file\nbenchmark_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib as plt\n# Loop over each row and plot the num_model_params vs validation_accuracy\nmarkers=[\".\",\",\",\"o\",\"v\",\"^\",\"<\",\">\",\"1\",\"2\",\"3\",\"4\",\"8\",\"s\",\"p\",\"P\",\"*\",\"h\",\"H\",\"+\",\"x\",\"X\",\"D\",\"d\",\"|\",\"_\",4,5,6,7,8,9,10,11]\nplt.figure(figsize=(7,5))\nfor row in benchmark_df.itertuples():\n    plt.scatter(row.num_model_params, row.validation_loss, label=row.model_name, marker=markers[row.Index], s=150, linewidths=2)\nplt.xscale('log')\nplt.xlabel('Number of Parameters in Model')\nplt.ylabel('Validation Accuracy after 3 Epochs')\nplt.title('Accuracy vs Model Size')\nplt.legend(bbox_to_anchor=(1, 1), loc='upper left'); # Move legend out of the plot","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}