{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"thanks to https://www.kaggle.com/xhlulu/siim-covid-19-convert-to-jpg-256px  \ntrain: https://www.kaggle.com/h053473666/siim-covid19-efnb7-train-study","metadata":{}},{"cell_type":"code","source":"!conda install '/kaggle/input/pydicom-conda-helper/libjpeg-turbo-2.1.0-h7f98852_0.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/libgcc-ng-9.3.0-h2828fa1_19.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/gdcm-2.8.9-py37h500ead1_1.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/conda-4.10.1-py37h89c1867_0.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/certifi-2020.12.5-py37h89c1867_1.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/openssl-1.1.1k-h7f98852_0.tar.bz2' -c conda-forge -y","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:28:46.116442Z","iopub.execute_input":"2021-08-08T20:28:46.116885Z","iopub.status.idle":"2021-08-08T20:30:11.791779Z","shell.execute_reply.started":"2021-08-08T20:28:46.116788Z","shell.execute_reply":"2021-08-08T20:30:11.790597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nsys.path.append(\"../input/leondgarse-keras-efficientnet-v2-test/Keras_efficientnet_v2_test\")\nsys.path.append(\"../input/leondgarse-keras-efficientnet-v2-test/Keras_efficientnet_v2_test/Keras_efficientnet_v2\")","metadata":{"execution":{"iopub.status.busy":"2021-08-08T20:30:22.14065Z","iopub.execute_input":"2021-08-08T20:30:22.141102Z","iopub.status.idle":"2021-08-08T20:30:22.148275Z","shell.execute_reply.started":"2021-08-08T20:30:22.141062Z","shell.execute_reply":"2021-08-08T20:30:22.146533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\nfrom PIL import Image\nimport pandas as pd\nfrom tqdm.auto import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-12T08:43:49.545759Z","iopub.execute_input":"2021-07-12T08:43:49.546123Z","iopub.status.idle":"2021-07-12T08:43:49.557906Z","shell.execute_reply.started":"2021-07-12T08:43:49.546082Z","shell.execute_reply":"2021-07-12T08:43:49.557156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# .dcm to .png","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\ndef read_xray(path, voi_lut = True, fix_monochrome = True):\n    # Original from: https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n    dicom = pydicom.read_file(path)\n    \n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \n    # \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n               \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n        \n    return data","metadata":{"execution":{"iopub.status.busy":"2021-07-12T08:43:49.561082Z","iopub.execute_input":"2021-07-12T08:43:49.561393Z","iopub.status.idle":"2021-07-12T08:43:49.813517Z","shell.execute_reply.started":"2021-07-12T08:43:49.561362Z","shell.execute_reply":"2021-07-12T08:43:49.812345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resize(array, size, keep_ratio=False, resample=Image.LANCZOS):\n    # Original from: https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\n    im = Image.fromarray(array)\n    \n    if keep_ratio:\n        im.thumbnail((size, size), resample)\n    else:\n        im = im.resize((size, size), resample)\n    \n    return im","metadata":{"execution":{"iopub.status.busy":"2021-07-12T08:43:49.817031Z","iopub.execute_input":"2021-07-12T08:43:49.81736Z","iopub.status.idle":"2021-07-12T08:43:49.824851Z","shell.execute_reply.started":"2021-07-12T08:43:49.817333Z","shell.execute_reply":"2021-07-12T08:43:49.824071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nsplit = 'test'\nsave_dir = f'/kaggle/tmp/{split}/'\n\nos.makedirs(save_dir, exist_ok=True)\n\nsave_dir = f'/kaggle/tmp/{split}/study/'\nos.makedirs(save_dir, exist_ok=True)\n\npred_id_df = {\n    'id': []\n}\n\nfor dirname, _, filenames in tqdm(os.walk(f'../input/siim-covid19-detection/{split}')):\n    for idx, file in enumerate(filenames):\n        # set keep_ratio=True to have original aspect ratio\n        xray = read_xray(os.path.join(dirname, file))\n        im = resize(xray, size=1000)\n        study = dirname.split('/')[-2] + '_'+ str(idx)\n        pred_id_df['id'].append(study)\n        im.save(os.path.join(save_dir, study+'.png'))\n","metadata":{"execution":{"iopub.status.busy":"2021-07-12T08:43:49.826188Z","iopub.execute_input":"2021-07-12T08:43:49.826558Z","iopub.status.idle":"2021-07-12T08:58:01.075757Z","shell.execute_reply.started":"2021-07-12T08:43:49.826521Z","shell.execute_reply":"2021-07-12T08:58:01.074891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# predict","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\n\ndf = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')\nid_laststr_list  = []\nfor i in range(df.shape[0]):\n    id_laststr_list.append(df.loc[i,'id'][-1])\ndf['id_last_str'] = id_laststr_list\n\nstudy_len = df[df['id_last_str'] == 'y'].shape[0]","metadata":{"execution":{"iopub.status.busy":"2021-07-12T08:58:01.077228Z","iopub.execute_input":"2021-07-12T08:58:01.077869Z","iopub.status.idle":"2021-07-12T08:58:01.141559Z","shell.execute_reply.started":"2021-07-12T08:58:01.077821Z","shell.execute_reply":"2021-07-12T08:58:01.140735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_id_df = pd.DataFrame(pred_id_df)","metadata":{"execution":{"iopub.status.busy":"2021-07-12T08:58:01.142879Z","iopub.execute_input":"2021-07-12T08:58:01.143238Z","iopub.status.idle":"2021-07-12T08:58:01.148918Z","shell.execute_reply.started":"2021-07-12T08:58:01.143194Z","shell.execute_reply":"2021-07-12T08:58:01.14775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from Keras_efficientnet_v2 import efficientnet_v2\ndef build_aux_loss_model(weights_path):\n    tf.keras.backend.clear_session()\n    base = efficientnet_v2.EfficientNetV2(model_type=\"l\", input_shape=(None, None, 3), survivals=None, dropout=0.2, classes=21843, classifier_activation=None, pretrained=None)\n    \n    base_notop = tf.keras.models.Model(base.inputs[0], base.layers[-4].output)\n        \n    intermediary_layer = base_notop.get_layer(\"add_72\").output\n    \n    classification_head = tf.keras.layers.GlobalAveragePooling2D()(base_notop.output)\n    classification_head = tf.keras.layers.Dropout(0.2)(classification_head)\n    classification_head = tf.keras.layers.Dense(4, activation='softmax', name = 'classification')(classification_head)\n\n    aux_head = tf.keras.layers.Conv2D(128, kernel_size = 3, padding='same')(intermediary_layer)\n    aux_head = tf.keras.layers.BatchNormalization()(aux_head)\n    aux_head = tf.keras.layers.ReLU()(aux_head)\n    aux_head = tf.keras.layers.Conv2D(128, kernel_size = 3, padding='same')(aux_head)\n    aux_head = tf.keras.layers.BatchNormalization()(aux_head)\n    aux_head = tf.keras.layers.ReLU()(aux_head)\n    aux_head = tf.keras.layers.Conv2D(1, kernel_size=1, padding= 'valid', name = 'aux_head')(aux_head)\n\n    model = tf.keras.Model(base_notop.inputs, [classification_head,aux_head])\n    model.load_weights(weights_path)\n    return model","metadata":{"execution":{"iopub.status.busy":"2021-08-08T20:30:46.954217Z","iopub.execute_input":"2021-08-08T20:30:46.954612Z","iopub.status.idle":"2021-08-08T20:30:46.967818Z","shell.execute_reply.started":"2021-08-08T20:30:46.954577Z","shell.execute_reply":"2021-08-08T20:30:46.966127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install /kaggle/input/kerasapplications -q\n!pip install /kaggle/input/efficientnet-keras-source-code/ -q --no-deps\n\nimport os\n\nimport efficientnet.tfkeras as efn\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\n\ndef auto_select_accelerator():\n    try:\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n        print(\"Running on TPU:\", tpu.master())\n    except ValueError:\n        strategy = tf.distribute.get_strategy()\n    print(f\"Running on {strategy.num_replicas_in_sync} replicas\")\n\n    return strategy\n\n\ndef build_decoder(with_labels=True, target_size=(300, 300), ext='jpg'):\n    def decode(path):\n        file_bytes = tf.io.read_file(path)\n        if ext == 'png':\n            img = tf.image.decode_png(file_bytes, channels=3)\n        elif ext in ['jpg', 'jpeg']:\n            img = tf.image.decode_jpeg(file_bytes, channels=3)\n        else:\n            raise ValueError(\"Image extension not supported\")\n\n        img = tf.cast(img, tf.float32) / 255.0\n        img = tf.image.resize(img, target_size)\n\n        return img\n\n    def decode_with_labels(path, label):\n        return decode(path), label\n\n    return decode_with_labels if with_labels else decode\n\n\ndef build_augmenter(with_labels=True):\n    def augment(img):\n        img = tf.image.random_flip_left_right(img)\n        img = tf.image.random_flip_up_down(img)\n        return img\n\n    def augment_with_labels(img, label):\n        return augment(img), label\n\n    return augment_with_labels if with_labels else augment\n\n\ndef build_dataset(paths, labels=None, bsize=32, cache=True,\n                  decode_fn=None, augment_fn=None,\n                  augment=True, repeat=True, shuffle=1024, \n                  cache_dir=\"\"):\n    if cache_dir != \"\" and cache is True:\n        os.makedirs(cache_dir, exist_ok=True)\n\n    if decode_fn is None:\n        decode_fn = build_decoder(labels is not None)\n\n    if augment_fn is None:\n        augment_fn = build_augmenter(labels is not None)\n\n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = paths if labels is None else (paths, labels)\n\n    dset = tf.data.Dataset.from_tensor_slices(slices)\n    dset = dset.map(decode_fn, num_parallel_calls=AUTO)\n    dset = dset.cache(cache_dir) if cache else dset\n    dset = dset.map(augment_fn, num_parallel_calls=AUTO) if augment else dset\n    dset = dset.repeat() if repeat else dset\n    dset = dset.shuffle(shuffle) if shuffle else dset\n    dset = dset.batch(bsize).prefetch(AUTO)\n\n    return dset\n\n#COMPETITION_NAME = \"siim-cov19-test-img512-study-600\"\nstrategy = auto_select_accelerator()\nBATCH_SIZE = strategy.num_replicas_in_sync * 16\n\nIMSIZE = (224, 240, 260, 300, 380, 456, 528, 512)\n\n#load_dir = f\"/kaggle/input/{COMPETITION_NAME}/\"\nsub_df = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')\nsub_df = sub_df[:study_len]\ntest_paths = f'/kaggle/tmp/{split}/study/' + pred_id_df['id'] +'.png'\n\npred_id_df['negative'] = 0\npred_id_df['typical'] = 0\npred_id_df['indeterminate'] = 0\npred_id_df['atypical'] = 0\n\n\nlabel_cols = pred_id_df.columns[1:]\n\ntest_decoder = build_decoder(with_labels=False, target_size=(IMSIZE[7], IMSIZE[7]), ext='png')\ndtest = build_dataset(\n    test_paths, bsize=BATCH_SIZE, repeat=False, \n    shuffle=False, augment=False, cache=False,\n    decode_fn=test_decoder\n)\n\nwith strategy.scope():\n    \n    models = []\n    \n    models0 = build_aux_loss_model(\n        '../input/effnetv2-last/model0.h5'\n    )\n    models1 = build_aux_loss_model(\n        '../input/effnetv2-last/model1.h5'\n    )\n    models2 = build_aux_loss_model(\n        '../input/effnetv2-last/model2.h5'\n    )\n    models3 = build_aux_loss_model(\n        '../input/effnetv2-last/model3.h5'\n    )\n    models4 = build_aux_loss_model(\n        '../input/effnetv2-last/model4.h5'\n    )\n    \n    models.append(models0)\n    models.append(models1)\n    models.append(models2)\n    models.append(models3)\n    models.append(models4)\n\n    \n    \n    \npred_id_df[label_cols] = sum([model.predict(dtest, verbose=1)[0] for model in models]) / len(models)","metadata":{"execution":{"iopub.status.busy":"2021-07-12T09:12:43.945961Z","iopub.execute_input":"2021-07-12T09:12:43.946346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_id_df[['id','helper_id']] = pred_id_df.id.str.split(\"_\",expand=True,)\npred_id_df","metadata":{"execution":{"iopub.status.busy":"2021-07-12T09:00:41.565674Z","iopub.status.idle":"2021-07-12T09:00:41.566065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_id_df = pred_id_df.groupby('id').mean()\npred_id_df","metadata":{"execution":{"iopub.status.busy":"2021-07-12T09:00:41.567039Z","iopub.status.idle":"2021-07-12T09:00:41.567481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_id_df.index","metadata":{"execution":{"iopub.status.busy":"2021-07-12T09:00:41.568351Z","iopub.status.idle":"2021-07-12T09:00:41.568912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_id_df['PredictionString1'] = 0\npred_id_df['id'] = pred_id_df.index\npred_id_df['id'] = pred_id_df['id'].apply(lambda s : s+'_study')\npred_id_df.index = pred_id_df['id']\npred_id_df.drop(axis=1,columns='id',inplace=True)","metadata":{"execution":{"iopub.status.busy":"2021-07-12T09:00:41.570164Z","iopub.status.idle":"2021-07-12T09:00:41.570723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_id_df","metadata":{"execution":{"iopub.status.busy":"2021-07-12T09:00:41.571863Z","iopub.status.idle":"2021-07-12T09:00:41.572551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.merge(df, pred_id_df, on = 'id', how = 'left')\ndf","metadata":{"execution":{"iopub.status.busy":"2021-07-12T09:00:41.573914Z","iopub.status.idle":"2021-07-12T09:00:41.57448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# study string","metadata":{}},{"cell_type":"code","source":"for i in range(study_len):\n    negative = df.loc[i,'negative']\n    typical = df.loc[i,'typical']\n    indeterminate = df.loc[i,'indeterminate']\n    atypical = df.loc[i,'atypical']\n    df.loc[i, 'PredictionString'] = f'negative {negative} 0 0 1 1 typical {typical} 0 0 1 1 indeterminate {indeterminate} 0 0 1 1 atypical {atypical} 0 0 1 1'","metadata":{"execution":{"iopub.status.busy":"2021-07-12T09:00:41.575646Z","iopub.status.idle":"2021-07-12T09:00:41.57623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df[['id', 'PredictionString']]\n\ndf.to_csv('submission.csv',index=False)\ndf","metadata":{"execution":{"iopub.status.busy":"2021-07-12T09:00:41.577347Z","iopub.status.idle":"2021-07-12T09:00:41.577905Z"},"trusted":true},"execution_count":null,"outputs":[]}]}