{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"## For resnext,seresnext ... etc\n!pip install git+https://github.com/qubvel/classification_models.git","metadata":{"execution":{"iopub.status.busy":"2021-08-09T19:45:02.217656Z","iopub.execute_input":"2021-08-09T19:45:02.218077Z","iopub.status.idle":"2021-08-09T19:45:17.498097Z","shell.execute_reply.started":"2021-08-09T19:45:02.218039Z","shell.execute_reply":"2021-08-09T19:45:17.497174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for keras\nfrom classification_models.keras import Classifiers\n\nClassifiers.models_names()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T19:45:17.499868Z","iopub.execute_input":"2021-08-09T19:45:17.500375Z","iopub.status.idle":"2021-08-09T19:45:24.369302Z","shell.execute_reply.started":"2021-08-09T19:45:17.500331Z","shell.execute_reply":"2021-08-09T19:45:24.367812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport re\nimport warnings\nimport random\nimport sklearn.exceptions\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom sklearn.metrics import roc_auc_score\nfrom time import perf_counter\nfrom tensorflow.keras import backend as K\nfrom tqdm.notebook import tqdm\nfrom kaggle_datasets import KaggleDatasets\nfrom glob import glob\n\n\nwarnings.filterwarnings('ignore', category=DeprecationWarning)\nwarnings.filterwarnings('ignore', category=FutureWarning)\nwarnings.filterwarnings(\"ignore\", category=sklearn.exceptions.UndefinedMetricWarning)\n\nRANDOM_SEED = 42\nCOMPETITION_DATASET_PATH = \"../input/g2net-gravitational-wave-detection\"\nPRETRAINED_MODEL_PATH = \"../input/gwave-seresnet50-baseline\"\nQUANTILE = 0.7\nFOLDS = (0, 1, 2, 3)\nIMG_SIZES = 256\nBATCH_SIZES = 32\n\ndef seed_everything(seed=RANDOM_SEED):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    random.seed(seed)\n    tf.random.set_seed(seed)\n\nseed_everything()","metadata":{"papermill":{"duration":10.406747,"end_time":"2021-06-12T15:10:18.648284","exception":false,"start_time":"2021-06-12T15:10:08.241537","status":"completed"},"tags":[],"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-09T19:45:24.371846Z","iopub.execute_input":"2021-08-09T19:45:24.372341Z","iopub.status.idle":"2021-08-09T19:45:25.348816Z","shell.execute_reply.started":"2021-08-09T19:45:24.372289Z","shell.execute_reply":"2021-08-09T19:45:25.347634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# From https://www.kaggle.com/xhlulu/ranzcr-efficientnet-tpu-training\ndef auto_select_accelerator():\n    TPU_DETECTED = False\n    try:\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n        print(\"Running on TPU:\", tpu.master())\n        TPU_DETECTED =True\n    except ValueError:\n        strategy = tf.distribute.get_strategy()\n    print(f\"Running on {strategy.num_replicas_in_sync} replicas\")\n    \n    return strategy, TPU_DETECTED","metadata":{"papermill":{"duration":0.02909,"end_time":"2021-06-12T15:10:18.696808","exception":false,"start_time":"2021-06-12T15:10:18.667718","status":"completed"},"tags":[],"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-09T19:45:25.350863Z","iopub.execute_input":"2021-08-09T19:45:25.351691Z","iopub.status.idle":"2021-08-09T19:45:25.359852Z","shell.execute_reply.started":"2021-08-09T19:45:25.351629Z","shell.execute_reply":"2021-08-09T19:45:25.358320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BACKBONE = 'seresnet50'","metadata":{"execution":{"iopub.status.busy":"2021-08-09T19:45:25.361423Z","iopub.execute_input":"2021-08-09T19:45:25.361810Z","iopub.status.idle":"2021-08-09T19:45:25.380946Z","shell.execute_reply.started":"2021-08-09T19:45:25.361773Z","shell.execute_reply":"2021-08-09T19:45:25.379616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"4\"></a>\n# TPU Configuration & Utils scripts","metadata":{"papermill":{"duration":0.017986,"end_time":"2021-06-12T15:10:18.733021","exception":false,"start_time":"2021-06-12T15:10:18.715035","status":"completed"},"tags":[]}},{"cell_type":"code","source":"strategy, TPU_DETECTED = auto_select_accelerator()\nAUTO = tf.data.experimental.AUTOTUNE\nREPLICAS = strategy.num_replicas_in_sync","metadata":{"papermill":{"duration":5.397929,"end_time":"2021-06-12T15:10:24.197997","exception":false,"start_time":"2021-06-12T15:10:18.800068","status":"completed"},"tags":[],"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-09T19:45:31.736848Z","iopub.execute_input":"2021-08-09T19:45:31.737378Z","iopub.status.idle":"2021-08-09T19:45:31.749453Z","shell.execute_reply.started":"2021-08-09T19:45:31.737341Z","shell.execute_reply":"2021-08-09T19:45:31.747835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_set_paths(idxs, path_prefix: str ='cqt-g2net-test', file_prefix: str = 'test', sep: str='-'):\n    files = []\n    for i,k in tqdm(idxs):\n        GCS_PATH = KaggleDatasets().get_gcs_path(f'{path_prefix}-{i}{sep}{k}')\n        files.extend(np.sort(np.array(tf.io.gfile.glob(GCS_PATH + f'/{file_prefix}*.tfrec'))).tolist())\n    print('Detected', len(files), file_prefix, 'files')\n    return files","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-09T19:45:32.560670Z","iopub.execute_input":"2021-08-09T19:45:32.561231Z","iopub.status.idle":"2021-08-09T19:45:32.567994Z","shell.execute_reply.started":"2021-08-09T19:45:32.561189Z","shell.execute_reply":"2021-08-09T19:45:32.566671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files_train_g = get_set_paths([(0, 1), (2, 3), (4, 5), (6, 7), \n                               (8, 9), (10, 11), (12, 13), (14, 15)], \n                              path_prefix='cqt-g2net-v2', file_prefix='train', sep='-')\nfiles_test_g = get_set_paths([(0, 1), (2, 3), (4, 5), (6, 7)], file_prefix='test')","metadata":{"papermill":{"duration":1.815172,"end_time":"2021-06-12T15:10:26.070321","exception":false,"start_time":"2021-06-12T15:10:24.255149","status":"completed"},"tags":[],"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-09T19:45:34.333800Z","iopub.execute_input":"2021-08-09T19:45:34.334213Z","iopub.status.idle":"2021-08-09T19:45:51.662200Z","shell.execute_reply.started":"2021-08-09T19:45:34.334177Z","shell.execute_reply":"2021-08-09T19:45:51.661048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Reading Tfrecords","metadata":{"papermill":{"duration":0.020506,"end_time":"2021-06-12T15:10:26.31495","exception":false,"start_time":"2021-06-12T15:10:26.294444","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def read_labeled_tfrecord(example):\n    tfrec_format = {\n        'image'                        : tf.io.FixedLenFeature([], tf.string),\n        'image_id'                     : tf.io.FixedLenFeature([], tf.string),\n        'target'                       : tf.io.FixedLenFeature([], tf.int64)\n    }           \n    example = tf.io.parse_single_example(example, tfrec_format)\n    return prepare_image(example['image']), tf.reshape(tf.cast(example['target'], tf.float32), [1])\n\n\ndef read_unlabeled_tfrecord(example, return_image_id):\n    tfrec_format = {\n        'image'                        : tf.io.FixedLenFeature([], tf.string),\n        'image_id'                     : tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, tfrec_format)\n    return prepare_image(example['image']), example['image_id'] if return_image_id else 0\n\n \ndef prepare_image(img, dim=IMG_SIZES):    \n    img = tf.image.resize(tf.image.decode_png(img, channels=3), size=(dim, dim))\n    img = tf.cast(img, tf.float32) / 255.0\n    img = tf.reshape(img, [dim,dim, 3])\n            \n    return img\n\ndef count_data_items(fileids):\n    n = [int(re.compile(r\"-([0-9]*)\\.\").search(fileid).group(1)) \n         for fileid in fileids]\n    return np.sum(n)","metadata":{"papermill":{"duration":0.046323,"end_time":"2021-06-12T15:10:26.380685","exception":false,"start_time":"2021-06-12T15:10:26.334362","status":"completed"},"tags":[],"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-09T19:45:51.663809Z","iopub.execute_input":"2021-08-09T19:45:51.664330Z","iopub.status.idle":"2021-08-09T19:45:51.677195Z","shell.execute_reply.started":"2021-08-09T19:45:51.664290Z","shell.execute_reply":"2021-08-09T19:45:51.675787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Dataset Creation","metadata":{"papermill":{"duration":0.01899,"end_time":"2021-06-12T15:10:26.419124","exception":false,"start_time":"2021-06-12T15:10:26.400134","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def get_dataset(files, shuffle = False, repeat = False, \n                labeled=True, return_image_ids=True, batch_size=16, dim=IMG_SIZES):\n    \n    ds = tf.data.TFRecordDataset(files, num_parallel_reads=AUTO)\n    ds = ds.cache()\n    \n    if repeat:\n        ds = ds.repeat()\n    \n    if shuffle: \n        ds = ds.shuffle(1024*2)\n        opt = tf.data.Options()\n        opt.experimental_deterministic = False\n        ds = ds.with_options(opt)\n        \n    if labeled: \n        ds = ds.map(read_labeled_tfrecord, num_parallel_calls=AUTO)\n    else:\n        ds = ds.map(lambda example: read_unlabeled_tfrecord(example, return_image_ids), \n                    num_parallel_calls=AUTO)      \n    \n    ds = ds.batch(batch_size * REPLICAS)\n    ds = ds.prefetch(AUTO)\n    return ds","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-09T19:45:51.679518Z","iopub.execute_input":"2021-08-09T19:45:51.679934Z","iopub.status.idle":"2021-08-09T19:45:51.697595Z","shell.execute_reply.started":"2021-08-09T19:45:51.679897Z","shell.execute_reply":"2021-08-09T19:45:51.696494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Build Model","metadata":{}},{"cell_type":"code","source":"def build_model(size,path,backbone):\n    inp = tf.keras.layers.Input(shape=(size, size,3))\n    \n    ResNeXt, preprocess_input = Classifiers.get(backbone)\n    base_net = ResNeXt(include_top = False, input_shape=(size,size,3), weights=path)\n\n    x = base_net(inp)    \n    x = tf.keras.layers.GlobalAvgPool2D()(x)\n    x = tf.keras.layers.Dropout(0.)(x)\n    x = tf.keras.layers.Dense(1,activation='sigmoid')(x)\n    \n    model = tf.keras.Model(inputs=inp, outputs=x)\n    loss = tf.keras.losses.BinaryCrossentropy() \n    model.compile(optimizer='adam',loss=loss,metrics=['AUC'])\n    return model","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-09T19:45:51.699254Z","iopub.execute_input":"2021-08-09T19:45:51.699682Z","iopub.status.idle":"2021-08-09T19:45:51.712898Z","shell.execute_reply.started":"2021-08-09T19:45:51.699647Z","shell.execute_reply":"2021-08-09T19:45:51.711623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"5\"></a>\n# Search best quantile","metadata":{"papermill":{"duration":0.019514,"end_time":"2021-06-12T15:10:26.55885","exception":false,"start_time":"2021-06-12T15:10:26.539336","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def predict(paths, is_label=False):\n    pred = []; ids = []\n\n    ds = get_dataset(paths,labeled=False,return_image_ids=False,\n                repeat=False,shuffle=False,dim=IMG_SIZES,batch_size=BATCH_SIZES*2)\n        \n    for fold in FOLDS:\n    \n        print('#'*50); print('--> FOLD',fold+1);\n        start_time = perf_counter()\n    \n        K.clear_session()\n    \n        with strategy.scope():\n            model = build_model(IMG_SIZES, None , BACKBONE)\n            print('\\t-->Loading model...')\n            model.load_weights(f'{PRETRAINED_MODEL_PATH}/fold-{fold}.h5')\n            print('\\t<--Model loaded.')\n    \n        print('\\t-->Start Predict...')\n    \n        pred.append(model.predict(ds, verbose=0).flatten())      \n        print('\\t<--Predict finished.')\n        print('<-- FOLD',fold+1, f'finished; duration = {perf_counter() - start_time} s')\n    \n    if is_label:\n        ds = get_dataset(paths,labeled=True,return_image_ids=False,\n                repeat=False,shuffle=False,dim=IMG_SIZES,batch_size=BATCH_SIZES*2)\n        ids = np.array([target.numpy() for _, target in tqdm(ds.unbatch())]).flatten()\n    else:\n        ds = get_dataset(paths,labeled=False,return_image_ids=True,\n                repeat=False,shuffle=False,dim=IMG_SIZES,batch_size=BATCH_SIZES*2)\n        ids = np.array([target.numpy().decode(\"utf-8\") for _, target in tqdm(ds.unbatch())]).flatten()\n    return pred, ids","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-09T19:45:51.714683Z","iopub.execute_input":"2021-08-09T19:45:51.715059Z","iopub.status.idle":"2021-08-09T19:45:51.731751Z","shell.execute_reply.started":"2021-08-09T19:45:51.715027Z","shell.execute_reply":"2021-08-09T19:45:51.730501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"6\"></a>\n# Inference","metadata":{}},{"cell_type":"code","source":"pred, ids = predict(np.array(files_test_g), False)","metadata":{"papermill":{"duration":21.234649,"end_time":"2021-06-12T18:39:46.151391","exception":false,"start_time":"2021-06-12T18:39:24.916742","status":"completed"},"tags":[],"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-09T19:45:51.733663Z","iopub.execute_input":"2021-08-09T19:45:51.734036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.mean(pred,axis = 0).shape","metadata":{"execution":{"iopub.status.busy":"2021-08-09T19:44:55.798569Z","iopub.status.idle":"2021-08-09T19:44:55.799116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(f'{COMPETITION_DATASET_PATH}/sample_submission.csv')\nsub['id'] = ids\nsub['target'] = np.mean(pred,axis = 0)\nsub = sub.sort_values('id') \nsub.head()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-09T19:44:55.800027Z","iopub.status.idle":"2021-08-09T19:44:55.800591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('submission.csv', index=False)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-09T19:44:55.801464Z","iopub.status.idle":"2021-08-09T19:44:55.801999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"7\"></a>\n# Next steps\n* Add TTA Inference","metadata":{}}]}