{"metadata":{"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":59093,"databundleVersionId":7469972},{"sourceType":"datasetVersion","sourceId":7752462,"datasetId":4382744,"databundleVersionId":7853112},{"sourceType":"datasetVersion","sourceId":7698585,"datasetId":4493636,"databundleVersionId":7797034},{"sourceType":"datasetVersion","sourceId":7942214,"datasetId":4486710,"databundleVersionId":8051088}],"dockerImageVersionId":30646,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true},"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"print('Initialize ENV')\nRUN_TRAINED_MODEL = True\nRUN_ON_KAGGLE = True\nLOAD_KAGGLE_SPECTROGRAMS = False\nFOR_SUBMISSION = True\nMIX = True","metadata":{"execution":{"iopub.status.busy":"2024-03-25T23:09:47.806951Z","iopub.execute_input":"2024-03-25T23:09:47.807306Z","iopub.status.idle":"2024-03-25T23:09:47.812839Z","shell.execute_reply.started":"2024-03-25T23:09:47.807281Z","shell.execute_reply":"2024-03-25T23:09:47.811799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('import libs')\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nimport os, gc\nos.environ[\"CUDA_VISIBLE_DEVICES\"]=\"0,1\"\nimport time\nimport math\nimport random\nimport json\nimport tensorflow as tf\nimport pandas as pd, numpy as np","metadata":{"execution":{"iopub.status.busy":"2024-03-25T22:09:27.157278Z","iopub.execute_input":"2024-03-25T22:09:27.157935Z","iopub.status.idle":"2024-03-25T22:09:41.164007Z","shell.execute_reply.started":"2024-03-25T22:09:27.157906Z","shell.execute_reply":"2024-03-25T22:09:41.162888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Config GPU')\ngpus = tf.config.list_physical_devices('GPU')\nif len(gpus)<=1: \n    strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n    print(f'Using {len(gpus)} GPU')\nelse: \n    strategy = tf.distribute.MirroredStrategy()\n    print(f'Using {len(gpus)} GPUs')\n","metadata":{"execution":{"iopub.status.busy":"2024-03-25T22:10:20.321205Z","iopub.execute_input":"2024-03-25T22:10:20.322375Z","iopub.status.idle":"2024-03-25T22:10:21.397196Z","shell.execute_reply.started":"2024-03-25T22:10:20.322342Z","shell.execute_reply":"2024-03-25T22:10:21.396307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Mix Precision')\nif MIX:\n    tf.config.optimizer.set_experimental_options({\"auto_mixed_precision\": True})\n    print('Mixed precision enabled')\nelse:\n    print('Using full precision')","metadata":{"execution":{"iopub.status.busy":"2024-03-25T22:10:24.039933Z","iopub.execute_input":"2024-03-25T22:10:24.040311Z","iopub.status.idle":"2024-03-25T22:10:24.046940Z","shell.execute_reply.started":"2024-03-25T22:10:24.040283Z","shell.execute_reply":"2024-03-25T22:10:24.045872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('set up file path')\nif RUN_ON_KAGGLE:\n    print('run on kaggle!!!')\n    FILE_BASE_PATH = '/kaggle/input/hms-harmful-brain-activity-classification/'\n    MY_FILE_BASE_PATH = '/kaggle/input/hms-egg-spectrograms/'\n    #MY_FILE_BASE_PATH = '/kaggle/input/hms-eeg-sptrograms-fd/'\nelse:\n    print('run on local!!!')\n    FILE_BASE_PATH = 'C:\\\\Work\\\\projects\\\\ML\\\\jupyter_book\\\\HMS - Harmful Brain Activity Classification\\\\'\n    MY_FILE_BASE_PATH = 'C:\\\\Work\\\\projects\\\\ML\\\\jupyter_book\\\\HMS - Harmful Brain Activity Classification\\\\'","metadata":{"execution":{"iopub.status.busy":"2024-03-25T22:10:26.257738Z","iopub.execute_input":"2024-03-25T22:10:26.258142Z","iopub.status.idle":"2024-03-25T22:10:26.264172Z","shell.execute_reply.started":"2024-03-25T22:10:26.258113Z","shell.execute_reply":"2024-03-25T22:10:26.263219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('load train.csv')\ndf = pd.read_csv(FILE_BASE_PATH + \"train.csv\")\nTARGETS = df.columns[-6:]\nprint('Train shape:', df.shape )\nprint('Targets', list(TARGETS))\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-25T22:10:28.957143Z","iopub.execute_input":"2024-03-25T22:10:28.957499Z","iopub.status.idle":"2024-03-25T22:10:29.243838Z","shell.execute_reply.started":"2024-03-25T22:10:28.957463Z","shell.execute_reply":"2024-03-25T22:10:29.242727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Select Trained Data and Labels')\nif not RUN_TRAINED_MODEL:\n    #Only use train dat with total vote count > 10\n    df1 = df[df['seizure_vote'] + df['lpd_vote'] + df['gpd_vote'] + \n             df['lrda_vote'] + df['grda_vote'] + df['other_vote'] > 10]\n\n    train = df1.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n        {'spectrogram_id':'first','spectrogram_label_offset_seconds':'min'})\n    train.columns = ['spec_id','min']\n\n    tmp = df1.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n        {'spectrogram_label_offset_seconds':'max'})\n    train['max'] = tmp\n\n    tmp = df1.groupby('eeg_id')[TARGETS].agg('sum')\n    for t in TARGETS:\n        train[t] = tmp[t].values\n\n    y_data = train[TARGETS].values\n    y_data = y_data / y_data.sum(axis=1,keepdims=True)\n    train[TARGETS] = y_data\n\n    tmp = df1.groupby('eeg_id')[['expert_consensus']].agg('first')\n    train['target'] = tmp\n\n    train = train.reset_index()\n    print('Train non-overlapp eeg_id shape:', train.shape )\n\n    train.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-25T22:10:31.567869Z","iopub.execute_input":"2024-03-25T22:10:31.568224Z","iopub.status.idle":"2024-03-25T22:10:31.632638Z","shell.execute_reply.started":"2024-03-25T22:10:31.568197Z","shell.execute_reply":"2024-03-25T22:10:31.631561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nprint('Load Train Spectrograms')\nif not RUN_TRAINED_MODEL:\n    if LOAD_KAGGLE_SPECTROGRAMS:\n        if RUN_ON_KAGGLE:\n            TRAIN_SPEC_PATH = FILE_BASE_PATH + \"train_spectrograms/\"\n        else: \n            TRAIN_SPEC_PATH = FILE_BASE_PATH + \"train_spectrograms\\\\\"\n\n        files = os.listdir(TRAIN_SPEC_PATH)\n        print(f'There are {len(files)} spectrogram parquets')\n\n        spectrograms_ids = np.empty((0), dtype = np.int64)\n        spectrograms_images = np.empty((0, 256, 400))\n        spectrograms_labels = np.empty((0,6))\n        spectrograms_experts = np.empty((0), dtype = np.string_)\n        for i,f in enumerate(files):\n            if i%100==0:print(i,', ',end='')\n            name = int(f.split('.')[0])\n            tmp2 = train[train[\"spec_id\"]==name]\n            if tmp2.shape[0] > 0:\n                tmp = pd.read_parquet(f'{TRAIN_SPEC_PATH}{f}')\n                spectrograms_labels = np.insert(spectrograms_labels, spectrograms_labels.shape[0], \n                                                    tmp2[0:1][TARGETS], axis=0)\n                spectrograms_images = np.insert(spectrograms_images, spectrograms_images.shape[0], \n                                                    tmp.iloc[0:256,1:].values, axis=0)\n                spectrograms_ids = np.insert(spectrograms_ids, spectrograms_ids.shape[0], [name], axis=0)\n                spectrograms_experts = np.insert(spectrograms_experts, spectrograms_experts.shape[0], \n                                                    tmp2[0:1][\"target\"], axis=0)\n        if not FOR_SUBMISSION:\n            with open('spectrograms_labels.npy', 'wb') as fspectrograms_labels:\n                np.save(fspectrograms_labels, spectrograms_labels)\n\n            with open('spectrograms_images.npy', 'wb') as fspectrograms_images:\n                np.save(fspectrograms_images, spectrograms_images)\n\n            with open('spectrograms_ids.npy', 'wb') as fspectrograms_ids:\n                np.save(fspectrograms_ids, spectrograms_ids)\n\n            with open('spectrograms_experts.npy', 'wb') as fspectrograms_experts:\n                np.save(fspectrograms_experts, spectrograms_experts)\n            \n    else:\n        with open(MY_FILE_BASE_PATH + 'spectrograms_labels.npy', 'rb') as fspectrograms_labels:\n            spectrograms_labels = np.load(fspectrograms_labels)\n    \n        with open(MY_FILE_BASE_PATH + 'spectrograms_images.npy', 'rb') as fspectrograms_images:\n            spectrograms_images = np.load(fspectrograms_images)\n    \n        with open(MY_FILE_BASE_PATH + 'spectrograms_ids.npy', 'rb') as fspectrograms_ids:\n            spectrograms_ids = np.load(fspectrograms_ids)\n    \n        with open(MY_FILE_BASE_PATH + 'spectrograms_experts.npy', 'rb') as fspectrograms_experts:\n            spectrograms_experts = np.load(fspectrograms_experts)","metadata":{"execution":{"iopub.status.busy":"2024-03-25T22:10:36.104747Z","iopub.execute_input":"2024-03-25T22:10:36.105104Z","iopub.status.idle":"2024-03-25T22:11:04.000224Z","shell.execute_reply.started":"2024-03-25T22:10:36.105077Z","shell.execute_reply":"2024-03-25T22:11:03.999200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('re-size spectrograms from 256x400x1 to 256x100x4')\ndef egg_spec_img_resize(spec_imags):\n    #spec_images_c3 = np.zeros((spec_imags.shape[0], 64, 200, 3), dtype='float32')\n    \n    #spec_images_c3[:,:, 0:200, 0] = spec_imags[:,96:160, 0:200]\n    #spec_images_c3[:,:, 0:200, 1] = spec_imags[:,96:160, 200:400]\n    #spec_images_c3[:,:, 0:100, 2] = spec_imags[:,96:160, 100:200]\n    #spec_images_c3[:,:, 100:200, 2] = spec_imags[:,96:160, 300:400]\n    \n    spec_images_c3 = np.zeros((spec_imags.shape[0], 128, 200, 3), dtype='float32')\n    \n    spec_images_c3[:,:, 0:200, 0] = spec_imags[:,64:192, 0:200]\n    spec_images_c3[:,:, 0:200, 1] = spec_imags[:,64:192, 200:400]\n    spec_images_c3[:,:, 0:100, 2] = spec_imags[:,64:192, 100:200]\n    spec_images_c3[:,:, 100:200, 2] = spec_imags[:,64:192, 300:400]\n    \n    return spec_images_c3\n\nif not RUN_TRAINED_MODEL:\n    spectrograms_images_c3 = egg_spec_img_resize(spectrograms_images)\n    del spectrograms_images\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-25T22:11:10.742351Z","iopub.execute_input":"2024-03-25T22:11:10.743147Z","iopub.status.idle":"2024-03-25T22:11:12.586358Z","shell.execute_reply.started":"2024-03-25T22:11:10.743117Z","shell.execute_reply":"2024-03-25T22:11:12.585594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Normaliztion spectrogram values to 0 ... 255')    \ndef egg_spec_img_normalization(spec_images):\n    # LOG TRANSFORM SPECTROGRAM\n    spec_images = np.clip(spec_images,np.exp(-4),np.exp(8))\n    spec_images = np.log(spec_images)\n    \n    spec_images = np.nan_to_num(spec_images, nan=0.0)    \n\n    ep = 1e-6\n    rows = spec_images.shape[0]\n    for ij in range(0, rows):\n        minv = spec_images[ij].flatten().min()\n        maxv = spec_images[ij].flatten().max()\n        spec_images[ij] = 255 * (spec_images[ij] - minv) / (maxv - minv + ep)\n        \n    return spec_images\n\nif not RUN_TRAINED_MODEL:\n    spectrograms_images_c3 = egg_spec_img_normalization(spectrograms_images_c3)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-25T22:11:15.447122Z","iopub.execute_input":"2024-03-25T22:11:15.447729Z","iopub.status.idle":"2024-03-25T22:11:19.685997Z","shell.execute_reply.started":"2024-03-25T22:11:15.447700Z","shell.execute_reply":"2024-03-25T22:11:19.685156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Set Up model EfficientNet')\nimport tensorflow as tf\nimport keras\nfrom keras import layers, layers, models\nfrom keras.applications import EfficientNetB2\n\ndef build_model_EfficientNet():\n    \n    with strategy.scope():\n        inputs = layers.Input(shape=(128, 200, 3))\n        model = EfficientNetB2(include_top=False, input_tensor=inputs)\n\n        # Rebuild top\n        x = layers.GlobalAveragePooling2D(name=\"avg_pool\")(model.output)\n        #x = layers.BatchNormalization()(x)\n        #top_dropout_rate = 0.2\n        #x = layers.Dropout(top_dropout_rate, name=\"top_dropout\")(x)\n        outputs = layers.Dense(6, activation=\"softmax\")(x)\n\n        model = keras.Model(inputs, outputs, name=\"EfficientNet\")\n        optimizer = keras.optimizers.Adam(learning_rate=1e-4)\n        model.compile(optimizer=optimizer,\n                          loss='kullback_leibler_divergence',\n                          #metrics=[\"accuracy\"]\n                          metrics=[\"categorical_accuracy\"]\n                          #metrics=[keras.metrics.KLDivergence()]\n        )\n    return model\n\nif not RUN_TRAINED_MODEL:\n    modelEfficientNet = build_model_EfficientNet()\n    modelEfficientNet.summary()","metadata":{"execution":{"iopub.status.busy":"2024-03-25T22:11:23.536918Z","iopub.execute_input":"2024-03-25T22:11:23.537598Z","iopub.status.idle":"2024-03-25T22:11:30.700690Z","shell.execute_reply.started":"2024-03-25T22:11:23.537566Z","shell.execute_reply":"2024-03-25T22:11:30.699816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Set Up model ResNet')\nimport tensorflow as tf\nimport keras\nfrom keras import layers, layers, models\nfrom keras.applications import ResNet101\n\ndef build_model_ResNet():\n    \n    with strategy.scope():\n        inputs = layers.Input(shape=(128, 200, 3))\n        model = ResNet101(include_top=False, input_tensor=inputs)\n\n        # Rebuild top\n        x = layers.GlobalAveragePooling2D(name=\"avg_pool\")(model.output)\n        #x = layers.BatchNormalization()(x)\n        #top_dropout_rate = 0.2\n        #x = layers.Dropout(top_dropout_rate, name=\"top_dropout\")(x)\n        outputs = layers.Dense(6, activation=\"softmax\")(x)\n\n        model = keras.Model(inputs, outputs, name=\"ResNet50\")\n        optimizer = keras.optimizers.Adam(learning_rate=1e-4)\n        model.compile(optimizer=optimizer,\n                          loss='kullback_leibler_divergence',\n                          #metrics=[\"accuracy\"]\n                          metrics=[\"categorical_accuracy\"]\n                          #metrics=[keras.metrics.KLDivergence()]\n        )\n    return model\n\nif not RUN_TRAINED_MODEL:\n    modelResNet = build_model_ResNet()\n    modelResNet.summary()","metadata":{"execution":{"iopub.status.busy":"2024-03-25T22:11:49.905974Z","iopub.execute_input":"2024-03-25T22:11:49.906582Z","iopub.status.idle":"2024-03-25T22:11:58.845415Z","shell.execute_reply.started":"2024-03-25T22:11:49.906551Z","shell.execute_reply":"2024-03-25T22:11:58.844714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Set Up model DenseNet')\nimport tensorflow as tf\nimport keras\nfrom keras import layers, layers, models\nfrom keras.applications import DenseNet169\n\ndef build_model_DenseNet():\n    \n    with strategy.scope():\n        inputs = layers.Input(shape=(128, 200, 3))\n        model = DenseNet169(include_top=False, input_tensor=inputs)\n\n        # Rebuild top\n        x = layers.GlobalAveragePooling2D(name=\"avg_pool\")(model.output)\n        #x = layers.BatchNormalization()(x)\n        #top_dropout_rate = 0.2\n        #x = layers.Dropout(top_dropout_rate, name=\"top_dropout\")(x)\n        outputs = layers.Dense(6, activation=\"softmax\")(x)\n\n        model = keras.Model(inputs, outputs, name=\"DenseNet169\")\n        optimizer = keras.optimizers.Adam(learning_rate=1e-4)\n        model.compile(optimizer=optimizer,\n                          loss='kullback_leibler_divergence',\n                          #metrics=[\"accuracy\"]\n                          metrics=[\"categorical_accuracy\"]\n                          #metrics=[keras.metrics.KLDivergence()]\n        )\n    return model\n\nif not RUN_TRAINED_MODEL:\n    modelDenseNet = build_model_DenseNet()\n    modelDenseNet.summary()","metadata":{"execution":{"iopub.status.busy":"2024-03-25T22:12:21.117177Z","iopub.execute_input":"2024-03-25T22:12:21.117981Z","iopub.status.idle":"2024-03-25T22:12:33.204768Z","shell.execute_reply.started":"2024-03-25T22:12:21.117946Z","shell.execute_reply":"2024-03-25T22:12:33.203873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom sklearn.model_selection import train_test_split, KFold\nimport math as ops\n\ndef train_split(spectrograms_images_c3):\n    X_train, X_test, y_train, y_test = train_test_split(spectrograms_images_c3, spectrograms_labels, test_size = 0.1)\n\n    del spectrograms_images_c3\n    gc.collect()\n    \n    return X_train, X_test, y_train, y_test\n\ndef train_model(model, X_train, y_train):\n    folds = list(KFold(n_splits=5, shuffle=True, random_state=1).split(X_train, y_train))\n    \n    \n    def scheduler(epoch):\n        return [1e-4,1e-4,1e-4,1e-5,1e-5][epoch]\n            \n    callback = keras.callbacks.LearningRateScheduler(scheduler)\n    \n    epochs = 3\n    batch_size = 10\n    for j, (train_index, val_index) in enumerate(folds):\n\n        print('\\nFold ',j)\n        #X_train_cv = X_train[train_index]\n        #y_train_cv = y_train[train_index]\n        #X_valid_cv = X_train[val_index]\n        #y_valid_cv = y_train[val_index]\n\n        model.fit(\n                    X_train[train_index], y_train[train_index],\n                    steps_per_epoch=len(X_train[train_index])/batch_size,\n                    epochs=epochs,\n                    batch_size=batch_size, \n                    shuffle=True,\n                    verbose=1,\n                    #callbacks=[callback],\n                    validation_data = (X_train[val_index], y_train[val_index])\n                 )\n        print(model.evaluate(X_train[val_index], y_train[val_index]))\n        \n    return model\n\n    \nif not RUN_TRAINED_MODEL:\n    X_train, X_test, y_train, y_test = train_split(spectrograms_images_c3)\n    \n    modelEfficientNet = train_model(modelEfficientNet, X_train, y_train)\n    if not FOR_SUBMISSION: \n        modelEfficientNet.save('modelEfficientNet.keras')\n        \n    modelResNet = train_model(modelResNet, X_train, y_train)    \n    if not FOR_SUBMISSION: \n        modelResNet.save('modelResNet.keras')\n        \n    modelDenseNet = train_model(modelDenseNet, X_train, y_train)    \n    if not FOR_SUBMISSION: \n        modelDenseNet.save('modelDenseNet.keras')\n","metadata":{"execution":{"iopub.status.busy":"2024-03-25T22:14:05.466492Z","iopub.execute_input":"2024-03-25T22:14:05.467345Z","iopub.status.idle":"2024-03-25T22:56:27.132561Z","shell.execute_reply.started":"2024-03-25T22:14:05.467314Z","shell.execute_reply":"2024-03-25T22:56:27.131476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not RUN_TRAINED_MODEL:\n    pred_test1 = modelResNet.predict(X_test)\n    kl = tf.keras.losses.KLDivergence()\n    kl(y_test, pred_test1).numpy()\n    print(kl(y_test, pred_test1).numpy())","metadata":{"execution":{"iopub.status.busy":"2024-03-25T23:07:13.512849Z","iopub.execute_input":"2024-03-25T23:07:13.513926Z","iopub.status.idle":"2024-03-25T23:07:20.367616Z","shell.execute_reply.started":"2024-03-25T23:07:13.513893Z","shell.execute_reply":"2024-03-25T23:07:20.366697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not RUN_TRAINED_MODEL:\n    pred_test2 = modelEfficientNet.predict(X_test)\n    kl = tf.keras.losses.KLDivergence()\n    kl(y_test, pred_test2).numpy()\n    print(kl(y_test, pred_test2).numpy())","metadata":{"execution":{"iopub.status.busy":"2024-03-25T23:07:23.709198Z","iopub.execute_input":"2024-03-25T23:07:23.709556Z","iopub.status.idle":"2024-03-25T23:07:29.991084Z","shell.execute_reply.started":"2024-03-25T23:07:23.709528Z","shell.execute_reply":"2024-03-25T23:07:29.990195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not RUN_TRAINED_MODEL:\n    pred_test3 = modelDenseNet.predict(X_test)\n    kl = tf.keras.losses.KLDivergence()\n    kl(y_test, pred_test3).numpy()\n    print(kl(y_test, pred_test3).numpy())","metadata":{"execution":{"iopub.status.busy":"2024-03-25T23:07:39.373805Z","iopub.execute_input":"2024-03-25T23:07:39.374683Z","iopub.status.idle":"2024-03-25T23:07:47.445880Z","shell.execute_reply.started":"2024-03-25T23:07:39.374653Z","shell.execute_reply":"2024-03-25T23:07:47.444870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not RUN_TRAINED_MODEL:\n    pred_test = (pred_test1 + pred_test2 + pred_test3)/3.0\n    kl = tf.keras.losses.KLDivergence()\n    print(kl(y_test, pred_test).numpy())","metadata":{"execution":{"iopub.status.busy":"2024-03-25T23:08:00.927133Z","iopub.execute_input":"2024-03-25T23:08:00.927917Z","iopub.status.idle":"2024-03-25T23:08:00.939153Z","shell.execute_reply.started":"2024-03-25T23:08:00.927874Z","shell.execute_reply":"2024-03-25T23:08:00.938126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n%%time\nif RUN_ON_KAGGLE:\n    TEST_SPEC_PATH = FILE_BASE_PATH + \"test_spectrograms/\"\nelse:\n    TEST_SPEC_PATH = FILE_BASE_PATH + \"test_spectrograms\\\\\"\nfiles = os.listdir(TEST_SPEC_PATH)\nprint(f'There are {len(files)} spectrogram parquets')\n\ntest_spectrograms_ids = np.empty((0), dtype = np.int64)\ntest_spectrograms_images = np.empty((0, 256, 400))\ntest_spectrograms_labels = np.empty((0,6))\ntest_spectrograms_experts = np.empty((0), dtype = np.string_)\nfor i,f in enumerate(files):\n    if i%100==0:print(i,', ',end='')\n    name = int(f.split('.')[0])\n    tmp = pd.read_parquet(f'{TEST_SPEC_PATH}{f}')\n    test_spectrograms_images = np.insert(test_spectrograms_images, test_spectrograms_images.shape[0], \n                                                tmp.iloc[0:256,1:].values, axis=0)\n    test_spectrograms_ids = np.insert(test_spectrograms_ids, test_spectrograms_ids.shape[0], [name], axis=0)\n    ","metadata":{"execution":{"iopub.status.busy":"2024-03-25T23:08:17.349195Z","iopub.execute_input":"2024-03-25T23:08:17.349792Z","iopub.status.idle":"2024-03-25T23:08:17.668805Z","shell.execute_reply.started":"2024-03-25T23:08:17.349762Z","shell.execute_reply":"2024-03-25T23:08:17.667878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_spectrograms_images_c3 = egg_spec_img_resize(test_spectrograms_images)\ntest_spectrograms_images_c3 = egg_spec_img_normalization(test_spectrograms_images_c3)","metadata":{"execution":{"iopub.status.busy":"2024-03-25T23:08:21.347073Z","iopub.execute_input":"2024-03-25T23:08:21.347410Z","iopub.status.idle":"2024-03-25T23:08:21.353189Z","shell.execute_reply.started":"2024-03-25T23:08:21.347384Z","shell.execute_reply":"2024-03-25T23:08:21.352202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if RUN_TRAINED_MODEL:\n    if RUN_ON_KAGGLE:\n        MODEL_PATH = MY_FILE_BASE_PATH\n    else:\n        MODEL_PATH = MY_FILE_BASE_PATH\n    print('Load model modelResNet')\n    modelResNet = tf.keras.models.load_model(MODEL_PATH + 'modelResNet.keras')\n    pred_test_spectrograms_images_c3_1 = modelResNet.predict(test_spectrograms_images_c3)\n    \n    print('Load model modelEfficientNet')\n    modelEfficientNet = tf.keras.models.load_model(MODEL_PATH + 'modelEfficientNet.keras')\n    pred_test_spectrograms_images_c3_2 = modelEfficientNet.predict(test_spectrograms_images_c3)\n    \n    print('Load model modelDenseNet')\n    modelDenseNet = tf.keras.models.load_model(MODEL_PATH + 'modelDenseNet.keras')\n    pred_test_spectrograms_images_c3_3 = modelDenseNet.predict(test_spectrograms_images_c3)\n    \n    pred_test_spectrograms_images_c3 = (pred_test_spectrograms_images_c3_1 + pred_test_spectrograms_images_c3_2 + pred_test_spectrograms_images_c3_3)/3.0\n    \nimport random\n\ntest = pd.read_csv(FILE_BASE_PATH + 'test.csv')\n\neeg_ids = np.empty((0), dtype = np.int64)\nfor id in test_spectrograms_ids:\n    eeg_id = test[test['spectrogram_id'] == id]['eeg_id']\n    if eeg_id.shape[0] > 0:\n        eeg_ids = np.insert(eeg_ids, eeg_ids.shape[0], [eeg_id.values[0]], axis = 0)\n    \nprint('eeg_ids.shape=', eeg_ids.shape)\n\nsubmission = pd.DataFrame({'eeg_id':eeg_ids})\n\npred_test_spectrograms_images_c3 = np.nan_to_num(pred_test_spectrograms_images_c3, nan=0.0)\n\nsubmission[TARGETS] = pred_test_spectrograms_images_c3\nsubmission.to_csv('submission.csv',index=False)\nprint('submissionn',submission.shape)\nprint(submission.head())","metadata":{"execution":{"iopub.status.busy":"2024-03-25T23:08:24.386973Z","iopub.execute_input":"2024-03-25T23:08:24.387991Z","iopub.status.idle":"2024-03-25T23:08:25.129987Z","shell.execute_reply.started":"2024-03-25T23:08:24.387951Z","shell.execute_reply":"2024-03-25T23:08:25.128738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\na=np.array([[0,1],[2,1],[3,1],[4,1]])\na.resize((2,2))\nprint(a)","metadata":{},"execution_count":null,"outputs":[]}]}