{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.10","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":67356,"databundleVersionId":8006601},{"sourceType":"datasetVersion","sourceId":8860486,"datasetId":5318563,"databundleVersionId":9020503},{"sourceType":"datasetVersion","sourceId":8275617,"datasetId":4914065,"databundleVersionId":8404778}],"dockerImageVersionId":30514,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# BELKA 1DCNN Model Ensemble (20 Folds * 25 Epoch)\n\nInspired by [AH](https://www.kaggle.com/code/ahmedelfazouan/belka-1dcnn-starter-with-all-data).\n\nCopied from [JIADI WANG](https://www.kaggle.com/code/hugowjd/belka-1dcnn-with-all-data-15-folds-20-epoch) for environment setup (2023-06-15).\n\nIdea:\n* Encode smiles into padded list of numbers (length=142).\n* Fit 1DCnn model on processed data.\n* Train the model with 20 folds, each with 25 epoch.\n* Store model weights in each fold.\n* Load pretrain model weights to get an ensemble result.","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2024-07-04T20:41:32.761198Z","iopub.execute_input":"2024-07-04T20:41:32.761715Z","iopub.status.idle":"2024-07-04T20:41:32.782935Z","shell.execute_reply.started":"2024-07-04T20:41:32.761688Z","shell.execute_reply":"2024-07-04T20:41:32.782034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install fastparquet -q","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-07-04T20:41:32.784407Z","iopub.execute_input":"2024-07-04T20:41:32.784697Z","iopub.status.idle":"2024-07-04T20:41:45.799693Z","shell.execute_reply.started":"2024-07-04T20:41:32.784673Z","shell.execute_reply":"2024-07-04T20:41:45.798685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\nimport os\nimport pickle\nimport random\nimport joblib\nfrom tqdm import tqdm\nimport tensorflow as tf\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import average_precision_score as APS","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-07-04T20:41:45.800997Z","iopub.execute_input":"2024-07-04T20:41:45.801278Z","iopub.status.idle":"2024-07-04T20:41:54.276622Z","shell.execute_reply.started":"2024-07-04T20:41:45.801252Z","shell.execute_reply":"2024-07-04T20:41:54.275838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    PREPROCESS = False\n    PRETRAINED = True\n    \n    EPOCHS = 25\n    BATCH_SIZE = 4096\n    LR = 1e-3\n    WD = 0.05\n    \n    # Number of folds\n    NBR_FOLDS = 20\n\n    # Get weights 4 at a time\n#     SELECTED_FOLDS = [i for i in range(4)]\n#     SELECTED_FOLDS = [i for i in range(4, 8)]\n#     SELECTED_FOLDS = [i for i in range(8, 12)]\n#     SELECTED_FOLDS = [i for i in range(12, 16)]\n#     SELECTED_FOLDS = [i for i in range(16, 20)]\n\n    SEED = 2024\n    LEN = 142","metadata":{"execution":{"iopub.status.busy":"2024-07-04T20:41:54.279224Z","iopub.execute_input":"2024-07-04T20:41:54.280268Z","iopub.status.idle":"2024-07-04T20:41:54.285256Z","shell.execute_reply.started":"2024-07-04T20:41:54.280230Z","shell.execute_reply":"2024-07-04T20:41:54.284309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seeds(seed):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n    tf.random.set_seed(seed)\n    np.random.seed(seed)\n\nset_seeds(seed=CFG.SEED)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-07-04T20:41:54.286363Z","iopub.execute_input":"2024-07-04T20:41:54.286648Z","iopub.status.idle":"2024-07-04T20:41:54.297562Z","shell.execute_reply.started":"2024-07-04T20:41:54.286602Z","shell.execute_reply":"2024-07-04T20:41:54.296792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\n\n# Detect hardware, return appropriate distribution strategy\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect(tpu=\"local\") # \"local\" for 1VM TPU\n    strategy = tf.distribute.TPUStrategy(tpu)\n    print(\"Running on TPU\")\n    print(\"REPLICAS: \", strategy.num_replicas_in_sync)\nexcept tf.errors.NotFoundError:\n    strategy = tf.distribute.get_strategy()\n    print(\"Not on TPU\")","metadata":{"execution":{"iopub.status.busy":"2024-07-04T20:41:54.298635Z","iopub.execute_input":"2024-07-04T20:41:54.298916Z","iopub.status.idle":"2024-07-04T20:41:56.636519Z","shell.execute_reply.started":"2024-07-04T20:41:54.298893Z","shell.execute_reply":"2024-07-04T20:41:56.635518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"code","source":"if CFG.PREPROCESS:\n    # Similar to ASCII\n    enc = {'l': 1, 'y': 2, '@': 3, '3': 4, 'H': 5, 'S': 6, 'F': 7, 'C': 8, 'r': 9, 's': 10, '/': 11, 'c': 12, 'o': 13,\n           '+': 14, 'I': 15, '5': 16, '(': 17, '2': 18, ')': 19, '9': 20, 'i': 21, '#': 22, '6': 23, '8': 24, '4': 25, '=': 26,\n           '1': 27, 'O': 28, '[': 29, 'D': 30, 'B': 31, ']': 32, 'N': 33, '7': 34, 'n': 35, '-': 36}\n    \n    def encode_smile(smile):\n        tmp = [enc[i] for i in smile]\n        # Pad encoded list with 0s\n        tmp = tmp + [0]*(CFG.LEN-len(tmp))\n        return np.array(tmp).astype(np.uint8)\n    \n    # Every smile comes in a set of three (protein binds)\n    # Only process the unique smiles\n    train_raw = pd.read_parquet('/kaggle/input/leash-BELKA/train.parquet')\n    smiles = train_raw[train_raw['protein_name']=='BRD4']['molecule_smiles'].values\n    assert (smiles!=train_raw[train_raw['protein_name']=='HSA']['molecule_smiles'].values).sum() == 0\n    assert (smiles!=train_raw[train_raw['protein_name']=='sEH']['molecule_smiles'].values).sum() == 0\n\n    # Speed up\n    smiles_enc = joblib.Parallel(n_jobs=96)(joblib.delayed(encode_smile)(smile) for smile in tqdm(smiles))\n    smiles_enc = np.stack(smiles_enc)\n    \n    train = pd.DataFrame(smiles_enc, columns = [f'enc{i}' for i in range(CFG.LEN)])\n    train['bind1'] = train_raw[train_raw['protein_name']=='BRD4']['binds'].values\n    train['bind2'] = train_raw[train_raw['protein_name']=='HSA']['binds'].values\n    train['bind3'] = train_raw[train_raw['protein_name']=='sEH']['binds'].values\n    train.to_parquet('train_enc.parquet')\n\n    # Similiar process for test set\n    test_raw = pd.read_parquet('/kaggle/input/leash-BELKA/test.parquet')\n    smiles = test_raw['molecule_smiles'].values\n\n    smiles_enc = joblib.Parallel(n_jobs=96)(joblib.delayed(encode_smile)(smile) for smile in tqdm(smiles))\n    smiles_enc = np.stack(smiles_enc)\n    \n    test = pd.DataFrame(smiles_enc, columns = [f'enc{i}' for i in range(CFG.LEN)])\n    test.to_parquet('test_enc.parquet')\n\nelse:\n    if not CFG.PRETRAINED:\n        # Preprocessed data by AH\n        # Reference: https://www.kaggle.com/datasets/ahmedelfazouan/belka-enc-dataset/code\n        train = pd.read_parquet('/kaggle/input/belka-enc-dataset/train_enc.parquet')\n        \n    test = pd.read_parquet('/kaggle/input/belka-enc-dataset/test_enc.parquet')","metadata":{"execution":{"iopub.status.busy":"2024-07-04T20:41:56.639719Z","iopub.execute_input":"2024-07-04T20:41:56.640019Z","iopub.status.idle":"2024-07-04T20:41:58.165940Z","shell.execute_reply.started":"2024-07-04T20:41:56.639991Z","shell.execute_reply":"2024-07-04T20:41:58.165171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"code","source":"# 1D-CNN model\ndef my_model():\n    with strategy.scope():\n        NUM_FILTERS = 32\n        hidden_dim = 128\n        enc_len = 36\n\n        # Define the input layer with a specific shape and data type\n        inputs = tf.keras.layers.Input(shape=(CFG.LEN,), dtype='int32')\n        \n        # Add an Embedding layer to convert input integers to dense vectors of fixed size\n        x = tf.keras.layers.Embedding(\n            input_dim=enc_len, \n            output_dim=hidden_dim, \n            input_length=CFG.LEN, \n            mask_zero = True\n        )(inputs)\n        \n        # Add Conv1D layers (filters are increase each layer)\n        x = tf.keras.layers.Conv1D(\n            filters=NUM_FILTERS, \n            kernel_size=3,  \n            activation='relu', \n            padding='valid',  \n            strides=1\n        )(x)\n        x = tf.keras.layers.Conv1D(\n            filters=NUM_FILTERS*2, \n            kernel_size=3,  \n            activation='relu', \n            padding='valid',  \n            strides=1)(x)\n        x = tf.keras.layers.Conv1D(\n            filters=NUM_FILTERS*3, \n            kernel_size=3, \n            activation='relu', \n            padding='valid', \n            strides=1\n        )(x)\n        \n        # Downsample the input by taking the maximum value over time steps\n        x = tf.keras.layers.GlobalMaxPooling1D()(x)\n\n        # Add Dense and Droupout layers\n        x = tf.keras.layers.Dense(1024, activation='relu')(x)\n        x = tf.keras.layers.Dropout(0.1)(x)\n        x = tf.keras.layers.Dense(1024, activation='relu')(x)\n        x = tf.keras.layers.Dropout(0.1)(x)\n        x = tf.keras.layers.Dense(512, activation='relu')(x)\n        x = tf.keras.layers.Dropout(0.1)(x)\n\n        # Multi-label Classification\n        outputs = tf.keras.layers.Dense(3, activation='sigmoid')(x)\n\n        # Compile model\n        model = tf.keras.models.Model(inputs = inputs, outputs = outputs)\n        optimizer = tf.keras.optimizers.AdamW(learning_rate=CFG.LR, weight_decay=CFG.WD)\n        loss = 'binary_crossentropy'\n        weighted_metrics = [tf.keras.metrics.AUC(curve='PR', name = 'avg_precision')]\n        model.compile(\n        loss=loss,\n        optimizer=optimizer,\n        weighted_metrics=weighted_metrics,\n        )\n        \n        return model\n    \nmodel = my_model()\n\nmodel.summary()\ntf.keras.backend.clear_session()","metadata":{"execution":{"iopub.status.busy":"2024-07-04T20:41:58.166843Z","iopub.execute_input":"2024-07-04T20:41:58.167175Z","iopub.status.idle":"2024-07-04T20:41:58.776047Z","shell.execute_reply.started":"2024-07-04T20:41:58.167144Z","shell.execute_reply":"2024-07-04T20:41:58.775138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train","metadata":{}},{"cell_type":"code","source":"%time\n\nFEATURES = [f'enc{i}' for i in range(CFG.LEN)] # The first 142 encoded columns as features\nTARGETS = ['bind1', 'bind2', 'bind3'] # Three types of binds as targets\n\n# 20 fold -> only train 1/20 of the entire dataset\nskf = StratifiedKFold(n_splits = CFG.NBR_FOLDS, shuffle=True, random_state=999) \n                                                                                    \nif CFG.PRETRAINED:\n    print(\"Model Pretrained. Skip Training process.\")\nelse:\n    for fold,(train_idx, valid_idx) in enumerate(skf.split(train, train[TARGETS].sum(1))):\n        if fold in CFG.SELECTED_FOLDS:\n            print(f\"Training on fold {fold}\")\n            X_train = train.loc[train_idx, FEATURES]\n            y_train = train.loc[train_idx, TARGETS]\n            X_val = train.loc[valid_idx, FEATURES]\n            y_val = train.loc[valid_idx, TARGETS]\n\n            # Stops training when validation loss stops improving\n            es = tf.keras.callbacks.EarlyStopping(patience=5, monitor=\"val_loss\", mode='min', verbose=1)\n            # Save the model with the best validation loss\n            checkpoint = tf.keras.callbacks.ModelCheckpoint(\n                monitor='val_loss', \n                filepath=f\"model-{fold}.weights.h5\",          \n                save_best_only=True, \n                save_weights_only=True, \n                mode='min'\n            )\n            # Reduce learning rate when validation loss plateaus\n            reduce_lr_loss = tf.keras.callbacks.ReduceLROnPlateau(\n                monitor='val_loss', \n                factor=0.05, \n                patience=5, \n                verbose=1\n            )\n            \n            # Train\n            model = my_model()\n            history = model.fit(\n                X_train, y_train,\n                validation_data=(X_val, y_val),\n                epochs=CFG.EPOCHS,\n                callbacks=[checkpoint, reduce_lr_loss, es],\n                batch_size=CFG.BATCH_SIZE,\n                verbose=1,\n            )\n            \n            model.load_weights(f\"model-{fold}.weights.h5\")\n            oof = model.predict(X_val, batch_size=2*CFG.BATCH_SIZE)\n            print('fold :', fold, 'CV score =', APS(y_val, oof, average='micro'))\n","metadata":{"execution":{"iopub.status.busy":"2024-07-04T20:41:58.777665Z","iopub.execute_input":"2024-07-04T20:41:58.777955Z","iopub.status.idle":"2024-07-04T20:41:58.794718Z","shell.execute_reply.started":"2024-07-04T20:41:58.777930Z","shell.execute_reply":"2024-07-04T20:41:58.793747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference","metadata":{}},{"cell_type":"code","source":"all_preds = []\nfor fold in range(CFG.NBR_FOLDS):\n    weight_path = f'/kaggle/input/belka-1dcnn-20-folds-25-epoch-model-weights/model-{fold}.weights.h5'\n    print(f\"Loading weight model-{fold}.weights.h5\")\n    \n    model = my_model()\n    model.load_weights(weight_path)\n    preds = model.predict(test, batch_size=2*CFG.BATCH_SIZE)\n    all_preds.append(preds)\n    \npreds = np.mean(all_preds, 0)","metadata":{"execution":{"iopub.status.busy":"2024-07-04T20:41:58.797275Z","iopub.execute_input":"2024-07-04T20:41:58.797577Z","iopub.status.idle":"2024-07-04T20:44:45.306436Z","shell.execute_reply.started":"2024-07-04T20:41:58.797547Z","shell.execute_reply":"2024-07-04T20:44:45.305514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds[:5]","metadata":{"execution":{"iopub.status.busy":"2024-07-04T20:44:45.307740Z","iopub.execute_input":"2024-07-04T20:44:45.308121Z","iopub.status.idle":"2024-07-04T20:44:45.316450Z","shell.execute_reply.started":"2024-07-04T20:44:45.308084Z","shell.execute_reply":"2024-07-04T20:44:45.315508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"tst = pd.read_parquet('/kaggle/input/leash-BELKA/test.parquet')\n\ntst['binds'] = 0.0\ntst.loc[tst['protein_name']=='BRD4', 'binds'] = preds[(tst['protein_name']=='BRD4').values, 0]\ntst.loc[tst['protein_name']=='HSA', 'binds'] = preds[(tst['protein_name']=='HSA').values, 1]\ntst.loc[tst['protein_name']=='sEH', 'binds'] = preds[(tst['protein_name']=='sEH').values, 2]\n\ntst[['id', 'binds']].to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2024-07-04T20:44:45.317445Z","iopub.execute_input":"2024-07-04T20:44:45.317705Z","iopub.status.idle":"2024-07-04T20:44:54.425677Z","shell.execute_reply.started":"2024-07-04T20:44:45.317683Z","shell.execute_reply":"2024-07-04T20:44:54.424920Z"},"trusted":true},"execution_count":null,"outputs":[]}]}