{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\nWas inspired by notebook:  [BELKA 1DCNN Starter with all data ](https://www.kaggle.com/code/ahmedelfazouan/belka-1dcnn-starter-with-all-data) \n\nand paper: [Convolutional neural network based on SMILES representation of compounds for detecting chemical motif](https://bmcbioinformatics.biomedcentral.com/articles/10.1186/s12859-018-2523-5)","metadata":{}},{"cell_type":"markdown","source":"# Encoding","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nimport joblib\nfrom tqdm import tqdm\n\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, Conv1D, MaxPooling1D, AveragePooling1D, GlobalMaxPooling1D, Dense, Embedding\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom tensorflow.keras.layers import Layer\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom sklearn.metrics import average_precision_score as APS\n","metadata":{"execution":{"iopub.status.busy":"2024-06-04T14:01:16.072000Z","iopub.execute_input":"2024-06-04T14:01:16.073723Z","iopub.status.idle":"2024-06-04T14:01:30.118529Z","shell.execute_reply.started":"2024-06-04T14:01:16.073651Z","shell.execute_reply":"2024-06-04T14:01:30.116776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\n\ntry: # detect TPUs\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect() # TPU detection\n    strategy = tf.distribute.TPUStrategy(tpu)\nexcept ValueError: # detect GPUs\n    strategy = tf.distribute.MirroredStrategy() # for GPU or multi-GPU machines\n    #strategy = tf.distribute.get_strategy() # default strategy that works on CPU and single GPU\n    #strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy() # for clusters of multi-GPU machines\n\nprint(\"Number of accelerators: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2024-06-04T14:01:30.121076Z","iopub.execute_input":"2024-06-04T14:01:30.121888Z","iopub.status.idle":"2024-06-04T14:01:30.172845Z","shell.execute_reply.started":"2024-06-04T14:01:30.121843Z","shell.execute_reply":"2024-06-04T14:01:30.171310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Importing only a part of the rows, because taking the whole dataset eats up all the RAM. ","metadata":{}},{"cell_type":"code","source":"enc = {'l': 1, 'y': 2, '@': 3, '3': 4, 'H': 5, 'S': 6, 'F': 7, 'C': 8, 'r': 9, 's': 10, '/': 11, 'c': 12, 'o': 13,\n           '+': 14, 'I': 15, '5': 16, '(': 17, '2': 18, ')': 19, '9': 20, 'i': 21, '#': 22, '6': 23, '8': 24, '4': 25, '=': 26,\n           '1': 27, 'O': 28, '[': 29, 'D': 30, 'B': 31, ']': 32, 'N': 33, '7': 34, 'n': 35, '-': 36}","metadata":{"execution":{"iopub.status.busy":"2024-06-04T14:01:30.174619Z","iopub.execute_input":"2024-06-04T14:01:30.175082Z","iopub.status.idle":"2024-06-04T14:01:30.184486Z","shell.execute_reply.started":"2024-06-04T14:01:30.175044Z","shell.execute_reply":"2024-06-04T14:01:30.182811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def encode_smile(smile, max_len=142):\n# Loop trough all chars in passed smile and take integer \n# corresponding to the char\n    encoded = [enc[char] for char in smile]\n\n# Pad zeros if the encoded string is shorter than 142 \n    encoded += [0] * (max_len - len(encoded))  \n    return encoded","metadata":{"execution":{"iopub.status.busy":"2024-06-04T14:01:30.187361Z","iopub.execute_input":"2024-06-04T14:01:30.188531Z","iopub.status.idle":"2024-06-04T14:01:30.197254Z","shell.execute_reply.started":"2024-06-04T14:01:30.188491Z","shell.execute_reply":"2024-06-04T14:01:30.196033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_raw = pd.read_csv('/kaggle/input/leash-BELKA/test.csv')\nsmiles = test_raw['molecule_smiles'].values\ndel test_raw\n\n\nencoded_smiles = joblib.Parallel(n_jobs=-2)(joblib.delayed(encode_smile)(smile) for smile in tqdm(smiles))\nencoded_smiles = np.array(encoded_smiles)\n\ntest = pd.DataFrame(encoded_smiles, columns =[f'enc{i}' for i in range(142)])\ndel encoded_smiles","metadata":{"execution":{"iopub.status.busy":"2024-06-04T14:01:30.198508Z","iopub.execute_input":"2024-06-04T14:01:30.198906Z","iopub.status.idle":"2024-06-04T14:02:45.257133Z","shell.execute_reply.started":"2024-06-04T14:01:30.198872Z","shell.execute_reply":"2024-06-04T14:02:45.255669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_raw = pd.read_csv('/kaggle/input/leash-BELKA/train.csv', nrows = 2400000) # 12000000","metadata":{"execution":{"iopub.status.busy":"2024-06-04T14:02:45.259034Z","iopub.execute_input":"2024-06-04T14:02:45.259578Z","iopub.status.idle":"2024-06-04T14:02:45.267371Z","shell.execute_reply.started":"2024-06-04T14:02:45.259528Z","shell.execute_reply":"2024-06-04T14:02:45.265777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We exctract molecule_smiles w.r.t. single protein_name. ","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_data(train_raw):\n    # \n    \n    smiles = train_raw[train_raw['protein_name']=='BRD4']['molecule_smiles'].values\n    \n    \n    \n    # \n    encoded_smiles = joblib.Parallel(n_jobs=-1)(joblib.delayed(encode_smile)(smile) for smile in tqdm(smiles))\n    encoded_smiles = np.array(encoded_smiles)\n    data_train = pd.DataFrame(encoded_smiles, columns =[f'enc{i}' for i in range(142)])\n    del encoded_smiles\n    del smiles\n    data_train['bind1'] = train_raw[train_raw['protein_name']=='BRD4']['binds'].values\n    data_train['bind2'] = train_raw[train_raw['protein_name']=='HSA']['binds'].values\n    data_train['bind3']  = train_raw[train_raw['protein_name']=='sEH']['binds'].values\n\n    zero_binds_condition = (data_train['bind1'] == 0) & (data_train['bind2'] == 0) & (data_train['bind3'] == 0)\n\n    # Invert the condition to get rows where at least one binds column is 1\n    keep_rows_condition = ~zero_binds_condition\n\n    # Filter the DataFrame based on the inverted condition\n    filtered_df =  data_train[keep_rows_condition]\n\n    sampled_zero_binds_df = data_train[zero_binds_condition].sample( 3*len(filtered_df), random_state=42)\n    \n    \n    return pd.concat([filtered_df, sampled_zero_binds_df], ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2024-06-04T14:02:45.269092Z","iopub.execute_input":"2024-06-04T14:02:45.269688Z","iopub.status.idle":"2024-06-04T14:02:45.282748Z","shell.execute_reply.started":"2024-06-04T14:02:45.269650Z","shell.execute_reply":"2024-06-04T14:02:45.281259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntest_raw = pd.read_csv('/kaggle/input/leash-BELKA/test.csv')\nsmiles = test_raw['molecule_smiles'].values\ndel test_raw\n\n\nencoded_smiles = joblib.Parallel(n_jobs=-2)(joblib.delayed(encode_smile)(smile) for smile in tqdm(smiles))\nencoded_smiles = np.array(encoded_smiles)\n\ntest = pd.DataFrame(encoded_smiles, columns =[f'enc{i}' for i in range(142)])\ndel encoded_smiles","metadata":{"execution":{"iopub.status.busy":"2024-06-04T14:02:45.284477Z","iopub.execute_input":"2024-06-04T14:02:45.284893Z","iopub.status.idle":"2024-06-04T14:03:52.385808Z","shell.execute_reply.started":"2024-06-04T14:02:45.284858Z","shell.execute_reply":"2024-06-04T14:03:52.384073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"\n\ndef make_model():\n    with strategy.scope():\n        inputs = tf.keras.layers.Input(shape=(142,), dtype = 'int32')\n        x = tf.keras.layers.Embedding(input_dim=36, output_dim=128,\n                                    input_length=142 )(inputs)\n\n        x = tf.keras.layers.Conv1D(filters=32, kernel_size=3, strides=1, \n                   padding= 'valid', activation='relu')(x)\n        #x = tf.keras.layers.AveragePooling1D(pool_size=21, strides=1, padding='same')(x)\n        \n        x = tf.keras.layers.Conv1D(filters=64, kernel_size=3, strides=1,\n                   padding='valid', activation='relu')(x)\n        \n        x = tf.keras.layers.Conv1D(filters=96, kernel_size=3, strides=1,\n                   padding='valid', activation='relu')(x)\n        #x = tf.keras.layers.AveragePooling1D(pool_size=5, strides=1, padding='same')(x)\n           \n        x = tf.keras.layers.GlobalMaxPooling1D()(x)\n        \n        x = tf.keras.layers.Dense(1024, activation='relu')(x)\n        x = tf.keras.layers.Dropout(0.1)(x)\n        x = tf.keras.layers.Dense(1024, activation='relu')(x)\n        x = tf.keras.layers.Dropout(0.1)(x)\n        x = tf.keras.layers.Dense(512, activation='relu')(x)\n        x = tf.keras.layers.Dropout(0.1)(x)\n\n        \n        outputs = tf.keras.layers.Dense(3, activation='sigmoid')(x)  # Assuming binary classification  \n        model = tf.keras.models.Model(inputs=inputs, outputs=outputs)\n        model.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=0.001,weight_decay=0.01),\n                  loss='binary_crossentropy' ,\n                  metrics=[tf.keras.metrics.AUC(curve='PR', name = 'avg_precision')])\n    \n        return model\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-04T14:09:14.482716Z","iopub.execute_input":"2024-06-04T14:09:14.483402Z","iopub.status.idle":"2024-06-04T14:09:14.500546Z","shell.execute_reply.started":"2024-06-04T14:09:14.483359Z","shell.execute_reply":"2024-06-04T14:09:14.499242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_model(all_preds,skf,data_train,y_cols):\n    for fold,(train_id, test_id) in enumerate(skf.split(data_train, data_train[y_cols].sum(1))):\n\n        X_train = data_train.loc[train_id, X_cols]\n        y_train = data_train.loc[train_id, y_cols]\n        X_val = data_train.loc[test_id, X_cols]\n        y_val = data_train.loc[test_id, y_cols]\n\n        y_train_np = y_train.values\n\n    # Compute class weights for each label separately\n        class_weights = {}\n        for i in range(y_train_np.shape[1]):\n        # compute_class_weight expects a 1D array, so we provide y_train[:, i]\n            class_weight = compute_class_weight(class_weight='balanced', classes=[0, 1], y=y_train_np[:, i])\n            class_weights[i] = class_weight[1]\n\n\n\n        checkpoint = tf.keras.callbacks.ModelCheckpoint(\n            monitor='val_loss', filepath=f\"/kaggle/working/model_{fold}.weights.h5\",\n            save_best_only=True, save_weights_only=True,\n            mode='min')\n\n        reduce_lr_loss = tf.keras.callbacks.ReduceLROnPlateau(\n            monitor='val_loss', factor=0.05, patience=5, verbose=1)\n        model = make_model()\n        history = model.fit(\n                X_train, y_train,\n                validation_data=(X_val, y_val),\n                epochs=35,\n                callbacks=[checkpoint, reduce_lr_loss, early_stopping],\n                class_weight=class_weights,\n                batch_size=8192,\n                verbose=1,\n            )\n\n        # Load the best weight and use them for prediction\n        model.load_weights(f\"/kaggle/working/model_{fold}.weights.h5\")\n        oof = model.predict(X_val, batch_size = 2048)\n        print('fold :', fold, 'CV score =', APS(y_val, oof, average = 'micro'))\n\n        preds = model.predict(test, batch_size = 8192)\n        all_preds.append(preds)\n        \n    return all_preds\n        \n        ","metadata":{"execution":{"iopub.status.busy":"2024-06-04T14:03:52.408121Z","iopub.execute_input":"2024-06-04T14:03:52.409399Z","iopub.status.idle":"2024-06-04T14:03:52.425772Z","shell.execute_reply.started":"2024-06-04T14:03:52.409314Z","shell.execute_reply":"2024-06-04T14:03:52.424373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_cols = [f'enc{i}' for i in range(142)]\ny_cols = ['bind1', 'bind2', 'bind3']\n\nearly_stopping = tf.keras.callbacks.EarlyStopping(patience=5, monitor=\"val_loss\", mode='min', verbose=1)\n\nskf = StratifiedKFold(n_splits=3 ,shuffle = True, random_state = 42)\nall_preds = []","metadata":{"execution":{"iopub.status.busy":"2024-06-04T14:03:52.427519Z","iopub.execute_input":"2024-06-04T14:03:52.428053Z","iopub.status.idle":"2024-06-04T14:03:52.447336Z","shell.execute_reply.started":"2024-06-04T14:03:52.428002Z","shell.execute_reply":"2024-06-04T14:03:52.445750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chunksize = 12000000   \n\n# Import one chunk at a time\nfor chunk in pd.read_csv('/kaggle/input/leash-BELKA/train.csv', chunksize=chunksize,nrows = 60000000   ):\n    \n    # Get data and encode data\n    data_train = make_data(chunk)\n    \n    # Each data_train has ~ 150k rows \n    \n    # Train data\n    all_preds = train_model(all_preds,skf,data_train,y_cols)","metadata":{"execution":{"iopub.status.busy":"2024-06-04T14:09:47.794741Z","iopub.execute_input":"2024-06-04T14:09:47.796049Z","iopub.status.idle":"2024-06-04T14:29:27.997858Z","shell.execute_reply.started":"2024-06-04T14:09:47.796002Z","shell.execute_reply":"2024-06-04T14:29:27.995417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction = np.mean(all_preds, 0)","metadata":{"execution":{"iopub.status.busy":"2024-06-04T14:29:36.433005Z","iopub.execute_input":"2024-06-04T14:29:36.433517Z","iopub.status.idle":"2024-06-04T14:29:36.500536Z","shell.execute_reply.started":"2024-06-04T14:29:36.433473Z","shell.execute_reply":"2024-06-04T14:29:36.498907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/leash-BELKA/test.csv')\ntest['binds'] = 0\ntest.loc[test['protein_name']=='BRD4', 'binds'] = prediction[(test['protein_name']=='BRD4').values, 0]\ntest.loc[test['protein_name']=='HSA', 'binds'] = prediction[(test['protein_name']=='HSA').values, 1]\ntest.loc[test['protein_name']=='sEH', 'binds'] = prediction[(test['protein_name']=='sEH').values, 2]\ntest[['id', 'binds']].to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2024-06-04T14:29:38.459029Z","iopub.execute_input":"2024-06-04T14:29:38.459581Z","iopub.status.idle":"2024-06-04T14:29:51.148257Z","shell.execute_reply.started":"2024-06-04T14:29:38.459532Z","shell.execute_reply":"2024-06-04T14:29:51.146724Z"},"trusted":true},"execution_count":null,"outputs":[]}]}