{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":59094,"databundleVersionId":7010844,"sourceType":"competition"},{"sourceId":5835808,"sourceType":"datasetVersion","datasetId":3354626}],"dockerImageVersionId":30559,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# NLP Regression  \nThis is just a simple experiment with NLP and to get low MRRMSE(Mean Rowwise Root Mean Sqared Error)\n\nWe're going to use SMILES embedding.\n\nSteps to do above:\n1. Preprocess the data \n2. Make a TextVectorizer and embedding\n3. Build a Model\n4. Visualize, Evaluate and Repeat \n\nLet's go...\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"!pip install -U tensorflow==2.14.0","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:51:13.576088Z","iopub.execute_input":"2023-11-14T10:51:13.576375Z","iopub.status.idle":"2023-11-14T10:52:13.464036Z","shell.execute_reply.started":"2023-11-14T10:51:13.576349Z","shell.execute_reply":"2023-11-14T10:52:13.462826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:52:13.466448Z","iopub.execute_input":"2023-11-14T10:52:13.466773Z","iopub.status.idle":"2023-11-14T10:52:18.771159Z","shell.execute_reply.started":"2023-11-14T10:52:13.466732Z","shell.execute_reply":"2023-11-14T10:52:18.769827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.__version__","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:52:18.773078Z","iopub.execute_input":"2023-11-14T10:52:18.776003Z","iopub.status.idle":"2023-11-14T10:52:18.801097Z","shell.execute_reply.started":"2023-11-14T10:52:18.775949Z","shell.execute_reply":"2023-11-14T10:52:18.784711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Read data","metadata":{}},{"cell_type":"code","source":"data = pd.read_parquet(\"/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet\")\nid_map = pd.read_csv(\"/kaggle/input/open-problems-single-cell-perturbations/id_map.csv\")\nsample_submission = pd.read_csv(\"/kaggle/input/open-problems-single-cell-perturbations/sample_submission.csv\")\nsample_columns = sample_submission.columns\nsample_columns = sample_columns[1:]\nRNDST1=271828\nRNDST2=314159","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:52:18.802970Z","iopub.execute_input":"2023-11-14T10:52:18.803821Z","iopub.status.idle":"2023-11-14T10:52:28.117271Z","shell.execute_reply.started":"2023-11-14T10:52:18.803772Z","shell.execute_reply":"2023-11-14T10:52:28.116451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" **🔑Tip:** Stare at the data for a while to get some insights and ideas to run experiments.","metadata":{}},{"cell_type":"code","source":"data.head(20)","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:52:28.120357Z","iopub.execute_input":"2023-11-14T10:52:28.121097Z","iopub.status.idle":"2023-11-14T10:52:28.164884Z","shell.execute_reply.started":"2023-11-14T10:52:28.121058Z","shell.execute_reply":"2023-11-14T10:52:28.163910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Aux functions ","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.callbacks import ModelCheckpoint\n\ndef create_model_checkpoint(filepath, monitor='val_mae', save_best_only=True,\n                            save_weights_only=True, mode='auto', verbose=0):\n    \"\"\"\n    Create a ModelCheckpoint callback for saving the best model weights during training.\n\n    Args:\n        filepath (str): Filepath to save the best weights.\n        monitor (str): Metric to monitor (e.g., 'val_loss' or 'val_mae').\n        save_best_only (bool): Save only the best weights.\n        save_weights_only (bool): Save only the model's weights, not the entire model.\n        mode (str): One of {'auto', 'min', 'max'}. In 'min' mode, it saves when the monitored metric decreases.\n        verbose (int): Verbosity mode. 0 = silent, 1 = progress bar, 2 = one line per epoch.\n\n    Returns:\n        keras.callbacks.ModelCheckpoint: ModelCheckpoint callback.\n    \"\"\"\n    checkpoint = ModelCheckpoint(\n        filepath=filepath,\n        monitor=monitor,\n        save_best_only=save_best_only,\n        save_weights_only=save_weights_only,\n        mode=mode,\n        verbose=verbose\n    )\n    return checkpoint\n","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:52:28.166267Z","iopub.execute_input":"2023-11-14T10:52:28.166564Z","iopub.status.idle":"2023-11-14T10:52:28.193734Z","shell.execute_reply.started":"2023-11-14T10:52:28.166532Z","shell.execute_reply":"2023-11-14T10:52:28.192946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_training_history(history, metrics):\n    \"\"\"\n    Plot training history curves for loss and evaluation metrics on the same line.\n\n    Args:\n        history (keras.callbacks.History): Training history object.\n        metrics (list): List of metric names to plot.\n\n    Returns:\n        None\n    \"\"\"\n    loss = history.history['loss']\n    val_loss = history.history['val_loss']\n\n    epochs = range(len(loss))\n\n    plt.figure(figsize=(12, 6))\n\n    # Plot loss\n    plt.subplot(1, 2, 1)\n    plt.plot(epochs, loss, label='Training Loss', color=\"blue\")\n    plt.plot(epochs, val_loss, label='Validation Loss', color=\"red\")\n    plt.title('Loss')\n    plt.xlabel('Epochs')\n    plt.legend()\n\n    # Plot specified evaluation metrics on the same line\n    for metric in metrics:\n        train_metric_name = f'Training {metric.capitalize()}'\n        val_metric_name = f'Validation {metric.capitalize()}'\n        train_metric = history.history[metric]\n        val_metric = history.history['val_' + metric]\n\n        plt.subplot(1, 2, 2)\n        plt.plot(epochs, train_metric, label=train_metric_name, color=\"green\")\n        plt.plot(epochs, val_metric, label=val_metric_name, color=\"orange\")\n\n    plt.title('Metrics')\n    plt.xlabel('Epochs')\n    plt.legend(loc='upper right')\n\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:52:28.194980Z","iopub.execute_input":"2023-11-14T10:52:28.195315Z","iopub.status.idle":"2023-11-14T10:52:28.204729Z","shell.execute_reply.started":"2023-11-14T10:52:28.195284Z","shell.execute_reply":"2023-11-14T10:52:28.203738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_columns(data, id_map):\n    sm_name_to_smiles = data.set_index('sm_name')['SMILES'].to_dict()\n    sm_lincs_id = data.set_index('sm_name')[\"sm_lincs_id\"].to_dict()\n\n    id_map['SMILES'] = id_map['sm_name'].map(sm_name_to_smiles)\n    id_map['sm_lincs_id'] = id_map['sm_name'].map(sm_lincs_id)\n\n    return id_map","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:52:28.205949Z","iopub.execute_input":"2023-11-14T10:52:28.206320Z","iopub.status.idle":"2023-11-14T10:52:28.218748Z","shell.execute_reply.started":"2023-11-14T10:52:28.206289Z","shell.execute_reply":"2023-11-14T10:52:28.217882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mean_absolute_error\n\ndef calculate_mae_and_mrrmse(model, data, y_true, scaler=None):\n    \"\"\"\n    Calculate Mean Absolute Error (MAE) and Mean Rowwise Root Mean Squared Error (MRRMSE).\n\n    Parameters:\n    - model: The trained  model.\n    - data: The input data for prediction.\n    - y_true: The true target values.\n    - scaler: The scaler used for data normalization.\n\n    Returns:\n    - None\n    \"\"\"\n    # Predict using the model\n    y_pred_original = model.predict(data, batch_size=1)\n\n    if scaler:\n        \n        y_pred = scaler.inverse_transform(y_pred_original)\n        y_true = scaler.inverse_transform(y_true)\n        # Calculate Mean Absolute Error (MAE)\n        mae = mean_absolute_error(y_true , y_pred)\n\n        # Calculate Mean Rowwise Root Mean Squared Error (MRRMSE)\n        rowwise_rmse = np.sqrt(np.mean(np.square(y_true - y_pred), axis=1))\n        mrrmse_score = np.mean(rowwise_rmse)\n    else:\n       # Calculate Mean Absolute Error (MAE)\n        mae = mean_absolute_error(y_true , y_pred_original)\n\n        # Calculate Mean Rowwise Root Mean Squared Error (MRRMSE)\n        rowwise_rmse = np.sqrt(np.mean(np.square(y_true - y_pred_original), axis=1))\n        mrrmse_score = np.mean(rowwise_rmse)\n    # Print the results\n    print(f\"Mean Absolute Error (MAE): {mae}\")\n    print(f\"Mean Rowwise Root Mean Squared Error (MRRMSE): {mrrmse_score}\")","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:52:28.220015Z","iopub.execute_input":"2023-11-14T10:52:28.220382Z","iopub.status.idle":"2023-11-14T10:52:28.569390Z","shell.execute_reply.started":"2023-11-14T10:52:28.220350Z","shell.execute_reply":"2023-11-14T10:52:28.568632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.model_selection import KFold, StratifiedKFold\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import mean_absolute_error\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, Flatten, concatenate, GaussianNoise, Concatenate\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom scipy.stats import t\n\ndef train_and_predict(model, X_train_list, y_train, X_val_list, y_val, X_test_list, scaler, epochs, early_stopping=10):\n    # Define Early Stopping callback\n    early_stopping_callback = EarlyStopping(monitor='val_loss', patience=early_stopping, restore_best_weights=True)\n\n    # Train the model with early stopping\n    history = model.fit(X_train_list, y_train, epochs=epochs, verbose=0, \n                        validation_data=(X_val_list, y_val), callbacks=[early_stopping_callback])\n\n    # Make predictions on the test set\n    preds = model.predict(X_test_list, batch_size=1)\n    preds = scaler.inverse_transform(preds)\n\n    return history, preds\n\n\ndef plot_training_history_kfold(histories_and_folds):\n    fig, axes = plt.subplots(nrows=2, ncols=1, figsize=(15, 10))\n\n    for history, fold in histories_and_folds:\n        color = plt.cm.jet(fold / len(histories_and_folds))\n\n        # Plot Training Loss\n        train_loss = history.history['loss']\n        min_train_loss_epoch = np.argmin(train_loss)\n        axes[0].plot(train_loss, label=f'Fold {fold + 1}', color=color)\n        axes[0].axvline(min_train_loss_epoch, linestyle='--', color=color, label=f'Min Train Loss (Fold {fold + 1}, Epoch {min_train_loss_epoch + 1})')\n\n        # Plot Validation Loss\n        val_loss = history.history['val_loss']\n        min_val_loss_epoch = np.argmin(val_loss)\n        axes[1].plot(val_loss, label=f'Fold {fold + 1}', color=color)\n        axes[1].axvline(min_val_loss_epoch, linestyle='--', color=color, label=f'Min Val Loss (Fold {fold + 1}, Epoch {min_val_loss_epoch + 1})')\n\n    axes[0].set_xlabel('Epochs')\n    axes[0].set_ylabel('Training Loss')\n    axes[0].set_title('Training Loss Across Folds')\n    axes[0].legend()\n\n    axes[1].set_xlabel('Epochs')\n    axes[1].set_ylabel('Validation Loss')\n    axes[1].set_title('Validation Loss Across Folds')\n    axes[1].legend()\n\n    plt.tight_layout()\n    plt.show()\n\ndef k_fold_predict(create_model_func, features_list=[], full_labels=None, test_data=None, num_folds=7, random_state=None, scaler=None, optimizer=None, epochs=100, early_stopping=10, dropout=0.7):\n    # Initialize lists to store the predictions\n    all_preds = []\n    all_histories = []\n    val_losses = []\n\n    # Initialize the KFold object\n    kf = StratifiedKFold(n_splits=num_folds, shuffle=True, random_state=random_state)\n\n    # Loop through the K folds\n    for fold, (train_index, val_index) in enumerate(kf.split(full_labels, features_list[1].argmax(axis=1))):\n        # Convert indices to integers and split the data\n        train_index = train_index.astype(int)\n        val_index = val_index.astype(int)\n\n        X_train_list = [features[train_index] for features in features_list]\n        X_val_list = [features[val_index] for features in features_list]\n        y_train = full_labels[train_index]\n        y_val = full_labels[val_index]\n\n        # Create and compile the model\n        model = create_model_func(input_shape=[X_train.shape[1:] for X_train in X_train_list], optimizer=optimizer, dropout_rate=dropout)\n\n        # Train the model and get predictions on the test set\n        history, preds = train_and_predict(model, X_train_list, y_train, X_val_list, y_val, test_data, scaler, epochs=epochs, early_stopping=early_stopping)\n\n        # Print validation metric for every batch\n        print(f'Fold {fold + 1} - Mean Validation Loss for each batch:')\n        print(np.mean(history.history['val_loss']))\n\n        # Store validation loss for this fold\n        val_losses.append(np.min(history.history['val_loss']))\n\n        # Store predictions for this fold\n        all_preds.append(preds)\n        all_histories.append((history, fold))\n\n    # Calculate mean and confidence interval for best val_losses\n    mean_val_loss = np.mean(val_losses)\n    confidence_interval = t.interval(0.95, len(val_losses) - 1, loc=mean_val_loss, scale=np.std(val_losses) / np.sqrt(len(val_losses)))\n\n    print(f\"\\nMean of Best Validation Losses: {mean_val_loss}\")\n    print(f\"Confidence Interval (95%): {confidence_interval}\")\n\n    plot_training_history_kfold(all_histories)\n    return all_preds\n\n# Example usage:\n# final_test_data = [np.array(final_test_smiles), final_cell_type]\n# all_preds = k_fold_predict(build_model_4, features_list=[full_train_chars, full_cell_type], full_labels=full_labels,\n#                            test_data=final_test_data, num_folds=5, random_state=42, scaler=scaler, optimizer=Adam(learning_rate=0.001))\n","metadata":{"execution":{"iopub.status.busy":"2023-11-14T11:15:52.801116Z","iopub.execute_input":"2023-11-14T11:15:52.801494Z","iopub.status.idle":"2023-11-14T11:15:52.823744Z","shell.execute_reply.started":"2023-11-14T11:15:52.801462Z","shell.execute_reply":"2023-11-14T11:15:52.822948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing data ","metadata":{"execution":{"iopub.status.busy":"2023-10-10T09:17:28.606984Z","iopub.execute_input":"2023-10-10T09:17:28.607326Z","iopub.status.idle":"2023-10-10T09:17:28.611868Z","shell.execute_reply.started":"2023-10-10T09:17:28.607301Z","shell.execute_reply":"2023-10-10T09:17:28.610695Z"}}},{"cell_type":"code","source":"# Shuffle data because we use full features for final training \ndata = data.sample(frac=1.0, random_state=RNDST2)","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:17.063366Z","iopub.execute_input":"2023-11-14T10:59:17.064120Z","iopub.status.idle":"2023-11-14T10:59:17.112867Z","shell.execute_reply.started":"2023-11-14T10:59:17.064089Z","shell.execute_reply":"2023-11-14T10:59:17.111902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"add_columns(data, id_map)","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:17.282284Z","iopub.execute_input":"2023-11-14T10:59:17.282635Z","iopub.status.idle":"2023-11-14T10:59:17.373195Z","shell.execute_reply.started":"2023-11-14T10:59:17.282607Z","shell.execute_reply":"2023-11-14T10:59:17.372263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_type_feature = pd.DataFrame(data[\"cell_type\"], columns=[\"cell_type\"])\nsmiles_feature = pd.DataFrame(data[\"SMILES\"], columns=[\"SMILES\"])\nlabels = data.drop([\"cell_type\",\"sm_name\",\"sm_lincs_id\",\"SMILES\",\"control\"], axis=1)\n\n# for test\ntest_feature_smiles = pd.DataFrame(id_map[\"SMILES\"], columns=[\"SMILES\"])\ntest_feature_cell_type = pd.DataFrame(id_map[\"cell_type\"], columns=[\"cell_type\"])","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:17.470545Z","iopub.execute_input":"2023-11-14T10:59:17.470899Z","iopub.status.idle":"2023-11-14T10:59:17.514487Z","shell.execute_reply.started":"2023-11-14T10:59:17.470872Z","shell.execute_reply":"2023-11-14T10:59:17.513453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_type_feature.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:17.711776Z","iopub.execute_input":"2023-11-14T10:59:17.712742Z","iopub.status.idle":"2023-11-14T10:59:17.722380Z","shell.execute_reply.started":"2023-11-14T10:59:17.712704Z","shell.execute_reply":"2023-11-14T10:59:17.721232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"smiles_feature.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:18.879367Z","iopub.execute_input":"2023-11-14T10:59:18.879733Z","iopub.status.idle":"2023-11-14T10:59:18.888263Z","shell.execute_reply.started":"2023-11-14T10:59:18.879703Z","shell.execute_reply":"2023-11-14T10:59:18.887195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_feature_cell_type.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:19.360319Z","iopub.execute_input":"2023-11-14T10:59:19.360665Z","iopub.status.idle":"2023-11-14T10:59:19.369272Z","shell.execute_reply.started":"2023-11-14T10:59:19.360637Z","shell.execute_reply":"2023-11-14T10:59:19.368272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_feature_smiles.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:19.895801Z","iopub.execute_input":"2023-11-14T10:59:19.896154Z","iopub.status.idle":"2023-11-14T10:59:19.904499Z","shell.execute_reply.started":"2023-11-14T10:59:19.896126Z","shell.execute_reply":"2023-11-14T10:59:19.903456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:20.340766Z","iopub.execute_input":"2023-11-14T10:59:20.341601Z","iopub.status.idle":"2023-11-14T10:59:20.367246Z","shell.execute_reply.started":"2023-11-14T10:59:20.341570Z","shell.execute_reply":"2023-11-14T10:59:20.366285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Normalize labels\nfrom sklearn.preprocessing import StandardScaler\n\nscaler = StandardScaler()\nnorm_label = scaler.fit_transform(labels)","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:21.245617Z","iopub.execute_input":"2023-11-14T10:59:21.246417Z","iopub.status.idle":"2023-11-14T10:59:21.781473Z","shell.execute_reply.started":"2023-11-14T10:59:21.246386Z","shell.execute_reply":"2023-11-14T10:59:21.780413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Onehotencode\nfrom sklearn.preprocessing import OneHotEncoder\n\nencoder = OneHotEncoder()\none_hot_celltype = encoder.fit_transform(cell_type_feature)\none_hot_test_cell_type = encoder.transform(test_feature_cell_type)","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:22.036371Z","iopub.execute_input":"2023-11-14T10:59:22.036721Z","iopub.status.idle":"2023-11-14T10:59:22.044135Z","shell.execute_reply.started":"2023-11-14T10:59:22.036693Z","shell.execute_reply":"2023-11-14T10:59:22.043296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_cell_type, temp_cell_type, train_labels, temp_labels = train_test_split(one_hot_celltype.toarray(),\n                                                                            norm_label,\n                                                                            test_size=0.3, random_state=RNDST2)\n\nval_cell_type, test_cell_type, val_labels , test_labels = train_test_split(temp_cell_type, temp_cell_type, test_size=0.6, random_state=RNDST2)","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:24.675994Z","iopub.execute_input":"2023-11-14T10:59:24.676340Z","iopub.status.idle":"2023-11-14T10:59:24.805754Z","shell.execute_reply.started":"2023-11-14T10:59:24.676312Z","shell.execute_reply":"2023-11-14T10:59:24.804981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_smiles, temp_smiles, train_labels, temp_labels = train_test_split(smiles_feature[\"SMILES\"].to_numpy(),\n                                                                            norm_label,\n                                                                            test_size=0.3, random_state=RNDST2)\n\nval_smiles, test_smiles, val_labels , test_labels = train_test_split(temp_smiles, temp_labels, test_size=0.6, random_state=RNDST2)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:25.023734Z","iopub.execute_input":"2023-11-14T10:59:25.024083Z","iopub.status.idle":"2023-11-14T10:59:25.156000Z","shell.execute_reply.started":"2023-11-14T10:59:25.024055Z","shell.execute_reply":"2023-11-14T10:59:25.155199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"full_smiles = smiles_feature[\"SMILES\"].values \nfull_cell_type = one_hot_celltype.toarray()\nfull_labels = norm_label","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:26.732822Z","iopub.execute_input":"2023-11-14T10:59:26.733312Z","iopub.status.idle":"2023-11-14T10:59:26.738716Z","shell.execute_reply.started":"2023-11-14T10:59:26.733271Z","shell.execute_reply":"2023-11-14T10:59:26.737682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_test_smiles = test_feature_smiles[\"SMILES\"].values\nfinal_cell_type = one_hot_test_cell_type.toarray()","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:27.015338Z","iopub.execute_input":"2023-11-14T10:59:27.016180Z","iopub.status.idle":"2023-11-14T10:59:27.020729Z","shell.execute_reply.started":"2023-11-14T10:59:27.016147Z","shell.execute_reply":"2023-11-14T10:59:27.019821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"full_smiles[:3]","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:27.252847Z","iopub.execute_input":"2023-11-14T10:59:27.253637Z","iopub.status.idle":"2023-11-14T10:59:27.259255Z","shell.execute_reply.started":"2023-11-14T10:59:27.253605Z","shell.execute_reply":"2023-11-14T10:59:27.258201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check the lengths\nlen(train_smiles), len(train_smiles), len(val_smiles), len(val_smiles), len(test_smiles), len(test_smiles)","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:52:29.572143Z","iopub.status.idle":"2023-11-14T10:52:29.572564Z","shell.execute_reply.started":"2023-11-14T10:52:29.572346Z","shell.execute_reply":"2023-11-14T10:52:29.572367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Split chars \nWe will split the `SMILES` into chars and then vectorize.","metadata":{}},{"cell_type":"code","source":"# Function to split sentences into characters\ndef split_chars(text):\n    return \" \".join(list(text))","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:35.228090Z","iopub.execute_input":"2023-11-14T10:59:35.228436Z","iopub.status.idle":"2023-11-14T10:59:35.233061Z","shell.execute_reply.started":"2023-11-14T10:59:35.228407Z","shell.execute_reply":"2023-11-14T10:59:35.232110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = \"CC(C)c1cc(C(=O)N2Cc3ccc(CN4CCN(C)CC4)cc3C2)c(O)cc1O\"\nsplit_chars(sample)","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:35.511002Z","iopub.execute_input":"2023-11-14T10:59:35.511386Z","iopub.status.idle":"2023-11-14T10:59:35.517416Z","shell.execute_reply.started":"2023-11-14T10:59:35.511356Z","shell.execute_reply":"2023-11-14T10:59:35.516494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Split the smiles","metadata":{}},{"cell_type":"code","source":"train_char_smiles = [split_chars(feature) for feature in train_smiles]\nval_char_smiles = [split_chars(feature) for feature in val_smiles]\ntest_char_smiles = [split_chars(feature) for feature in test_smiles]","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:36.417735Z","iopub.execute_input":"2023-11-14T10:59:36.418095Z","iopub.status.idle":"2023-11-14T10:59:36.425282Z","shell.execute_reply.started":"2023-11-14T10:59:36.418067Z","shell.execute_reply":"2023-11-14T10:59:36.424239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For final training\nfull_train_smiles = [split_chars(feature) for feature in full_smiles]","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:36.826315Z","iopub.execute_input":"2023-11-14T10:59:36.826673Z","iopub.status.idle":"2023-11-14T10:59:36.832875Z","shell.execute_reply.started":"2023-11-14T10:59:36.826644Z","shell.execute_reply":"2023-11-14T10:59:36.831972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For prediction\nfinal_test_smiles = [split_chars(feature) for feature in final_test_smiles]","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:37.175910Z","iopub.execute_input":"2023-11-14T10:59:37.176282Z","iopub.status.idle":"2023-11-14T10:59:37.181956Z","shell.execute_reply.started":"2023-11-14T10:59:37.176252Z","shell.execute_reply":"2023-11-14T10:59:37.180972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Char vectorizer","metadata":{}},{"cell_type":"code","source":"# What's the average character length?\nchar_len_smiles = [len(feature) for feature in full_smiles]\nmean_char_smiles = np.mean(char_len_smiles)\nmean_char_smiles","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:38.178199Z","iopub.execute_input":"2023-11-14T10:59:38.178540Z","iopub.status.idle":"2023-11-14T10:59:38.185886Z","shell.execute_reply.started":"2023-11-14T10:59:38.178495Z","shell.execute_reply":"2023-11-14T10:59:38.184985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  Find what character length covers 95% of sequences\noutput_seq_char_len = int(np.percentile(char_len_smiles, 95))\noutput_seq_char_len","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:38.633984Z","iopub.execute_input":"2023-11-14T10:59:38.634821Z","iopub.status.idle":"2023-11-14T10:59:38.645222Z","shell.execute_reply.started":"2023-11-14T10:59:38.634780Z","shell.execute_reply":"2023-11-14T10:59:38.644242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check the distribution of our sequences at character-level\nplt.hist(char_len_smiles, bins=7);","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:39.148117Z","iopub.execute_input":"2023-11-14T10:59:39.148449Z","iopub.status.idle":"2023-11-14T10:59:39.448761Z","shell.execute_reply.started":"2023-11-14T10:59:39.148424Z","shell.execute_reply":"2023-11-14T10:59:39.447848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Get all the unique characters in smiles","metadata":{"execution":{"iopub.status.busy":"2023-10-10T09:36:47.578256Z","iopub.execute_input":"2023-10-10T09:36:47.578652Z","iopub.status.idle":"2023-10-10T09:36:47.585849Z","shell.execute_reply.started":"2023-10-10T09:36:47.578623Z","shell.execute_reply":"2023-10-10T09:36:47.5846Z"}}},{"cell_type":"code","source":"unique_characters = []\n\n# Iterate over the \"features\" and extract unique characters\nfor feature in full_train_smiles:\n    unique_characters.extend(set(feature))\n\n# Remove duplicates by converting the list to a set and then back to a list\nunique_characters = list(set(unique_characters))\n\n# Print the list of unique characters\nprint(unique_characters)","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:41.932367Z","iopub.execute_input":"2023-11-14T10:59:41.933108Z","iopub.status.idle":"2023-11-14T10:59:41.940324Z","shell.execute_reply.started":"2023-11-14T10:59:41.933071Z","shell.execute_reply":"2023-11-14T10:59:41.939426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(unique_characters)","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:42.340624Z","iopub.execute_input":"2023-11-14T10:59:42.340983Z","iopub.status.idle":"2023-11-14T10:59:42.346908Z","shell.execute_reply.started":"2023-11-14T10:59:42.340952Z","shell.execute_reply":"2023-11-14T10:59:42.346016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.layers.experimental.preprocessing import TextVectorization\n\n# Create char-level token vectorizer instance\nNUM_CHAR_TOKENS = len(unique_characters) + 2 \nchar_vectorizer_smiles = TextVectorization(max_tokens=NUM_CHAR_TOKENS,\n                                    output_sequence_length=output_seq_char_len,\n                                    standardize=None,\n                                    split=\"character\",\n                                    name=\"char_vectorizer_SMILES\")\n\n# Adapt character vectorizer to training characters\nchar_vectorizer_smiles.adapt(train_char_smiles)","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:42.830896Z","iopub.execute_input":"2023-11-14T10:59:42.831293Z","iopub.status.idle":"2023-11-14T10:59:44.830677Z","shell.execute_reply.started":"2023-11-14T10:59:42.831261Z","shell.execute_reply":"2023-11-14T10:59:44.829822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the config of our char vectorizer\nchar_vectorizer_smiles.get_config()","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:44.832256Z","iopub.execute_input":"2023-11-14T10:59:44.832532Z","iopub.status.idle":"2023-11-14T10:59:44.839524Z","shell.execute_reply.started":"2023-11-14T10:59:44.832507Z","shell.execute_reply":"2023-11-14T10:59:44.838641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\n\n# Test out character vectorizer\nrandom_train_feature = random.choice(train_char_smiles)\nprint(f\"Charified text:\\n{random_train_feature}\")\nprint(f\"\\nLength of feature: {len(random_train_feature.split())}\")\nvectorized_feature = char_vectorizer_smiles([random_train_feature])\nprint(f\"\\nVectorized feature:\\n{vectorized_feature}\")\nprint(f\"\\nLength of vectorized feature: {len(vectorized_feature[0])}\")","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:44.840668Z","iopub.execute_input":"2023-11-14T10:59:44.841011Z","iopub.status.idle":"2023-11-14T10:59:45.855377Z","shell.execute_reply.started":"2023-11-14T10:59:44.840977Z","shell.execute_reply":"2023-11-14T10:59:45.854435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"You'll notice sequences with a length shorter than 193 (output_seq_char_length) get padded with zeros on the end, this ensures all sequences passed to our model are the same length.","metadata":{}},{"cell_type":"code","source":"# Check character vocabulary characteristics\nchar_vocab = char_vectorizer_smiles.get_vocabulary()\nprint(f\"Number of different characters in character vocab: {len(char_vocab)}\")\nprint(f\"5 most common characters: {char_vocab[:5]}\")\nprint(f\"5 least common characters: {char_vocab[-5:]}\")","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:45.857546Z","iopub.execute_input":"2023-11-14T10:59:45.857977Z","iopub.status.idle":"2023-11-14T10:59:45.866450Z","shell.execute_reply.started":"2023-11-14T10:59:45.857919Z","shell.execute_reply":"2023-11-14T10:59:45.865403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Char embedding","metadata":{}},{"cell_type":"code","source":"# Create char embedding layer\nchar_embedding = tf.keras.layers.Embedding(input_dim=NUM_CHAR_TOKENS, # number of different characters\n                              output_dim=16, # embedding dimension of each character \n                              name=\"char_embedding_SMILES\")","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:49.376109Z","iopub.execute_input":"2023-11-14T10:59:49.376452Z","iopub.status.idle":"2023-11-14T10:59:49.384091Z","shell.execute_reply.started":"2023-11-14T10:59:49.376423Z","shell.execute_reply":"2023-11-14T10:59:49.383066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test out character embedding layer\nprint(f\"Charified text (before vectorization and embedding):\\n{random_train_feature}\\n\")\nchar_embed_example = char_embedding(char_vectorizer_smiles([random_train_feature]))\nprint(f\"Embedded chars (after vectorization and embedding):\\n{char_embed_example}\\n\")\nprint(f\"Character embedding shape: {char_embed_example.shape}\")","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:49.624144Z","iopub.execute_input":"2023-11-14T10:59:49.624547Z","iopub.status.idle":"2023-11-14T10:59:49.655443Z","shell.execute_reply.started":"2023-11-14T10:59:49.624513Z","shell.execute_reply":"2023-11-14T10:59:49.654344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert list to array\ntrain_chars = np.array(train_char_smiles)\nval_chars = np.array(val_char_smiles)\ntest_chars = np.array(test_char_smiles)\nfull_train_chars = np.array(full_train_smiles)","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:50.000537Z","iopub.execute_input":"2023-11-14T10:59:50.001254Z","iopub.status.idle":"2023-11-14T10:59:50.007020Z","shell.execute_reply.started":"2023-11-14T10:59:50.001220Z","shell.execute_reply":"2023-11-14T10:59:50.006158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Building models","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.utils import plot_model\nfrom tensorflow.keras import Model\nfrom tensorflow.keras.layers import LSTM, Dense, concatenate, Conv1D, GlobalMaxPooling1D, Input, Flatten, GaussianNoise, GlobalAveragePooling1D","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:51.142541Z","iopub.execute_input":"2023-11-14T10:59:51.143594Z","iopub.status.idle":"2023-11-14T10:59:51.149833Z","shell.execute_reply.started":"2023-11-14T10:59:51.143551Z","shell.execute_reply":"2023-11-14T10:59:51.148774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.layers import Input, Conv1D, Dense, GlobalMaxPooling1D, concatenate\nfrom tensorflow.keras.models import Model\n\ndef build_smiles_input(input_shape=(1,)):\n    # Define the SMILES input\n    smiles_input = Input(shape=input_shape, dtype=\"string\", name=\"smiles_input\")\n    char_vectors = char_vectorizer_smiles(smiles_input)\n    char_embeddings = char_embedding(char_vectors)\n    conv_layer = Conv1D(64, kernel_size=5, padding=\"same\", activation=\"relu\")(char_embeddings)\n    smiles_output = GlobalMaxPooling1D()(conv_layer)\n    return smiles_input, smiles_output\n\ndef build_cell_type_input(cell_type_shape=(6,)):\n    # Define the cell type input\n    cell_type_input = Input(shape=cell_type_shape, name=\"cell_type_input\")\n    cell_type_output = Dense(32, activation=\"relu\")(cell_type_input)\n    return cell_type_input, cell_type_output\n\ndef build_neural_network(concatenated):\n    # Define the neural network architecture\n    x = Dense(128, activation=\"relu\")(concatenated)\n    output = Dense(18211, activation=\"linear\")(x)\n    return output","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:54.682899Z","iopub.execute_input":"2023-11-14T10:59:54.683689Z","iopub.status.idle":"2023-11-14T10:59:54.691705Z","shell.execute_reply.started":"2023-11-14T10:59:54.683658Z","shell.execute_reply":"2023-11-14T10:59:54.690805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model 2: Custom neural net ","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.layers import Input, Flatten, concatenate, GaussianNoise, Dense, Dropout\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom tensorflow.keras.layers import BatchNormalization\n\ndef mean_rowwise_rmse(y_true, y_pred):\n    # Calculate RMSE for each row\n    rowwise_rmse = tf.sqrt(tf.reduce_mean(tf.square(y_true - y_pred), axis=1))\n    \n    # Return the mean of rowwise RMSE\n    return tf.reduce_mean(rowwise_rmse)\n\ndef build_model_3(input_shape=(1,), cell_type_shape=(6,), embedding_dim=128, dropout_rate=0.9, optimizer=None):\n    # SMILES input\n    smiles_input = Input(shape=(1,), dtype=\"string\", name=\"smiles_input\")\n    char_vectors = char_vectorizer_smiles(smiles_input)\n    char_embeddings = char_embedding(char_vectors)\n    embed_flatten = Flatten()(char_embeddings)\n\n    # Cell type input\n    cell_type_input = Input(shape=cell_type_shape, dtype=tf.float32, name=\"cell_type_input\")\n\n    # Concatenate SMILES embedding and cell type data\n    concatenated_data = Concatenate(name=\"concatenate\")([embed_flatten, cell_type_input])\n\n    # Apply Gaussian Noise\n    x = GaussianNoise(0.09)(concatenated_data)\n\n    # Neural network layers\n    x = Dense(512, activation='elu')(x)\n    x = BatchNormalization()(x)\n#     x = Dropout(dropout_rate)(x)\n\n    x = Dense(256, activation='elu')(x)\n#     x = BatchNormalization()(x)\n#     x = Dropout(dropout_rate)(x)\n\n    x = Dense(128, activation='elu')(x)\n    x = BatchNormalization()(x)\n    x = Dropout(dropout_rate)(x)\n\n    x = Dense(256, activation='elu')(x)\n#     x = BatchNormalization()(x)\n#     x = Dropout(dropout_rate)(x)\n\n    x = Dense(512, activation='elu')(x)\n    x = BatchNormalization()(x)\n    x = Dropout(dropout_rate)(x)\n\n    # Output layer\n    output = Dense(18211, activation='linear')(x)\n\n    # Model creation\n    model = Model(inputs=[smiles_input, cell_type_input], outputs=output, name=\"custom_model\")\n\n    # Compile the model\n    if optimizer:\n        model.compile(loss=mean_rowwise_rmse, optimizer=optimizer, metrics=[\"mae\"])\n    else:\n        model.compile(loss=mean_rowwise_rmse, optimizer=tf.keras.optimizers.Lion(\n    ), metrics=[\"mae\"])\n        \n\n    return model\n","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:59:56.715554Z","iopub.execute_input":"2023-11-14T10:59:56.716224Z","iopub.status.idle":"2023-11-14T10:59:56.728845Z","shell.execute_reply.started":"2023-11-14T10:59:56.716193Z","shell.execute_reply":"2023-11-14T10:59:56.727829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_model(build_model_3(), show_shapes=True, show_layer_activations=True)","metadata":{"execution":{"iopub.status.busy":"2023-11-14T11:00:00.727039Z","iopub.execute_input":"2023-11-14T11:00:00.727866Z","iopub.status.idle":"2023-11-14T11:00:01.274777Z","shell.execute_reply.started":"2023-11-14T11:00:00.727835Z","shell.execute_reply":"2023-11-14T11:00:01.273824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_2 = build_model_3()\nmodel_2_history = model_2.fit(\n    x=[train_chars, train_cell_type],\n    y=train_labels,\n    epochs=40,\n    verbose=0,\n    validation_data=([val_chars, val_cell_type], val_labels),\n    callbacks=[create_model_checkpoint(\"model_2\")])","metadata":{"execution":{"iopub.status.busy":"2023-11-14T11:00:07.872597Z","iopub.execute_input":"2023-11-14T11:00:07.872970Z","iopub.status.idle":"2023-11-14T11:00:36.018612Z","shell.execute_reply.started":"2023-11-14T11:00:07.872938Z","shell.execute_reply":"2023-11-14T11:00:36.017709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_2.load_weights(\"model_2\")\nprint(\"Scores on test:\\n\")\ncalculate_mae_and_mrrmse(model=model_2, data=[test_chars, test_cell_type], y_true=test_labels, scaler=scaler)\nprint(\"\\nScores on full data:\\n\")\ncalculate_mae_and_mrrmse(model=model_2, data=[full_train_chars, full_cell_type], y_true=full_labels, scaler=scaler)\nprint(\"\\nPlot training and validation curves:\\n\")\nplot_training_history(model_2_history, metrics=[\"mae\"])","metadata":{"execution":{"iopub.status.busy":"2023-11-14T11:00:36.020797Z","iopub.execute_input":"2023-11-14T11:00:36.021426Z","iopub.status.idle":"2023-11-14T11:00:39.801679Z","shell.execute_reply.started":"2023-11-14T11:00:36.021391Z","shell.execute_reply":"2023-11-14T11:00:39.800660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.__version__","metadata":{"execution":{"iopub.status.busy":"2023-11-14T11:00:39.802893Z","iopub.execute_input":"2023-11-14T11:00:39.803207Z","iopub.status.idle":"2023-11-14T11:00:39.809116Z","shell.execute_reply.started":"2023-11-14T11:00:39.803180Z","shell.execute_reply":"2023-11-14T11:00:39.808163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_sub(res, idx=0):\n    df = pd.DataFrame(np.mean(res, axis = 0), columns=sample_columns)\n    df.insert(0, 'id', range(255))\n    df.to_csv(f\"submission_{idx}.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-14T11:00:39.811286Z","iopub.execute_input":"2023-11-14T11:00:39.811619Z","iopub.status.idle":"2023-11-14T11:00:39.819843Z","shell.execute_reply.started":"2023-11-14T11:00:39.811587Z","shell.execute_reply":"2023-11-14T11:00:39.818897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"full_train_chars","metadata":{"execution":{"iopub.status.busy":"2023-11-14T11:52:10.332098Z","iopub.execute_input":"2023-11-14T11:52:10.332963Z","iopub.status.idle":"2023-11-14T11:52:10.347018Z","shell.execute_reply.started":"2023-11-14T11:52:10.332920Z","shell.execute_reply":"2023-11-14T11:52:10.346007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #dropout 0.3\n# res = k_fold_predict(build_model_3, \n#                      features_list=[full_train_chars, np.array(full_cell_type)], \n#                      full_labels=full_labels, \n#                      num_folds=10, \n#                      random_state=RNDST1, \n#                      epochs=300, \n#                      scaler=scaler,\n#                      test_data=[np.array(final_test_smiles), final_cell_type], \n#                      early_stopping=200,\n#                     dropout=0.35)\n# build_sub(res, idx=0)","metadata":{"execution":{"iopub.status.busy":"2023-11-14T11:15:58.665966Z","iopub.execute_input":"2023-11-14T11:15:58.666323Z","iopub.status.idle":"2023-11-14T11:26:01.062634Z","shell.execute_reply.started":"2023-11-14T11:15:58.666297Z","shell.execute_reply":"2023-11-14T11:26:01.061423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #dropout 0.3\n# res = k_fold_predict(build_model_3, \n#                      features_list=[full_train_chars, np.array(full_cell_type)], \n#                      full_labels=full_labels, \n#                      num_folds=10, \n#                      random_state=RNDST1, \n#                      epochs=300, \n#                      scaler=scaler,\n#                      test_data=[np.array(final_test_smiles), final_cell_type], \n#                      early_stopping=200,\n#                     dropout=0.375)\n# build_sub(res, idx=1)","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:52:29.617117Z","iopub.status.idle":"2023-11-14T10:52:29.617446Z","shell.execute_reply.started":"2023-11-14T10:52:29.617286Z","shell.execute_reply":"2023-11-14T10:52:29.617302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #dropout 0.3\n# res = k_fold_predict(build_model_3, \n#                      features_list=[full_train_chars, np.array(full_cell_type)], \n#                      full_labels=full_labels, \n#                      num_folds=10, \n#                      random_state=RNDST1, \n#                      epochs=300, \n#                      scaler=scaler,\n#                      test_data=[np.array(final_test_smiles), final_cell_type], \n#                      early_stopping=200,\n#                      dropout=0.4)\n# build_sub(res, idx=2)","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:52:29.619060Z","iopub.status.idle":"2023-11-14T10:52:29.619378Z","shell.execute_reply.started":"2023-11-14T10:52:29.619218Z","shell.execute_reply":"2023-11-14T10:52:29.619233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dropout 0.7\nres = k_fold_predict(build_model_3, \n                     features_list=[full_train_chars, np.array(full_cell_type)], \n                     full_labels=full_labels, \n                     num_folds=10, \n                     random_state=RNDST1, \n                     epochs=300, \n                     scaler=scaler,\n                     test_data=[np.array(final_test_smiles), final_cell_type], \n                     early_stopping=200,\n                    dropout=0.425)\nbuild_sub(res, idx=3)","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:52:29.620618Z","iopub.status.idle":"2023-11-14T10:52:29.620978Z","shell.execute_reply.started":"2023-11-14T10:52:29.620787Z","shell.execute_reply":"2023-11-14T10:52:29.620803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #dropout 0.9\n# res = k_fold_predict(build_model_3, \n#                      features_list=[full_train_chars, np.array(full_cell_type)], \n#                      full_labels=full_labels, \n#                      num_folds=10, \n#                      random_state=RNDST1, \n#                      epochs=300, \n#                      scaler=scaler,\n#                      test_data=[np.array(final_test_smiles), final_cell_type], \n#                      early_stopping=200,\n#                     dropout=0.45)\n# build_sub(res, idx=4)","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:52:29.622232Z","iopub.status.idle":"2023-11-14T10:52:29.622574Z","shell.execute_reply.started":"2023-11-14T10:52:29.622402Z","shell.execute_reply":"2023-11-14T10:52:29.622418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Make Prediction","metadata":{}},{"cell_type":"markdown","source":"We'll train `Model_2` with full train data for final time to make prediction. ","metadata":{}},{"cell_type":"code","source":"smiles_input = Input(shape=(1,), dtype=\"string\", name=\"smiles_input\")\ncelltype_input = Input(shape=(6,), dtype=tf.float32, name=\"cell_type_input\")\n\nchar_vectors = char_vectorizer_smiles(smiles_input)\nchar_embeddings = char_embedding(char_vectors)\nembed_flatten = Flatten()(char_embeddings)\nconcatenated_data = concatenate([embed_flatten, celltype_input], name=\"concatenate\")\nx = GaussianNoise(0.09)(concatenated_data)\nx = Dense(512, activation='elu')(x)\nx = Dense(256, activation='elu')(x)\nx = Dense(128, activation='elu')(x)\nx = Dense(256, activation='elu')(x)\nx = Dense(512, activation='elu')(x)\noutput = Dense(18211, activation='linear')(x)\n\nmodel = Model(inputs=[smiles_input, celltype_input], outputs=output)\nmodel.compile(loss=\"mae\", optimizer=tf.keras.optimizers.Adam(learning_rate=0.00098), metrics=[\"mae\"])\n\nhistory = model.fit(x=[full_train_chars, np.array(full_cell_type)], y=full_labels,\n                              epochs=100,\n                              verbose=0,\n                              validation_data=([test_chars, test_cell_type], test_labels))\n\nprint(\"Scores on test:\\n\")\ncalculate_mae_and_mrrmse(model=model, data=[test_chars, test_cell_type], y_true=test_labels, scaler=scaler)\nprint(\"\\nScores on full data:\\n\")\ncalculate_mae_and_mrrmse(model=model, data=[full_train_chars, full_cell_type], y_true=full_labels, scaler=scaler)\nprint(\"\\nPlot training and validation curves:\\n\")\nplot_training_history(history, metrics=[\"mae\"])\n\nsample_columns = sample_submission.columns\nsample_columns = sample_columns[1:]\n\npreds = model.predict([np.array(final_test_smiles), final_cell_type], batch_size=1)\npreds = scaler.inverse_transform(preds)","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:52:29.623811Z","iopub.status.idle":"2023-11-14T10:52:29.624164Z","shell.execute_reply.started":"2023-11-14T10:52:29.623993Z","shell.execute_reply":"2023-11-14T10:52:29.624009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame(np.mean(res, axis = 0), columns=sample_columns)\ndf.insert(0, 'id', range(255))\ndf.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-14T10:52:29.624990Z","iopub.status.idle":"2023-11-14T10:52:29.625301Z","shell.execute_reply.started":"2023-11-14T10:52:29.625148Z","shell.execute_reply":"2023-11-14T10:52:29.625162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> **Note:** The model is not at all tuned. So try tuning hyperparams and learning rate to get better LB score.  And let me know how it goes. ","metadata":{}},{"cell_type":"markdown","source":"# Improving Model\n* Experiment with different `SMILES` embedding dimensions, observe and find the best.\n* Explore various layer types such as `Attention`, `BatchNormalization` `Bidirectional`, `RNN`...\n* Use different model architecture.\n* Fine-tune model hyperparameters and learning rates for optimization.\n* Test out alternative loss functions to determine their impact on model performance.\n* Evaluate the use of `MinMax` Scaler instead of `StandardScaler` for scaling the labels\n* Try vectorization or embedding for the `cell_type`\n*  Increase the number of features and create a more complex model to capture additional patterns.","metadata":{}}]}