{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":59094,"databundleVersionId":7010844,"sourceType":"competition"}],"dockerImageVersionId":30580,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Pytorch and TF Pipelines\n\nThis notebook shows part of [our solution](https://www.kaggle.com/code/alexandervc/op2-u900-team-blend).\n\nIn this notebook I will show you one example from my part **0.587 blend (0.767 private)** because I used different ideas with different hyperparameters but with the same training pipeline.\n\nAlso, in this example I trained models with inverse (multiplied by -1) targets.\n\nOne of the most important points is that in my pipeline I used a multi-label stratified cross-validation scheme by cell type, and OOF improved and LB improved for me.\n\nThe features were very simple: cell type label encoding, sm name label encoding, SMILES character vectorization.\nWhen I tried to create numeric features from categorical features, it didn't work well.\nFor the targets, I used StandardScaler.\n\nFor the pytorch model I used SophiaG optimizer and for the keras model I used Lion. The hyperparameters changed from version to version.\n\nTherefore, I trained the models within this pipeline in several ways:\n* With the original (not inverted) targets.\n* With the LSTM layer after smile embeddings.\n* With sigmoidal range activation as the model output.\n* With multiplication of outputs by coefs.\n* With pseudolabels from these models blends.\n\nA combination of these models was used to get blend prediction.","metadata":{}},{"cell_type":"code","source":"# https://github.com/trent-b/iterative-stratification/blob/master/iterstrat/ml_stratifiers.py\n\nimport numpy as np\n\nfrom sklearn.utils import check_random_state\nfrom sklearn.utils.validation import _num_samples, check_array\nfrom sklearn.utils.multiclass import type_of_target\n\nfrom sklearn.model_selection._split import _BaseKFold, _RepeatedSplits, \\\n    BaseShuffleSplit, _validate_shuffle_split\n\n\ndef IterativeStratification(labels, r, random_state):\n    \"\"\"This function implements the Iterative Stratification algorithm described\n    in the following paper:\n    Sechidis K., Tsoumakas G., Vlahavas I. (2011) On the Stratification of\n    Multi-Label Data. In: Gunopulos D., Hofmann T., Malerba D., Vazirgiannis M.\n    (eds) Machine Learning and Knowledge Discovery in Databases. ECML PKDD\n    2011. Lecture Notes in Computer Science, vol 6913. Springer, Berlin,\n    Heidelberg.\n    \"\"\"\n\n    n_samples = labels.shape[0]\n    test_folds = np.zeros(n_samples, dtype=int)\n\n    # Calculate the desired number of examples at each subset\n    c_folds = r * n_samples\n\n    # Calculate the desired number of examples of each label at each subset\n    c_folds_labels = np.outer(r, labels.sum(axis=0))\n\n    labels_not_processed_mask = np.ones(n_samples, dtype=bool)\n\n    while np.any(labels_not_processed_mask):\n        # Find the label with the fewest (but at least one) remaining examples,\n        # breaking ties randomly\n        num_labels = labels[labels_not_processed_mask].sum(axis=0)\n\n        # Handle case where only all-zero labels are left by distributing\n        # across all folds as evenly as possible (not in original algorithm but\n        # mentioned in the text). (By handling this case separately, some\n        # code redundancy is introduced; however, this approach allows for\n        # decreased execution time when there are a relatively large number\n        # of all-zero labels.)\n        if num_labels.sum() == 0:\n            sample_idxs = np.where(labels_not_processed_mask)[0]\n\n            for sample_idx in sample_idxs:\n                fold_idx = np.where(c_folds == c_folds.max())[0]\n\n                if fold_idx.shape[0] > 1:\n                    fold_idx = fold_idx[random_state.choice(fold_idx.shape[0])]\n\n                test_folds[sample_idx] = fold_idx\n                c_folds[fold_idx] -= 1\n\n            break\n\n        label_idx = np.where(num_labels == num_labels[np.nonzero(num_labels)].min())[0]\n        if label_idx.shape[0] > 1:\n            label_idx = label_idx[random_state.choice(label_idx.shape[0])]\n\n        sample_idxs = np.where(np.logical_and(labels[:, label_idx].flatten(), labels_not_processed_mask))[0]\n\n        for sample_idx in sample_idxs:\n            # Find the subset(s) with the largest number of desired examples\n            # for this label, breaking ties by considering the largest number\n            # of desired examples, breaking further ties randomly\n            label_folds = c_folds_labels[:, label_idx]\n            fold_idx = np.where(label_folds == label_folds.max())[0]\n\n            if fold_idx.shape[0] > 1:\n                temp_fold_idx = np.where(c_folds[fold_idx] ==\n                                         c_folds[fold_idx].max())[0]\n                fold_idx = fold_idx[temp_fold_idx]\n\n                if temp_fold_idx.shape[0] > 1:\n                    fold_idx = fold_idx[random_state.choice(temp_fold_idx.shape[0])]\n\n            test_folds[sample_idx] = fold_idx\n            labels_not_processed_mask[sample_idx] = False\n\n            # Update desired number of examples\n            c_folds_labels[fold_idx, labels[sample_idx]] -= 1\n            c_folds[fold_idx] -= 1\n\n    return test_folds\n\n\nclass MultilabelStratifiedKFold(_BaseKFold):\n    \"\"\"Multilabel stratified K-Folds cross-validator\n    Provides train/test indices to split multilabel data into train/test sets.\n    This cross-validation object is a variation of KFold that returns\n    stratified folds for multilabel data. The folds are made by preserving\n    the percentage of samples for each label.\n    Parameters\n    ----------\n    n_splits : int, default=3\n        Number of folds. Must be at least 2.\n    shuffle : boolean, optional\n        Whether to shuffle each stratification of the data before splitting\n        into batches.\n    random_state : int, RandomState instance or None, optional, default=None\n        If int, random_state is the seed used by the random number generator;\n        If RandomState instance, random_state is the random number generator;\n        If None, the random number generator is the RandomState instance used\n        by `np.random`. Unlike StratifiedKFold that only uses random_state\n        when ``shuffle`` == True, this multilabel implementation\n        always uses the random_state since the iterative stratification\n        algorithm breaks ties randomly.\n    Examples\n    --------\n    >>> from iterstrat.ml_stratifiers import MultilabelStratifiedKFold\n    >>> import numpy as np\n    >>> X = np.array([[1,2], [3,4], [1,2], [3,4], [1,2], [3,4], [1,2], [3,4]])\n    >>> y = np.array([[0,0], [0,0], [0,1], [0,1], [1,1], [1,1], [1,0], [1,0]])\n    >>> mskf = MultilabelStratifiedKFold(n_splits=2, random_state=0)\n    >>> mskf.get_n_splits(X, y)\n    2\n    >>> print(mskf)  # doctest: +NORMALIZE_WHITESPACE\n    MultilabelStratifiedKFold(n_splits=2, random_state=0, shuffle=False)\n    >>> for train_index, test_index in mskf.split(X, y):\n    ...    print(\"TRAIN:\", train_index, \"TEST:\", test_index)\n    ...    X_train, X_test = X[train_index], X[test_index]\n    ...    y_train, y_test = y[train_index], y[test_index]\n    TRAIN: [0 3 4 6] TEST: [1 2 5 7]\n    TRAIN: [1 2 5 7] TEST: [0 3 4 6]\n    Notes\n    -----\n    Train and test sizes may be slightly different in each fold.\n    See also\n    --------\n    RepeatedMultilabelStratifiedKFold: Repeats Multilabel Stratified K-Fold\n    n times.\n    \"\"\"\n\n    def __init__(self, n_splits=3, *, shuffle=False, random_state=None):\n        super(MultilabelStratifiedKFold, self).__init__(n_splits=n_splits, shuffle=shuffle, random_state=random_state)\n\n    def _make_test_folds(self, X, y):\n        y = np.asarray(y, dtype=bool)\n        type_of_target_y = type_of_target(y)\n\n        if type_of_target_y != 'multilabel-indicator':\n            raise ValueError(\n                'Supported target type is: multilabel-indicator. Got {!r} instead.'.format(type_of_target_y))\n\n        num_samples = y.shape[0]\n\n        rng = check_random_state(self.random_state)\n        indices = np.arange(num_samples)\n\n        if self.shuffle:\n            rng.shuffle(indices)\n            y = y[indices]\n\n        r = np.asarray([1 / self.n_splits] * self.n_splits)\n\n        test_folds = IterativeStratification(labels=y, r=r, random_state=rng)\n\n        return test_folds[np.argsort(indices)]\n\n    def _iter_test_masks(self, X=None, y=None, groups=None):\n        test_folds = self._make_test_folds(X, y)\n        for i in range(self.n_splits):\n            yield test_folds == i\n\n    def split(self, X, y, groups=None):\n        \"\"\"Generate indices to split data into training and test set.\n        Parameters\n        ----------\n        X : array-like, shape (n_samples, n_features)\n            Training data, where n_samples is the number of samples\n            and n_features is the number of features.\n            Note that providing ``y`` is sufficient to generate the splits and\n            hence ``np.zeros(n_samples)`` may be used as a placeholder for\n            ``X`` instead of actual training data.\n        y : array-like, shape (n_samples, n_labels)\n            The target variable for supervised learning problems.\n            Multilabel stratification is done based on the y labels.\n        groups : object\n            Always ignored, exists for compatibility.\n        Returns\n        -------\n        train : ndarray\n            The training set indices for that split.\n        test : ndarray\n            The testing set indices for that split.\n        Notes\n        -----\n        Randomized CV splitters may return different results for each call of\n        split. You can make the results identical by setting ``random_state``\n        to an integer.\n        \"\"\"\n        y = check_array(y, ensure_2d=False, dtype=None)\n        return super(MultilabelStratifiedKFold, self).split(X, y, groups)\n\n\nclass RepeatedMultilabelStratifiedKFold(_RepeatedSplits):\n    \"\"\"Repeated Multilabel Stratified K-Fold cross validator.\n    Repeats Mulilabel Stratified K-Fold n times with different randomization\n    in each repetition.\n    Parameters\n    ----------\n    n_splits : int, default=5\n        Number of folds. Must be at least 2.\n    n_repeats : int, default=10\n        Number of times cross-validator needs to be repeated.\n    random_state : None, int or RandomState, default=None\n        Random state to be used to generate random state for each\n        repetition as well as randomly breaking ties within the iterative\n        stratification algorithm.\n    Examples\n    --------\n    >>> from iterstrat.ml_stratifiers import RepeatedMultilabelStratifiedKFold\n    >>> import numpy as np\n    >>> X = np.array([[1,2], [3,4], [1,2], [3,4], [1,2], [3,4], [1,2], [3,4]])\n    >>> y = np.array([[0,0], [0,0], [0,1], [0,1], [1,1], [1,1], [1,0], [1,0]])\n    >>> rmskf = RepeatedMultilabelStratifiedKFold(n_splits=2, n_repeats=2,\n    ...     random_state=0)\n    >>> for train_index, test_index in rmskf.split(X, y):\n    ...     print(\"TRAIN:\", train_index, \"TEST:\", test_index)\n    ...     X_train, X_test = X[train_index], X[test_index]\n    ...     y_train, y_test = y[train_index], y[test_index]\n    ...\n    TRAIN: [0 3 4 6] TEST: [1 2 5 7]\n    TRAIN: [1 2 5 7] TEST: [0 3 4 6]\n    TRAIN: [0 1 4 5] TEST: [2 3 6 7]\n    TRAIN: [2 3 6 7] TEST: [0 1 4 5]\n    See also\n    --------\n    RepeatedStratifiedKFold: Repeats (Non-multilabel) Stratified K-Fold\n    n times.\n    \"\"\"\n    def __init__(self, n_splits=5, *, n_repeats=10, random_state=None):\n        super(RepeatedMultilabelStratifiedKFold, self).__init__(\n            MultilabelStratifiedKFold, n_repeats=n_repeats, random_state=random_state,\n            n_splits=n_splits)\n\n\nclass MultilabelStratifiedShuffleSplit(BaseShuffleSplit):\n    \"\"\"Multilabel Stratified ShuffleSplit cross-validator\n    Provides train/test indices to split data into train/test sets.\n    This cross-validation object is a merge of MultilabelStratifiedKFold and\n    ShuffleSplit, which returns stratified randomized folds for multilabel\n    data. The folds are made by preserving the percentage of each label.\n    Note: like the ShuffleSplit strategy, multilabel stratified random splits\n    do not guarantee that all folds will be different, although this is\n    still very likely for sizeable datasets.\n    Parameters\n    ----------\n    n_splits : int, default 10\n        Number of re-shuffling & splitting iterations.\n    test_size : float, int, None, optional\n        If float, should be between 0.0 and 1.0 and represent the proportion\n        of the dataset to include in the test split. If int, represents the\n        absolute number of test samples. If None, the value is set to the\n        complement of the train size. By default, the value is set to 0.1.\n        The default will change in version 0.21. It will remain 0.1 only\n        if ``train_size`` is unspecified, otherwise it will complement\n        the specified ``train_size``.\n    train_size : float, int, or None, default is None\n        If float, should be between 0.0 and 1.0 and represent the\n        proportion of the dataset to include in the train split. If\n        int, represents the absolute number of train samples. If None,\n        the value is automatically set to the complement of the test size.\n    random_state : int, RandomState instance or None, optional (default=None)\n        If int, random_state is the seed used by the random number generator;\n        If RandomState instance, random_state is the random number generator;\n        If None, the random number generator is the RandomState instance used\n        by `np.random`. Unlike StratifiedShuffleSplit that only uses\n        random_state when ``shuffle`` == True, this multilabel implementation\n        always uses the random_state since the iterative stratification\n        algorithm breaks ties randomly.\n    Examples\n    --------\n    >>> from iterstrat.ml_stratifiers import MultilabelStratifiedShuffleSplit\n    >>> import numpy as np\n    >>> X = np.array([[1,2], [3,4], [1,2], [3,4], [1,2], [3,4], [1,2], [3,4]])\n    >>> y = np.array([[0,0], [0,0], [0,1], [0,1], [1,1], [1,1], [1,0], [1,0]])\n    >>> msss = MultilabelStratifiedShuffleSplit(n_splits=3, test_size=0.5,\n    ...    random_state=0)\n    >>> msss.get_n_splits(X, y)\n    3\n    >>> print(mss)       # doctest: +ELLIPSIS\n    MultilabelStratifiedShuffleSplit(n_splits=3, random_state=0, test_size=0.5,\n                                     train_size=None)\n    >>> for train_index, test_index in msss.split(X, y):\n    ...    print(\"TRAIN:\", train_index, \"TEST:\", test_index)\n    ...    X_train, X_test = X[train_index], X[test_index]\n    ...    y_train, y_test = y[train_index], y[test_index]\n    TRAIN: [1 2 5 7] TEST: [0 3 4 6]\n    TRAIN: [2 3 6 7] TEST: [0 1 4 5]\n    TRAIN: [1 2 5 6] TEST: [0 3 4 7]\n    Notes\n    -----\n    Train and test sizes may be slightly different from desired due to the\n    preference of stratification over perfectly sized folds.\n    \"\"\"\n\n    def __init__(self, n_splits=10, *, test_size=\"default\", train_size=None,\n                 random_state=None):\n        super(MultilabelStratifiedShuffleSplit, self).__init__(\n            n_splits=n_splits, test_size=test_size, train_size=train_size, random_state=random_state)\n\n    def _iter_indices(self, X, y, groups=None):\n        n_samples = _num_samples(X)\n        y = check_array(y, ensure_2d=False, dtype=None)\n        y = np.asarray(y, dtype=bool)\n        type_of_target_y = type_of_target(y)\n\n        if type_of_target_y != 'multilabel-indicator':\n            raise ValueError(\n                'Supported target type is: multilabel-indicator. Got {!r} instead.'.format(\n                    type_of_target_y))\n\n        n_train, n_test = _validate_shuffle_split(n_samples, self.test_size,\n                                                  self.train_size)\n\n        n_samples = y.shape[0]\n        rng = check_random_state(self.random_state)\n        y_orig = y.copy()\n\n        r = np.array([n_train, n_test]) / (n_train + n_test)\n\n        for _ in range(self.n_splits):\n            indices = np.arange(n_samples)\n            rng.shuffle(indices)\n            y = y_orig[indices]\n\n            test_folds = IterativeStratification(labels=y, r=r, random_state=rng)\n\n            test_idx = test_folds[np.argsort(indices)] == 1\n            test = np.where(test_idx)[0]\n            train = np.where(~test_idx)[0]\n\n            yield train, test\n\n    def split(self, X, y, groups=None):\n        \"\"\"Generate indices to split data into training and test set.\n        Parameters\n        ----------\n        X : array-like, shape (n_samples, n_features)\n            Training data, where n_samples is the number of samples\n            and n_features is the number of features.\n            Note that providing ``y`` is sufficient to generate the splits and\n            hence ``np.zeros(n_samples)`` may be used as a placeholder for\n            ``X`` instead of actual training data.\n        y : array-like, shape (n_samples, n_labels)\n            The target variable for supervised learning problems.\n            Multilabel stratification is done based on the y labels.\n        groups : object\n            Always ignored, exists for compatibility.\n        Returns\n        -------\n        train : ndarray\n            The training set indices for that split.\n        test : ndarray\n            The testing set indices for that split.\n        Notes\n        -----\n        Randomized CV splitters may return different results for each call of\n        split. You can make the results identical by setting ``random_state``\n        to an integer.\n        \"\"\"\n        y = check_array(y, ensure_2d=False, dtype=None)\n        return super(MultilabelStratifiedShuffleSplit, self).split(X, y, groups)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T15:40:33.324917Z","iopub.execute_input":"2023-11-24T15:40:33.325753Z","iopub.status.idle":"2023-11-24T15:40:33.377204Z","shell.execute_reply.started":"2023-11-24T15:40:33.325704Z","shell.execute_reply":"2023-11-24T15:40:33.375722Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\n!git clone https://github.com/kyegomez/Sophia.git\n! python Sophia/setup.py install\n!rm Sophia/Sophia/__init__.py\nfrom Sophia.Sophia.Sophia import SophiaG","metadata":{"execution":{"iopub.status.busy":"2023-11-24T15:40:33.378888Z","iopub.execute_input":"2023-11-24T15:40:33.379864Z","iopub.status.idle":"2023-11-24T15:40:45.039842Z","shell.execute_reply.started":"2023-11-24T15:40:33.379805Z","shell.execute_reply":"2023-11-24T15:40:45.037964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.nn import functional as F\nimport torch.optim as optim\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import DataLoader, Dataset\nfrom typing import *\n\nimport tensorflow.keras.backend as K\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Input, Dense, Embedding, GaussianDropout, BatchNormalization, Concatenate, Flatten\nfrom tensorflow.keras.layers import GaussianNoise, Dropout\nfrom tensorflow.keras.layers.experimental.preprocessing import TextVectorization\nimport tensorflow_addons as tfa\nfrom sklearn.model_selection import StratifiedKFold, KFold, GroupKFold\nimport matplotlib.pyplot as plt\nimport warnings\nimport copy\nimport random\nimport os\nimport gc\nfrom sklearn.linear_model import Ridge, LinearRegression\nfrom sklearn.model_selection import KFold\nimport numpy as np\nimport pandas as pd\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.preprocessing import QuantileTransformer\nfrom sklearn.preprocessing import LabelEncoder, OneHotEncoder\nfrom sklearn.decomposition import TruncatedSVD, PCA\nfrom sklearn.preprocessing import StandardScaler\nimport category_encoders as ce\nfrom sklearn.manifold import TSNE\nfrom sklearn.metrics import mean_absolute_error, r2_score\nfrom tqdm import tqdm\nfrom sklearn.feature_extraction.text import TfidfVectorizer, HashingVectorizer\nimport scipy\n\nSEED = 42\nTSVD_SIZE = 768\nPCA_SIZE = 10\n\ntqdm.pandas()\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-24T15:40:45.042806Z","iopub.execute_input":"2023-11-24T15:40:45.043347Z","iopub.status.idle":"2023-11-24T15:41:02.037222Z","shell.execute_reply.started":"2023-11-24T15:40:45.043308Z","shell.execute_reply":"2023-11-24T15:41:02.035851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.utils.tensorboard import SummaryWriter\n# saves logs to ./runs/ directory\n# open tensorboard with terminal command: tensorboard --logdir=runs\nlogger = SummaryWriter()","metadata":{"execution":{"iopub.status.busy":"2023-11-24T15:41:02.038902Z","iopub.execute_input":"2023-11-24T15:41:02.039679Z","iopub.status.idle":"2023-11-24T15:41:02.072397Z","shell.execute_reply.started":"2023-11-24T15:41:02.039644Z","shell.execute_reply":"2023-11-24T15:41:02.07072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.environ[\"CUDA_LAUNCH_BLOCKING\"] = \"1\"","metadata":{"execution":{"iopub.status.busy":"2023-11-24T15:41:02.074095Z","iopub.execute_input":"2023-11-24T15:41:02.074467Z","iopub.status.idle":"2023-11-24T15:41:02.079705Z","shell.execute_reply.started":"2023-11-24T15:41:02.074437Z","shell.execute_reply":"2023-11-24T15:41:02.078591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# set random seed\ndef set_seed(seed: int = 42) -> None:\n    np.random.seed(seed)\n    random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    # When running on the CuDNN backend, two further options must be set\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n    # Set a fixed value for the hash seed\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    print(f\"Random seed set as {seed}\")\n    return\n\nset_seed(SEED)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T15:41:02.080955Z","iopub.execute_input":"2023-11-24T15:41:02.082691Z","iopub.status.idle":"2023-11-24T15:41:02.104543Z","shell.execute_reply.started":"2023-11-24T15:41:02.082645Z","shell.execute_reply":"2023-11-24T15:41:02.102871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Preparation","metadata":{}},{"cell_type":"markdown","source":"## Read Data","metadata":{}},{"cell_type":"code","source":"df = pd.read_parquet(\"/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet\")\ndf.drop([\"control\"], axis=1, inplace=True)\n\nx_train = df.iloc[:, :4]\ny_train = df.iloc[:, 4:]\ny_train = (-1) * y_train.values\n\ndel df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-11-24T15:41:02.109285Z","iopub.execute_input":"2023-11-24T15:41:02.11045Z","iopub.status.idle":"2023-11-24T15:41:05.654359Z","shell.execute_reply.started":"2023-11-24T15:41:02.110398Z","shell.execute_reply":"2023-11-24T15:41:05.652968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train","metadata":{"execution":{"iopub.status.busy":"2023-11-24T15:41:05.655997Z","iopub.execute_input":"2023-11-24T15:41:05.656363Z","iopub.status.idle":"2023-11-24T15:41:05.683231Z","shell.execute_reply.started":"2023-11-24T15:41:05.656333Z","shell.execute_reply":"2023-11-24T15:41:05.681636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train","metadata":{"execution":{"iopub.status.busy":"2023-11-24T15:41:05.684896Z","iopub.execute_input":"2023-11-24T15:41:05.685928Z","iopub.status.idle":"2023-11-24T15:41:05.695996Z","shell.execute_reply.started":"2023-11-24T15:41:05.685888Z","shell.execute_reply":"2023-11-24T15:41:05.694475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test = pd.read_csv(\"/kaggle/input/open-problems-single-cell-perturbations/id_map.csv\")\nx_test.drop([\"id\"], axis=1, inplace=True)\nx_test","metadata":{"execution":{"iopub.status.busy":"2023-11-24T15:41:05.724137Z","iopub.execute_input":"2023-11-24T15:41:05.725344Z","iopub.status.idle":"2023-11-24T15:41:05.752816Z","shell.execute_reply.started":"2023-11-24T15:41:05.725289Z","shell.execute_reply":"2023-11-24T15:41:05.751502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_columns(data, id_map):\n    sm_name_to_smiles = data.set_index('sm_name')['SMILES'].to_dict()\n    sm_lincs_id = data.set_index('sm_name')[\"sm_lincs_id\"].to_dict()\n\n    id_map['SMILES'] = id_map['sm_name'].map(sm_name_to_smiles)\n    id_map['sm_lincs_id'] = id_map['sm_name'].map(sm_lincs_id)\n\n    return id_map\n\nx_test = add_columns(x_train, x_test)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T15:41:05.754409Z","iopub.execute_input":"2023-11-24T15:41:05.754823Z","iopub.status.idle":"2023-11-24T15:41:05.773307Z","shell.execute_reply.started":"2023-11-24T15:41:05.754792Z","shell.execute_reply":"2023-11-24T15:41:05.771654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create Cross Validation Features","metadata":{}},{"cell_type":"code","source":"le_cell_type = LabelEncoder()\nx_train[\"cell_type\"] = le_cell_type.fit_transform(x_train[\"cell_type\"])\nx_test[\"cell_type\"] = le_cell_type.transform(x_test[\"cell_type\"])\n\nle_sm_name = LabelEncoder()\nx_train[\"sm_name\"] = le_sm_name.fit_transform(x_train[\"sm_name\"])\nx_test[\"sm_name\"] = le_sm_name.transform(x_test[\"sm_name\"])","metadata":{"execution":{"iopub.status.busy":"2023-11-24T15:41:05.775495Z","iopub.execute_input":"2023-11-24T15:41:05.77597Z","iopub.status.idle":"2023-11-24T15:41:05.789789Z","shell.execute_reply.started":"2023-11-24T15:41:05.775933Z","shell.execute_reply":"2023-11-24T15:41:05.788243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prepare indexes for AmbrosM scheme:\n# For AmbrosM scheme: \nfolds_index_data_AmbrosM = [ ]\ntrain_sm_names = ['Idelalisib', 'Crizotinib', 'Linagliptin', 'Palbociclib', 'Dabrafenib', 'Alvocidib', 'LDN 193189', 'R428', 'Porcn Inhibitor III', \n  'Belinostat', 'Foretinib', 'MLN 2238', 'Penfluridol', 'Dactolisib', 'O-Demethylated Adapalene', 'Oprozomib (ONX 0912)', 'CHIR-99021']\nlist_fold_ids =  ['NK cells', 'T cells CD4+', 'T cells CD8+', 'T regulatory cells'] \nfor fold_id in list_fold_ids:\n    mask_va = (x_train.cell_type == fold_id) & ~x_train.sm_name.isin(train_sm_names)\n    mask_tr = ~mask_va # 485 or 487 training rows\n    IX_train = np.where( mask_tr > 0 )[0]\n    IX_test = np.where( mask_va > 0 )[0]\n    #print(fold_id,  len(IX_test), type(IX_test), IX_test[:3], len(IX_train), type(IX_train), IX_train[:3]  )\n    folds_index_data_AmbrosM.append( [IX_train, IX_test ])\n\n# For MT CV scheme:\nfolds_index_data_MT = []\nfold_to_compounds = {0: ['Alvocidib', 'Belinostat', 'Foretinib', 'LDN 193189',  'Linagliptin', 'O-Demethylated Adapalene'],\n 1: ['Dabrafenib', 'Dactolisib', 'Idelalisib', 'MLN 2238', 'Palbociclib', 'Porcn Inhibitor III'],\n 2: ['CHIR-99021', 'Crizotinib', 'Oprozomib (ONX 0912)', 'Penfluridol',  'R428']}\nfor fold_id in [0,1,2]:\n    mask_va = x_train['cell_type'].isin(['Myeloid cells', 'B cells']) & x_train['sm_name'].isin(fold_to_compounds[fold_id])\n    mask_tr = ~mask_va    \n    IX_train = np.where( mask_tr > 0 )[0]\n    IX_test = np.where( mask_va > 0 )[0]\n    #print(fold_id,  len(IX_test), type(IX_test), IX_test[:3], len(IX_train), type(IX_train), IX_train[:3]  )\n    folds_index_data_MT.append( [IX_train, IX_test ])    \n    \n    \ndict_folds_Antonina = {0: [4, 15, 18, 41, 47, 53, 56, 57, 59, 61, 62, 66, 78, 83, 86, 93, 100, 110, 115, 116, 117, 120, 123, 131, 148, 152, 161, 185, 189, 193, 197, 198, 205, 207, 209, 210, 213, 218, 219, 225, 228, 230, 250, 252, 257, 263, 286, 292, 294, 304, 306, 310, 325, 327, 336, 338, 339, 350, 351, 356, 357, 360, 362, 369, 379, 385, 395, 419, 424, 430, 432, 434, 438, 441, 443, 445, 448, 452, 456, 465, 467, 471, 472, 474, 498, 508, 523, 534, 542, 544, 574, 581, 584, 589, 601, 607], 1: [1, 3, 5, 14, 24, 38, 40, 49, 50, 54, 55, 60, 79, 82, 87, 102, 112, 114, 118, 119, 127, 132, 146, 149, 166, 179, 183, 184, 186, 190, 192, 201, 203, 204, 216, 227, 242, 253, 261, 264, 271, 289, 295, 296, 299, 300, 309, 323, 324, 329, 333, 337, 354, 387, 390, 393, 406, 408, 415, 416, 420, 421, 422, 436, 442, 449, 453, 458, 462, 463, 469, 484, 487, 489, 490, 492, 497, 501, 505, 506, 509, 513, 515, 516, 546, 548, 566, 569, 585, 587, 588, 592, 593, 595, 600, 605], 2: [0, 16, 19, 44, 48, 51, 52, 58, 63, 65, 70, 81, 85, 88, 89, 92, 111, 113, 122, 125, 135, 137, 139, 150, 164, 175, 191, 202, 206, 211, 217, 220, 221, 229, 239, 243, 247, 248, 251, 258, 259, 260, 267, 270, 274, 281, 288, 290, 291, 303, 322, 326, 328, 332, 347, 359, 364, 370, 383, 384, 389, 392, 401, 423, 428, 454, 455, 461, 464, 494, 496, 504, 510, 511, 524, 525, 526, 530, 540, 541, 551, 565, 568, 570, 572, 573, 575, 577, 579, 580, 597, 598, 603, 606, 610, 611, 613], 3: [21, 22, 26, 36, 42, 45, 46, 64, 67, 69, 71, 80, 84, 90, 101, 103, 134, 136, 140, 147, 151, 160, 163, 165, 167, 174, 176, 177, 180, 181, 182, 187, 188, 195, 196, 199, 200, 212, 215, 226, 231, 240, 241, 249, 255, 256, 262, 265, 269, 285, 287, 297, 298, 307, 312, 319, 321, 330, 340, 342, 344, 345, 348, 358, 366, 368, 382, 391, 400, 403, 404, 405, 425, 427, 431, 433, 435, 444, 450, 451, 457, 459, 460, 473, 485, 499, 500, 502, 503, 507, 512, 514, 527, 528, 529, 531, 543, 547, 553, 554, 564, 571, 576, 583, 586, 591, 596, 602, 604, 608, 609], 4: [2, 6, 7, 17, 20, 23, 25, 27, 28, 29, 37, 39, 43, 68, 91, 121, 124, 126, 128, 129, 130, 133, 138, 153, 162, 178, 194, 208, 214, 222, 223, 224, 232, 244, 245, 246, 254, 266, 268, 272, 273, 282, 283, 284, 293, 301, 302, 305, 308, 311, 320, 331, 334, 335, 341, 343, 346, 349, 352, 353, 355, 361, 363, 365, 367, 377, 378, 380, 381, 386, 388, 394, 396, 397, 398, 399, 402, 407, 417, 418, 426, 429, 437, 439, 440, 446, 447, 466, 468, 470, 481, 482, 483, 486, 488, 491, 493, 495, 532, 533, 545, 549, 550, 552, 555, 562, 563, 567, 578, 582, 590, 594, 599, 612], 5: [8, 9, 10, 11, 12, 13, 30, 31, 32, 33, 34, 35, 72, 73, 74, 75, 76, 77, 94, 95, 96, 97, 98, 99, 104, 105, 106, 107, 108, 109, 141, 142, 143, 144, 145, 154, 155, 156, 157, 158, 159, 168, 169, 170, 171, 172, 173, 233, 234, 235, 236, 237, 238, 275, 276, 277, 278, 279, 280, 313, 314, 315, 316, 317, 318, 371, 372, 373, 374, 375, 376, 409, 410, 411, 412, 413, 414, 475, 476, 477, 478, 479, 480, 517, 518, 519, 520, 521, 522, 535, 536, 537, 538, 539, 556, 557, 558, 559, 560, 561]}\n\nfolds_index_data_Antonina = []    \nfor i in range(5):  \n    l_valid = np.array( dict_folds_Antonina[i] )\n    l_train = np.array( [k for k in range(614) if k not in l_valid ] )\n    folds_index_data_Antonina.append( [l_train, l_valid ])    \nprint( len(folds_index_data_Antonina ) )","metadata":{"execution":{"iopub.status.busy":"2023-11-24T15:41:05.825762Z","iopub.execute_input":"2023-11-24T15:41:05.826385Z","iopub.status.idle":"2023-11-24T15:41:05.90625Z","shell.execute_reply.started":"2023-11-24T15:41:05.826353Z","shell.execute_reply":"2023-11-24T15:41:05.904583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random \nfrom sklearn.model_selection import KFold\n\nclass KFold_custom:\n    '''\n    Class with similar to sklearn \"KFold\" class, interface.\n    Support of CV schemes proposed by AmbrosM, MT, Kishan and standard sklearn KFold\n    Supports options to return IX_train,IX_valid,IX_test - triplet - if valid_size > 0\n    Examples:\n    kf = KFold_custom('AmbrosM'):     \n    for i_fold,(IX_train,IX_test)   in enumerate( kf.split()) :\n        pass\n    kf = KFold_custom('AmbrosM', valid_size = 0.1 )     \n    for i_fold,(IX_train ,IX_valid,IX_test)   in enumerate( kf.split()) :\n        pass\n        \n    kf = KFold_custom(CV_scheme = 'Random') # wrapper for sklearn Kfolds with shuffle = True\n    kf = KFold_custom(CV_scheme = CV_scheme, random_state = 42 ) # fixing random_state ensures reprodicibility of folds \n    '''\n    def __init__(self, CV_scheme,  valid_size = 0, n_splits=5, random_state = 42, verbose = 0 ):\n        self.CV_scheme = CV_scheme\n        self.CV_scheme_inf =  CV_scheme\n        self.valid_size = np.clip(valid_size,0,1)\n        self.random_state = random_state\n\n        self.folds_index_data = [ [np.arange(0), np.arange(0), np.arange(0)] ] # Example data\n        self.n_splits = 1 # Example data\n        if CV_scheme == 'Kishan1':\n            # One fold scheme which contains train, valid, test parts \n            IX_train_Kishan1 = np.arange(429)\n            IX_val_Kishan1 = np.arange(429,521)\n            IX_test_Kishan1 = np.arange(521,614)\n            self.n_splits = 1\n            self.folds_index_data =  [   [IX_train_Kishan1, IX_val_Kishan1, IX_test_Kishan1]  ]\n        elif CV_scheme == 'Tonya':\n            self.n_splits = 5\n            self.folds_index_data = folds_index_data_Antonina\n        elif CV_scheme == 'AmbrosM':\n            self.n_splits = 4\n            self.folds_index_data = folds_index_data_AmbrosM\n        elif CV_scheme == 'MT':\n            self.n_splits = 3\n            self.folds_index_data = folds_index_data_MT\n        elif 'Random'.lower() in CV_scheme.lower():\n            self.n_splits = n_splits\n            self.CV_scheme_inf = 'Random_'+str(n_splits) +'_'+str(random_state )\n            kf = KFold(n_splits=n_splits, random_state = random_state, shuffle=True )#,\n            self.folds_index_data = list( kf.split( np.arange(614) ) )\n        elif 'Full'.lower() in CV_scheme.lower():\n            # Just return the full set as both train and test - it useful to re-train the model on the entire data\n            IX_train_full = np.arange(614)\n            IX_test_full = np.arange(614)\n            self.n_splits = 1\n            self.folds_index_data = [   [IX_train_full, IX_test_full]  ]\n        else:\n            s = 'Uncrecognized CV_scheme ' + str(CV_scheme) \n            raise ValueError(s)\n\n        self.verbose = verbose \n        if verbose >= 10:\n            print(self.CV_scheme, self.n_splits, len(self.folds_index_data) )\n            #print(self.folds_index_data)\n            \n    def split(self, X=None):\n        '''\n        X - NOT used, just for compatibility with sklearn \n        '''\n        for item in self.folds_index_data:\n            if self.CV_scheme in ['Kishan1']:\n                if self.valid_size > 0:\n                    yield item[0],item[1],item[2]\n                else :\n                    yield np.array( list(set(item[0])|set(item[1]))  ) , item[2] # Return train, test only \n            elif self.CV_scheme in ['Tonya', 'AmbrosM','MT', 'Random', 'Full']:\n                if self.valid_size == 0:\n                    yield item[0],item[1]\n                else:\n                    # Split \"full-train\"->( real-train, valid ) \n                    index_for_real_train_part = int(  (1-self.valid_size) * len(item[0]) )\n                    if self.random_state is None:\n                        IX_train =  item[0][:index_for_real_train_part]\n                        IX_valid =  item[0][index_for_real_train_part:]\n                    elif self.random_state == -1:\n                        p = np.random.permutation(len(item[0])) \n                        IX_train =  item[0][p][:index_for_real_train_part]\n                        IX_valid =  item[0][p][index_for_real_train_part:]\n                    else:\n                        np.random.seed(self.random_state) # Set temporary random seed\n                        p = np.random.permutation(len(item[0])) \n                        np.random.seed(random.randint(0,30000)) # Randomize seed again - use Python random, not numpy       \n                        IX_train =  item[0][p][:index_for_real_train_part]\n                        IX_valid =  item[0][p][index_for_real_train_part:]\n                    yield IX_train, IX_valid, item[1]\n                    \n        \n    def get_list_main_CV_schemes(self):\n        return ['AmbrosM','MT','Kishan1','Random']\n    def get_n_splits(self, X=np.array([])):\n        return self.n_splits","metadata":{"execution":{"iopub.status.busy":"2023-11-24T15:41:05.907954Z","iopub.execute_input":"2023-11-24T15:41:05.908361Z","iopub.status.idle":"2023-11-24T15:41:05.936017Z","shell.execute_reply.started":"2023-11-24T15:41:05.908329Z","shell.execute_reply":"2023-11-24T15:41:05.93437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.layers.experimental.preprocessing import TextVectorization\n\n# Create char-level token vectorizer instance\nNUM_CHAR_TOKENS = 35\noutput_seq_char_len = 100\nsmiles_voc_size = 34\nchar_vectorizer_smiles = TextVectorization(max_tokens=NUM_CHAR_TOKENS,\n                                    output_sequence_length=output_seq_char_len,\n                                    standardize=None,\n                                    split=\"character\",\n                                    name=\"char_vectorizer_SMILES\")\n\n# # Adapt character vectorizer to training characters\nchar_vectorizer_smiles.adapt(x_train[\"SMILES\"].values)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T15:41:05.937542Z","iopub.execute_input":"2023-11-24T15:41:05.937937Z","iopub.status.idle":"2023-11-24T15:41:06.521754Z","shell.execute_reply.started":"2023-11-24T15:41:05.937907Z","shell.execute_reply":"2023-11-24T15:41:06.520227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"char_vectorizer_smiles.get_config()","metadata":{"execution":{"iopub.status.busy":"2023-11-24T15:41:06.523593Z","iopub.execute_input":"2023-11-24T15:41:06.525147Z","iopub.status.idle":"2023-11-24T15:41:06.536927Z","shell.execute_reply.started":"2023-11-24T15:41:06.525093Z","shell.execute_reply":"2023-11-24T15:41:06.535074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"strat_gr = pd.get_dummies(x_train[\"cell_type\"])\n\nCV_scheme = \"Multilabel CT Stratified\"\nprint(CV_scheme)\nkf = MultilabelStratifiedKFold(n_splits=8, random_state=42, shuffle=True)\n\nfolds_data = []\n\n# Iterate through each fold\nfor fold, (trn_ind, val_ind) in enumerate(tqdm(kf.split(x_train, strat_gr))):\n    torch.cuda.empty_cache()\n    gc.collect()\n    \n    print(f'Fold {fold}')\n    \n    # get train and validation data\n    x_tr = x_train.iloc[trn_ind].reset_index(drop=True)\n    x_val = x_train.iloc[val_ind].reset_index(drop=True)\n    orig_y_tr = y_train[trn_ind]\n    orig_y_val = y_train[val_ind]\n    x_te = x_test.copy()\n    \n    # use standard scaler\n    labels_scaler = StandardScaler()\n\n    y_tr = labels_scaler.fit_transform(orig_y_tr)\n    y_val = labels_scaler.transform(orig_y_val)\n    \n    # tfidf\n    tr_smiles = x_tr[\"SMILES\"].apply(lambda x: \" \".join(list(x))).values\n    val_smiles = x_val[\"SMILES\"].apply(lambda x: \" \".join(list(x))).values\n    te_smiles = x_test[\"SMILES\"].apply(lambda x: \" \".join(list(x))).values\n    \n    # char vectorization\n    char_vectorizer_smiles = TextVectorization(max_tokens=NUM_CHAR_TOKENS,\n                                    output_sequence_length=output_seq_char_len,\n                                    standardize=None,\n                                    split=\"character\",\n                                    name=\"char_vectorizer_SMILES\")\n    char_vectorizer_smiles.adapt(tr_smiles)\n    \n    tr_smiles_char_vec_data = char_vectorizer_smiles(tr_smiles)\n    val_smiles_char_vec_data = char_vectorizer_smiles(val_smiles)\n    te_smiles_char_vec_data = char_vectorizer_smiles(te_smiles)\n    \n    # drop columns\n    x_tr.drop([\"SMILES\", \"sm_lincs_id\"], axis=1, inplace=True)\n    x_val.drop([\"SMILES\", \"sm_lincs_id\"], axis=1, inplace=True)\n    x_te.drop([\"SMILES\", \"sm_lincs_id\"], axis=1, inplace=True)\n    \n    # save\n    x_tr = np.concatenate([\n        np.array(tr_smiles_char_vec_data),\n        x_tr.values,\n    ], axis=1)\n    x_val = np.concatenate([\n        np.array(val_smiles_char_vec_data),\n        x_val.values,\n    ], axis=1)\n    x_te = np.concatenate([\n        np.array(te_smiles_char_vec_data),\n        x_te.values,\n    ], axis=1)\n    \n    folds_data.append({\n        \"x_tr\": x_tr, \n        \"x_val\": x_val, \n        \"x_te\": x_te,\n        \"orig_y_tr\": orig_y_tr, \n        \"orig_y_val\": orig_y_val, \n        \"y_tr\": y_tr, \n        \"y_val\": y_val,\n        \"trn_ind\": trn_ind,\n        \"val_ind\": val_ind,\n        \"scaler\": labels_scaler\n    })","metadata":{"execution":{"iopub.status.busy":"2023-11-24T15:43:21.492399Z","iopub.execute_input":"2023-11-24T15:43:21.492957Z","iopub.status.idle":"2023-11-24T15:43:31.917284Z","shell.execute_reply.started":"2023-11-24T15:43:21.492914Z","shell.execute_reply":"2023-11-24T15:43:31.915145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_type_voc_size = len(le_cell_type.classes_)\nsm_name_voc_size = len(le_sm_name.classes_)\n\ncell_type_voc_size, sm_name_voc_size","metadata":{"execution":{"iopub.status.busy":"2023-11-24T15:44:05.476785Z","iopub.execute_input":"2023-11-24T15:44:05.477201Z","iopub.status.idle":"2023-11-24T15:44:05.487158Z","shell.execute_reply.started":"2023-11-24T15:44:05.477171Z","shell.execute_reply":"2023-11-24T15:44:05.485561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Make Pytorch Dataset","metadata":{}},{"cell_type":"code","source":"class DataRetriever(Dataset):\n    \"\"\"\n    PyTorch dataset for DataLoader.\n    \n    Args:\n        x_data (List[Any]): List of input data samples (e.g., protein sequences).\n        labels (List[Any], optional): List of labels (e.g., binary or multilabel targets). Default is None.\n        mode (str): One of ('train', 'val', 'test') indicating the mode of the dataset. Default is 'train'.\n\n    Attributes:\n        x_data (List[Any]): List of input data samples (e.g., protein sequences).\n        labels (List[Any], optional): List of labels (e.g., binary or multilabel targets). Default is None.\n        mode (str): Mode of the dataset, one of ('train', 'val', 'test').\n        len_ (int): Length of the dataset.\n\n    Methods:\n        __len__(): Return the length of the dataset.\n        __getitem__(index): Retrieve a data sample and its label (if available) by index.\n    \"\"\"\n\n    def __init__(self, x_data: List[Any], labels: List[Any] = None, mode: str = 'train'):\n        \"\"\"\n        Initialize DataRetriever with input data, labels (if available), and dataset mode.\n\n        Args:\n            x_data (List[Any]): List of input data samples (e.g., protein sequences).\n            labels (List[Any], optional): List of labels (e.g., binary or multilabel targets). Default is None.\n            mode (str): One of ('train', 'val', 'test') indicating the mode of the dataset. Default is 'train'.\n\n        Raises:\n            NameError: If the provided mode is not one of ('train', 'val', 'test').\n        \"\"\"\n        super(DataRetriever, self).__init__()\n\n        self.x_data = x_data\n        self.labels = labels\n        self.mode = mode\n\n        DATA_MODES = ('train', 'val', 'test')\n        if self.mode not in DATA_MODES:\n            raise NameError(f\"{self.mode} is not correct; correct modes: {DATA_MODES}\")\n\n        self.len_ = len(self.x_data)\n\n    def __len__(self):\n        \"\"\"\n        Get the length of the dataset.\n\n        Returns:\n            int: Length of the dataset.\n        \"\"\"\n        return self.len_\n\n    def __getitem__(self, index: int) -> Tuple[Any, Any]:\n        \"\"\"\n        Retrieve a data sample and its label (if available) by index.\n\n        Args:\n            index (int): Index of the data sample to retrieve.\n\n        Returns:\n            Tuple[Any, Any]: A tuple containing the data sample (x) and its label (y) if available.\n                If in 'test' mode, returns only the data sample (x).\n\n        Note:\n            If the dataset is in 'test' mode, the returned tuple contains only the data sample (x).\n            Otherwise, it contains both the data sample (x) and its corresponding label (y).\n        \"\"\"\n        x = self.x_data[index]\n\n        if self.mode == 'test':\n            return x\n        else:\n            y = self.labels[index]\n            return x, y","metadata":{"execution":{"iopub.status.busy":"2023-11-24T15:44:09.531169Z","iopub.execute_input":"2023-11-24T15:44:09.531734Z","iopub.status.idle":"2023-11-24T15:44:09.545781Z","shell.execute_reply.started":"2023-11-24T15:44:09.531693Z","shell.execute_reply":"2023-11-24T15:44:09.544479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Utils","metadata":{}},{"cell_type":"markdown","source":"## Training Loops","metadata":{}},{"cell_type":"code","source":"def train_epoch(model, dataloader, optimizer, metric, criterion, device):\n    \"\"\"\n    Perform one epoch of training for the model.\n\n    Args:\n        model (nn.Module): The PyTorch model to train.\n        dataloader (DataLoader): DataLoader for the training data.\n        optimizer (torch.optim.Optimizer): Optimizer for updating model parameters.\n        criterion (nn.Module): Loss function used for training.\n        device (torch.device): Device on which the data and model reside.\n\n    Returns:\n        float: Average training loss for the epoch.\n    \"\"\"\n    model.train()\n    epoch_loss = 0\n    epoch_metric = 0\n    num_batches = len(dataloader)\n\n    for i, data in enumerate(dataloader, 0):\n        inputs, targets = data\n        inputs = inputs.to(device).float()\n        targets = targets.to(device).float()\n\n        optimizer.zero_grad()\n        outputs_class = model(inputs)\n\n        loss = criterion(outputs_class, targets)\n        loss.backward()\n\n        optimizer.step()\n        epoch_loss += loss.item()\n        epoch_metric += metric(targets.cpu().numpy(), outputs_class.detach().cpu().numpy())\n\n    return epoch_loss / num_batches, epoch_metric / num_batches\n\ndef eval_epoch(model, dataloader, metric, criterion, device):\n    \"\"\"\n    Perform one epoch of evaluation for the model.\n\n    Args:\n        model (nn.Module): The PyTorch model to evaluate.\n        dataloader (DataLoader): DataLoader for the validation data.\n        criterion (nn.Module): Loss function used for evaluation.\n        device (torch.device): Device on which the data and model reside.\n\n    Returns:\n        float: Average evaluation loss for the epoch.\n        float: Average macro F1 score for the epoch.\n    \"\"\"\n    model.eval()\n    epoch_loss = 0\n    epoch_metric = 0\n    num_batches = len(dataloader)\n\n    with torch.no_grad():\n        for i, data in enumerate(dataloader, 0):\n            inputs, targets = data\n            inputs = inputs.to(device).float()\n            targets = targets.to(device).float()\n\n            outputs_class = model(inputs)\n\n            loss = criterion(outputs_class, targets)\n\n            epoch_loss += loss.item()\n            epoch_metric += metric(targets.cpu().numpy(), outputs_class.cpu().numpy())\n\n    return epoch_loss / num_batches, epoch_metric / num_batches","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-11-24T15:44:12.282645Z","iopub.execute_input":"2023-11-24T15:44:12.283943Z","iopub.status.idle":"2023-11-24T15:44:12.298626Z","shell.execute_reply.started":"2023-11-24T15:44:12.283903Z","shell.execute_reply":"2023-11-24T15:44:12.296947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def sigmoid_numpy(value):\n    \"\"\"\n    Apply sigmoid function element-wise to a numpy array.\n\n    Args:\n        value (np.ndarray): Input array.\n\n    Returns:\n        np.ndarray: Output array after applying sigmoid function.\n    \"\"\"\n    return 1/(1 + np.exp(-value))","metadata":{"execution":{"iopub.status.busy":"2023-11-24T15:44:14.161167Z","iopub.execute_input":"2023-11-24T15:44:14.161699Z","iopub.status.idle":"2023-11-24T15:44:14.168554Z","shell.execute_reply.started":"2023-11-24T15:44:14.161663Z","shell.execute_reply":"2023-11-24T15:44:14.166916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_training_history(history, metrics):\n    \"\"\"\n    Plot training history curves for loss and evaluation metrics on the same line.\n\n    Args:\n        history (keras.callbacks.History): Training history object.\n        metrics (list): List of metric names to plot.\n\n    Returns:\n        None\n    \"\"\"\n    loss = history.history['loss']\n    val_loss = history.history['val_loss']\n\n    epochs = range(len(loss))\n\n    plt.figure(figsize=(12, 6))\n\n    # Plot loss\n    plt.subplot(1, 2, 1)\n    plt.plot(epochs, loss, label='Training Loss', color=\"blue\")\n    plt.plot(epochs, val_loss, label='Validation Loss', color=\"red\")\n    plt.title('Loss')\n    plt.xlabel('Epochs')\n    plt.legend()\n\n    # Plot specified evaluation metrics on the same line\n    for metric in metrics:\n        train_metric_name = f'Training {metric.capitalize()}'\n        val_metric_name = f'Validation {metric.capitalize()}'\n        train_metric = history.history[metric]\n        val_metric = history.history['val_' + metric]\n\n        plt.subplot(1, 2, 2)\n        plt.plot(epochs, train_metric, label=train_metric_name, color=\"green\")\n        plt.plot(epochs, val_metric, label=val_metric_name, color=\"orange\")\n\n    plt.title('Metrics')\n    plt.xlabel('Epochs')\n    plt.legend(loc='upper right')\n\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-24T15:44:14.625765Z","iopub.execute_input":"2023-11-24T15:44:14.626288Z","iopub.status.idle":"2023-11-24T15:44:14.638654Z","shell.execute_reply.started":"2023-11-24T15:44:14.62625Z","shell.execute_reply":"2023-11-24T15:44:14.637387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Metrics","metadata":{}},{"cell_type":"code","source":"import numpy as np\n\ndef mrrmse(y_true: np.ndarray, y_pred: np.ndarray) -> float:\n    \"\"\"\n    Calculate Mean Rowwise Root Mean Squared Error.\n\n    Parameters:\n    - y_true (np.ndarray): The ground truth values.\n    - y_pred (np.ndarray): The predicted values.\n\n    Returns:\n    - float: Mean Rowwise Root Mean Squared Error.\n    \"\"\"\n    colwise_mse = np.mean((y_true - y_pred)**2, axis=1)\n    mean_rowwise_root_mse = np.mean(np.sqrt(colwise_mse))\n\n    return mean_rowwise_root_mse","metadata":{"execution":{"iopub.status.busy":"2023-11-24T15:44:15.853575Z","iopub.execute_input":"2023-11-24T15:44:15.855323Z","iopub.status.idle":"2023-11-24T15:44:15.862899Z","shell.execute_reply.started":"2023-11-24T15:44:15.855273Z","shell.execute_reply":"2023-11-24T15:44:15.861439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Losses","metadata":{}},{"cell_type":"code","source":"class MeanRowwiseRMSELoss(tf.keras.losses.Loss):\n    def __init__(self):\n        super().__init__()\n    def call(self, y_true, y_pred):\n        \"\"\"\n        Calculate Mean Rowwise Root Mean Squared Error.\n\n        Parameters:\n        - y_true (tf.Tensor): The ground truth values.\n        - y_pred (tf.Tensor): The predicted values.\n\n        Returns:\n        - tf.Tensor: Mean Rowwise Root Mean Squared Error.\n        \"\"\"\n        colwise_mse = tf.reduce_mean(tf.square(y_true - y_pred), axis=1)\n        mean_rowwise_root_mse = tf.reduce_mean(tf.sqrt(colwise_mse))\n\n        return mean_rowwise_root_mse","metadata":{"execution":{"iopub.status.busy":"2023-11-24T15:44:17.536545Z","iopub.execute_input":"2023-11-24T15:44:17.537051Z","iopub.status.idle":"2023-11-24T15:44:17.54456Z","shell.execute_reply.started":"2023-11-24T15:44:17.537014Z","shell.execute_reply":"2023-11-24T15:44:17.543352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class MRRMSELoss(nn.Module):\n    def __init__(self):\n        super().__init__()\n    def forward(self, y_true: torch.Tensor, y_pred: torch.Tensor) -> torch.Tensor:\n        \"\"\"\n        Forward pass to calculate Mean Rowwise Root Mean Squared Error.\n\n        Parameters:\n        - y_true (torch.Tensor): The ground truth values.\n        - y_pred (torch.Tensor): The predicted values.\n\n        Returns:\n        - torch.Tensor: Mean Rowwise Root Mean Squared Error.\n        \"\"\"\n        colwise_mse = torch.mean((y_true - y_pred)**2, dim=1)\n        mean_rowwise_root_mse = torch.mean(torch.sqrt(colwise_mse))\n\n        return mean_rowwise_root_mse","metadata":{"execution":{"iopub.status.busy":"2023-11-24T15:44:18.567779Z","iopub.execute_input":"2023-11-24T15:44:18.568534Z","iopub.status.idle":"2023-11-24T15:44:18.576431Z","shell.execute_reply.started":"2023-11-24T15:44:18.568497Z","shell.execute_reply":"2023-11-24T15:44:18.575071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow.keras.backend as K\n\ndef negative_correlation_loss(y_true, y_pred):\n    \"\"\"Negative correlation loss function for Keras\n    \n    Precondition:\n    y_true.mean(axis=1) == 0\n    y_true.std(axis=1) == 1\n    \n    Returns:\n    -1 = perfect positive correlation\n    1 = totally negative correlation\n    \"\"\"\n    my = K.mean(tf.convert_to_tensor(y_pred), axis=1)\n    my = tf.tile(tf.expand_dims(my, axis=1), (1, y_true.shape[1]))\n    ym = y_pred - my\n    r_num = K.sum(tf.multiply(y_true, ym), axis=1)\n    r_den = tf.sqrt(K.sum(K.square(ym), axis=1) * float(y_true.shape[-1]))\n    r = tf.reduce_mean(r_num / r_den)\n    return - r","metadata":{"execution":{"iopub.status.busy":"2023-11-24T15:44:19.649216Z","iopub.execute_input":"2023-11-24T15:44:19.649741Z","iopub.status.idle":"2023-11-24T15:44:19.660116Z","shell.execute_reply.started":"2023-11-24T15:44:19.649703Z","shell.execute_reply":"2023-11-24T15:44:19.658276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"markdown","source":"## Keras","metadata":{}},{"cell_type":"code","source":"def create_model(input_len, output_len, cell_type_voc_size, sm_name_voc_size):\n    \"\"\"\n    Create a Keras model with embedding layers for categorical features.\n\n    Parameters:\n    - input_len: Length of the input data (including the categorical features).\n    - output_len: Length of the output data (labels).\n    - cell_type_voc_size: Vocabulary size for the 'cell_type' categorical feature.\n    - sm_name_voc_size: Vocabulary size for the 'sm_name' categorical feature.\n\n    Returns:\n    - model: The Keras model.\n    \"\"\"\n    # Define input layer\n    input_layer = Input(shape=(input_len,))  # Assuming the first two columns are categorical\n\n    # Separate categorical and numerical features\n    smiles_input = input_layer[:, 0:output_seq_char_len]\n    cell_type_input = input_layer[:, output_seq_char_len:output_seq_char_len+1]\n    sm_name_input = input_layer[:, output_seq_char_len+1:output_seq_char_len+2]\n\n    cell_type_emb = Embedding(\n        input_dim=cell_type_voc_size,\n        output_dim=8,\n        input_length=1,\n    )(cell_type_input)\n\n    sm_name_emb = Embedding(\n        input_dim=sm_name_voc_size,\n        output_dim=8,\n        input_length=1,\n    )(sm_name_input)\n    \n    smiles_emb = Embedding(\n        input_dim=NUM_CHAR_TOKENS,\n        output_dim=16,\n        input_length=1,\n    )(smiles_input)\n\n    # Flatten the embeddings\n    cell_type_flat = Flatten()(cell_type_emb)\n    cell_type_flat = Dense(128, activation=\"elu\")(cell_type_flat)\n    \n    sm_name_flat = Flatten()(sm_name_emb)\n    sm_name_flat = Dense(128, activation=\"elu\")(sm_name_flat)\n    \n    smiles_flat = Flatten()(smiles_emb)\n    smiles_flat = Dense(256, activation=\"elu\")(smiles_flat)\n\n    # Concatenate embeddings\n    concatenated_features = Concatenate()([cell_type_flat, \n                                           sm_name_flat, \n                                           smiles_flat])\n    concatenated_features = GaussianNoise(0.09)(concatenated_features)\n\n    # Dense layers\n    x = Dense(1024, activation=\"elu\")(concatenated_features)\n    x = BatchNormalization()(x)\n    x = GaussianDropout(0.4)(x)\n    x = Dense(512, activation=\"elu\")(x)\n    x = Dropout(0.2)(x)\n    x = Dense(256, activation=\"elu\")(x)\n    x = Dropout(0.2)(x)\n    x = Dense(128, activation=\"elu\")(x)\n    x = Dropout(0.15)(x)\n    x = Dense(64, activation=\"elu\")(x)\n    out = Dense(output_len, activation=\"linear\")(x)  # Output layer for numerical features\n\n    # Define the model\n    model = tf.keras.models.Model(inputs=input_layer, outputs=out)\n\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-11-24T15:44:22.749446Z","iopub.execute_input":"2023-11-24T15:44:22.750983Z","iopub.status.idle":"2023-11-24T15:44:22.765661Z","shell.execute_reply.started":"2023-11-24T15:44:22.750937Z","shell.execute_reply":"2023-11-24T15:44:22.764255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_epochs = 200\nbatch_size = 128\nlr = 0.000064\n\nmodels = []\nkeras_oof = np.zeros(y_train.shape)\npreds = []\n\nkeras_scores = dict(\n    fold = [],\n    corr_rows = [],\n    corr_cols = [],\n    MAE = [],\n    MRRMSE = [],\n    R2 = []\n)\n\n# Iterate through each fold\nfor fold, data in enumerate(folds_data):\n    print(f'Training fold {fold}')\n    \n    x_tr = data[\"x_tr\"]\n    x_val = data[\"x_val\"]\n    x_te = data[\"x_te\"]\n    y_tr = data[\"y_tr\"]\n    y_val = data[\"y_val\"]\n    val_ind = data[\"val_ind\"]\n    labels_scaler = data[\"scaler\"]\n    \n    \n    n_repeats = 5\n    x_tr = np.repeat(x_tr, n_repeats, 0)\n    y_tr = np.repeat(y_tr, n_repeats, 0)\n    \n    input_len = x_tr.shape[-1]\n    output_len = y_tr.shape[-1]\n    \n    model = create_model(input_len, output_len, cell_type_voc_size, sm_name_voc_size)\n    \n    optim = tf.keras.optimizers.Lion(learning_rate=lr, weight_decay=0.01)\n    model.compile(\n        loss=\"mae\",\n        optimizer=optim,\n    )\n    \n    gc.collect()\n\n    filepath = f'best_nn_fold{fold}.h5'\n#     es = tf.keras.callbacks.EarlyStopping(patience=9000, mode='min', verbose=1) \n    checkpoint = tf.keras.callbacks.ModelCheckpoint(monitor='val_loss', filepath=filepath, save_best_only=True,save_weights_only=True,mode='min') \n    reduce_lr_loss = tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.9, patience=20, verbose=1)\n\n\n    history = model.fit(x_tr, \n              y_tr, \n              epochs=num_epochs, \n              batch_size=batch_size, \n              validation_data=(x_val, y_val), \n              callbacks=[checkpoint, reduce_lr_loss], \n              workers=2, \n              verbose=False)\n    \n    plot_training_history(history, [])\n\n    model.load_weights(filepath)\n    models.append(model)\n    \n    val_pred = model.predict(x_val, batch_size=512)\n    val_pred = labels_scaler.inverse_transform(val_pred)\n    keras_oof[val_ind] = val_pred\n    \n    pred = model.predict(x_te, batch_size=512)\n    pred = labels_scaler.inverse_transform(pred)\n    preds.append(pred)\n    \n    # calculate metrics with original labels\n    y_val = data[\"orig_y_val\"]\n    # Save metrics\n    list_corr_rows_fold = [np.corrcoef(y_val[i,:], val_pred[i,:])[0,1] for i in range(val_pred.shape[0])]\n    list_corr_columns_fold = [np.corrcoef(y_val[:,i], val_pred[:,i])[0,1] for i in range(val_pred.shape[1])]\n\n    keras_scores[\"fold\"].append(fold)\n    keras_scores[\"corr_rows\"].append(np.nanmean(list_corr_rows_fold))\n    keras_scores[\"corr_cols\"].append(np.nanmean(list_corr_columns_fold))\n    keras_scores[\"MAE\"].append(\n        mean_absolute_error(\n            y_val,\n            val_pred\n        )\n    )\n    keras_scores[\"MRRMSE\"].append(\n        mrrmse(\n            y_val,\n            val_pred\n        )\n    )\n    keras_scores[\"R2\"].append(\n            r2_score(\n                y_val,\n                val_pred\n            )\n        )","metadata":{"execution":{"iopub.status.busy":"2023-11-24T10:04:13.839765Z","iopub.execute_input":"2023-11-24T10:04:13.840156Z","iopub.status.idle":"2023-11-24T10:07:50.185719Z","shell.execute_reply.started":"2023-11-24T10:04:13.840122Z","shell.execute_reply":"2023-11-24T10:07:50.184418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save metrics\nlist_corr_rows_fold = [np.corrcoef(y_train[i,:], keras_oof[i,:])[0,1] for i in range(keras_oof.shape[0])]\nlist_corr_columns_fold = [np.corrcoef(y_train[:,i], keras_oof[:,i])[0,1] for i in range(keras_oof.shape[1])]\n\nkeras_scores[\"fold\"].append(\"all\")\nkeras_scores[\"corr_rows\"].append(np.nanmean(list_corr_rows_fold))\nkeras_scores[\"corr_cols\"].append(np.nanmean(list_corr_columns_fold))\nkeras_scores[\"MAE\"].append(\n    mean_absolute_error(\n        y_train,\n        keras_oof\n    )\n)\nkeras_scores[\"MRRMSE\"].append(\n    mrrmse(\n        y_train,\n        keras_oof\n    )\n)\nkeras_scores[\"R2\"].append(\n            r2_score(\n                y_train,\n                keras_oof\n            )\n        )\n\ndisplay(pd.DataFrame(keras_scores))","metadata":{"execution":{"iopub.status.busy":"2023-11-24T10:07:50.187615Z","iopub.execute_input":"2023-11-24T10:07:50.188023Z","iopub.status.idle":"2023-11-24T10:07:53.654896Z","shell.execute_reply.started":"2023-11-24T10:07:50.18799Z","shell.execute_reply":"2023-11-24T10:07:53.653726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"keras_preds = np.mean(preds, axis=0)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T10:32:29.959508Z","iopub.execute_input":"2023-11-24T10:32:29.959981Z","iopub.status.idle":"2023-11-24T10:32:30.031635Z","shell.execute_reply.started":"2023-11-24T10:32:29.959938Z","shell.execute_reply":"2023-11-24T10:32:30.030438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Pytorch","metadata":{}},{"cell_type":"code","source":"class GaussianNoise(nn.Module):\n    \"\"\"Gaussian noise regularizer.\n\n    Args:\n        sigma (float, optional): relative standard deviation used to generate the\n            noise. Relative means that it will be multiplied by the magnitude of\n            the value your are adding the noise to. This means that sigma can be\n            the same regardless of the scale of the vector.\n        is_relative_detach (bool, optional): whether to detach the variable before\n            computing the scale of the noise. If `False` then the scale of the noise\n            won't be seen as a constant but something to optimize: this will bias the\n            network to generate vectors with smaller values.\n    \"\"\"\n\n    def __init__(self, sigma=0.1, is_relative_detach=True):\n        super().__init__()\n        self.sigma = sigma\n        self.is_relative_detach = is_relative_detach\n        self.noise = torch.tensor(0.).to(device)\n\n    def forward(self, x):\n        if self.training and self.sigma != 0:\n            scale = self.sigma * x.detach() if self.is_relative_detach else self.sigma * x\n            sampled_noise = self.noise.repeat(*x.size()).normal_() * scale\n            x = x + sampled_noise\n        return x","metadata":{"execution":{"iopub.status.busy":"2023-11-24T10:21:02.582961Z","iopub.execute_input":"2023-11-24T10:21:02.583453Z","iopub.status.idle":"2023-11-24T10:21:02.595167Z","shell.execute_reply.started":"2023-11-24T10:21:02.583418Z","shell.execute_reply":"2023-11-24T10:21:02.592815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class NNModel(nn.Module):\n    \"\"\"\n    A convolutional model that encodes an input data into a lower dimensional latent space and then \n    do classification or regression. \n\n    Attributes:\n    -----------\n    encoder (nn.Sequential): a sequential neural network that performs the encoding of the input image.\n    classifier (nn.Sequential): a sequential head of neural network.\n\n    Methods:\n    --------\n    forward(x):\n        Passes the input tensor through the encoder, latent layers, and classifier networks \n        to classify input tensor.\n\n    \"\"\"\n    def __init__(self, num_classes: int, \n                 x_length: int,\n                 cell_type_voc_size: int,\n                 sm_name_voc_size: int,\n                ):\n        \"\"\"\n        Initializes the model.\n        \n        :param num_classes: number of output parameters.\n        :param x_length: number of features in input data.\n        :param cell_type_voc_size: Vocabulary size for the 'cell_type' categorical feature.\n        :param sm_name_voc_size: Vocabulary size for the 'sm_name' categorical feature.\n        \n        \"\"\"\n        super(NNModel, self).__init__()\n        self.num_classes = num_classes\n        self.x_length = x_length\n        self.cell_type_voc_size = cell_type_voc_size\n        self.sm_name_voc_size = sm_name_voc_size\n        \n        emb_size = 16\n        emb_lin_size = 128\n        \n        self.cell_type_emb = nn.Sequential(\n            nn.Embedding(\n                num_embeddings=cell_type_voc_size,\n                embedding_dim=emb_size\n            ),\n            nn.Flatten(start_dim=1),\n            nn.Linear(emb_size, emb_lin_size),\n            nn.ELU(),\n        )\n        \n        self.sm_name_emb = nn.Sequential(\n            nn.Embedding(\n                num_embeddings=sm_name_voc_size,\n                embedding_dim=4*emb_size\n            ),\n            nn.Flatten(start_dim=1),\n            nn.Linear(4*emb_size, emb_lin_size),\n            nn.ELU(),\n        )\n        \n        self.smiles_emb = nn.Sequential(\n            nn.Embedding(\n                num_embeddings=output_seq_char_len,\n                embedding_dim=emb_size,\n            ),\n            nn.Flatten(start_dim=1),\n            nn.Linear(output_seq_char_len * emb_size, 2*emb_lin_size),\n            nn.ELU(),\n        )\n        \n        self.gaus_noise = GaussianNoise(sigma=0.09)\n        \n        self.classifier = nn.Sequential(\n            nn.Linear(4 * emb_lin_size, 1024),\n            nn.SELU(),\n            nn.BatchNorm1d(1024),\n            nn.Dropout1d(0.5),\n            nn.Linear(1024, 512),\n            nn.SELU(),\n            nn.Dropout1d(0.4),\n            nn.Linear(512, 128),\n            nn.SELU(),\n            nn.Dropout1d(0.2),\n            nn.Linear(128, 32),\n            nn.SELU(),\n            nn.Linear(32, num_classes)\n        )\n\n    def forward(self, x):\n        \"\"\"\n        Passes the input tensor through the encoder and classifier networks.\n\n        :param x: The input tensor. Must be of size (batch_size, channels, length)\n\n        :return: The vector with classified data.\n        \n        \"\"\"\n        # get feature vector\n        x_smiles = x[:,:,:output_seq_char_len].long()\n        x_cell_type = x[:,:,output_seq_char_len:output_seq_char_len+1].long()\n        x_sm_name = x[:,:,output_seq_char_len+1:output_seq_char_len+2].long()\n        \n        x_cell_type = self.cell_type_emb(x_cell_type)\n        x_sm_name = self.sm_name_emb(x_sm_name)\n        x_smiles = self.smiles_emb(x_smiles)\n        \n        x = torch.cat([x_cell_type, x_sm_name, x_smiles], dim=1)\n        x = self.gaus_noise(x)\n        # classification\n        x_class = self.classifier(x)\n\n        # return\n        return x_class\n    \ndef init_weights(m):\n    \"\"\"\n    Initialize weights of a neural network module.\n\n    Args:\n        m (nn.Module): Neural network module to initialize weights for.\n\n    Note:\n        This function initializes the weights of linear layers using Xavier uniform initialization\n        and sets the bias terms to a constant value of 0.01.\n\n    Example:\n        model = MyModel()\n        model.apply(init_weights)\n    \"\"\"\n    if isinstance(m, nn.Linear):\n        torch.nn.init.xavier_uniform_(m.weight)\n        m.bias.data.fill_(0.01)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T10:21:10.718739Z","iopub.execute_input":"2023-11-24T10:21:10.719192Z","iopub.status.idle":"2023-11-24T10:21:10.739948Z","shell.execute_reply.started":"2023-11-24T10:21:10.719158Z","shell.execute_reply":"2023-11-24T10:21:10.738436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # clear cuda memory\n# torch.cuda.empty_cache()\n# gc.collect()\n\n\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\n# device = torch.device(\"cpu\")\n# criterion_class = MRRMSELoss()\ncriterion_class = nn.L1Loss()\nnum_epochs = 200\nbatch_size = 256\nlr = 0.0000096\nearly_stopping_patience = 500\n\n\n# models = []\ntorch_oof = np.zeros(y_train.shape)\npreds = []\n\ntorch_scores = dict(\n    fold = [],\n    corr_rows = [],\n    corr_cols = [],\n    MAE = [],\n    MRRMSE = [],\n    R2 = []\n)\n\n# Iterate through each fold\nfor fold, data in enumerate(folds_data):\n    print(f'Training fold {fold}')\n    \n    x_tr = data[\"x_tr\"]\n    x_val = data[\"x_val\"]\n    x_te = data[\"x_te\"]\n    y_tr = data[\"y_tr\"]\n    y_val = data[\"y_val\"]\n    val_ind = data[\"val_ind\"]\n    labels_scaler = data[\"scaler\"]\n    \n    x_tr_len = x_tr.shape[0]\n    \n    n_repeats = 7\n    x_tr = np.repeat(x_tr, n_repeats, 0)\n    y_tr = np.repeat(y_tr, n_repeats, 0)\n\n    \n    model = NNModel(num_classes=y_tr.shape[1], \n                    x_length=x_tr.shape[-1]-102,\n                    cell_type_voc_size=cell_type_voc_size,\n                    sm_name_voc_size=sm_name_voc_size)\n    model.to(device)\n    model.apply(init_weights)\n\n    optimizer = SophiaG(model.parameters(), lr, betas=(0.98, 0.99), rho = 0.005, weight_decay=1e-1)\n\n    tr_dataset = DataRetriever(np.expand_dims(x_tr, 1), y_tr, \"train\")\n    val_dataset = DataRetriever(np.expand_dims(x_val, 1), y_val, \"val\")\n\n    train_dataloader = DataLoader(tr_dataset,\n                                  batch_size=batch_size,\n                                  shuffle=True,\n                                  num_workers=2,\n                                  drop_last=False)\n    val_dataloader = DataLoader(val_dataset,\n                                 batch_size=batch_size,\n                                 shuffle=False,\n                                 num_workers=2,\n                                 drop_last=False)\n\n    best_loss = float(\"inf\")\n    best_score = float(\"-inf\")\n    best_model_state_dict = None\n    no_improvement_count = 0\n\n    # Initialize the learning rate scheduler\n    scheduler = ReduceLROnPlateau(optimizer, mode='min', factor=0.9, patience=20, verbose=True)\n\n    for epoch in range(num_epochs):\n        # Training\n        train_loss, train_score = train_epoch(model, \n                                 train_dataloader, \n                                 optimizer, \n                                 mrrmse, \n                                 criterion_class, \n                                 device)\n        logger.add_scalar('Loss/train', train_loss, epoch)\n        logger.add_scalar('Score/train', train_score, epoch)\n        \n        # Evaluation\n        val_loss, val_score = eval_epoch(model, \n                                      val_dataloader, \n                                      mrrmse, \n                                      criterion_class, \n                                      device)\n        logger.add_scalar('Loss/validation', val_loss, epoch)\n        logger.add_scalar('Score/validation', val_score, epoch)\n\n        print('Epoch [{}/{}], Train Loss: {:.4f}, Train Score: {:.4f}, Val Loss: {:.4f}, Val Score: {:.4f}'\n              .format(epoch + 1, num_epochs, train_loss, train_score, val_loss, val_score))\n\n        # Update the learning rate scheduler\n        scheduler.step(val_loss)\n        \n        if val_loss < best_loss:\n            best_loss = val_loss\n            best_model_state_dict = model.state_dict()\n            no_improvement_count = 0\n        else:\n            no_improvement_count += 1\n            if no_improvement_count >= early_stopping_patience:\n                print(f'Early stopping at epoch {epoch + 1} due to no improvement.')\n                break\n\n    if best_model_state_dict is not None:\n        best_model = NNModel(num_classes=y_tr.shape[1], \n                    x_length=x_tr.shape[-1]-102,\n                    cell_type_voc_size=cell_type_voc_size,\n                    sm_name_voc_size=sm_name_voc_size)\n        best_model.load_state_dict(best_model_state_dict)\n        best_model.eval().to(device)\n\n        # Get oof preds from the best model\n        with torch.no_grad():\n            val_pred = []\n            for i, data in enumerate(val_dataloader, 0):\n                inputs, targets = data\n                inputs = inputs.to(device).float()\n                targets = targets.to(device).float()\n\n                outputs_class = best_model(inputs)\n\n                loss = criterion_class(outputs_class, targets)\n\n                y_pred = outputs_class.cpu().numpy()\n                val_pred.extend(y_pred)\n                \n        val_pred = np.array(val_pred)\n        val_pred = labels_scaler.inverse_transform(val_pred)\n        torch_oof[val_ind] = val_pred\n        \n        te_dataset = DataRetriever(np.expand_dims(x_te, 1), mode=\"test\")\n        te_dataloader = DataLoader(te_dataset,\n                                      batch_size=2*batch_size,\n                                      shuffle=True,\n                                      num_workers=2,\n                                      drop_last=False)\n        \n        # Get preds from the best model\n        with torch.no_grad():\n            pred = []\n            for i, inputs in enumerate(te_dataloader, 0):\n                inputs = inputs.to(device).float()\n                \n                outputs_class = best_model(inputs)\n                \n                pred.extend(outputs_class.cpu().numpy())\n        \n        pred = np.array(pred)\n        pred = labels_scaler.inverse_transform(pred)\n        preds.append(pred)\n\n        # Save the best model's state dictionary\n        torch.save(best_model_state_dict, f\"pytorch_nn_fold{fold}_weights.pt\")\n        \n        # calculate metrics with original labels\n        y_val = y_train[val_ind]\n        # Save metrics\n        list_corr_rows_fold = [np.corrcoef(y_val[i,:], val_pred[i,:])[0,1] for i in range(val_pred.shape[0])]\n        list_corr_columns_fold = [np.corrcoef(y_val[:,i], val_pred[:,i])[0,1] for i in range(val_pred.shape[1])]\n        \n        torch_scores[\"fold\"].append(fold)\n        torch_scores[\"corr_rows\"].append(np.nanmean(list_corr_rows_fold))\n        torch_scores[\"corr_cols\"].append(np.nanmean(list_corr_columns_fold))\n        torch_scores[\"MAE\"].append(\n            mean_absolute_error(\n                y_val,\n                val_pred\n            )\n        )\n        torch_scores[\"MRRMSE\"].append(\n            mrrmse(\n                y_val,\n                val_pred\n            )\n        )\n        torch_scores[\"R2\"].append(\n            r2_score(\n                y_val,\n                val_pred\n            )\n        )\n\ntorch.cuda.empty_cache()\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-11-24T10:21:11.784483Z","iopub.execute_input":"2023-11-24T10:21:11.785944Z","iopub.status.idle":"2023-11-24T10:23:43.234172Z","shell.execute_reply.started":"2023-11-24T10:21:11.785888Z","shell.execute_reply":"2023-11-24T10:23:43.232712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save metrics\nlist_corr_rows_fold = [np.corrcoef(y_train[i,:], torch_oof[i,:])[0,1] for i in range(torch_oof.shape[0])]\nlist_corr_columns_fold = [np.corrcoef(y_train[:,i], torch_oof[:,i])[0,1] for i in range(torch_oof.shape[1])]\n\ntorch_scores[\"fold\"].append(\"all\")\ntorch_scores[\"corr_rows\"].append(np.nanmean(list_corr_rows_fold))\ntorch_scores[\"corr_cols\"].append(np.nanmean(list_corr_columns_fold))\ntorch_scores[\"MAE\"].append(\n    mean_absolute_error(\n        y_train,\n        torch_oof\n    )\n)\ntorch_scores[\"MRRMSE\"].append(\n    mrrmse(\n        y_train,\n        torch_oof\n    )\n)\ntorch_scores[\"R2\"].append(\n            r2_score(\n                y_train,\n                torch_oof\n            )\n        )","metadata":{"execution":{"iopub.status.busy":"2023-11-24T10:23:43.23665Z","iopub.execute_input":"2023-11-24T10:23:43.237519Z","iopub.status.idle":"2023-11-24T10:23:46.660973Z","shell.execute_reply.started":"2023-11-24T10:23:43.237479Z","shell.execute_reply":"2023-11-24T10:23:46.659314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(pd.DataFrame(torch_scores))","metadata":{"execution":{"iopub.status.busy":"2023-11-24T10:23:46.662954Z","iopub.execute_input":"2023-11-24T10:23:46.666656Z","iopub.status.idle":"2023-11-24T10:23:46.6876Z","shell.execute_reply.started":"2023-11-24T10:23:46.666588Z","shell.execute_reply":"2023-11-24T10:23:46.686189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torch_preds = np.mean(preds, axis=0)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T10:23:46.690561Z","iopub.execute_input":"2023-11-24T10:23:46.691737Z","iopub.status.idle":"2023-11-24T10:23:46.761343Z","shell.execute_reply.started":"2023-11-24T10:23:46.691688Z","shell.execute_reply":"2023-11-24T10:23:46.760036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"display(pd.DataFrame(torch_scores))","metadata":{"execution":{"iopub.status.busy":"2023-11-24T10:23:46.762849Z","iopub.execute_input":"2023-11-24T10:23:46.763219Z","iopub.status.idle":"2023-11-24T10:23:46.783562Z","shell.execute_reply.started":"2023-11-24T10:23:46.763188Z","shell.execute_reply":"2023-11-24T10:23:46.781912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(pd.DataFrame(keras_scores))","metadata":{"execution":{"iopub.status.busy":"2023-11-24T10:23:46.786074Z","iopub.execute_input":"2023-11-24T10:23:46.786559Z","iopub.status.idle":"2023-11-24T10:23:46.803875Z","shell.execute_reply.started":"2023-11-24T10:23:46.786527Z","shell.execute_reply":"2023-11-24T10:23:46.802209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torch_coef = 0.2\nkeras_coef = 0.8\n\nblend_oof = torch_coef * torch_oof + keras_coef * keras_oof\nblend_preds = torch_coef * torch_preds + keras_coef * keras_preds * (-1)\n\nblend_mae_score = mean_absolute_error(y_train, blend_oof)\nblend_mrrmse_score = mrrmse(y_train, blend_oof)\n\nprint(f\"Blend OOF MAE: {blend_mae_score:.6f}\")\nprint(f\"Blend OOF MRRMSE: {blend_mrrmse_score:.6f}\")","metadata":{"execution":{"iopub.status.busy":"2023-11-24T10:32:37.03639Z","iopub.execute_input":"2023-11-24T10:32:37.036854Z","iopub.status.idle":"2023-11-24T10:32:37.521375Z","shell.execute_reply.started":"2023-11-24T10:32:37.036821Z","shell.execute_reply":"2023-11-24T10:32:37.519822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(\"/kaggle/input/open-problems-single-cell-perturbations/sample_submission.csv\")\nsub","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_cols = sub.columns[1:]\n\nsub[label_cols] = blend_preds\nsub","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.save(\"keras_oof.npy\", keras_oof)\nnp.save(\"torch_oof.npy\", torch_oof)\nnp.save(\"blend_oof.npy\", blend_oof)\n\nnp.save(\"keras_preds.npy\", keras_preds)\nnp.save(\"torch_preds.npy\", torch_preds)\nnp.save(\"blend_preds.npy\", blend_preds)","metadata":{},"execution_count":null,"outputs":[]}]}