{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":59094,"databundleVersionId":7010844,"sourceType":"competition"},{"sourceId":120352,"sourceType":"modelInstanceVersion","modelInstanceId":101229,"modelId":125424},{"sourceId":254049,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":217211,"modelId":238929}],"dockerImageVersionId":30840,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **OPxMOE V2**","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\nimport torch \nimport numpy as np\nimport pandas as pd\nimport os\nimport random\nimport kagglehub\nfrom kaggle_secrets import UserSecretsClient\n\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f'\\nUsing {device}')\n\nseed = 42\nrandom.seed(seed)\nos.environ['PYTHONHASHSEED'] = str(seed)\nos.environ['TOKENIZERS_PARALLELISM'] = 'true'\nnp.random.seed(seed)\ntorch.manual_seed(seed)\nif torch.cuda.is_available():\n    torch.cuda.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\nprint('-----Seed Set!-----')\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2025-02-09T19:26:14.550981Z","iopub.execute_input":"2025-02-09T19:26:14.551298Z","iopub.status.idle":"2025-02-09T19:26:18.304724Z","shell.execute_reply.started":"2025-02-09T19:26:14.551273Z","shell.execute_reply":"2025-02-09T19:26:18.303815Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **Installing Requirements**","metadata":{}},{"cell_type":"code","source":"!pip install -q umap-learn rdkit captum git+https://github.com/samoturk/mol2vec selfies==2.1.1  simpletransformers==0.63.9 pandarallel==1.6.4 wandb==0.13.10","metadata":{"execution":{"iopub.status.busy":"2025-02-09T19:26:18.305948Z","iopub.execute_input":"2025-02-09T19:26:18.306591Z","iopub.status.idle":"2025-02-09T19:26:54.434216Z","shell.execute_reply.started":"2025-02-09T19:26:18.306534Z","shell.execute_reply":"2025-02-09T19:26:54.433209Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from gensim.models import word2vec\nmodel1 = word2vec.Word2Vec.load('/kaggle/input/mol2vec/pytorch/default/1/model_300dim.pkl')","metadata":{"execution":{"iopub.status.busy":"2025-02-09T19:26:54.436246Z","iopub.execute_input":"2025-02-09T19:26:54.436483Z","iopub.status.idle":"2025-02-09T19:27:10.055094Z","shell.execute_reply.started":"2025-02-09T19:26:54.436461Z","shell.execute_reply":"2025-02-09T19:27:10.054433Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_parquet('../input/open-problems-single-cell-perturbations/de_train.parquet')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2025-02-09T19:27:10.056270Z","iopub.execute_input":"2025-02-09T19:27:10.056482Z","iopub.status.idle":"2025-02-09T19:27:11.886908Z","shell.execute_reply.started":"2025-02-09T19:27:10.056463Z","shell.execute_reply":"2025-02-09T19:27:11.886118Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to convert SMILES to SELFIES\nfrom selfies import encoder\ndef smiles_to_selfies(smiles):\n    try:\n        return encoder(smiles)\n    except Exception as e:\n        print(f\"Error converting SMILES '{smiles}': {e}\") \n        return None  \n\n# Apply the function to create the 'SELFIES' column\ndf['SELFIES'] = df['SMILES'].apply(smiles_to_selfies)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-09T19:27:11.887784Z","iopub.execute_input":"2025-02-09T19:27:11.887995Z","iopub.status.idle":"2025-02-09T19:27:12.252261Z","shell.execute_reply.started":"2025-02-09T19:27:11.887972Z","shell.execute_reply":"2025-02-09T19:27:12.251612Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **Preprocessing Classes**","metadata":{}},{"cell_type":"code","source":"from abc import ABC, abstractmethod\nimport pandas as pd\nimport numpy as np\nfrom rdkit import Chem\nfrom rdkit.Chem import Descriptors\nfrom sklearn.preprocessing import OneHotEncoder, StandardScaler\nfrom gensim.models import word2vec\nfrom mol2vec.features import mol2alt_sentence, MolSentence\nfrom rdkit.Chem import AllChem\nfrom rdkit import DataStructs\nfrom rdkit.Chem import Descriptors, rdMolDescriptors, QED\nfrom rdkit.Chem.rdMolDescriptors import CalcTPSA, CalcNumRotatableBonds, CalcNumHBA, CalcNumHBD, CalcFractionCSP3\nfrom rdkit.Chem import BRICS, Recap\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import DataLoader, TensorDataset\nfrom sklearn.base import BaseEstimator, TransformerMixin\nfrom transformers import AutoModelForMaskedLM, AutoTokenizer , AutoModel\nfrom tqdm import tqdm\nimport os\nimport json\nimport pickle\nfrom IPython.display import clear_output\nfrom pandarallel import pandarallel\nfrom transformers import RobertaTokenizer, RobertaModel, RobertaConfig\n\nclass BaseEmbedding(ABC):\n    \"\"\"Abstract base class for embeddings with save/load functionality.\"\"\"\n    \n    def __init__(self):\n        self.embedding_dict = {}\n        self.metadata = {}\n    \n    @abstractmethod\n    def preprocess(self, df):\n        pass\n    \n    @abstractmethod\n    def get_embedding(self, df):\n        pass\n    \n    @property\n    @abstractmethod\n    def embedding_size(self):\n        pass\n    \n    def compute_and_store_embeddings(self, df, entity_column):\n        \"\"\"\n        Compute and store embeddings for unique entities in the specified column.\n        \n        Args:\n            df (pd.DataFrame): Input DataFrame\n            entity_column (str): Column name containing entities (e.g., 'cell_type' or 'sm_name')\n        \"\"\"\n        unique_entities = df[entity_column].unique()\n        for entity in unique_entities:\n            entity_df = df[df[entity_column] == entity].copy()\n            embedding = self.get_embedding(entity_df, fit=False)\n            # Store the mean embedding if there are multiple rows\n            self.embedding_dict[entity] = embedding.mean().values\n            \n    def get_entity_embedding(self, entity_name):\n        \"\"\"\n        Retrieve embedding for a specific entity.\n        \n        Args:\n            entity_name (str): Name of the entity\n            \n        Returns:\n            np.ndarray: Embedding vector for the entity\n        \"\"\"\n        if entity_name in self.embedding_dict:\n            return self.embedding_dict[entity_name]\n        else:\n            return np.zeros(self.embedding_size)\n    \n    def save_embeddings(self, filepath, metadata=None):\n        \"\"\"\n        Save embeddings and metadata to disk.\n        \n        Args:\n            filepath (str): Path to save the embeddings\n            metadata (dict, optional): Additional metadata to save\n        \"\"\"\n        directory = os.path.dirname(filepath)\n        if directory and not os.path.exists(directory):\n            os.makedirs(directory)\n            \n        # Convert numpy arrays to lists for JSON serialization\n        serializable_dict = {\n            k: v.tolist() if isinstance(v, (np.ndarray, torch.Tensor)) else v \n            for k, v in self.embedding_dict.items()\n        }\n        \n        # Prepare save data\n        save_data = {\n            'embeddings': serializable_dict,\n            'metadata': metadata or self.metadata,\n            'embedding_size': self.embedding_size,\n            'embedding_type': self.__class__.__name__\n        }\n        \n        # Save as pickle if the filepath ends with .pkl, otherwise save as JSON\n        if filepath.endswith('.pkl'):\n            with open(filepath, 'wb') as f:\n                pickle.dump(save_data, f)\n        else:\n            with open(filepath, 'w') as f:\n                json.dump(save_data, f)\n                \n    def load_embeddings(self, filepath):\n        \"\"\"\n        Load embeddings and metadata from disk.\n        \n        Args:\n            filepath (str): Path to load the embeddings from\n            \n        Returns:\n            bool: True if loading was successful\n        \"\"\"\n        try:\n            # Load pickle if the filepath ends with .pkl, otherwise load JSON\n            if filepath.endswith('.pkl'):\n                with open(filepath, 'rb') as f:\n                    save_data = pickle.load(f)\n            else:\n                with open(filepath, 'r') as f:\n                    save_data = json.load(f)\n            \n            # Convert lists back to numpy arrays\n            self.embedding_dict = {\n                k: np.array(v) if isinstance(v, list) else v \n                for k, v in save_data['embeddings'].items()\n            }\n            \n            self.metadata = save_data.get('metadata', {})\n            \n            # Verify embedding size matches\n            if save_data['embedding_size'] != self.embedding_size:\n                raise ValueError(\n                    f\"Loaded embedding size ({save_data['embedding_size']}) \"\n                    f\"doesn't match current embedding size ({self.embedding_size})\"\n                )\n            \n            return True\n            \n        except Exception as e:\n            print(f\"Error loading embeddings: {str(e)}\")\n            return False\n\n\nclass OneHotEmbedding(BaseEmbedding):\n    def __init__(self, columns):\n        super().__init__()\n        self.columns = columns\n        self.encoder = OneHotEncoder(sparse_output=False)\n        self._embedding_size = None\n\n    def preprocess(self, df):\n        return df[self.columns]\n\n    def get_embedding(self, df, fit=True):\n        if fit:\n            encoded_features = self.encoder.fit_transform(df[self.columns])\n        else:\n            encoded_features = self.encoder.transform(df[self.columns])\n        \n        self._embedding_size = encoded_features.shape[1]\n        encoded_df = pd.DataFrame(encoded_features, columns=self.encoder.get_feature_names_out(self.columns))\n        \n        # Store embeddings using base class functionality\n        for idx, row in df[self.columns].iterrows():\n            key = tuple(row.values)\n            self.embedding_dict[key] = encoded_df.loc[idx].values\n            \n        return encoded_df\n\n    @property\n    def embedding_size(self):\n        return self._embedding_size\n\n    \nclass SMILESEmbedding(BaseEmbedding):\n    def __init__(self):\n        super().__init__()\n        self.scaler = StandardScaler()\n\n    def preprocess(self, df):\n        return df\n\n    def create_molecule_embedding_dict(self, df):\n        \"\"\"\n        Create a dictionary of molecule embeddings.\n        \n        Args:\n            df (pd.DataFrame): DataFrame containing SMILES column\n        \"\"\"\n        self.compute_and_store_embeddings(df, 'SMILES')\n        \n    def get_embedding(self, df, fit=True):\n        smiles_info_list = df['SMILES'].apply(self.extract_smiles_info)\n        smiles_info_df = pd.DataFrame(smiles_info_list.tolist())\n        if fit:\n            smiles_info_df = self.scaler.fit_transform(smiles_info_df)\n        else:\n            smiles_info_df = self.scaler.transform(smiles_info_df)\n\n            \n        columns = [\n            'Molecular Weight', 'LogP', 'TPSA', 'Number of Atoms', 'Number of Bonds',\n            'Number of Rotatable Bonds', 'Number of Hydrogen Bond Acceptors', 'Number of Hydrogen Bond Donors',\n            'Number of Rings', 'Number of Aromatic Rings', 'Number of Stereocenters',\n            'Fraction of sp3 Carbons', 'Balaban J Index', 'Bertz CT', 'QED Score'\n        ]\n        \n        result_df = pd.DataFrame(smiles_info_df, columns=columns)\n        \n        # Store embeddings using base class functionality\n        for idx, smiles in enumerate(df['SMILES']):\n            self.embedding_dict[smiles] = result_df.loc[idx].values\n            \n        return result_df\n\n    @staticmethod\n    def extract_smiles_info(smiles):\n        if smiles is None or smiles == '':\n            return None\n        mol = Chem.MolFromSmiles(smiles)\n        if mol is None:\n            return None\n        info = {\n            'Molecular Weight': Descriptors.MolWt(mol),\n            'LogP': Descriptors.MolLogP(mol),\n            'TPSA': CalcTPSA(mol),\n            'Number of Atoms': mol.GetNumAtoms(),\n            'Number of Bonds': mol.GetNumBonds(),\n            'Number of Rotatable Bonds': CalcNumRotatableBonds(mol),\n            'Number of Hydrogen Bond Acceptors': CalcNumHBA(mol),\n            'Number of Hydrogen Bond Donors': CalcNumHBD(mol),\n            'Number of Rings': Descriptors.RingCount(mol),\n            'Number of Aromatic Rings': rdMolDescriptors.CalcNumAromaticRings(mol),\n            'Number of Stereocenters': len(Chem.FindMolChiralCenters(mol, includeUnassigned=True)),\n            'Fraction of sp3 Carbons': CalcFractionCSP3(mol),\n            'Balaban J Index': Descriptors.BalabanJ(mol),\n            'Bertz CT': Descriptors.BertzCT(mol),\n            'QED Score': QED.qed(mol)\n        }\n        return info\n\n    @property\n    def embedding_size(self):\n        return 15  \n\nclass Mol2VecEmbedding(BaseEmbedding):\n    def __init__(self, model_path):\n        super().__init__()\n        self.model = word2vec.Word2Vec.load(model_path)\n        self.keys = set(self.model.wv.key_to_index.keys())\n\n    def preprocess(self, df):\n        df['mol'] = df['SMILES'].apply(lambda x: Chem.MolFromSmiles(x))\n        df['sentence'] = df.apply(lambda x: MolSentence(mol2alt_sentence(x['mol'], 1)), axis=1)\n        return df\n\n    def get_embedding(self, df, fit=True):\n        df['vector'] = df['sentence'].apply(lambda sentence: self.sentence_to_vector(sentence))\n        vector_dim = len(self.model.wv.get_vector(next(iter(self.keys))))\n        vector_columns = [f'vector_{i}' for i in range(vector_dim)]\n        result_df = pd.DataFrame(df['vector'].tolist(), columns=vector_columns)\n        \n        # Store embeddings using base class functionality\n        for idx, smiles in enumerate(df['SMILES']):\n            self.embedding_dict[smiles] = result_df.loc[idx].values\n            \n        return result_df\n\n    def sentence_to_vector(self, sentence, unseen=False, unseen_vec=np.zeros(300)):\n        if unseen:\n            vec = sum([self.model.wv.get_vector(word) if word in self.keys else unseen_vec for word in sentence])\n        else:\n            vec = sum([self.model.wv.get_vector(word) for word in sentence if word in self.keys])\n        return vec\n\n    def create_molecule_embedding_dict(self, df):\n        \"\"\"\n        Create a dictionary of molecule embeddings.\n        \n        Args:\n            df (pd.DataFrame): DataFrame containing SMILES column\n        \"\"\"\n        self.compute_and_store_embeddings(df, 'SMILES')\n    \n    @property\n    def embedding_size(self):\n        return len(self.model.wv.get_vector(next(iter(self.keys))))\n    \n\nclass Autoencoder(nn.Module):\n    def __init__(self, input_size, hidden_size):\n        super(Autoencoder, self).__init__()\n        # Encoder\n        self.encoder = nn.Sequential(\n            nn.Linear(input_size, hidden_size),\n            nn.Sigmoid(),\n        )\n        # Decoder\n        self.decoder = nn.Sequential(\n            nn.Linear(hidden_size, input_size),\n            nn.Sigmoid()\n        )\n\n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return encoded, decoded\n\nclass TargetEmbedding(BaseEmbedding, nn.Module):\n    \"\"\"Target embedding class using autoencoder to learn compressed representations of medians.\"\"\"\n    \n    def __init__(self, df_train, emb_size=256, hidden_size=1024):\n        \"\"\"\n        Initialize the autoencoder target embedding class.\n        \n        Args:\n            df_train (pd.DataFrame): Training DataFrame for storing medians\n            emb_size (int): Size of the latent embedding for each component\n            hidden_size (int): Size of the hidden layer in encoder/decoder\n        \"\"\"\n        \n        BaseEmbedding.__init__(self)\n        nn.Module.__init__(self)\n        \n        self.emb_size = emb_size\n        self._embedding_size = emb_size * 2  # combined size of both embeddings\n        self.input_size = 18211  # Original dimension of median values\n        \n        # Encoder networks for cell type and small molecule embeddings\n        self.cell_type_encoder = nn.Sequential(\n            nn.Linear(self.input_size, hidden_size),\n            nn.GELU(),\n            nn.Linear(hidden_size, hidden_size),\n            nn.GELU(),\n            nn.Linear(hidden_size, emb_size)\n        )\n        \n        self.sm_encoder = nn.Sequential(\n            nn.Linear(self.input_size, hidden_size),\n            nn.GELU(),\n            nn.Linear(hidden_size, hidden_size),\n            nn.GELU(),\n            nn.Linear(hidden_size, emb_size)\n        )\n        \n        # Decoder networks for reconstruction\n        self.cell_type_decoder = nn.Sequential(\n            nn.Linear(emb_size, hidden_size),\n            nn.GELU(),\n            nn.Linear(hidden_size, hidden_size),\n            nn.GELU(),\n            nn.Linear(hidden_size, self.input_size)\n        )\n        \n        self.sm_decoder = nn.Sequential(\n            nn.Linear(emb_size, hidden_size),\n            nn.GELU(),\n            nn.Linear(hidden_size, hidden_size),\n            nn.GELU(),\n            nn.Linear(hidden_size, self.input_size)\n        )\n        \n        # Initialize dictionaries\n        self.cell_type_dict = {}\n        self.sm_dict = {}\n        \n        # Process training data\n        df_train = self.preprocess(df_train)\n        \n        # Store training data and compute medians\n        self.de_cell_type_train = df_train.iloc[:, [0] + list(range(5, df_train.shape[1]))]\n        self.de_sm_name_train = df_train.iloc[:, [1] + list(range(5, df_train.shape[1]))]\n        \n        # Rename columns for consistency\n        self.de_cell_type_train.columns = ['cell_type' if i == 0 else col \n                                         for i, col in enumerate(self.de_cell_type_train.columns)]\n        self.de_sm_name_train.columns = ['sm_name' if i == 0 else col \n                                         for i, col in enumerate(self.de_sm_name_train.columns)]\n        \n        # Compute medians from training data\n        self.cell_type_medians = self.de_cell_type_train.select_dtypes(include=['number']).groupby(self.de_cell_type_train['cell_type']).median()\n        self.sm_name_medians = self.de_sm_name_train.select_dtypes(include=['number']).groupby(self.de_sm_name_train['sm_name']).median()\n\n        \n        # Convert medians to tensors\n        self.cell_type_tensors = {k: torch.tensor(v).float() \n                                 for k, v in zip(self.cell_type_medians.index, \n                                               self.cell_type_medians.values)}\n        self.sm_tensors = {k: torch.tensor(v).float() \n                          for k, v in zip(self.sm_name_medians.index, \n                                        self.sm_name_medians.values)}\n        \n        self.__class__.name = \"TargetEmbedding\"\n        self.is_fitted = False\n    \n    def preprocess(self, df):\n        \"\"\"\n        Preprocess the input DataFrame.\n        \n        Args:\n            df (pd.DataFrame): Input DataFrame\n            \n        Returns:\n            pd.DataFrame: Preprocessed DataFrame\n        \"\"\"\n        # Copy the DataFrame to avoid modifying the original\n        df = df.copy()\n        \n        # Ensure required columns exist\n        required_cols = ['cell_type', 'sm_name']\n        missing_cols = [col for col in required_cols if col not in df.columns]\n        if missing_cols:\n            raise ValueError(f\"Missing required columns: {missing_cols}\")\n            \n        # Handle any missing values in gene expression columns\n        numeric_cols = df.select_dtypes(include=['float64', 'int64']).columns\n        df[numeric_cols] = df[numeric_cols].fillna(0)\n        \n        return df\n    \n    def train_autoencoder(self, num_epochs=100, batch_size=32, learning_rate=1e-3):\n        \"\"\"\n        Train the autoencoder to learn compressed representations of the medians.\n        \n        Args:\n            num_epochs (int): Number of training epochs\n            batch_size (int): Batch size for training\n            learning_rate (float): Learning rate for optimizer\n        \"\"\"\n        self.train()  # Set to training mode\n        optimizer = torch.optim.Adam(self.parameters(), lr=learning_rate)\n        criterion = nn.MSELoss()\n        \n        # Convert median dictionaries to tensors for training\n        cell_type_data = torch.stack(list(self.cell_type_tensors.values()))\n        sm_data = torch.stack(list(self.sm_tensors.values()))\n        \n        for epoch in range(num_epochs):\n            # Train cell type autoencoder\n            cell_type_encoded = self.cell_type_encoder(cell_type_data)\n            cell_type_decoded = self.cell_type_decoder(cell_type_encoded)\n            cell_type_loss = criterion(cell_type_decoded, cell_type_data)\n            \n            # Train small molecule autoencoder\n            sm_encoded = self.sm_encoder(sm_data)\n            sm_decoded = self.sm_decoder(sm_encoded)\n            sm_loss = criterion(sm_decoded, sm_data)\n            \n            # Combined loss\n            total_loss = cell_type_loss + sm_loss\n            \n            # Backpropagation\n            optimizer.zero_grad()\n            total_loss.backward()\n            optimizer.step()\n            \n            if (epoch + 1) % 10 == 0:\n                print(f'Epoch [{epoch+1}/{num_epochs}], '\n                      f'Loss: {total_loss.item():.4f}, '\n                      f'CT Loss: {cell_type_loss.item():.4f}, '\n                      f'SM Loss: {sm_loss.item():.4f}')\n        \n        self.eval()  # Set to evaluation mode\n        # After training, update the dictionaries with encoded values\n        with torch.no_grad():\n            for k, v in self.cell_type_tensors.items():\n                self.cell_type_dict[k] = self.cell_type_encoder(v.unsqueeze(0)).squeeze(0)\n            \n            for k, v in self.sm_tensors.items():\n                self.sm_dict[k] = self.sm_encoder(v.unsqueeze(0)).squeeze(0)\n        \n        self.is_fitted = True\n    \n    def get_embedding(self, df, fit=False):\n        \"\"\"\n        Get embeddings for the input data using the trained autoencoder.\n        \n        Args:\n            df (pd.DataFrame): Input DataFrame with cell_type and sm_name columns\n            fit (bool): If True, train the autoencoder before getting embeddings\n            \n        Returns:\n            pd.DataFrame: DataFrame containing the embedded features\n        \"\"\"\n        # Preprocess input data\n        df = self.preprocess(df)\n        \n        # Train autoencoder if fit=True and not already fitted\n        if fit and not self.is_fitted:\n            self.train_autoencoder()\n        \n        self.eval()  # Set to evaluation mode\n        with torch.no_grad():\n            cell_type_tensors = []\n            sm_tensors = []\n            \n            for _, row in df.iterrows():\n                # Get cell type embedding\n                ct = row['cell_type']\n                if ct in self.cell_type_dict:\n                    cell_type_tensors.append(self.cell_type_dict[ct])\n                else:\n                    cell_type_tensors.append(torch.zeros(self.emb_size))\n                \n                # Get small molecule embedding\n                sm = row['sm_name']\n                if sm in self.sm_dict:\n                    sm_tensors.append(self.sm_dict[sm])\n                else:\n                    sm_tensors.append(torch.zeros(self.emb_size))\n                \n                # Update the parent class embedding dictionary for both entities\n                # Store cell type embedding\n                if ct in self.cell_type_dict:\n                    self.embedding_dict[f\"cell_type_{ct}\"] = self.cell_type_dict[ct].numpy()\n                \n                # Store small molecule embedding\n                if sm in self.sm_dict:\n                    self.embedding_dict[f\"sm_name_{sm}\"] = self.sm_dict[sm].numpy()\n            \n            # Convert to tensors\n            cell_type_tensor = torch.stack(cell_type_tensors)\n            sm_tensor = torch.stack(sm_tensors)\n            \n            # Combine embeddings\n            combined_embedding = torch.cat([cell_type_tensor, sm_tensor], dim=1)\n        \n        # Create column names for the embedding DataFrame\n        ct_cols = [f'target_ct_emb_{i}' for i in range(self.emb_size)]\n        sm_cols = [f'target_sm_emb_{i}' for i in range(self.emb_size)]\n        all_cols = ct_cols + sm_cols\n        \n        return pd.DataFrame(combined_embedding.detach().numpy(), columns=all_cols, index=df.index)\n        \n    @property\n    def embedding_size(self):\n        \"\"\"Returns the size of the combined embedding.\"\"\"\n        return self._embedding_size\n    \nclass MorganFingerPrintEmbedding(BaseEmbedding):\n    def __init__(self, hidden_size=128):\n        super().__init__()\n        self.hidden_size = hidden_size\n        self.autoencoder = None  # This will store the trained autoencoder\n\n    def preprocess(self, df):\n        return df\n\n    def get_embedding(self, df, fit=True):\n        morgan_fp_list = df['SMILES'].apply(self.extract_morgan_fingerprint)\n        morgan_fp_array = np.stack(morgan_fp_list)\n\n        input_size = morgan_fp_array.shape[1]\n\n        if fit:\n            self.autoencoder = self.train_autoencoder(morgan_fp_array, input_size, self.hidden_size)\n        \n        with torch.no_grad():\n            compressed_fp = self.autoencoder.encoder(torch.tensor(morgan_fp_array, dtype=torch.float32)).numpy()\n\n        result_df = pd.DataFrame(compressed_fp, columns=[f'CompressedFP_{i}' for i in range(self.hidden_size)])\n        \n        # Store embeddings using base class functionality\n        for idx, smiles in enumerate(df['SMILES']):\n            self.embedding_dict[smiles] = result_df.loc[idx].values\n            \n        return result_df\n\n    def extract_morgan_fingerprint(self, smiles, radius=2, nBits=2048):\n        if smiles is None or smiles == '':\n            return None\n        mol = Chem.MolFromSmiles(smiles)\n        if mol is None:\n            return None\n        \n        morgan_gen = AllChem.GetMorganGenerator(radius=radius, fpSize=nBits)\n        fp = morgan_gen.GetFingerprint(mol)\n        \n        fp_array = np.zeros((nBits,))\n        DataStructs.ConvertToNumpyArray(fp, fp_array)\n        \n        return fp_array\n\n    def train_autoencoder(self, data, input_size, hidden_size, num_epochs=50, batch_size=32, learning_rate=0.001):\n        autoencoder = Autoencoder(input_size=input_size, hidden_size=hidden_size)\n        criterion = nn.MSELoss()\n        optimizer = torch.optim.Adam(autoencoder.parameters(), lr=learning_rate)\n\n        dataset = TensorDataset(torch.tensor(data, dtype=torch.float32))\n        dataloader = DataLoader(dataset, batch_size=batch_size, shuffle=True)\n\n        for epoch in range(num_epochs):\n            for batch in dataloader:\n                inputs = batch[0]\n                _, reconstructed = autoencoder(inputs)\n\n                loss = criterion(reconstructed, inputs)\n                optimizer.zero_grad()\n                loss.backward()\n                optimizer.step()\n\n            if (epoch + 1) % 10 == 0:\n                print(f'Epoch [{epoch + 1}/{num_epochs}], Loss: {loss.item():.4f}')\n\n        return autoencoder\n\n    @property\n    def embedding_size(self):\n        return self.hidden_size\n\n\nclass ChemBERTaEmbedding(BaseEmbedding):\n    def __init__(self, model_name=\"DeepChem/ChemBERTa-77M-MTR\", embedding_type='mean_pooling', padding=False):\n        super().__init__()\n        self.model_name = model_name\n        self.embedding_type = embedding_type  # 'cls' or 'mean_pooling'\n        self.padding = padding\n        self.model = AutoModelForMaskedLM.from_pretrained(self.model_name)\n        self.tokenizer = AutoTokenizer.from_pretrained(self.model_name)\n        self.model.eval()\n        self._embedding_size = None # Dynamically set during get_embedding\n\n\n    def preprocess(self, df):\n         return df # No specific pre-processing\n\n    def get_embedding(self, df, fit=True):\n        smiles_list = df['SMILES'].tolist()\n        embeddings_cls, embeddings_mean = self.featurize_ChemBERTa(smiles_list, padding=self.padding)\n\n        if self.embedding_type == 'cls':\n            embeddings = embeddings_cls\n            emb_cols = [f'ChemBERTa_cls_{i}' for i in range(embeddings.shape[1])]\n        elif self.embedding_type == 'mean_pooling':\n            embeddings = embeddings_mean\n            emb_cols = [f'ChemBERTa_mean_{i}' for i in range(embeddings.shape[1])]\n        else:\n            raise ValueError(\"embedding_type must be 'cls' or 'mean_pooling'\")\n\n        result_df = pd.DataFrame(embeddings, columns=emb_cols)\n        self._embedding_size = embeddings.shape[1]\n        # Store embeddings\n        for idx, smiles in enumerate(df['SMILES']):\n            self.embedding_dict[smiles] = result_df.loc[idx].values\n\n        return result_df\n\n    def featurize_ChemBERTa(self, smiles_list, padding=True):\n\n        embeddings_cls = []\n        embeddings_mean = []\n\n        with torch.no_grad():\n            for smiles in smiles_list: # Removed tqdm for batch processing\n                encoded_input = self.tokenizer(smiles, return_tensors=\"pt\", padding=padding, truncation=True)\n                model_output = self.model(**encoded_input)\n\n                embedding_cls = model_output[0][:, 0, :]  # CLS token embedding\n                embeddings_cls.append(embedding_cls)\n\n                embedding_mean = torch.mean(model_output[0], 1)  # Mean pooling\n                embeddings_mean.append(embedding_mean)\n        \n        # Stack the tensors and convert to numpy\n        embeddings_cls = torch.cat(embeddings_cls).numpy()\n        embeddings_mean = torch.cat(embeddings_mean).numpy()\n        return embeddings_cls, embeddings_mean\n    \n    @property\n    def embedding_size(self):\n        if self._embedding_size is None:\n          return 600  # Default, but will be correctly set during get_embedding\n        return self._embedding_size\n\nclass MolformerEmbedding(BaseEmbedding):\n    def __init__(self, model_path='ibm/MoLFormer-XL-both-10pct', padding=True):\n        super().__init__()\n        self.model_path = model_path\n        self.model = AutoModel.from_pretrained(model_path, trust_remote_code=True)\n        self.tokenizer = AutoTokenizer.from_pretrained(model_path, trust_remote_code=True)\n        self.model.eval()  # Ensure the model is in evaluation mode\n        self._embedding_size = None\n        self.padding = padding # padding should be True for this model\n\n    def preprocess(self, df):\n        return df  # No specific pre-processing needed\n\n    def get_embedding(self, df, fit=True):\n        smiles_list = df['SMILES'].tolist()\n        embeddings = self.featurize_Molformer(smiles_list)\n        emb_cols = [f'Molformer_{i}' for i in range(embeddings.shape[1])]\n\n        result_df = pd.DataFrame(embeddings, columns=emb_cols, index=df.index) # added index=df.index\n        self._embedding_size = embeddings.shape[1]\n        # Store embeddings\n        for idx, smiles in enumerate(df['SMILES']):\n            self.embedding_dict[smiles] = result_df.loc[idx].values\n\n        return result_df\n\n    def featurize_Molformer(self, smiles_list):\n        \"\"\"\n        Generates MolFormer embeddings for a list of SMILES strings.\n        \"\"\"\n        inputs = self.tokenizer(smiles_list, padding=self.padding, return_tensors=\"pt\")\n        with torch.no_grad():\n            outputs = self.model(**inputs)\n        # Use pooler_output for the sentence-level representation\n        embeddings = outputs.pooler_output.cpu().numpy()  # Move to CPU before converting to NumPy\n        return embeddings\n\n\n    @property\n    def embedding_size(self):\n        if self._embedding_size is None:\n          #  This needs to be determined dynamically after the first call to featurize, or\n          #  looked up from the model config if available.  A placeholder won't work well.\n          #  The correct embedding size (after inspecting the model) is 768 for\n          #  'ibm/MoLFormer-XL-both-10pct'. However, it's best to set this dynamically.\n          return 768 # correct for ibm/MoLFormer-XL-both-10pct, but should ideally be dynamic\n        return self._embedding_size\n\nclass SmoleBartEmbedding(BaseEmbedding):\n    def __init__(self, model_path=\"UdS-LSV/smole-bart\", padding=True, embedder=\"encoder\"):\n        super().__init__()\n        self.model_path = model_path\n        self.padding = padding  # whether to pad the SMILES sequences\n        self.embedder = embedder  # choose \"encoder\" or \"decoder\"\n        # load the model and tokenizer using trust_remote_code=True if needed\n        self.model = AutoModel.from_pretrained(model_path, trust_remote_code=True)\n        self.tokenizer = AutoTokenizer.from_pretrained(model_path, trust_remote_code=True)\n        self.model.eval()  # set the model to evaluation mode\n        self._embedding_size = None\n\n    def preprocess(self, df):\n        # No special preprocessing needed; simply return the dataframe.\n        return df\n\n    def get_embedding(self, df, fit=True):\n        # df is expected to have a column \"SMILES\"\n        smiles_list = df['SMILES'].tolist()\n        embeddings = self.featurize_smole_bart(smiles_list, embedder=self.embedder)\n        # Create column names based on the embedding dimension\n        emb_cols = [f\"SmoleBart_{i}\" for i in range(embeddings.shape[1])]\n        # Create a DataFrame with the embeddings (preserving the index)\n        result_df = pd.DataFrame(embeddings, columns=emb_cols, index=df.index)\n        # Set the embedding size based on the shape of embeddings\n        self._embedding_size = embeddings.shape[1]\n        # Store embeddings in the embedding dictionary\n        for idx, smiles in enumerate(df['SMILES']):\n            self.embedding_dict[smiles] = result_df.loc[idx].values\n        return result_df\n\n    def featurize_smole_bart(self, smiles_list, embedder=\"encoder\"):\n        # Tokenize the list of SMILES strings.\n        inputs = self.tokenizer(smiles_list, padding=self.padding, return_tensors=\"pt\")\n        with torch.no_grad():\n            outputs = self.model(**inputs)\n        # For a BART model, there is no pooler output by default.\n        # Use mean pooling on the appropriate output.\n        if embedder == \"encoder\":\n            encoder_out = outputs.encoder_last_hidden_state  # shape: [batch, seq_len, hidden_size]\n            embeddings = encoder_out.mean(dim=1)\n        elif embedder == \"decoder\":\n            decoder_out = outputs.last_hidden_state  # shape: [batch, seq_len, hidden_size]\n            embeddings = decoder_out.mean(dim=1)\n        else:\n            raise ValueError(\"embedder must be either 'encoder' or 'decoder'\")\n        return embeddings.cpu().numpy()\n\n    @property\n    def embedding_size(self):\n        # Return the computed embedding size; if not computed, default to 0 (or raise an error)\n        return self._embedding_size if self._embedding_size is not None else 0\n\n\n\n\nimport os\nimport torch\nimport pandas as pd\nfrom transformers import RobertaModel, RobertaConfig, RobertaTokenizer\nfrom pandarallel import pandarallel\n\nclass SelfFormerEmbedding(BaseEmbedding):\n    def __init__(self, model_name, tokenizer_path=None, padding=True, batch_size=32, nb_workers=5, device=None):\n        \"\"\"\n        Initialize the SelfFormerEmbedding instance.\n        \n        Args:\n            model_name (str): Path or identifier for the pretrained SELFormer model.\n            tokenizer_path (str, optional): If provided, load the tokenizer from this path.\n            padding (bool): Whether to apply padding when tokenizing.\n            batch_size (int): Number of samples to process in a single batch.\n            nb_workers (int): Number of parallel workers for pandarallel.\n            device (str, optional): Device to run the model on (\"cuda\" or \"cpu\"). If None, auto-detect.\n        \"\"\"\n        super().__init__()\n        \n        # Disable parallelism warnings\n        os.environ[\"TOKENIZERS_PARALLELISM\"] = \"false\"\n        os.environ[\"WANDB_DISABLED\"] = \"true\"\n        \n        # Set device (auto-detect if not provided)\n        self.device = device if device else (\"cuda\" if torch.cuda.is_available() else \"cpu\")\n        \n        self.model_name = model_name\n        self.padding = padding\n        self.batch_size = batch_size\n        self._embedding_size = None\n        \n        # Load model configuration with hidden states enabled\n        config = RobertaConfig.from_pretrained(model_name)\n        config.output_hidden_states = True\n        self.model = RobertaModel.from_pretrained(model_name, config=config).to(self.device)\n        self.model.eval()\n        \n        # Load tokenizer\n        self.tokenizer = RobertaTokenizer.from_pretrained(tokenizer_path or model_name)\n        \n        # Initialize parallel processing\n        pandarallel.initialize(nb_workers=nb_workers, progress_bar=True)\n\n    def preprocess(self, df):\n        \"\"\"Preprocessing step (if needed). Currently, it returns the dataframe unchanged.\"\"\"\n        return df\n\n    def get_embedding(self, df, fit=True):\n        \"\"\"\n        Generate embeddings for all SELFIES in the dataframe using batch processing.\n        \n        Args:\n            df (pd.DataFrame): Dataframe containing a \"SELFIES\" column.\n            fit (bool): Unused compatibility flag.\n        \n        Returns:\n            pd.DataFrame: DataFrame with computed embeddings.\n        \"\"\"\n        selfies_list = df[\"SELFIES\"].tolist()\n        embeddings = self.get_embeddings_batch(selfies_list, batch_size=self.batch_size)\n\n        if len(embeddings) > 0:\n            self._embedding_size = embeddings.shape[1]\n\n        # Create a result DataFrame\n        emb_cols = [f'SELFormer_{i}' for i in range(self._embedding_size)]\n        result_df = pd.DataFrame(embeddings, columns=emb_cols, index=df.index)\n\n        # Store embeddings in the dictionary\n        for idx, selfie in enumerate(df[\"SELFIES\"]):\n            self.embedding_dict[selfie] = result_df.loc[idx].values\n\n        # Drop the original SELFIES column\n        return result_df\n\n    @property\n    def embedding_size(self):\n        \"\"\"Return the embedding size; if not computed yet, return a default value.\"\"\"\n        return self._embedding_size if self._embedding_size is not None else 600\n\n    def get_embeddings_batch(self, selfies_list, batch_size=32):\n        \"\"\"\n        Compute embeddings for a batch of SELFIES strings.\n        \n        Args:\n            selfies_list (list of str): List of SELFIES strings.\n            batch_size (int): Batch size for processing.\n        \n        Returns:\n            numpy.ndarray: Array of embeddings.\n        \"\"\"\n        all_embeddings = []\n\n        for i in range(0, len(selfies_list), batch_size):\n            batch = selfies_list[i: i + batch_size]\n            encoded_input = self.tokenizer(\n                batch,\n                add_special_tokens=True,\n                max_length=512,\n                padding=\"max_length\",\n                truncation=True,\n                return_tensors=\"pt\"\n            ).to(self.device)\n\n            with torch.no_grad():\n                output = self.model(**encoded_input)\n\n            # Mean pooling over the sequence dimension\n            embeddings = torch.mean(output.last_hidden_state, dim=1)\n            all_embeddings.append(embeddings.cpu())\n\n        # Concatenate all batches and convert to NumPy\n        return torch.cat(all_embeddings, dim=0).numpy()\n\n        ","metadata":{"execution":{"iopub.status.busy":"2025-02-09T19:31:58.353766Z","iopub.execute_input":"2025-02-09T19:31:58.354155Z","iopub.status.idle":"2025-02-09T19:31:58.425585Z","shell.execute_reply.started":"2025-02-09T19:31:58.354120Z","shell.execute_reply":"2025-02-09T19:31:58.424766Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Preprocessor:\n    def __init__(self, embeddings):\n        self.embeddings = embeddings\n        self.embedding_indices = {}\n        self.latest_processed_data = None\n        self.sm_name_to_smiles = {}\n        self.unique_cell_types = set()\n        self.unique_sm_names = set()\n\n    def preprocess(self, df, fit=True):\n        \"\"\"\n        Process the input DataFrame through all embeddings and combine the results.\n\n        Args:\n            df (pd.DataFrame): Input DataFrame\n            fit (bool): Whether to fit the embeddings (train encoders, etc.)\n\n        Returns:\n            pd.DataFrame: Combined embedded features\n        \"\"\"\n\n        if fit:\n            if 'cell_type' in df.columns:\n                self.unique_cell_types.update(df['cell_type'].unique())\n            if 'sm_name' in df.columns:\n                self.unique_sm_names.update(df['sm_name'].unique())\n\n        if fit and 'sm_name' in df.columns and 'SMILES' in df.columns:\n            sm_name_smiles_map = df[['sm_name', 'SMILES']].drop_duplicates()\n            self.sm_name_to_smiles.update(\n                dict(zip(sm_name_smiles_map['sm_name'], sm_name_smiles_map['SMILES']))\n            )\n\n        processed_dfs = []\n        current_index = 0\n\n        for embedding in self.embeddings:\n            embedding_name = embedding.__class__.__name__\n            # Handle SMILES-based embeddings, including the new ones\n            smiles_based_embeddings = [\n                'SMILESEmbedding', 'MorganFingerPrintEmbedding', 'Mol2VecEmbedding',\n                'ChemBERTaEmbedding','MolformerEmbedding','SmoleBartEmbedding','SelfFormerEmbedding'\n            ]\n            if embedding_name in smiles_based_embeddings:\n                if 'SMILES' not in df.columns:\n                    # Create a copy to avoid modifying the original DataFrame\n                    df_with_smiles = df.copy()\n                    # Add SMILES column using the stored mapping\n                    df_with_smiles['SMILES'] = df_with_smiles['sm_name'].map(self.sm_name_to_smiles)\n                    # Check for missing SMILES *after* mapping\n                    if df_with_smiles['SMILES'].isnull().any():\n                        print(f\"Warning: Missing SMILES for some sm_names in {embedding_name}.  Filling with empty string.\")\n                        df_with_smiles['SMILES'] = df_with_smiles['SMILES'].fillna('')  # Or some other placeholder\n                    preprocessed_df = embedding.preprocess(df_with_smiles)\n                    embedded_df = embedding.get_embedding(preprocessed_df, fit=fit)\n                else:\n                     # Check for missing/empty SMILES *before* processing\n                    if df['SMILES'].isnull().any() or (df['SMILES'] == '').any():\n                        print(f\"Warning: Missing/Empty SMILES found in {embedding_name}. Filling with empty string.\")\n                        df_copy = df.copy() # Work on copy\n                        df_copy['SMILES'] = df_copy['SMILES'].fillna('')\n                        preprocessed_df = embedding.preprocess(df_copy)\n                        embedded_df = embedding.get_embedding(preprocessed_df, fit=fit)\n                    else:\n                        preprocessed_df = embedding.preprocess(df)\n                        embedded_df = embedding.get_embedding(preprocessed_df, fit=fit)\n            else:\n                preprocessed_df = embedding.preprocess(df)\n                embedded_df = embedding.get_embedding(preprocessed_df, fit=fit)\n\n\n            processed_dfs.append(embedded_df)\n\n\n            self.embedding_indices[embedding_name] = {\n                'start': current_index,\n                'end': current_index + embedding.embedding_size,\n                'columns': list(embedded_df.columns)  # Store column names\n            }\n            current_index += embedding.embedding_size\n\n        self.latest_processed_data = pd.concat(processed_dfs, axis=1)\n        return self.latest_processed_data\n    def get_entity_embeddings(self, cell_type=None, sm_name=None):\n        \"\"\"\n        Retrieves combined embeddings for a given cell type and/or small molecule.\n\n        Args:\n            cell_type (str, optional): The cell type.\n            sm_name (str, optional): The small molecule name.\n\n        Returns:\n            pd.DataFrame:  Combined embeddings.\n            dict: Individual embeddings from each embedding type.\n        \"\"\"\n\n        if cell_type is None and sm_name is None:\n            raise ValueError(\"At least one of cell_type or sm_name must be provided\")\n\n        individual_embeddings = {}\n        all_values = []\n        all_columns = []\n\n        for embedding in self.embeddings:\n            embedding_name = embedding.__class__.__name__\n            embedding_info = self.embedding_indices[embedding_name]\n\n            if not hasattr(embedding, 'get_entity_embedding'):\n                # Handle embeddings *without* get_entity_embedding\n                zeros = np.zeros(embedding.embedding_size)\n                all_values.append(zeros)\n                all_columns.extend(embedding_info['columns'])\n                continue\n            \n            entity_embedding = None\n\n            # TargetEmbedding (handles both cell_type and sm_name)\n            if hasattr(embedding, 'cell_type_dict') and cell_type is not None:\n                key = f\"cell_type_{cell_type}\"\n                ct_embedding = embedding.get_entity_embedding(key)\n                if ct_embedding is not None:  # Check for None\n                   individual_embeddings[f\"{embedding_name}_cell_type\"] = ct_embedding\n                   entity_embedding = ct_embedding\n\n            if hasattr(embedding, 'sm_dict') and sm_name is not None:\n                key = f\"sm_name_{sm_name}\"\n                sm_embedding = embedding.get_entity_embedding(key)\n                if sm_embedding is not None: # Check for None\n                    individual_embeddings[f\"{embedding_name}_sm\"] = sm_embedding\n                    if entity_embedding is None:\n                        entity_embedding = sm_embedding\n                    else:\n                        entity_embedding = np.concatenate([entity_embedding, sm_embedding])\n            \n            #SMILES Based Embeddings\n            smiles_based_embeddings = [\n                'SMILESEmbedding', 'MorganFingerPrintEmbedding', 'Mol2VecEmbedding',\n                'ChemBERTaEmbedding','MolformerEmbedding','SmoleBartEmbedding','SelfFormerEmbedding'\n            ]\n            if sm_name is not None and embedding_name in smiles_based_embeddings:\n                if sm_name in self.sm_name_to_smiles:\n                    smiles = self.sm_name_to_smiles[sm_name]\n                    entity_embedding = embedding.get_entity_embedding(smiles)\n                    if entity_embedding is not None:  # Check for None\n                        individual_embeddings[embedding_name] = entity_embedding\n                else:  # Handle missing SMILES\n                    entity_embedding = np.zeros(embedding.embedding_size)\n                    individual_embeddings[embedding_name] = entity_embedding\n                    print(f\"Warning:  SMILES not found for {sm_name} in {embedding_name}. Using zero embedding.\")\n            \n            # Fill with zeros if no embedding was found\n            if entity_embedding is None:\n                entity_embedding = np.zeros(embedding.embedding_size)\n                print(f\"Using zero embedding for {cell_type}/{sm_name} in {embedding_name}\")\n\n            all_values.append(entity_embedding)\n            all_columns.extend(embedding_info['columns'])\n\n        combined_vector = np.concatenate(all_values)\n        result_df = pd.DataFrame([combined_vector], columns=all_columns)\n        return result_df, individual_embeddings\n    def generate_expert_config(self, expert_specs):\n        \"\"\"\n        Generate expert configuration based on embedding combinations.\n\n        Args:\n            expert_specs (dict): Dictionary mapping expert names to lists of embedding names\n\n        Returns:\n            dict: Expert configuration with corresponding feature indices\n        \"\"\"\n        expert_config = {}\n        for expert, embedding_names in expert_specs.items():\n            indices = []\n            for embedding_name in embedding_names:\n                if embedding_name in self.embedding_indices:\n                    indices.extend(range(\n                        self.embedding_indices[embedding_name]['start'],\n                        self.embedding_indices[embedding_name]['end']\n                    ))\n                else:\n                    raise KeyError(f\"Embedding '{embedding_name}' not found in preprocessor\")\n            expert_config[expert] = indices\n        return expert_config\n\n    def get_embedding_info(self):\n        \"\"\"\n        Get information about available embeddings and their ranges.\n\n        Returns:\n            dict: Dictionary containing embedding information including column names\n        \"\"\"\n        return {\n            embedding_name: {\n                'size': self.embedding_indices[embedding_name]['end'] -\n                       self.embedding_indices[embedding_name]['start'],\n                'start_index': self.embedding_indices[embedding_name]['start'],\n                'end_index': self.embedding_indices[embedding_name]['end'],\n                'columns': self.embedding_indices[embedding_name]['columns']\n            }\n            for embedding_name in self.embedding_indices\n        }\n\n    def get_smiles_mapping(self):\n        \"\"\"\n        Get the dictionary mapping sm_names to SMILES strings.\n\n        Returns:\n            dict: Dictionary containing sm_name to SMILES mapping\n        \"\"\"\n        return self.sm_name_to_smiles.copy()","metadata":{"execution":{"iopub.status.busy":"2025-02-09T19:32:03.662108Z","iopub.execute_input":"2025-02-09T19:32:03.662425Z","iopub.status.idle":"2025-02-09T19:32:03.679244Z","shell.execute_reply.started":"2025-02-09T19:32:03.662396Z","shell.execute_reply":"2025-02-09T19:32:03.678510Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# onehot_embedding = OneHotEmbedding(columns=['cell_type', 'sm_name'])\ntarget_embedding = TargetEmbedding(df)\nsmiles_embedding = SMILESEmbedding()\nmol2vec_embedding = Mol2VecEmbedding('/kaggle/input/mol2vec/pytorch/default/1/model_300dim.pkl')\nmorgan_embedding = MorganFingerPrintEmbedding(hidden_size=256)\nchemberta_embedding = ChemBERTaEmbedding(embedding_type='mean_pooling', padding=False)\nmolformer_embedding = MolformerEmbedding(model_path='ibm/MoLFormer-XL-both-10pct', padding=True)\nsmolebart_embedding = SmoleBartEmbedding(model_path=\"UdS-LSV/smole-bart\")\nmodel_path = \"/kaggle/input/selfformer/transformers/default/1\"\n\nselformer_embedding = SelfFormerEmbedding(\n    model_name=model_path, \n    device=\"cuda\",  # Ensures GPU usage\n    batch_size=16\n)\n\n","metadata":{"execution":{"iopub.status.busy":"2025-02-09T19:32:21.022760Z","iopub.execute_input":"2025-02-09T19:32:21.023104Z","iopub.status.idle":"2025-02-09T19:32:35.101179Z","shell.execute_reply.started":"2025-02-09T19:32:21.023075Z","shell.execute_reply":"2025-02-09T19:32:35.100184Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preprocessor_train = Preprocessor([target_embedding, smiles_embedding, mol2vec_embedding, morgan_embedding,\n    chemberta_embedding, molformer_embedding,smolebart_embedding,selformer_embedding ])","metadata":{"execution":{"iopub.status.busy":"2025-02-09T19:32:48.379099Z","iopub.execute_input":"2025-02-09T19:32:48.379436Z","iopub.status.idle":"2025-02-09T19:32:48.384325Z","shell.execute_reply.started":"2025-02-09T19:32:48.379405Z","shell.execute_reply":"2025-02-09T19:32:48.383373Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_cols = ['cell_type','sm_name','sm_lincs_id','SMILES','control','SELFIES']\ntargets = df.drop(columns=target_cols)\nprocessed_df_train = preprocessor_train.preprocess(df, fit=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-09T19:32:50.596499Z","iopub.execute_input":"2025-02-09T19:32:50.596795Z","iopub.status.idle":"2025-02-09T19:36:02.163532Z","shell.execute_reply.started":"2025-02-09T19:32:50.596771Z","shell.execute_reply":"2025-02-09T19:36:02.162804Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# baselines","metadata":{}},{"cell_type":"code","source":"processed_df_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-09T19:36:02.164704Z","iopub.execute_input":"2025-02-09T19:36:02.164957Z","iopub.status.idle":"2025-02-09T19:36:02.204395Z","shell.execute_reply.started":"2025-02-09T19:36:02.164936Z","shell.execute_reply":"2025-02-09T19:36:02.203490Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom sklearn.neural_network import MLPRegressor\nfrom sklearn.decomposition import TruncatedSVD\nfrom sklearn.metrics import mean_squared_error, r2_score\nfrom sklearn.model_selection import train_test_split, GridSearchCV\n\n# Prepare data\nX = processed_df_train\ny = targets.values\n\n# Train-test split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Dimensionality reduction on the target variable y\ntsvd = TruncatedSVD(n_components=50)\ny_train_reduced = tsvd.fit_transform(y_train)\ny_test_reduced = tsvd.transform(y_test)\n\n # Train-test split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Dimensionality reduction on the target variable y \ntsvd = TruncatedSVD(n_components=50)\ny_train_reduced = tsvd.fit_transform(y_train)\ny_test_reduced = tsvd.transform(y_test)\n\n#best parameters using grid Search \nmodels = {\n    'Linear Regression': LinearRegression(),\n    'Decision Tree Regressor': DecisionTreeRegressor(\n        max_depth=7,\n        min_samples_split=10\n    ),\n    'K-Nearest Neighbors Regressor': KNeighborsRegressor(\n        n_neighbors=7,\n        weights='uniform'\n    ),\n    'Neural Network Regressor': MLPRegressor(\n        hidden_layer_sizes=(64, 32),\n        alpha=0.01,\n        max_iter=1000\n    )\n}\n\n# Train and evaluate models\nmodel_results = {}\nfor name, model in models.items():\n    print(f\"Training {name}...\")\n    \n    # Fit the model\n    model.fit(X_train, y_train_reduced)\n    \n    # Make predictions\n    train_preds_reduced = model.predict(X_train)\n    test_preds_reduced = model.predict(X_test)\n    \n    # Inverse transform the predictions to get the original scale\n    train_preds = tsvd.inverse_transform(train_preds_reduced)\n    test_preds = tsvd.inverse_transform(test_preds_reduced)\n    \n    # Calculate metrics\n    train_mse = mean_squared_error(y_train, train_preds)\n    test_mse = mean_squared_error(y_test, test_preds)\n    train_r2 = r2_score(y_train, train_preds)\n    test_r2 = r2_score(y_test, test_preds)\n    \n    model_results[name] = {\n        'train_mse': train_mse,\n        'test_mse': test_mse,\n        'train_r2': train_r2,\n        'test_r2': test_r2,\n        'model': model  # store the trained model\n    }\n    \n    print(f\"{name} Results:\")\n    print(f\"Train MSE: {train_mse:.4f}, Test MSE: {test_mse:.4f}\")\n    print(f\"Train R^2: {train_r2:.4f}, Test R^2: {test_r2:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-08T16:51:45.315968Z","iopub.execute_input":"2025-02-08T16:51:45.316312Z","iopub.status.idle":"2025-02-08T16:52:12.822062Z","shell.execute_reply.started":"2025-02-08T16:51:45.316283Z","shell.execute_reply":"2025-02-08T16:52:12.821153Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}