{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%pip install rdkit","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:19:44.073115Z","iopub.execute_input":"2023-09-16T04:19:44.073550Z","iopub.status.idle":"2023-09-16T04:20:02.794973Z","shell.execute_reply.started":"2023-09-16T04:19:44.073515Z","shell.execute_reply":"2023-09-16T04:20:02.793365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:yellow;display:inline-block;border-radius:5px;background-color:#007BA7;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:yellow;overflow:hidden;font-size:85%;letter-spacing:0.5px;margin:0\"><b> </b> Import Modules</p></div>","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom rdkit import Chem\nfrom rdkit.Chem import Draw\nfrom rdkit.Chem import PandasTools\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.preprocessing import StandardScaler\n\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport math\n\nrc = {\n    \"axes.facecolor\": \"#E6FFE6\",\n    \"figure.facecolor\": \"#E6FFE6\",\n    \"axes.edgecolor\": \"#000000\",\n    \"grid.color\": \"#EBEBE7\",\n    \"font.family\": \"serif\",\n    \"axes.labelcolor\": \"#000000\",\n    \"xtick.color\": \"#000000\",\n    \"ytick.color\": \"#000000\",\n    \"grid.alpha\": 0.4\n}\n\nsns.set(rc=rc)\n\nfrom colorama import Style, Fore\nred = Style.BRIGHT + Fore.RED\nblu = Style.BRIGHT + Fore.BLUE\nmgt = Style.BRIGHT + Fore.MAGENTA\ngld = Style.BRIGHT + Fore.YELLOW\nres = Style.RESET_ALL","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:20:07.204166Z","iopub.execute_input":"2023-09-16T04:20:07.204671Z","iopub.status.idle":"2023-09-16T04:20:09.128466Z","shell.execute_reply.started":"2023-09-16T04:20:07.204632Z","shell.execute_reply":"2023-09-16T04:20:09.126850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:yellow;display:inline-block;border-radius:5px;background-color:#007BA7;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:yellow;overflow:hidden;font-size:85%;letter-spacing:0.5px;margin:0\"><b> </b> Load the Dataset</p></div>","metadata":{}},{"cell_type":"code","source":"d_train = pd.read_parquet('/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet')\nd_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:20:15.127689Z","iopub.execute_input":"2023-09-16T04:20:15.128592Z","iopub.status.idle":"2023-09-16T04:20:18.159928Z","shell.execute_reply.started":"2023-09-16T04:20:15.128537Z","shell.execute_reply":"2023-09-16T04:20:18.158603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"da_train = pd.read_parquet('/kaggle/input/open-problems-single-cell-perturbations/adata_train.parquet')\nda_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:20:22.602504Z","iopub.execute_input":"2023-09-16T04:20:22.603019Z","iopub.status.idle":"2023-09-16T04:21:47.515150Z","shell.execute_reply.started":"2023-09-16T04:20:22.602980Z","shell.execute_reply":"2023-09-16T04:21:47.513662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map = pd.read_csv('/kaggle/input/open-problems-single-cell-perturbations/id_map.csv')\nid_map.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:21:50.349739Z","iopub.execute_input":"2023-09-16T04:21:50.350208Z","iopub.status.idle":"2023-09-16T04:21:50.386693Z","shell.execute_reply.started":"2023-09-16T04:21:50.350173Z","shell.execute_reply":"2023-09-16T04:21:50.385100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = pd.read_csv('/kaggle/input/open-problems-single-cell-perturbations/sample_submission.csv')\nsubmission_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:21:54.951681Z","iopub.execute_input":"2023-09-16T04:21:54.952228Z","iopub.status.idle":"2023-09-16T04:22:00.019603Z","shell.execute_reply.started":"2023-09-16T04:21:54.952185Z","shell.execute_reply":"2023-09-16T04:22:00.018427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:22:04.937016Z","iopub.execute_input":"2023-09-16T04:22:04.938654Z","iopub.status.idle":"2023-09-16T04:22:04.947530Z","shell.execute_reply.started":"2023-09-16T04:22:04.938595Z","shell.execute_reply":"2023-09-16T04:22:04.946015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d_train.info()","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:22:07.497119Z","iopub.execute_input":"2023-09-16T04:22:07.497609Z","iopub.status.idle":"2023-09-16T04:22:08.692108Z","shell.execute_reply.started":"2023-09-16T04:22:07.497575Z","shell.execute_reply":"2023-09-16T04:22:08.690524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d_train.columns","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:22:11.414690Z","iopub.execute_input":"2023-09-16T04:22:11.415342Z","iopub.status.idle":"2023-09-16T04:22:11.428457Z","shell.execute_reply.started":"2023-09-16T04:22:11.415290Z","shell.execute_reply":"2023-09-16T04:22:11.427099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for missing values in id_map\nid_map_missing = id_map.isnull().sum()\nprint(\"Missing values in id_map:\")\nprint(id_map_missing)","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:22:15.521352Z","iopub.execute_input":"2023-09-16T04:22:15.521774Z","iopub.status.idle":"2023-09-16T04:22:15.532158Z","shell.execute_reply.started":"2023-09-16T04:22:15.521741Z","shell.execute_reply":"2023-09-16T04:22:15.530547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for missing values in data\nd_train_missing = d_train.isnull().sum()\nprint(\"\\nMissing values in d_train:\")\nprint(d_train_missing)","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:22:18.626325Z","iopub.execute_input":"2023-09-16T04:22:18.626848Z","iopub.status.idle":"2023-09-16T04:22:18.679819Z","shell.execute_reply.started":"2023-09-16T04:22:18.626807Z","shell.execute_reply":"2023-09-16T04:22:18.678521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop duplicate rows based on all columns\nd_train.drop_duplicates(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:22:25.261676Z","iopub.execute_input":"2023-09-16T04:22:25.263074Z","iopub.status.idle":"2023-09-16T04:22:29.092739Z","shell.execute_reply.started":"2023-09-16T04:22:25.263022Z","shell.execute_reply":"2023-09-16T04:22:29.091653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d_train.describe().style.background_gradient(cmap='tab20c')","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:22:32.001232Z","iopub.execute_input":"2023-09-16T04:22:32.001951Z","iopub.status.idle":"2023-09-16T04:23:41.044509Z","shell.execute_reply.started":"2023-09-16T04:22:32.001906Z","shell.execute_reply":"2023-09-16T04:23:41.042301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:yellow;display:inline-block;border-radius:5px;background-color:#007BA7;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:yellow;overflow:hidden;font-size:85%;letter-spacing:0.5px;margin:0\"><b> </b>Exploratory Data Analysis (EDA)</p></div>","metadata":{}},{"cell_type":"code","source":"# Count the occurrences of each cell type\ncell_type_counts = d_train['cell_type'].value_counts()\n\n# Create a pie chart\nplt.figure(figsize=(10, 6))\nplt.pie(cell_type_counts, labels=cell_type_counts.index, autopct='%1.1f%%', startangle=140, colors=['#ff9999','#66b3ff','#99ff99','#ffcc99','#c2c2f0'])\nplt.title('Distribution of Cell Types', fontsize = 14, fontweight = 'bold', color = 'darkgreen')\nplt.axis('equal')  # Equal aspect ratio ensures the pie chart is circular.\nplt.savefig('Distribution of Cell Types.png')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:25:18.667467Z","iopub.execute_input":"2023-09-16T04:25:18.668205Z","iopub.status.idle":"2023-09-16T04:25:19.074582Z","shell.execute_reply.started":"2023-09-16T04:25:18.668165Z","shell.execute_reply":"2023-09-16T04:25:19.073188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.histplot(data=d_train, x='A1BG', hue='cell_type', kde=True)\nplt.title('Distribution of A1BG Gene Expression', fontsize = 14, fontweight = 'bold', color = 'darkgreen')\nplt.savefig('Distribution of A1BG Gene Expression.png')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:25:28.432533Z","iopub.execute_input":"2023-09-16T04:25:28.432939Z","iopub.status.idle":"2023-09-16T04:25:31.941351Z","shell.execute_reply.started":"2023-09-16T04:25:28.432910Z","shell.execute_reply":"2023-09-16T04:25:31.940223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to convert SMILES to RDKit molecule\ndef smiles_to_mol(smiles):\n    return Chem.MolFromSmiles(smiles)\n\n# Function to draw molecule from RDKit molecule object\ndef draw_molecule(mol, title, save_path):\n    img = Draw.MolToImage(mol)\n    img = img.resize((400, 400))  # Resize to a reasonable size\n    img.save(save_path)  # Save the image\n    plt.figure(figsize=(4, 4))\n    plt.imshow(img)\n    plt.title(title)\n    plt.axis('off')\n    plt.show()\n\n\n# Randomly select 10 different images\nsampled_df = d_train.sample(n=10, random_state=42)\n\n# Example usage of the functions\nfor i, row in sampled_df.iterrows():\n    mol = smiles_to_mol(row['SMILES'])\n    if mol:\n        image_path = f'{row[\"sm_name\"]}.png'  # Define the image path\n        draw_molecule(mol, row['sm_name'], image_path)","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:25:36.129149Z","iopub.execute_input":"2023-09-16T04:25:36.129561Z","iopub.status.idle":"2023-09-16T04:25:39.210391Z","shell.execute_reply.started":"2023-09-16T04:25:36.129530Z","shell.execute_reply":"2023-09-16T04:25:39.208843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a heatmap\nplt.figure(figsize=(10, 8))\nsns.heatmap(d_train.drop(['cell_type', 'sm_name', 'sm_lincs_id', 'SMILES', 'control'], axis=1))\nplt.title('Gene Expression Heatmap', fontsize = 14, fontweight = 'bold', color = 'darkgreen')\nplt.savefig('Gene Expression Heatmap.png')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:25:49.501458Z","iopub.execute_input":"2023-09-16T04:25:49.501899Z","iopub.status.idle":"2023-09-16T04:26:16.598154Z","shell.execute_reply.started":"2023-09-16T04:25:49.501865Z","shell.execute_reply":"2023-09-16T04:26:16.596837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a scatter plot\nplt.figure(figsize=(10, 6))\nsns.scatterplot(x='A1BG', y='A2M', data=d_train, hue='cell_type')\nplt.title('A1BG vs A2M Gene Expression', fontsize = 14, fontweight = 'bold', color = 'darkgreen')\nplt.savefig('A1BG vs A2M Gene Expression.png')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:26:22.562757Z","iopub.execute_input":"2023-09-16T04:26:22.563558Z","iopub.status.idle":"2023-09-16T04:26:23.527188Z","shell.execute_reply.started":"2023-09-16T04:26:22.563509Z","shell.execute_reply":"2023-09-16T04:26:23.525769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a box plot\nplt.figure(figsize=(10, 6))\nsns.boxplot(x='cell_type', y='A1BG', data=d_train)\nplt.title('A1BG Gene Expression Across Cell Types', fontsize = 14, fontweight = 'bold', color = 'darkgreen')\nplt.xticks(rotation=45)\nplt.savefig('A1BG Gene Expression Across Cell Types.png')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:26:27.617047Z","iopub.execute_input":"2023-09-16T04:26:27.617574Z","iopub.status.idle":"2023-09-16T04:26:28.151204Z","shell.execute_reply.started":"2023-09-16T04:26:27.617535Z","shell.execute_reply":"2023-09-16T04:26:28.149682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a violin plot\nplt.figure(figsize=(10, 6))\nsns.violinplot(x='cell_type', y='A2M', data=d_train)\nplt.title('A2M Gene Expression Across Cell Types', fontsize = 14, fontweight = 'bold', color = 'darkgreen')\nplt.xticks(rotation=45)\nplt.savefig('A2M Gene Expression Across Cell Types.png')\nplt.show()\n","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-09-16T04:26:32.963604Z","iopub.execute_input":"2023-09-16T04:26:32.964070Z","iopub.status.idle":"2023-09-16T04:26:33.730549Z","shell.execute_reply.started":"2023-09-16T04:26:32.964037Z","shell.execute_reply":"2023-09-16T04:26:33.729165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.pairplot(d_train[['A1BG', 'A2M', 'A2M-AS1', 'A2MP1']])\nplt.suptitle('Pair Plot of Gene Expressions', fontsize = 14, fontweight = 'bold', color = 'darkgreen')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:26:39.046177Z","iopub.execute_input":"2023-09-16T04:26:39.046610Z","iopub.status.idle":"2023-09-16T04:26:46.869112Z","shell.execute_reply.started":"2023-09-16T04:26:39.046577Z","shell.execute_reply":"2023-09-16T04:26:46.867890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(6, 4))\nsns.heatmap(d_train[['A1BG', 'A2M', 'A2M-AS1', 'A2MP1']].corr(), annot=True, cmap='coolwarm')\nplt.title('Correlation Heatmap', fontsize = 14, fontweight = 'bold', color = 'darkgreen')\nplt.savefig('Correlation Heatmap.png')\nplt.show()\n","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-09-16T04:27:07.473485Z","iopub.execute_input":"2023-09-16T04:27:07.474084Z","iopub.status.idle":"2023-09-16T04:27:08.066918Z","shell.execute_reply.started":"2023-09-16T04:27:07.474042Z","shell.execute_reply":"2023-09-16T04:27:08.065671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nsns.clustermap(d_train.drop(['cell_type', 'sm_name', 'sm_lincs_id', 'SMILES', 'control'], axis=1), cmap='coolwarm')\nplt.title('Gene Expression Clustermap', fontsize = 14, fontweight = 'bold', color = 'darkgreen')\nplt.savefig('Gene Expression Clustermap.png')\nplt.show()\n","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-09-16T04:27:14.703779Z","iopub.execute_input":"2023-09-16T04:27:14.704244Z","iopub.status.idle":"2023-09-16T04:30:00.092809Z","shell.execute_reply.started":"2023-09-16T04:27:14.704212Z","shell.execute_reply":"2023-09-16T04:30:00.091081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from rdkit import Chem\n\n# Define the substructure pattern using SMARTS notation\npatt = Chem.MolFromSmarts('ClccccF')\n\n# The `d_train` is your pandas DataFrame containing SMILES strings in a column named 'SMILES'\nsubms = []\n\nfor smi in d_train['SMILES']:\n    mol = Chem.MolFromSmiles(smi)\n    if mol:\n        hit_ats = list(mol.GetSubstructMatch(patt))\n        if hit_ats:\n            subms.append(mol)\n\n# Print the number of molecules with the substructure\nprint(len(subms))\n","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:30:05.562285Z","iopub.execute_input":"2023-09-16T04:30:05.562762Z","iopub.status.idle":"2023-09-16T04:30:05.736854Z","shell.execute_reply.started":"2023-09-16T04:30:05.562727Z","shell.execute_reply":"2023-09-16T04:30:05.735317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:yellow;display:inline-block;border-radius:5px;background-color:#007BA7;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:yellow;overflow:hidden;font-size:85%;letter-spacing:0.5px;margin:0\"><b> </b>Chemical Reaction Computations</p></div>\n\n* Use RDKit to perform a simple reaction computation using SMILES ","metadata":{}},{"cell_type":"code","source":"from rdkit import Chem\n\n# Define a function to perform a chemical reaction\ndef perform_reaction(reactant_smiles, product_smiles, reaction_type):\n    # Convert SMILES strings to RDKit molecules\n    reactant = Chem.MolFromSmiles(reactant_smiles)\n    product = Chem.MolFromSmiles(product_smiles)\n    \n    # This is just an example reaction (dummy reaction for illustration purposes)\n    reaction = Chem.CombineMols(reactant, product)\n    \n    return Chem.MolToSmiles(reaction)\n\n# Example usage\nreactant_smiles = \"CC(=O)OC1=CC=CC=C1C(=O)O\"\nproduct_smiles = \"CC(=O)OC1=CC=CC=C1C(=O)OC\"\n\nresult_smiles = perform_reaction(reactant_smiles, product_smiles, 'example_reaction')\nprint(f\"Resulting SMILES string: {result_smiles}\")","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:30:11.131472Z","iopub.execute_input":"2023-09-16T04:30:11.131958Z","iopub.status.idle":"2023-09-16T04:30:11.142058Z","shell.execute_reply.started":"2023-09-16T04:30:11.131922Z","shell.execute_reply":"2023-09-16T04:30:11.140478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:yellow;display:inline-block;border-radius:5px;background-color:#007BA7;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:yellow;overflow:hidden;font-size:85%;letter-spacing:0.5px;margin:0\"><b> </b>Chemical Property Calculation</p></div>\n\n\n* Compute various chemical properties like molecular weight, logP, etc., for each compound.","metadata":{}},{"cell_type":"code","source":"from rdkit import Chem\nfrom rdkit.Chem import Descriptors\nimport pandas as pd\n\n# DataFrame 'd_train with a column 'SMILES'\nd_train['Molecule'] = d_train['SMILES'].apply(Chem.MolFromSmiles)\n\n# Check if the 'Molecule' column is correctly created\nprint(d_train.head())\n\n# Calculate molecular weight for each compound\nd_train['MolecularWeight'] = d_train['Molecule'].apply(Descriptors.MolWt)\n\n# Print the molecular weights\nprint(d_train['MolecularWeight'])","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:30:17.870541Z","iopub.execute_input":"2023-09-16T04:30:17.870992Z","iopub.status.idle":"2023-09-16T04:30:18.126250Z","shell.execute_reply.started":"2023-09-16T04:30:17.870948Z","shell.execute_reply":"2023-09-16T04:30:18.124690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:yellow;display:inline-block;border-radius:5px;background-color:#007BA7;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:yellow;overflow:hidden;font-size:85%;letter-spacing:0.5px;margin:0\"><b> </b>Chemical Similarity and Clustering</p></div>\n\n\n* Use RDKit to calculate chemical similarities and a clustering algorithm to group similar compounds.\n\n","metadata":{}},{"cell_type":"code","source":"from rdkit import Chem\nfrom rdkit.Chem import DataStructs\nfrom rdkit.Chem import MACCSkeys\nfrom sklearn.cluster import AgglomerativeClustering\n\n# Convert SMILES to RDKit molecules\nd_train['Molecule'] = d_train['SMILES'].apply(Chem.MolFromSmiles)\n\n# Calculate MACCS fingerprints for each molecule\nd_train['Fingerprint'] = d_train['Molecule'].apply(MACCSkeys.GenMACCSKeys)\n\n# Calculate Tanimoto similarity between fingerprints\nsimilarity_matrix = []\nfor i, fp1 in enumerate(d_train['Fingerprint']):\n    similarities = [DataStructs.FingerprintSimilarity(fp1, fp2) for fp2 in d_train['Fingerprint']]\n    similarity_matrix.append(similarities)\n\n# Perform hierarchical clustering\nclustering = AgglomerativeClustering(n_clusters=3).fit(similarity_matrix)\nd_train['Cluster'] = clustering.labels_\n\n# Print the clustering labels\nprint(\"Clustering Labels:\")\nprint(d_train['Cluster'])","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:30:24.321320Z","iopub.execute_input":"2023-09-16T04:30:24.322676Z","iopub.status.idle":"2023-09-16T04:30:27.903744Z","shell.execute_reply.started":"2023-09-16T04:30:24.322620Z","shell.execute_reply":"2023-09-16T04:30:27.902846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:yellow;display:inline-block;border-radius:5px;background-color:#007BA7;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:yellow;overflow:hidden;font-size:85%;letter-spacing:0.5px;margin:0\"><b> </b>Enrichment Analysis</p></div>\n\n* Enrichment analysis typically involves comparing a set of compounds against a database of chemical classes or substructures to identify over-represented features.\n\n    * Example: Enrichment Analysis for Substructure \"Aromatic Rings\"","metadata":{}},{"cell_type":"code","source":"from rdkit.Chem import MACCSkeys\n\n# Define a SMARTS pattern for aromatic rings\nsmarts_pattern = Chem.MolFromSmarts('[c,C]')\n\n# Calculate MACCS fingerprints for each molecule\nd_train['Fingerprint'] = d_train['Molecule'].apply(MACCSkeys.GenMACCSKeys)\n\n# Count the number of compounds with aromatic rings\nd_train['Has_Aromatic_Ring'] = d_train['Molecule'].apply(lambda mol: mol.HasSubstructMatch(smarts_pattern))\n\n# Perform enrichment analysis\nenrichment_count = d_train['Has_Aromatic_Ring'].sum()\ntotal_compounds = len(d_train)\nenrichment_ratio = enrichment_count / total_compounds\n\nprint(f\"Enrichment of Aromatic Rings: {enrichment_ratio:.2%}\")","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:30:32.655073Z","iopub.execute_input":"2023-09-16T04:30:32.655536Z","iopub.status.idle":"2023-09-16T04:30:33.592887Z","shell.execute_reply.started":"2023-09-16T04:30:32.655503Z","shell.execute_reply":"2023-09-16T04:30:33.591467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:yellow;display:inline-block;border-radius:5px;background-color:#007BA7;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:yellow;overflow:hidden;font-size:85%;letter-spacing:0.5px;margin:0\"><b> </b>Statistical Testing</p></div>\n\n* Statistical testing involves comparing groups of compounds to identify significant differences. This requires specific hypotheses and appropriate statistical tests.\n\n    * Example: t-test for Difference in Means","metadata":{}},{"cell_type":"code","source":"from scipy.stats import ttest_ind\n\ngroup_1 = d_train[d_train['cell_type'] == 'NK cells']\ngroup_2 = d_train[d_train['cell_type'] == 'T cells CD4+']\n\nstatistic, p_value = ttest_ind(group_1['A1BG'], group_2['A1BG'])\n\nif p_value < 0.05:\n    print(f\"Statistically significant difference (p-value = {p_value:.4f})\")\nelse:\n    print(f\"No statistically significant difference (p-value = {p_value:.4f})\")\n","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:30:39.400688Z","iopub.execute_input":"2023-09-16T04:30:39.401310Z","iopub.status.idle":"2023-09-16T04:30:39.442339Z","shell.execute_reply.started":"2023-09-16T04:30:39.401262Z","shell.execute_reply":"2023-09-16T04:30:39.441388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# List of columns to drop\ncolumns_to_drop = ['Molecule', 'MolecularWeight', 'Fingerprint', 'Cluster', 'Has_Aromatic_Ring']\n\n# Drop the specified columns\nd_train = d_train.drop(columns=columns_to_drop)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:30:47.528649Z","iopub.execute_input":"2023-09-16T04:30:47.529057Z","iopub.status.idle":"2023-09-16T04:30:47.580546Z","shell.execute_reply.started":"2023-09-16T04:30:47.529025Z","shell.execute_reply":"2023-09-16T04:30:47.579228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d_train = pd.read_parquet('/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet')","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:31:15.306932Z","iopub.execute_input":"2023-09-16T04:31:15.307466Z","iopub.status.idle":"2023-09-16T04:31:18.022170Z","shell.execute_reply.started":"2023-09-16T04:31:15.307429Z","shell.execute_reply":"2023-09-16T04:31:18.020257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop duplicate rows based on all columns\nd_train.drop_duplicates(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:31:23.166731Z","iopub.execute_input":"2023-09-16T04:31:23.168375Z","iopub.status.idle":"2023-09-16T04:31:28.642055Z","shell.execute_reply.started":"2023-09-16T04:31:23.168301Z","shell.execute_reply":"2023-09-16T04:31:28.640488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# List of SMILES strings to drop\nsmiles_to_drop = [\n    'SMILES_O=C(c1ccc(Nc2nccc(-c3cc4ccccc4s3)n2)cc1)N1CCC(N2CCCC2)CC1',\n    'SMILES_O=C(c1ccc(OCCN2CCCCC2)cc1)c1c(-c2ccc(O)cc2)sc2cc(O)ccc12',\n    'SMILES_O=C1CC2(CCCC2)CC(=O)N1CCCCN1CCN(c2ncccn2)CC1',\n    'SMILES_O=C1NC(=O)C(c2cnc3ccccn23)=C1c1cn2c3c(cc(F)cc13)CN(C(=O)N1CCCCC1)CC2',\n    'SMILES_O=C1NC(=O)[C@@H](c2cn3c4c(cccc24)CCC3)[C@@H]1c1c[nH]c2ccccc12',\n    'SMILES_O=C1c2cccc3cccc(c23)C(=O)N1CCCCCC(O)=NO',\n    'SMILES_OC1(c2ccc(Cl)c(C(F)(F)F)c2)CCN(CCCC(c2ccc(F)cc2)c2ccc(F)cc2)CC1',\n    'SMILES_OCCCNc1cc(-c2ccnc(Nc3cccc(Cl)c3)n2)ccn1',\n    'SMILES_c1cc(OCCN2CCCCC2)cc(-c2[nH]nc3ccc(-c4nc[nH]n4)cc23)c1',\n    'SMILES_c1ccc2c(-c3cnn4cc(-c5ccc(N6CCNCC6)cc5)cnc34)ccnc2c1'\n]\n\n# Create a mask for rows to drop\nmask = ~d_train['SMILES'].isin(smiles_to_drop)\n\n# Apply the mask to the DataFrame\nd_train = d_train[mask]","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-09-16T04:31:34.273376Z","iopub.execute_input":"2023-09-16T04:31:34.274728Z","iopub.status.idle":"2023-09-16T04:31:34.345401Z","shell.execute_reply.started":"2023-09-16T04:31:34.274677Z","shell.execute_reply":"2023-09-16T04:31:34.343792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:yellow;display:inline-block;border-radius:5px;background-color:#007BA7;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:yellow;overflow:hidden;font-size:85%;letter-spacing:0.5px;margin:0\"><b> </b>Build Model and Prediction</p></div>","metadata":{}},{"cell_type":"code","source":"target_column = 'A1BG'\nX_train = d_train.drop(columns=[target_column])  # Assuming 'd_train' is your DataFrame\ny_train = d_train[target_column]","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:31:45.635873Z","iopub.execute_input":"2023-09-16T04:31:45.638402Z","iopub.status.idle":"2023-09-16T04:31:45.701436Z","shell.execute_reply.started":"2023-09-16T04:31:45.638345Z","shell.execute_reply":"2023-09-16T04:31:45.699661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d_train = d_train.drop(columns=['cell_type', 'sm_name'])","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:31:52.576969Z","iopub.execute_input":"2023-09-16T04:31:52.577507Z","iopub.status.idle":"2023-09-16T04:31:52.634354Z","shell.execute_reply.started":"2023-09-16T04:31:52.577472Z","shell.execute_reply":"2023-09-16T04:31:52.632788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = d_train.drop('A1BG', axis=1)\ny_train = d_train['A1BG']","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:31:57.150668Z","iopub.execute_input":"2023-09-16T04:31:57.151169Z","iopub.status.idle":"2023-09-16T04:31:57.217824Z","shell.execute_reply.started":"2023-09-16T04:31:57.151120Z","shell.execute_reply":"2023-09-16T04:31:57.216310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_columns = d_train.select_dtypes(include=['object']).columns.tolist()\n\n# Perform one-hot encoding\nd_train_encoded = pd.get_dummies(d_train, columns=categorical_columns)\n\n# Define target variable and features\nX_train = d_train_encoded.drop('A1BG', axis=1)\ny_train = d_train_encoded['A1BG']\n\n# Now, you can proceed to fit the model\nmodel = LinearRegression()\nmodel.fit(X_train, y_train)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:32:00.960103Z","iopub.execute_input":"2023-09-16T04:32:00.960612Z","iopub.status.idle":"2023-09-16T04:32:05.306693Z","shell.execute_reply.started":"2023-09-16T04:32:00.960579Z","shell.execute_reply":"2023-09-16T04:32:05.304099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_selection import SelectKBest, mutual_info_regression\n\n# X_train and y_train are your training data\nnum_features_to_select = 100  # Adjust this as needed\nselector = SelectKBest(score_func=mutual_info_regression, k=num_features_to_select)\nX_train_selected = selector.fit_transform(X_train, y_train)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:32:08.355921Z","iopub.execute_input":"2023-09-16T04:32:08.357590Z","iopub.status.idle":"2023-09-16T04:33:39.577492Z","shell.execute_reply.started":"2023-09-16T04:32:08.357543Z","shell.execute_reply":"2023-09-16T04:33:39.576037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import Ridge\n\nalpha = 1.0  # Regularization strength (adjust as needed)\nridge_model = Ridge(alpha=alpha)\nridge_model.fit(X_train_selected, y_train)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:33:43.757037Z","iopub.execute_input":"2023-09-16T04:33:43.757561Z","iopub.status.idle":"2023-09-16T04:33:43.786203Z","shell.execute_reply.started":"2023-09-16T04:33:43.757522Z","shell.execute_reply":"2023-09-16T04:33:43.784097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.decomposition import PCA\n\n#  X_train_selected from feature selection\nnum_components = 100  # Adjust this as needed\npca = PCA(n_components=num_components)\nX_train_pca = pca.fit_transform(X_train_selected)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:33:48.854954Z","iopub.execute_input":"2023-09-16T04:33:48.856485Z","iopub.status.idle":"2023-09-16T04:33:48.884584Z","shell.execute_reply.started":"2023-09-16T04:33:48.856414Z","shell.execute_reply":"2023-09-16T04:33:48.881988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\n\n#  X_train_pca from dimensionality reduction\nrf_model = RandomForestRegressor(n_estimators=100, random_state=42)  # Adjust parameters as needed\nrf_model.fit(X_train_pca, y_train)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:33:53.268522Z","iopub.execute_input":"2023-09-16T04:33:53.269032Z","iopub.status.idle":"2023-09-16T04:33:58.413004Z","shell.execute_reply.started":"2023-09-16T04:33:53.269000Z","shell.execute_reply":"2023-09-16T04:33:58.411437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score\n\n# Perform cross-validation\ncv_scores = cross_val_score(rf_model, X_train_pca, y_train, cv=5, scoring='neg_mean_squared_error')\n\n# Calculate RMSE\nrmse_scores = (-cv_scores)**0.5\nmean_rmse = rmse_scores.mean()\nprint(f\"Mean RMSE: {mean_rmse}\")","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:34:03.487678Z","iopub.execute_input":"2023-09-16T04:34:03.488209Z","iopub.status.idle":"2023-09-16T04:34:22.192782Z","shell.execute_reply.started":"2023-09-16T04:34:03.488108Z","shell.execute_reply":"2023-09-16T04:34:22.191475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mrrmse(y_true, y_pred):\n    rmse = np.sqrt(np.mean((y_true - y_pred)**2))\n    return rmse / np.mean(y_true)\n\n# The X_train_pca, y_train, and rf_model from previous steps\ncv_scores = cross_val_score(rf_model, X_train_pca, y_train, cv=5, scoring=mrrmse)\n\n# Calculate mean MRRMSE\nmean_mrrmse = np.mean(cv_scores)\nprint(f\"Mean MRRMSE: {mean_mrrmse}\")","metadata":{"execution":{"iopub.status.busy":"2023-09-16T04:34:27.051791Z","iopub.execute_input":"2023-09-16T04:34:27.052261Z","iopub.status.idle":"2023-09-16T04:34:45.675270Z","shell.execute_reply.started":"2023-09-16T04:34:27.052227Z","shell.execute_reply":"2023-09-16T04:34:45.674236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# gene_columns is a list of gene names\ngene_columns = X_train.columns.tolist()\n\n# Create a DataFrame with zeros\nsubmission_df = pd.DataFrame(0, index=range(5), columns=['id'] + gene_columns)\n\n# Assign ids to the 'id' column\nsubmission_df['id'] = range(5)\n\n# Save the submission file\nsubmission_df.to_csv('submission.csv', index=False)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\">\"Your positive feedback and upvote mean a lot! It motivates me to create more valuable content and helps others discover it too. Let's build a thriving community of knowledge-sharing. Thank you for your support! 😊\"</div>","metadata":{}}]}