{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <div style=\"color:magenta;display:inline-block;border-radius:5px;background-color:#E6FFE6;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:magenta;overflow:hidden;font-size:90%;letter-spacing:0.5px;margin:0\"><b> </b> Import Modules</p></div>","metadata":{}},{"cell_type":"code","source":"%%capture\n!pip install rdkit","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:30:25.020237Z","iopub.execute_input":"2023-09-25T07:30:25.020790Z","iopub.status.idle":"2023-09-25T07:30:40.336369Z","shell.execute_reply.started":"2023-09-25T07:30:25.020755Z","shell.execute_reply":"2023-09-25T07:30:40.334866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom rdkit import Chem\nfrom rdkit.Chem import Draw\nfrom rdkit.Chem import AllChem\nfrom rdkit.Chem.Draw import IPythonConsole\nfrom rdkit import Chem\nfrom rdkit.Chem import AllChem\nfrom sklearn.manifold import TSNE\nfrom sklearn.decomposition import PCA\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport math\n\nrc = {\n    \"axes.facecolor\": \"#E6FFE6\",\n    \"figure.facecolor\": \"#E6FFE6\",\n    \"axes.edgecolor\": \"#000000\",\n    \"grid.color\": \"#EBEBE7\",\n    \"font.family\": \"serif\",\n    \"axes.labelcolor\": \"#000000\",\n    \"xtick.color\": \"#000000\",\n    \"ytick.color\": \"#000000\",\n    \"grid.alpha\": 0.4\n}\n\nsns.set(rc=rc)\n\nfrom colorama import Style, Fore\nred = Style.BRIGHT + Fore.RED\nblu = Style.BRIGHT + Fore.BLUE\nmgt = Style.BRIGHT + Fore.MAGENTA\ngld = Style.BRIGHT + Fore.YELLOW\nres = Style.RESET_ALL","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:30:48.700013Z","iopub.execute_input":"2023-09-25T07:30:48.700496Z","iopub.status.idle":"2023-09-25T07:30:50.553107Z","shell.execute_reply.started":"2023-09-25T07:30:48.700458Z","shell.execute_reply":"2023-09-25T07:30:50.551833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:magenta;display:inline-block;border-radius:5px;background-color:#E6FFE6;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:magenta;overflow:hidden;font-size:90%;letter-spacing:0.5px;margin:0\"><b> </b>Load data💾</p></div>","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/open-problems-single-cell-perturbations/adata_obs_meta.csv')\ndf.head().style.set_properties(**{'background-color':'lightgreen','color':'black','border-color':'#8b8c8c'})","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:30:56.206870Z","iopub.execute_input":"2023-09-25T07:30:56.207787Z","iopub.status.idle":"2023-09-25T07:30:57.521575Z","shell.execute_reply.started":"2023-09-25T07:30:56.207726Z","shell.execute_reply":"2023-09-25T07:30:57.520278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for missing values\nmissing_values = df.isnull().sum()\nprint(\"Missing Values:\")\nprint(missing_values)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:31:01.788147Z","iopub.execute_input":"2023-09-25T07:31:01.788721Z","iopub.status.idle":"2023-09-25T07:31:01.924699Z","shell.execute_reply.started":"2023-09-25T07:31:01.788686Z","shell.execute_reply":"2023-09-25T07:31:01.923226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:magenta;display:inline-block;border-radius:5px;background-color:#E6FFE6;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:magenta;overflow:hidden;font-size:90%;letter-spacing:0.5px;margin:0\"><b> </b> Exploratory Data Analysis (EDA)📊</p></div>","metadata":{}},{"cell_type":"code","source":"# Count plot for 'donor_id'\nplt.figure(figsize=(10, 6))\nsns.countplot(data=df, x='donor_id', palette='Set3')\nplt.title(\"Distribution of Donors\",fontsize = 15, fontweight = 'bold', color = 'magenta')\nplt.xlabel(\"Donor ID\",fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.ylabel(\"Frequency\",fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.xticks(rotation=45)\nplt.savefig('Distribution of Donors.png')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:31:05.992952Z","iopub.execute_input":"2023-09-25T07:31:05.993347Z","iopub.status.idle":"2023-09-25T07:31:06.647399Z","shell.execute_reply.started":"2023-09-25T07:31:05.993317Z","shell.execute_reply":"2023-09-25T07:31:06.646269Z"},"_kg_hide-input":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"🟧 This plot displays the frequency of each donor in the dataset. Each bar represents a different donor, and the height of the bar indicates how many samples belong to each donor.\n\n🟧 This visualization will give you an overview of how the samples are distributed across different donors in your dataset.","metadata":{}},{"cell_type":"code","source":"#sns.set_theme(style=\"whitegrid\")\n\n# Histogram for 'dose_uM'\nplt.figure(figsize=(10, 6))\nsns.histplot(data=df, x='dose_uM', bins=20, color='skyblue')\nplt.title(\"Distribution of Dose (uM)\",fontsize = 15, fontweight = 'bold', color = 'magenta')\nplt.xlabel(\"Dose (uM)\",fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.ylabel(\"Frequency\",fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.savefig('Distribution of Dose (uM).png')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:31:11.935119Z","iopub.execute_input":"2023-09-25T07:31:11.935812Z","iopub.status.idle":"2023-09-25T07:31:12.633230Z","shell.execute_reply.started":"2023-09-25T07:31:11.935778Z","shell.execute_reply":"2023-09-25T07:31:12.631686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"🟧 This plot shows the distribution of doses (in uM). It gives an idea of how the doses are spread across the dataset.\n","metadata":{}},{"cell_type":"code","source":"# Bar plot for 'timepoint_hr'\nplt.figure(figsize=(10, 6))\nsns.barplot(data=df, x='timepoint_hr', y='dose_uM', errorbar=None, palette='viridis')\nplt.title(\"Average Dose (uM) at Different Timepoints\",fontsize = 15, fontweight = 'bold', color = 'magenta')\nplt.xlabel(\"Timepoint (hours)\",fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.ylabel(\"Average Dose (uM)\",fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.savefig('Average Dose (uM) at Different Timepoints.png')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:31:40.616852Z","iopub.execute_input":"2023-09-25T07:31:40.617281Z","iopub.status.idle":"2023-09-25T07:31:40.992815Z","shell.execute_reply.started":"2023-09-25T07:31:40.617247Z","shell.execute_reply":"2023-09-25T07:31:40.991550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"🟧 This plot displays the average dose (in uM) at different timepoints. It helps visualize how doses vary with time.","metadata":{}},{"cell_type":"code","source":"# Box plot for 'dose_uM'\nplt.figure(figsize=(10, 6))\nsns.boxplot(data=df, y='dose_uM', color='lightcoral')\nplt.title(\"Distribution of Dose (uM)\",fontsize = 15, fontweight = 'bold', color = 'magenta')\nplt.ylabel(\"Dose (uM)\",fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.savefig('Distribution of Dose (uM).png')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:31:45.687158Z","iopub.execute_input":"2023-09-25T07:31:45.687657Z","iopub.status.idle":"2023-09-25T07:31:46.319546Z","shell.execute_reply.started":"2023-09-25T07:31:45.687620Z","shell.execute_reply":"2023-09-25T07:31:46.318426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"🟧 The plot provides a summary of the distribution of doses, showing quartiles, median, and any potential outliers.\n","metadata":{}},{"cell_type":"code","source":"# Pair plot for 'dose_uM' and 'timepoint_hr'\nsns.pairplot(df[['dose_uM', 'timepoint_hr']], height=2, diag_kind='kde', palette='plasma')\nplt.suptitle(\"Dose (uM) and Timepoint (hours)\",fontsize = 15, fontweight = 'bold', color = 'magenta')\nplt.savefig('Dose (uM) and Timepoint (hours).png')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:31:50.159310Z","iopub.execute_input":"2023-09-25T07:31:50.160152Z","iopub.status.idle":"2023-09-25T07:31:56.005006Z","shell.execute_reply.started":"2023-09-25T07:31:50.160116Z","shell.execute_reply":"2023-09-25T07:31:56.003689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"🟧 This plot displays the relationships and distributions of 'dose_uM' and 'timepoint_hr'. The diagonal shows kernel density estimates.\n","metadata":{}},{"cell_type":"code","source":"# Count plot for 'cell_type'\nplt.figure(figsize=(10, 6))\nsns.countplot(data=df, y='cell_type', palette='pastel')\nplt.title(\"Distribution of Cell Types\",fontsize = 15, fontweight = 'bold', color = 'magenta')\nplt.xlabel(\"Frequency\",fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.ylabel(\"Cell Type\",fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.savefig('Distribution of Cell Types.png')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:32:00.835787Z","iopub.execute_input":"2023-09-25T07:32:00.836498Z","iopub.status.idle":"2023-09-25T07:32:01.501563Z","shell.execute_reply.started":"2023-09-25T07:32:00.836461Z","shell.execute_reply":"2023-09-25T07:32:01.500334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"🟧 This plot counts the occurrences of each cell type, giving an overview of the distribution of cell types in the dataset.\nHeatmap for correlations:","metadata":{}},{"cell_type":"code","source":"# Heatmap for correlations\ncorrelation_matrix = df[['dose_uM', 'timepoint_hr']].corr()\nplt.figure(figsize=(8, 6))\nsns.heatmap(correlation_matrix, annot=True, cmap='coolwarm')\nplt.title(\"Correlation Heatmap\",fontsize = 15, fontweight = 'bold', color = 'magenta')\nplt.savefig('Correlation Heatmap.png')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:32:06.943392Z","iopub.execute_input":"2023-09-25T07:32:06.943780Z","iopub.status.idle":"2023-09-25T07:32:07.482762Z","shell.execute_reply.started":"2023-09-25T07:32:06.943746Z","shell.execute_reply":"2023-09-25T07:32:07.481634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"🟧 This heatmap visualizes the correlation between 'dose_uM' and 'timepoint_hr'. Values closer to 1 indicate a stronger positive correlation.\n","metadata":{}},{"cell_type":"code","source":"# Pie chart for 'control'\ncontrol_counts = df['control'].value_counts()\nplt.figure(figsize=(8, 8))\nplt.pie(control_counts, labels=control_counts.index, autopct='%1.1f%%', colors=['#ff9999','#66b3ff'])\nplt.title(\"Control Distribution\",fontsize = 15, fontweight = 'bold', color = 'magenta')\nplt.savefig('Control Distribution.png')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:32:11.740802Z","iopub.execute_input":"2023-09-25T07:32:11.741170Z","iopub.status.idle":"2023-09-25T07:32:11.992368Z","shell.execute_reply.started":"2023-09-25T07:32:11.741142Z","shell.execute_reply":"2023-09-25T07:32:11.991188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"🟧 The pie chart displays the distribution of 'control' values (True or False) as a percentage of the total.\n","metadata":{}},{"cell_type":"code","source":"# Scatter plot for 'dose_uM' vs 'timepoint_hr'\nplt.figure(figsize=(10, 6))\nsns.scatterplot(data=df, x='dose_uM', y='timepoint_hr', hue='control', palette='viridis')\nplt.title(\"Dose (uM) vs Timepoint (hours)\",fontsize = 15, fontweight = 'bold', color = 'magenta')\nplt.xlabel(\"Dose (uM)\",fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.ylabel(\"Timepoint (hours)\",fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.savefig('Dose (uM) vs Timepoint (hours).png')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:32:19.634762Z","iopub.execute_input":"2023-09-25T07:32:19.635621Z","iopub.status.idle":"2023-09-25T07:32:36.230762Z","shell.execute_reply.started":"2023-09-25T07:32:19.635584Z","shell.execute_reply":"2023-09-25T07:32:36.229416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"🟧 This scatter plot visualizes the relationship between 'dose_uM' and 'timepoint_hr'. Different colors represent different control values.\n","metadata":{}},{"cell_type":"code","source":"# Violin plot for 'dose_uM' by 'control'\nplt.figure(figsize=(10, 6))\nsns.violinplot(data=df, x='control', y='dose_uM', palette='husl')\nplt.title(\"Dose (uM) by Control\",fontsize = 15, fontweight = 'bold', color = 'magenta')\nplt.xlabel(\"Control\",fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.ylabel(\"Dose (uM)\",fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.savefig('Dose (uM) by Control.png')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:32:47.792901Z","iopub.execute_input":"2023-09-25T07:32:47.793732Z","iopub.status.idle":"2023-09-25T07:32:48.438112Z","shell.execute_reply.started":"2023-09-25T07:32:47.793686Z","shell.execute_reply":"2023-09-25T07:32:48.436976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"🟧 This plot shows the distribution of doses for each control category, allowing for a comparison of distributions.\n","metadata":{}},{"cell_type":"markdown","source":"# <div style=\"color:magenta;display:inline-block;border-radius:5px;background-color:#E6FFE6;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:magenta;overflow:hidden;font-size:90%;letter-spacing:0.5px;margin:0\"><b> </b>Feature Extraction</p></div>\n\n\n🟦 Convert SMILES strings into a numerical representation suitable for T-SNE and PCA. For simplicity, we'll use Morgan fingerprints.\n\n<h2 style=\"color: #4CAF50;\">Morgan Fingerprints</h2> \n<p style=\"color: darkblue; font-size: 16px;\">Morgan fingerprints,also known as circular fingerprints or Extended Connectivity Fingerprints (ECFP), are a type of molecular representation used in chemoinformatics and computational chemistry. They encode information about the structural features of a molecule.</p>\n\n<p style=\"color: darkblue; font-size: 16px;\">Specifically, Morgan fingerprints are generated by considering the presence or absence of substructures (circular substructures or \"ECFP\" refers to Extended Connectivity of Functional Groups) within a certain radius around each atom in a molecule. The result is a fixed-length binary vector (bitstring) that encodes the presence or absence of specific substructures.</p>\n\n<p style=\"color: darkblue; font-size: 16px;\">These fingerprints are widely used in tasks such as similarity searching, compound clustering, and machine learning-based predictive modeling in the field of chemoinformatics. They are popular because they can be used to represent molecules in a format suitable for machine learning algorithms.</p>\n\n<p style=\"color: darkblue; font-size: 16px;\">The radius parameter in Morgan fingerprints determines how many bonds away from each atom the algorithm considers when generating the fingerprints. Larger radii capture more global structural information, but also result in longer fingerprint vectors.</p>\n\n<p style=\"color: darkblue; font-size: 16px;\">In summary, Morgan fingerprints are a way to represent the structural features of a molecule in a format suitable for computational analysis and machine learning applications in chemistry and drug discovery.</p>\n","metadata":{}},{"cell_type":"markdown","source":"<html>\n<body>\n    <span style=\"color: blue; font-weight: bold; font-size: 20px;\">T-SNE Visualization:</span>\n    <ol>\n       <li style=\"color: purple; font-size: 18px;\">Apply T-SNE to the features for dimensionality reduction and visualization.\n</li>\n    </ol>\n\n\n","metadata":{}},{"cell_type":"code","source":"# Feature Extraction (Morgan Fingerprints)\ndf['Morgan_Fingerprints'] = df['SMILES'].apply(lambda x: AllChem.GetMorganFingerprintAsBitVect(Chem.MolFromSmiles(x), 2, nBits=1024))\n","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:32:55.898164Z","iopub.execute_input":"2023-09-25T07:32:55.898590Z","iopub.status.idle":"2023-09-25T07:34:14.880694Z","shell.execute_reply.started":"2023-09-25T07:32:55.898557Z","shell.execute_reply":"2023-09-25T07:34:14.879532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert 'Morgan_Fingerprints' to a NumPy array\n#import numpy as np\n#fingerprints_array = np.array(df['Morgan_Fingerprints'].tolist())\n\n# T-SNE Visualization\n#tsne = TSNE(n_components=2, random_state=42)\n#features_tsne = tsne.fit_transform(fingerprints_array)\n\n#plt.figure(figsize=(10, 6))\n#sns.scatterplot(x=features_tsne[:, 0], y=features_tsne[:, 1])\n#plt.title(\"T-SNE Visualization\", fontweight = 'bold', color = 'magenta')\n#plt.savefig('T-SNE Visualization.png')\n#plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<html>\n<body>\n    <span style=\"color: blue; font-weight: bold; font-size: 20px;\">PCA Visualization:</span>\n    <ol>\n       <li style=\"color: purple; font-size: 18px;\">Apply PCA to the features for dimensionality reduction and visualization.\n</li>\n    </ol>","metadata":{}},{"cell_type":"code","source":"# PCA Visualization\npca = PCA(n_components=2, random_state=42)\nfeatures_pca = pca.fit_transform(list(df['Morgan_Fingerprints']))\n\nplt.figure(figsize=(10, 6))\nsns.scatterplot(x=features_pca[:, 0], y=features_pca[:, 1])\nplt.title(\"PCA Visualization\", fontweight = 'bold', color = 'magenta')\nplt.savefig('PCA Visualization.png')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:34:20.459351Z","iopub.execute_input":"2023-09-25T07:34:20.459849Z","iopub.status.idle":"2023-09-25T07:43:15.237080Z","shell.execute_reply.started":"2023-09-25T07:34:20.459807Z","shell.execute_reply":"2023-09-25T07:43:15.235786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<html>\n<body>\n    <span style=\"color: blue; font-weight: bold; font-size: 20px;\">Molecule Visualizations:</span>\n    <ol>\n       <li style=\"color: purple; font-size: 18px;\">Generate images for a subset of molecules using RDKit.\n</li>\n    </ol>\n","metadata":{}},{"cell_type":"code","source":"# Add a new column 'Molecule' with RDKit Mol objects\ndf['Molecule'] = df['SMILES'].apply(Chem.MolFromSmiles)\n\n# Visualize a subset of molecules\nsample_molecules = df['Molecule'].sample(4)  # Change 5 to the desired number of molecules to visualize\n\nfor mol in sample_molecules:\n    img = Draw.MolToImage(mol)\n    display(img)","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:43:29.921338Z","iopub.execute_input":"2023-09-25T07:43:29.921752Z","iopub.status.idle":"2023-09-25T07:44:23.039086Z","shell.execute_reply.started":"2023-09-25T07:43:29.921718Z","shell.execute_reply":"2023-09-25T07:44:23.038262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:44:31.378002Z","iopub.execute_input":"2023-09-25T07:44:31.378680Z","iopub.status.idle":"2023-09-25T07:44:31.409442Z","shell.execute_reply.started":"2023-09-25T07:44:31.378642Z","shell.execute_reply":"2023-09-25T07:44:31.408415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Define the train set (T, NK cells + 10% of Myeloid and B cells)\ntrain_set = df[df['cell_type'].isin(['T cells CD4+', 'T cells CD8+', 'Regulatory T cells', 'NK cells'])]\nmyeloid_b_cells = df[df['cell_type'].isin(['B cells naive', 'Myeloid dendritic cells resting'])]\n\n# Check if myeloid_b_cells has any samples\nif len(myeloid_b_cells) > 0:\n    myeloid_b_cells_sampled = myeloid_b_cells.sample(frac=0.1, random_state=42)\n    train_set = pd.concat([train_set, myeloid_b_cells_sampled])\n\n    # Define the public test set (50 randomly selected compounds in B and myeloid cells)\n    public_test_set = myeloid_b_cells.drop(myeloid_b_cells_sampled.index).sample(n=50, random_state=42)\n\n    # Define the private test set (79 randomly selected compounds in B and myeloid cells)\n    private_test_set = myeloid_b_cells.drop(myeloid_b_cells_sampled.index).drop(public_test_set.index)\n\n    # Check shapes of the splits\n    print(f\"Train set shape: {train_set.shape}\")\n    print(f\"Public test set shape: {public_test_set.shape}\")\n    print(f\"Private test set shape: {private_test_set.shape}\")\nelse:\n    print(\"myeloid_b_cells has no samples.\")\n","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:44:44.906134Z","iopub.execute_input":"2023-09-25T07:44:44.906564Z","iopub.status.idle":"2023-09-25T07:44:44.995649Z","shell.execute_reply.started":"2023-09-25T07:44:44.906532Z","shell.execute_reply":"2023-09-25T07:44:44.994510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2 = pd.read_parquet('../input/open-problems-single-cell-perturbations/de_train.parquet')\ndf2.tail()","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:46:13.480043Z","iopub.execute_input":"2023-09-25T07:46:13.481455Z","iopub.status.idle":"2023-09-25T07:46:14.764845Z","shell.execute_reply.started":"2023-09-25T07:46:13.481411Z","shell.execute_reply":"2023-09-25T07:46:14.763380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop the 'SMILES' column\ndf2 = df2.drop(columns=['SMILES'])","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:46:27.164944Z","iopub.execute_input":"2023-09-25T07:46:27.165444Z","iopub.status.idle":"2023-09-25T07:46:27.225155Z","shell.execute_reply.started":"2023-09-25T07:46:27.165401Z","shell.execute_reply":"2023-09-25T07:46:27.223714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:magenta;display:inline-block;border-radius:5px;background-color:#E6FFE6;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:magenta;overflow:hidden;font-size:90%;letter-spacing:0.5px;margin:0\"><b> </b>Build Model and Prediction</p></div>\n","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Assuming 'sm_name' and 'sm_lincs_id' are categorical variables\ndf2 = pd.get_dummies(df2, columns=['sm_name', 'sm_lincs_id','cell_type'])\n\n\n# Separate features and target variable\nX = df2.drop(['control'], axis=1)\ny = df2['control']\n","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:46:31.817117Z","iopub.execute_input":"2023-09-25T07:46:31.817594Z","iopub.status.idle":"2023-09-25T07:46:31.995458Z","shell.execute_reply.started":"2023-09-25T07:46:31.817558Z","shell.execute_reply":"2023-09-25T07:46:31.994015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split the data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:46:39.151220Z","iopub.execute_input":"2023-09-25T07:46:39.151646Z","iopub.status.idle":"2023-09-25T07:46:39.212691Z","shell.execute_reply.started":"2023-09-25T07:46:39.151614Z","shell.execute_reply":"2023-09-25T07:46:39.211626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Build Models\n\n# Linear Regression\nfrom sklearn.linear_model import LinearRegression\nlinear_reg_model = LinearRegression()\n\n# Logistic Regression\nfrom sklearn.linear_model import LogisticRegression\nlogistic_reg_model = LogisticRegression()\n\n# Decision Tree\nfrom sklearn.tree import DecisionTreeClassifier\ndecision_tree_model = DecisionTreeClassifier()\n\n# Random Forest\nfrom sklearn.ensemble import RandomForestClassifier\nrandom_forest_model = RandomForestClassifier()\n\n# Support Vector Machine (SVM)\nfrom sklearn.svm import SVC\nsvm_model = SVC()\n\n# K-Nearest Neighbors (KNN)\nfrom sklearn.neighbors import KNeighborsClassifier\nknn_model = KNeighborsClassifier()\n\n# K-Means Clustering\nfrom sklearn.cluster import KMeans\nkmeans_model = KMeans(n_clusters=2)  # Assuming 2 clusters\n\n# Naive Bayes\nfrom sklearn.naive_bayes import GaussianNB\nnaive_bayes_model = GaussianNB()\n\n# Neural Network (Multi-layer Perceptron)\nfrom sklearn.neural_network import MLPClassifier\nnn_model = MLPClassifier(max_iter=1000)  # Assuming 1000 iterations\n","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:46:43.415711Z","iopub.execute_input":"2023-09-25T07:46:43.416115Z","iopub.status.idle":"2023-09-25T07:46:43.709751Z","shell.execute_reply.started":"2023-09-25T07:46:43.416085Z","shell.execute_reply":"2023-09-25T07:46:43.708487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the Models\n\nlinear_reg_model.fit(X_train, y_train)\nlogistic_reg_model.fit(X_train, y_train)\ndecision_tree_model.fit(X_train, y_train)\nrandom_forest_model.fit(X_train, y_train)\nsvm_model.fit(X_train, y_train)\nknn_model.fit(X_train, y_train)\nkmeans_model.fit(X_train)\nnaive_bayes_model.fit(X_train, y_train)\nnn_model.fit(X_train, y_train)\n\n\n# Make Predictions\n\nlinear_reg_predictions = linear_reg_model.predict(X_test)\nlogistic_reg_predictions = logistic_reg_model.predict(X_test)\ndecision_tree_predictions = decision_tree_model.predict(X_test)\nrandom_forest_predictions = random_forest_model.predict(X_test)\nsvm_predictions = svm_model.predict(X_test)\nknn_predictions = knn_model.predict(X_test)\nkmeans_predictions = kmeans_model.predict(X_test)\nnaive_bayes_predictions = naive_bayes_model.predict(X_test)\nnn_predictions = nn_model.predict(X_test)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:46:48.427302Z","iopub.execute_input":"2023-09-25T07:46:48.427820Z","iopub.status.idle":"2023-09-25T07:47:14.412249Z","shell.execute_reply.started":"2023-09-25T07:46:48.427777Z","shell.execute_reply":"2023-09-25T07:47:14.410925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, mean_absolute_error, mean_squared_error, r2_score\n\n# Define a function to print results\ndef print_results(model_name, y_true, y_pred):\n    print(f\"Results for {model_name}:\")\n    \n    if isinstance(y_pred[0], int):  # Classification\n        accuracy = accuracy_score(y_true, y_pred)\n        precision = precision_score(y_true, y_pred)\n        recall = recall_score(y_true, y_pred)\n        f1 = f1_score(y_true, y_pred)\n        \n        print(f\"Accuracy: {accuracy}\")\n        print(f\"Precision: {precision}\")\n        print(f\"Recall: {recall}\")\n        print(f\"F1 Score: {f1}\")\n    else:  # Regression\n        mae = mean_absolute_error(y_true, y_pred)\n        mse = mean_squared_error(y_true, y_pred)\n        r2 = r2_score(y_true, y_pred)\n        \n        print(f\"Mean Absolute Error: {mae}\")\n        print(f\"Mean Squared Error: {mse}\")\n        print(f\"R-squared (R2): {r2}\")\n\n# Define model names and predictions\nmodel_names = [\"Linear Regression\", \"Logistic Regression\", \"Decision Tree\", \n               \"Random Forest\", \"SVM\", \"KNN\", \"K-Means\", \n               \"Naive Bayes\", \"Neural Network\"]\npredictions = [linear_reg_predictions, logistic_reg_predictions.astype(int), \n               decision_tree_predictions.astype(int), random_forest_predictions.astype(int), \n               svm_predictions.astype(int), knn_predictions.astype(int), \n               kmeans_predictions.astype(int), naive_bayes_predictions.astype(int), \n               nn_predictions.astype(int)]\n\n# Iterate through models and print results\nfor model_name, y_pred in zip(model_names, predictions):\n    print_results(model_name, y_test, y_pred)\n    print(\"\\n\" + \"=\"*30 + \"\\n\")  # Add a separator for better visibility\n","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:47:22.806459Z","iopub.execute_input":"2023-09-25T07:47:22.806859Z","iopub.status.idle":"2023-09-25T07:47:22.832680Z","shell.execute_reply.started":"2023-09-25T07:47:22.806828Z","shell.execute_reply":"2023-09-25T07:47:22.830798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import necessary libraries\nimport plotly.graph_objects as go\n\n# Define model names and evaluation metrics\nmodel_names = [\"Linear Regression\", \"Logistic Regression\", \"Decision Tree\", \n               \"Random Forest\", \"SVM\", \"KNN\", \"K-Means\", \n               \"Naive Bayes\", \"Neural Network\"]\naccuracy_scores = [0.875, 0.9, 0.7, 0.8, 0.8, 0.9, 0.333, 0.9, 0.9]\nprecision_scores = [None, 0.909, 0.667, 0.75, 0.75, 0.909, 0.0, 0.909, 0.909]\nrecall_scores = [None, 0.833, 0.667, 0.833, 0.833, 0.833, 0.0, 0.833, 0.833]\nf1_scores = [None, 0.869, 0.667, 0.789, 0.789, 0.869, 0.0, 0.869, 0.869]\n\n# Create a bar chart\nfig = go.Figure()\nfig.add_trace(go.Bar(x=model_names, y=accuracy_scores, name='Accuracy', marker_color='blue'))\nfig.add_trace(go.Bar(x=model_names, y=precision_scores, name='Precision', marker_color='green'))\nfig.add_trace(go.Bar(x=model_names, y=recall_scores, name='Recall', marker_color='orange'))\nfig.add_trace(go.Bar(x=model_names, y=f1_scores, name='F1 Score', marker_color='red'))\n\n# Update the layout\nfig.update_layout(barmode='group', title='Model Evaluation Metrics', xaxis_title='Model Name', yaxis_title='Score')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:47:30.516481Z","iopub.execute_input":"2023-09-25T07:47:30.516911Z","iopub.status.idle":"2023-09-25T07:47:31.030109Z","shell.execute_reply.started":"2023-09-25T07:47:30.516878Z","shell.execute_reply":"2023-09-25T07:47:31.028687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:magenta;display:inline-block;border-radius:5px;background-color:#E6FFE6;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:magenta;overflow:hidden;font-size:90%;letter-spacing:0.5px;margin:0\"><b> </b>Submission</p></div>\n","metadata":{}},{"cell_type":"code","source":"submission_df = pd.read_csv('/kaggle/input/open-problems-single-cell-perturbations/sample_submission.csv')\nsubmission_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:48:36.012243Z","iopub.execute_input":"2023-09-25T07:48:36.012678Z","iopub.status.idle":"2023-09-25T07:48:39.354783Z","shell.execute_reply.started":"2023-09-25T07:48:36.012643Z","shell.execute_reply":"2023-09-25T07:48:39.353117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df['id'] = submission_df.index\n\n# Create a DataFrame for submission\nsubmission = pd.DataFrame(submission_df)\n\n# Save the submission DataFrame to a CSV file\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:48:43.659260Z","iopub.execute_input":"2023-09-25T07:48:43.660683Z","iopub.status.idle":"2023-09-25T07:48:47.578402Z","shell.execute_reply.started":"2023-09-25T07:48:43.660609Z","shell.execute_reply":"2023-09-25T07:48:47.577154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2023-09-25T07:48:50.615769Z","iopub.execute_input":"2023-09-25T07:48:50.616262Z","iopub.status.idle":"2023-09-25T07:48:50.657901Z","shell.execute_reply.started":"2023-09-25T07:48:50.616197Z","shell.execute_reply":"2023-09-25T07:48:50.656556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\">\"Your positive feedback and upvote mean a lot! It motivates me to create more valuable content and helps others discover it too. Let's build a thriving community of knowledge-sharing. Thank you for your support! 😊\"</div>","metadata":{}}]}