{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# This is an example of autoencoder to predict gene expression from known combinations to unknown ones with drug taxnomic infromation.","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport seaborn as sns\nimport lightgbm as lgb\nimport matplotlib.pyplot as plt\nimport random\n\nfrom tqdm import tqdm\nfrom sklearn.preprocessing import StandardScaler\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nfrom torch.nn.functional import relu, softplus, gelu\nfrom torch.nn import Linear, Module, Dropout, MSELoss, CrossEntropyLoss, BatchNorm1d\nfrom torch.utils.data import DataLoader, TensorDataset\n\nfrom sklearn.metrics import mean_squared_error as mse\nfrom sklearn.metrics import mean_absolute_error as mae\nfrom sklearn.metrics import mean_absolute_percentage_error as mape\nfrom sklearn.metrics.pairwise import cosine_similarity\nfrom scipy.stats import pearsonr","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:47:54.114080Z","iopub.execute_input":"2023-10-08T17:47:54.114415Z","iopub.status.idle":"2023-10-08T17:47:54.121427Z","shell.execute_reply.started":"2023-10-08T17:47:54.114389Z","shell.execute_reply":"2023-10-08T17:47:54.120548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install rdkit","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-10-08T17:47:54.123325Z","iopub.execute_input":"2023-10-08T17:47:54.123933Z","iopub.status.idle":"2023-10-08T17:48:02.972855Z","shell.execute_reply.started":"2023-10-08T17:47:54.123904Z","shell.execute_reply":"2023-10-08T17:48:02.971516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from rdkit import Chem\nfrom rdkit.Chem import AllChem","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-10-08T17:48:02.977292Z","iopub.execute_input":"2023-10-08T17:48:02.977725Z","iopub.status.idle":"2023-10-08T17:48:02.983653Z","shell.execute_reply.started":"2023-10-08T17:48:02.977673Z","shell.execute_reply":"2023-10-08T17:48:02.982608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!nvidia-smi","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:48:02.984992Z","iopub.execute_input":"2023-10-08T17:48:02.987531Z","iopub.status.idle":"2023-10-08T17:48:03.990664Z","shell.execute_reply.started":"2023-10-08T17:48:02.987472Z","shell.execute_reply":"2023-10-08T17:48:03.989403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\ndevice","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:48:03.993770Z","iopub.execute_input":"2023-10-08T17:48:03.994391Z","iopub.status.idle":"2023-10-08T17:48:04.000789Z","shell.execute_reply.started":"2023-10-08T17:48:03.994363Z","shell.execute_reply":"2023-10-08T17:48:03.999828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_parquet('/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet')\ndf","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:48:04.002159Z","iopub.execute_input":"2023-10-08T17:48:04.002750Z","iopub.status.idle":"2023-10-08T17:48:06.368710Z","shell.execute_reply.started":"2023-10-08T17:48:04.002679Z","shell.execute_reply":"2023-10-08T17:48:06.367785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Make a morgan fingerprinting from SMILES","metadata":{}},{"cell_type":"code","source":"radius = 2\nnBits = 2048\n\ndef get_fp(smiles_string):\n    mol = Chem.MolFromSmiles(smiles_string)\n    fps = AllChem.GetMorganFingerprintAsBitVect(mol, radius, nBits=nBits)\n    return fps","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:48:06.369880Z","iopub.execute_input":"2023-10-08T17:48:06.370763Z","iopub.status.idle":"2023-10-08T17:48:06.376304Z","shell.execute_reply.started":"2023-10-08T17:48:06.370730Z","shell.execute_reply":"2023-10-08T17:48:06.375369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"conv = {i:list(get_fp(i)) for i in set(df.SMILES)}\nconv = pd.DataFrame(conv).T.reset_index()\nconv = conv.rename(columns={'index': 'SMILES'})\nconv.iloc[:, 1:] = conv.iloc[:, 1:].astype(int)\nconv","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:48:06.377488Z","iopub.execute_input":"2023-10-08T17:48:06.377836Z","iopub.status.idle":"2023-10-08T17:48:06.815605Z","shell.execute_reply.started":"2023-10-08T17:48:06.377807Z","shell.execute_reply":"2023-10-08T17:48:06.814686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Make a validation (cell_types which are in test data)","metadata":{}},{"cell_type":"code","source":"df_val = df[(df['cell_type'] == 'B cells') | (df['cell_type'] == 'Myeloid cells')]\ndf_val = df_val[df_val.control == False].reset_index(drop=True)\ndf_val.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:48:06.817121Z","iopub.execute_input":"2023-10-08T17:48:06.817722Z","iopub.status.idle":"2023-10-08T17:48:06.847689Z","shell.execute_reply.started":"2023-10-08T17:48:06.817680Z","shell.execute_reply":"2023-10-08T17:48:06.846741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Merge fingerprinting","metadata":{}},{"cell_type":"code","source":"df = df.merge(conv)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:48:06.849137Z","iopub.execute_input":"2023-10-08T17:48:06.849760Z","iopub.status.idle":"2023-10-08T17:48:06.914327Z","shell.execute_reply.started":"2023-10-08T17:48:06.849730Z","shell.execute_reply":"2023-10-08T17:48:06.913417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Generate training data (cell_type which are not in test data and drugs in validation data)","metadata":{}},{"cell_type":"code","source":"df_train = df[(df.sm_name.isin(list(set(df_val.sm_name)))) & ((df.cell_type != 'B cells') & (df.cell_type != 'Myeloid cells'))]\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:48:06.918096Z","iopub.execute_input":"2023-10-08T17:48:06.918353Z","iopub.status.idle":"2023-10-08T17:48:06.948199Z","shell.execute_reply.started":"2023-10-08T17:48:06.918332Z","shell.execute_reply":"2023-10-08T17:48:06.947238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_val = pd.DataFrame(df_train.sm_name).merge(df_val).sort_values('sm_name').reset_index(drop=True)\ndf_val.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:48:06.949627Z","iopub.execute_input":"2023-10-08T17:48:06.949987Z","iopub.status.idle":"2023-10-08T17:48:06.998884Z","shell.execute_reply.started":"2023-10-08T17:48:06.949955Z","shell.execute_reply":"2023-10-08T17:48:06.997991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.concat([df_train, df_train]).sort_values(['sm_name'])\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:48:07.000154Z","iopub.execute_input":"2023-10-08T17:48:07.001042Z","iopub.status.idle":"2023-10-08T17:48:07.034975Z","shell.execute_reply.started":"2023-10-08T17:48:07.001011Z","shell.execute_reply":"2023-10-08T17:48:07.034060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sum((np.array(df_train.sm_name) != np.array(df_val.sm_name)))","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:48:07.036234Z","iopub.execute_input":"2023-10-08T17:48:07.037179Z","iopub.status.idle":"2023-10-08T17:48:07.044239Z","shell.execute_reply.started":"2023-10-08T17:48:07.037144Z","shell.execute_reply":"2023-10-08T17:48:07.043142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define 3 linear layres' encoder and 3 linear layres' encoder","metadata":{}},{"cell_type":"code","source":"class Autoencoder(nn.Module):\n    def __init__(self, input_dim, hidden_dim, output_dim, train=True, dropout_prob=0.2):\n        super(Autoencoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, hidden_dim*3),\n            nn.BatchNorm1d(hidden_dim*3),\n            nn.GELU(),\n            nn.Dropout(dropout_prob if train else 0), \n            nn.Linear(hidden_dim*3, hidden_dim*2),\n            nn.BatchNorm1d(hidden_dim*2),\n            nn.GELU(),\n            nn.Dropout(dropout_prob if train else 0),\n            nn.Linear(hidden_dim*2, hidden_dim),\n            nn.BatchNorm1d(hidden_dim),\n            nn.GELU(),\n            nn.Dropout(dropout_prob if train else 0)\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(hidden_dim, hidden_dim*2),\n            nn.BatchNorm1d(hidden_dim*2),\n            nn.GELU(),\n            nn.Dropout(dropout_prob if train else 0),\n            nn.Linear(hidden_dim*2, hidden_dim*3),\n            nn.BatchNorm1d(hidden_dim*3),\n            nn.GELU(),\n            nn.Dropout(dropout_prob if train else 0),\n            nn.Linear(hidden_dim*3, output_dim)\n        )\n\n    def forward(self, x):\n        x = self.encoder(x)\n        x = self.decoder(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:48:07.045763Z","iopub.execute_input":"2023-10-08T17:48:07.046529Z","iopub.status.idle":"2023-10-08T17:48:07.058515Z","shell.execute_reply.started":"2023-10-08T17:48:07.046403Z","shell.execute_reply":"2023-10-08T17:48:07.057634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_tensor = torch.tensor(df_train.iloc[:, 5:].values).float().to(device)\nexpect_tensor = torch.tensor(df_val.iloc[:, 5:].values).float().to(device)","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:48:07.059644Z","iopub.execute_input":"2023-10-08T17:48:07.060894Z","iopub.status.idle":"2023-10-08T17:48:09.744662Z","shell.execute_reply.started":"2023-10-08T17:48:07.060860Z","shell.execute_reply":"2023-10-08T17:48:09.743681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_tensor.shape","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:48:09.746183Z","iopub.execute_input":"2023-10-08T17:48:09.746531Z","iopub.status.idle":"2023-10-08T17:48:09.752819Z","shell.execute_reply.started":"2023-10-08T17:48:09.746484Z","shell.execute_reply":"2023-10-08T17:48:09.751892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"expect_tensor.shape","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:48:09.754159Z","iopub.execute_input":"2023-10-08T17:48:09.755263Z","iopub.status.idle":"2023-10-08T17:48:09.764420Z","shell.execute_reply.started":"2023-10-08T17:48:09.755233Z","shell.execute_reply":"2023-10-08T17:48:09.763409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define dimentions","metadata":{}},{"cell_type":"code","source":"input_dim = input_tensor.shape[1]\nhidden_dim = 128\noutput_dim = expect_tensor.shape[1]\nbatch_size = 64","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:48:09.766082Z","iopub.execute_input":"2023-10-08T17:48:09.766680Z","iopub.status.idle":"2023-10-08T17:48:09.773621Z","shell.execute_reply.started":"2023-10-08T17:48:09.766651Z","shell.execute_reply":"2023-10-08T17:48:09.772764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = TensorDataset(input_tensor, expect_tensor)\ndataloader = DataLoader(dataset, batch_size=batch_size, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:48:09.774959Z","iopub.execute_input":"2023-10-08T17:48:09.775538Z","iopub.status.idle":"2023-10-08T17:48:09.784722Z","shell.execute_reply.started":"2023-10-08T17:48:09.775493Z","shell.execute_reply":"2023-10-08T17:48:09.783862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"autoencoder = Autoencoder(input_dim, hidden_dim, output_dim).to(device)\ncriterion = nn.MSELoss().to(device)\noptimizer = optim.Adam(autoencoder.parameters(), lr=0.01)\n\nnum_epochs = 500\nfor epoch in range(num_epochs):\n    total_loss = 0.0\n    res_list = []\n    for batch_input, batch_expect in dataloader:\n        outputs = autoencoder(batch_input)\n        loss = criterion(outputs, batch_expect)\n        \n        res_list.append(outputs) \n        \n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n        \n        total_loss += loss.item()\n    \n    if (epoch+1) % 10 == 0:\n        average_loss = total_loss / len(dataloader)\n        print(f'Epoch [{epoch+1}/{num_epochs}], Average Loss: {average_loss:.4f}')","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:48:52.890952Z","iopub.execute_input":"2023-10-08T17:48:52.891296Z","iopub.status.idle":"2023-10-08T17:49:00.474368Z","shell.execute_reply.started":"2023-10-08T17:48:52.891270Z","shell.execute_reply":"2023-10-08T17:49:00.473312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating concatnated matrix from autoencoder","metadata":{}},{"cell_type":"code","source":"res = torch.cat(res_list)\nres = pd.DataFrame(res.detach().cpu().numpy())\nres = pd.concat([df_train.iloc[:, :5].reset_index(drop=True), res], axis=1).reset_index(drop=True)\nres","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:49:00.476262Z","iopub.execute_input":"2023-10-08T17:49:00.476932Z","iopub.status.idle":"2023-10-08T17:49:00.521836Z","shell.execute_reply.started":"2023-10-08T17:49:00.476898Z","shell.execute_reply":"2023-10-08T17:49:00.520469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Here we defined return_metrics function to check which cell_type is similar to B cells and M cells","metadata":{}},{"cell_type":"code","source":"def return_metrics(name, x, y):\n    return [name, mse(x, y), mae(x, y), mape(x, y)]","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:49:00.523073Z","iopub.execute_input":"2023-10-08T17:49:00.523396Z","iopub.status.idle":"2023-10-08T17:49:00.527765Z","shell.execute_reply.started":"2023-10-08T17:49:00.523369Z","shell.execute_reply":"2023-10-08T17:49:00.526892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result = []\nfor i in set(df_val.cell_type):\n    for l in set(res.cell_type):\n        result.append(return_metrics(\n            (l, ' => ', i),\n            df_val[(df_val.cell_type == i) & (res.cell_type == l)].iloc[:, 5:].values,\n            res[(df_val.cell_type == i) & (res.cell_type == l)].iloc[:, 5:].values\n        ))","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:49:00.530137Z","iopub.execute_input":"2023-10-08T17:49:00.531167Z","iopub.status.idle":"2023-10-08T17:49:00.613476Z","shell.execute_reply.started":"2023-10-08T17:49:00.531135Z","shell.execute_reply":"2023-10-08T17:49:00.612557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(result)","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:49:00.614775Z","iopub.execute_input":"2023-10-08T17:49:00.615096Z","iopub.status.idle":"2023-10-08T17:49:00.628375Z","shell.execute_reply.started":"2023-10-08T17:49:00.615067Z","shell.execute_reply":"2023-10-08T17:49:00.627296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# B cells => T regulatory cells, M cells => T cells CD8+","metadata":{}},{"cell_type":"code","source":"id_map = pd.read_csv('/kaggle/input/open-problems-single-cell-perturbations/id_map.csv')\nid_map.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:49:00.629953Z","iopub.execute_input":"2023-10-08T17:49:00.630846Z","iopub.status.idle":"2023-10-08T17:49:00.648127Z","shell.execute_reply.started":"2023-10-08T17:49:00.630815Z","shell.execute_reply":"2023-10-08T17:49:00.647234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create test data","metadata":{}},{"cell_type":"code","source":"id_map.loc[id_map['cell_type'] == 'B cells', 'cell_type'] = 'T regulatory cells'\nid_map.loc[id_map['cell_type'] == 'Myeloid cells', 'cell_type'] = 'T cells CD8+'\nid_map.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:49:00.649399Z","iopub.execute_input":"2023-10-08T17:49:00.649735Z","iopub.status.idle":"2023-10-08T17:49:00.661541Z","shell.execute_reply.started":"2023-10-08T17:49:00.649696Z","shell.execute_reply":"2023-10-08T17:49:00.660312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = id_map.merge(df)\ntest_df","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:49:00.663088Z","iopub.execute_input":"2023-10-08T17:49:00.663457Z","iopub.status.idle":"2023-10-08T17:49:00.768389Z","shell.execute_reply.started":"2023-10-08T17:49:00.663429Z","shell.execute_reply":"2023-10-08T17:49:00.767376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set(range(255)) - set(test_df.id)","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:49:00.769678Z","iopub.execute_input":"2023-10-08T17:49:00.770654Z","iopub.status.idle":"2023-10-08T17:49:00.777218Z","shell.execute_reply.started":"2023-10-08T17:49:00.770623Z","shell.execute_reply":"2023-10-08T17:49:00.776169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Generate prediction from AE","metadata":{}},{"cell_type":"code","source":"autoencoder.eval()\n\nwith torch.no_grad():\n    eval_outputs = autoencoder(torch.tensor(test_df.iloc[:, 6:].values).float().to(device))","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:49:00.780006Z","iopub.execute_input":"2023-10-08T17:49:00.780836Z","iopub.status.idle":"2023-10-08T17:49:00.866485Z","shell.execute_reply.started":"2023-10-08T17:49:00.780802Z","shell.execute_reply":"2023-10-08T17:49:00.865532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = eval_outputs.detach().cpu().numpy()\npred = pd.DataFrame(pred, columns=list(df_val.columns[5:]))\npred = pd.concat([test_df[['cell_type', 'sm_name']], pred], axis=1)\npred","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:49:00.868060Z","iopub.execute_input":"2023-10-08T17:49:00.868376Z","iopub.status.idle":"2023-10-08T17:49:00.914172Z","shell.execute_reply.started":"2023-10-08T17:49:00.868347Z","shell.execute_reply":"2023-10-08T17:49:00.913306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_res = id_map.merge(pred, how='left').fillna(0)\npred_res = pred_res.drop(['cell_type', 'sm_name'], axis=1)\npred_res","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:49:00.963095Z","iopub.execute_input":"2023-10-08T17:49:00.963712Z","iopub.status.idle":"2023-10-08T17:49:01.033952Z","shell.execute_reply.started":"2023-10-08T17:49:00.963681Z","shell.execute_reply":"2023-10-08T17:49:01.032974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_res.to_csv('ae_mfp.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:49:01.100516Z","iopub.execute_input":"2023-10-08T17:49:01.100921Z","iopub.status.idle":"2023-10-08T17:49:06.268396Z","shell.execute_reply.started":"2023-10-08T17:49:01.100897Z","shell.execute_reply":"2023-10-08T17:49:06.267164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Emsemble","metadata":{}},{"cell_type":"code","source":"sub_chain = pd.read_csv('../input/2-op2-regressor-chain/submission.csv')\nsub_chain.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:49:06.270555Z","iopub.execute_input":"2023-10-08T17:49:06.271159Z","iopub.status.idle":"2023-10-08T17:49:09.687005Z","shell.execute_reply.started":"2023-10-08T17:49:06.271122Z","shell.execute_reply":"2023-10-08T17:49:09.686101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Thanks to **@mehrankazeminia @somayyehgholami** https://www.kaggle.com/code/mehrankazeminia/1-op2-eda-linearsvr-regressorchain","metadata":{}},{"cell_type":"code","source":"sub_import1 = pd.read_csv('../input/op2-603/op2_603.csv')\nsub_import1.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:49:09.688466Z","iopub.execute_input":"2023-10-08T17:49:09.689027Z","iopub.status.idle":"2023-10-08T17:49:12.443740Z","shell.execute_reply.started":"2023-10-08T17:49:09.688995Z","shell.execute_reply":"2023-10-08T17:49:12.442799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Thanks to: **@vendekagonlabs**  https://www.kaggle.com/code/vendekagonlabs/jax-autoencoder-quickstart","metadata":{}},{"cell_type":"code","source":"sub_import2 = pd.read_csv('../input/op2-720/op2_720.csv')\nsub_import2 = sub_import2[sub_import1.columns]\nsub_import2.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:49:12.445975Z","iopub.execute_input":"2023-10-08T17:49:12.446565Z","iopub.status.idle":"2023-10-08T17:49:15.641625Z","shell.execute_reply.started":"2023-10-08T17:49:12.446531Z","shell.execute_reply":"2023-10-08T17:49:15.640364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Thanks to: **@kishanvavdara**  https://www.kaggle.com/code/kishanvavdara/neural-network-regression","metadata":{}},{"cell_type":"code","source":"sub_import3 = pd.read_csv('../input/op2-604/submission_df.csv')\nsub_import3.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:49:15.643277Z","iopub.execute_input":"2023-10-08T17:49:15.643678Z","iopub.status.idle":"2023-10-08T17:49:18.165845Z","shell.execute_reply.started":"2023-10-08T17:49:15.643643Z","shell.execute_reply":"2023-10-08T17:49:18.164925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_new = (sub_import1 *0.4) + (sub_import2 *0.1) + (sub_import3 *0.25) + (sub_chain *0.1) + (pred_res *0.25) \n# sub_new.id = sub_new.id.astype(int)\nsub_new","metadata":{"execution":{"iopub.status.busy":"2023-10-08T18:01:11.781165Z","iopub.execute_input":"2023-10-08T18:01:11.781733Z","iopub.status.idle":"2023-10-08T18:01:11.946482Z","shell.execute_reply.started":"2023-10-08T18:01:11.781694Z","shell.execute_reply":"2023-10-08T18:01:11.945577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('/kaggle/input/open-problems-single-cell-perturbations/sample_submission.csv')\nsub","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:57:43.158410Z","iopub.execute_input":"2023-10-08T17:57:43.158798Z","iopub.status.idle":"2023-10-08T17:57:45.384136Z","shell.execute_reply.started":"2023-10-08T17:57:43.158771Z","shell.execute_reply":"2023-10-08T17:57:45.383033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.iloc[:, 1:] = sub_new.iloc[:, 1:]\nsub","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:58:08.432054Z","iopub.execute_input":"2023-10-08T17:58:08.432385Z","iopub.status.idle":"2023-10-08T17:58:10.347780Z","shell.execute_reply.started":"2023-10-08T17:58:08.432360Z","shell.execute_reply":"2023-10-08T17:58:10.346871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-10-08T17:49:18.324002Z","iopub.execute_input":"2023-10-08T17:49:18.324609Z","iopub.status.idle":"2023-10-08T17:49:25.978590Z","shell.execute_reply.started":"2023-10-08T17:49:18.324575Z","shell.execute_reply":"2023-10-08T17:49:25.977554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Next step.\n- HPTuning\n- Increase/decrease layer\n- Change activation function\n- Use other layers such as 1D Conv\n- Use more complicated models such as diffusion and VAE\n- Use more biological knowledges like Gene Ontology\n- Dig into raw data more\n\n## If you have any suggestions, lmk!","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}