{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59094,"databundleVersionId":7010844,"sourceType":"competition"}],"dockerImageVersionId":30558,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-30T18:02:39.736766Z","iopub.execute_input":"2023-11-30T18:02:39.737275Z","iopub.status.idle":"2023-11-30T18:02:39.763919Z","shell.execute_reply.started":"2023-11-30T18:02:39.737233Z","shell.execute_reply":"2023-11-30T18:02:39.763008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This notebook is written in pytorch with the elements from Kishan Vavdara https://www.kaggle.com/code/kishanvavdara/neural-network-regression","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n#import tensorflow as tf\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport sklearn","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:02:44.267128Z","iopub.execute_input":"2023-11-30T18:02:44.268122Z","iopub.status.idle":"2023-11-30T18:02:44.274102Z","shell.execute_reply.started":"2023-11-30T18:02:44.268056Z","shell.execute_reply":"2023-11-30T18:02:44.272964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"traindata =   pd.read_parquet(\"/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet\")\nid_map = pd.read_csv(\"/kaggle/input/open-problems-single-cell-perturbations/id_map.csv\")\nsample_submission = pd.read_csv(\"/kaggle/input/open-problems-single-cell-perturbations/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:02:46.969309Z","iopub.execute_input":"2023-11-30T18:02:46.970245Z","iopub.status.idle":"2023-11-30T18:02:53.094634Z","shell.execute_reply.started":"2023-11-30T18:02:46.970204Z","shell.execute_reply":"2023-11-30T18:02:53.093437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#https://www.kaggle.com/code/kishanvavdara/neural-network-regression","metadata":{"execution":{"iopub.status.busy":"2023-11-28T08:38:16.858430Z","iopub.execute_input":"2023-11-28T08:38:16.858936Z","iopub.status.idle":"2023-11-28T08:38:16.864264Z","shell.execute_reply.started":"2023-11-28T08:38:16.858896Z","shell.execute_reply":"2023-11-28T08:38:16.862812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"traindata.head()\n#614 rows and columns related to 18,211 genes","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:02:53.096584Z","iopub.execute_input":"2023-11-30T18:02:53.096934Z","iopub.status.idle":"2023-11-30T18:02:53.132849Z","shell.execute_reply.started":"2023-11-30T18:02:53.096905Z","shell.execute_reply":"2023-11-30T18:02:53.131482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:02:53.718425Z","iopub.execute_input":"2023-11-30T18:02:53.718819Z","iopub.status.idle":"2023-11-30T18:02:53.730307Z","shell.execute_reply.started":"2023-11-30T18:02:53.718788Z","shell.execute_reply":"2023-11-30T18:02:53.728940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"traindata.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:02:58.397389Z","iopub.execute_input":"2023-11-30T18:02:58.397823Z","iopub.status.idle":"2023-11-30T18:02:58.446363Z","shell.execute_reply.started":"2023-11-30T18:02:58.397789Z","shell.execute_reply":"2023-11-30T18:02:58.445240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_sm= traindata['sm_name'].unique()\nprint(f\"number of unique perturbations {len(unique_sm)}\")\nunique_sm","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:03:01.649272Z","iopub.execute_input":"2023-11-30T18:03:01.649698Z","iopub.status.idle":"2023-11-30T18:03:01.663753Z","shell.execute_reply.started":"2023-11-30T18:03:01.649664Z","shell.execute_reply":"2023-11-30T18:03:01.662797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"traindata['cell_type'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:03:05.330111Z","iopub.execute_input":"2023-11-30T18:03:05.331418Z","iopub.status.idle":"2023-11-30T18:03:05.341396Z","shell.execute_reply.started":"2023-11-30T18:03:05.331375Z","shell.execute_reply":"2023-11-30T18:03:05.339810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"traindata[traindata['control']==True].count","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:03:07.392675Z","iopub.execute_input":"2023-11-30T18:03:07.393395Z","iopub.status.idle":"2023-11-30T18:03:07.418051Z","shell.execute_reply.started":"2023-11-30T18:03:07.393349Z","shell.execute_reply":"2023-11-30T18:03:07.416892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#feature columns and target columns, target consists only gene columns\nfeature_cols =['cell_type','sm_name']\ntarget_cols = ['cell_type','sm_name','sm_lincs_id','SMILES','control']\ntargets = traindata.drop(columns=target_cols)\nfeatures = pd.DataFrame(traindata,columns=feature_cols)\nfeatures","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:03:11.864480Z","iopub.execute_input":"2023-11-30T18:03:11.864910Z","iopub.status.idle":"2023-11-30T18:03:11.921194Z","shell.execute_reply.started":"2023-11-30T18:03:11.864878Z","shell.execute_reply":"2023-11-30T18:03:11.920161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:03:15.254685Z","iopub.execute_input":"2023-11-30T18:03:15.255128Z","iopub.status.idle":"2023-11-30T18:03:15.295273Z","shell.execute_reply.started":"2023-11-30T18:03:15.255085Z","shell.execute_reply":"2023-11-30T18:03:15.294359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testdata = pd.DataFrame(id_map,columns=feature_cols) \n#testdata.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:03:19.537540Z","iopub.execute_input":"2023-11-30T18:03:19.538361Z","iopub.status.idle":"2023-11-30T18:03:19.544583Z","shell.execute_reply.started":"2023-11-30T18:03:19.538325Z","shell.execute_reply":"2023-11-30T18:03:19.543361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets.values.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:03:21.826037Z","iopub.execute_input":"2023-11-30T18:03:21.826461Z","iopub.status.idle":"2023-11-30T18:03:21.834830Z","shell.execute_reply.started":"2023-11-30T18:03:21.826429Z","shell.execute_reply":"2023-11-30T18:03:21.833317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#one hot encoding\nfeatures['sm_name'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:03:23.498221Z","iopub.execute_input":"2023-11-30T18:03:23.498705Z","iopub.status.idle":"2023-11-30T18:03:23.508639Z","shell.execute_reply.started":"2023-11-30T18:03:23.498667Z","shell.execute_reply":"2023-11-30T18:03:23.506989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\none_hot = OneHotEncoder()\nfeatures_array = one_hot.fit(features)\n#features_array.transform(features_array)","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:03:28.251469Z","iopub.execute_input":"2023-11-30T18:03:28.251969Z","iopub.status.idle":"2023-11-30T18:03:28.261567Z","shell.execute_reply.started":"2023-11-30T18:03:28.251931Z","shell.execute_reply":"2023-11-30T18:03:28.259691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_one_hot=one_hot.transform(features)\none_hot_test = one_hot.transform(testdata)\nprint(features_one_hot.toarray().shape,one_hot_test.toarray().shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:03:30.467866Z","iopub.execute_input":"2023-11-30T18:03:30.468394Z","iopub.status.idle":"2023-11-30T18:03:30.483018Z","shell.execute_reply.started":"2023-11-30T18:03:30.468354Z","shell.execute_reply":"2023-11-30T18:03:30.481547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Split the data into 70% training, 15% validation, and 15% testing\nX_train, X_temp, y_train, y_temp = train_test_split(features_one_hot, targets.values, test_size=0.3, shuffle=False)\nX_val, X_test, y_val, y_test = train_test_split(X_temp, y_temp, test_size=0.5, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:03:32.756348Z","iopub.execute_input":"2023-11-30T18:03:32.756755Z","iopub.status.idle":"2023-11-30T18:03:32.847843Z","shell.execute_reply.started":"2023-11-30T18:03:32.756723Z","shell.execute_reply":"2023-11-30T18:03:32.846313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"featurespace = features_one_hot.toarray()\ntargetsspace = targets.values\ntargetsspace.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:03:35.263095Z","iopub.execute_input":"2023-11-30T18:03:35.263507Z","iopub.status.idle":"2023-11-30T18:03:35.272230Z","shell.execute_reply.started":"2023-11-30T18:03:35.263474Z","shell.execute_reply":"2023-11-30T18:03:35.270989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.utils.data import DataLoader,Dataset\nclass dataset(Dataset):\n    def __init__(self,X_train,y_train):\n        self.X_train=X_train\n        self.y_train=y_train\n    def __len__(self):\n        return len(self.X_train)\n    def __getitem__(self,idx):\n        return self.X_train[idx], self.y_train[idx]","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:03:37.505581Z","iopub.execute_input":"2023-11-30T18:03:37.506119Z","iopub.status.idle":"2023-11-30T18:03:37.514368Z","shell.execute_reply.started":"2023-11-30T18:03:37.506047Z","shell.execute_reply":"2023-11-30T18:03:37.512897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nfeatures_t= torch.tensor(featurespace)\ntarget_t=torch.tensor(targetsspace)\ndata=dataset(features_t,target_t)\ntarget_t.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:03:39.739833Z","iopub.execute_input":"2023-11-30T18:03:39.740359Z","iopub.status.idle":"2023-11-30T18:03:39.807240Z","shell.execute_reply.started":"2023-11-30T18:03:39.740318Z","shell.execute_reply":"2023-11-30T18:03:39.805952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nclass Net(torch.nn.Module):\n    def __init__(self,input_size,output_size):\n        super(Net,self).__init__()\n        self.input_size=input_size\n        self.output_size=output_size\n        self.hidden_size=hidden_size\n        self.fc0 = nn.Linear(self.input_size,self.hidden_size)\n        self.fc1=nn.Linear(self.hidden_size,self.output_size)\n       \n    def forward(self,idx):\n        h1=F.relu(self.fc0(idx))\n        target_out=F.tanh(self.fc1(h1))\n        return target_out","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:03:41.801778Z","iopub.execute_input":"2023-11-30T18:03:41.802443Z","iopub.status.idle":"2023-11-30T18:03:41.809869Z","shell.execute_reply.started":"2023-11-30T18:03:41.802408Z","shell.execute_reply":"2023-11-30T18:03:41.808922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mean_absolute_error\n\ndef RMSE_rowwise_loss(model, data, y_true):\n    \"\"\"\n    Calculate Mean Absolute Error (MAE) and Mean Rowwise Root Mean Squared Error (MRRMSE).\n\n    Parameters:\n    - model: The trained  model.\n    - data: The input data for prediction.\n    - y_true: The true target values.\n    \"\"\"\n    model.eval()\n    #print('data',data.shape)\n    y_pred_original = model(data)\n    mae = mean_absolute_error(y_true , y_pred_original.detach().numpy())\n    rowwise_rmse = np.sqrt(np.mean(np.square(y_true - y_pred_original.detach().numpy()).numpy(), axis=1))\n    mrrmse_score = np.mean(rowwise_rmse)\n    return mrrmse_score","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:03:44.881792Z","iopub.execute_input":"2023-11-30T18:03:44.882603Z","iopub.status.idle":"2023-11-30T18:03:44.889794Z","shell.execute_reply.started":"2023-11-30T18:03:44.882569Z","shell.execute_reply":"2023-11-30T18:03:44.888458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batchsize=64\ndataloader = torch.utils.data.DataLoader(data,batch_size=batchsize,shuffle=True)\nfor batch,(x,y) in enumerate(dataloader):\n    print('batch',batch,x.shape,y.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:03:47.058576Z","iopub.execute_input":"2023-11-30T18:03:47.058970Z","iopub.status.idle":"2023-11-30T18:03:47.310824Z","shell.execute_reply.started":"2023-11-30T18:03:47.058940Z","shell.execute_reply":"2023-11-30T18:03:47.309634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch.optim as optim\nfrom torch.autograd import Variable\ninput_size = 152\noutput_size = 18211\nhidden_size = 500\nepochs = 150\nmodel=Net(input_size,output_size)\n#loss_function = RMSE_rowwise_loss(model,x, y)\noptimizer = optim.Adam(model.parameters(),lr=0.001)\nfor epoch in range(0,epochs):\n    for batch, (feature_train,target_train) in enumerate(dataloader):\n       # print('in the epoch',epoch,feature_train.shape,target_train.shape)\n        optimizer.zero_grad()\n        loss = RMSE_rowwise_loss(model,feature_train.float(), target_train.float())\n        loss=Variable(torch.tensor(loss),requires_grad=True)\n        loss.backward()\n        optimizer.step()\n    if epoch % 5 == 0:\n          #loss, current = loss.item(), batch * len(inp)\n        print(f\"epoch: {epoch + 1} loss: {loss}\")#[{current}/{size}] loss: {loss}\")","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:03:49.553548Z","iopub.execute_input":"2023-11-30T18:03:49.554192Z","iopub.status.idle":"2023-11-30T18:04:59.523508Z","shell.execute_reply.started":"2023-11-30T18:03:49.554160Z","shell.execute_reply":"2023-11-30T18:04:59.522205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lost is not lowering! Needs improvement.","metadata":{}},{"cell_type":"code","source":"model.eval()\ntarget_pred = model(torch.Tensor(one_hot_test.toarray()))","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:04:59.525608Z","iopub.execute_input":"2023-11-30T18:04:59.529469Z","iopub.status.idle":"2023-11-30T18:04:59.604004Z","shell.execute_reply.started":"2023-11-30T18:04:59.529416Z","shell.execute_reply":"2023-11-30T18:04:59.603123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_columns = sample_submission.columns\nsample_columns= sample_columns[1:]\nsubmission_df = pd.DataFrame(target_pred.detach().numpy(), columns=sample_columns)","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:05:05.655817Z","iopub.execute_input":"2023-11-30T18:05:05.656246Z","iopub.status.idle":"2023-11-30T18:05:05.663294Z","shell.execute_reply.started":"2023-11-30T18:05:05.656216Z","shell.execute_reply":"2023-11-30T18:05:05.661666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.insert(0, 'id', range(255))","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:05:08.379557Z","iopub.execute_input":"2023-11-30T18:05:08.379977Z","iopub.status.idle":"2023-11-30T18:05:08.390480Z","shell.execute_reply.started":"2023-11-30T18:05:08.379946Z","shell.execute_reply":"2023-11-30T18:05:08.389139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:05:11.904870Z","iopub.execute_input":"2023-11-30T18:05:11.906527Z","iopub.status.idle":"2023-11-30T18:05:11.960646Z","shell.execute_reply.started":"2023-11-30T18:05:11.906481Z","shell.execute_reply":"2023-11-30T18:05:11.959472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:05:15.779927Z","iopub.execute_input":"2023-11-30T18:05:15.780396Z","iopub.status.idle":"2023-11-30T18:05:15.821430Z","shell.execute_reply.started":"2023-11-30T18:05:15.780364Z","shell.execute_reply":"2023-11-30T18:05:15.820132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:17:22.568552Z","iopub.execute_input":"2023-11-30T18:17:22.569751Z","iopub.status.idle":"2023-11-30T18:17:32.126034Z","shell.execute_reply.started":"2023-11-30T18:17:22.569701Z","shell.execute_reply":"2023-11-30T18:17:32.124775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!zip submission_preds.zip /kaggle/working/submission.csv","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:17:37.441562Z","iopub.execute_input":"2023-11-30T18:17:37.441997Z","iopub.status.idle":"2023-11-30T18:17:44.649762Z","shell.execute_reply.started":"2023-11-30T18:17:37.441963Z","shell.execute_reply":"2023-11-30T18:17:44.648171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-11-30T18:12:53.286327Z","iopub.execute_input":"2023-11-30T18:12:53.286733Z","iopub.status.idle":"2023-11-30T18:12:53.338194Z","shell.execute_reply.started":"2023-11-30T18:12:53.286702Z","shell.execute_reply":"2023-11-30T18:12:53.336346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}