{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# keeps the plots in one place. calls image as static pngs\n%matplotlib inline \nimport matplotlib.pyplot as plt # side-stepping mpl backend\nimport matplotlib.gridspec as gridspec # subplots\nimport mpld3 as mpl\n\n\n\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-30T01:15:18.699430Z","iopub.execute_input":"2022-12-30T01:15:18.700217Z","iopub.status.idle":"2022-12-30T01:15:18.824205Z","shell.execute_reply.started":"2022-12-30T01:15:18.700099Z","shell.execute_reply":"2022-12-30T01:15:18.823005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Import models from scikit learn module:\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.linear_model import Ridge\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error\nfrom sklearn.linear_model import RidgeCV\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.model_selection import KFold \nfrom sklearn.metrics import r2_score\nfrom sklearn import metrics","metadata":{"execution":{"iopub.status.busy":"2022-12-30T01:15:18.826517Z","iopub.execute_input":"2022-12-30T01:15:18.826827Z","iopub.status.idle":"2022-12-30T01:15:19.976549Z","shell.execute_reply.started":"2022-12-30T01:15:18.826802Z","shell.execute_reply":"2022-12-30T01:15:19.975555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load data","metadata":{}},{"cell_type":"code","source":"df_targets = pd.read_hdf('/kaggle/input/open-problems-multimodal/train_cite_targets.h5')\ndf_meta = pd.read_csv('/kaggle/input/feature-shop-for-multimodal-singlecell-competition/_citeseq_meta_all_text_also.csv')\ndf_meta = df_meta[df_meta['Train0OrTest1']==0]\n# df_meta = pd.read_csv('/kaggle/input/open-problems-multimodal/metadata.csv')","metadata":{"execution":{"iopub.status.busy":"2022-12-30T01:15:19.978055Z","iopub.execute_input":"2022-12-30T01:15:19.978605Z","iopub.status.idle":"2022-12-30T01:15:21.016241Z","shell.execute_reply.started":"2022-12-30T01:15:19.978571Z","shell.execute_reply":"2022-12-30T01:15:21.015016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare data","metadata":{}},{"cell_type":"code","source":"df_m= df_meta[['cell_id','HSC cell_type', 'EryP cell_type', 'NeuP cell_type', 'MasP cell_type', 'MkP cell_type','MoP cell_type','BP cell_type']].set_index('cell_id')\ndf_m\n","metadata":{"execution":{"iopub.status.busy":"2022-12-30T01:15:21.020604Z","iopub.execute_input":"2022-12-30T01:15:21.020946Z","iopub.status.idle":"2022-12-30T01:15:21.065598Z","shell.execute_reply.started":"2022-12-30T01:15:21.020918Z","shell.execute_reply":"2022-12-30T01:15:21.064831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_scores = pd.DataFrame()\n\nn_splits_for_cross_valdition = 2\n\nindex = df_meta['cell_id'].tolist() \n\ntarget_name =['CD86','CD274' ,'CD270','CD155' ,'CD112','CD47','CD48','CD40','CD154','CD52','CD3','CD8','CD56','CD19','CD33','CD11c','HLA-A-B-C','CD45RA','CD123','CD7','CD105','CD49f','CD194','CD4','CD44','CD14','CD16','CD25',\n'CD45RO','CD279','TIGIT','Mouse-IgG1','Mouse-IgG2a','Mouse-IgG2b','Rat-IgG2b','CD20','CD335','CD31','Podoplanin','CD146','IgM','CD5','CD195','CD32',\n'CD196','CD185','CD103','CD69','CD62L','CD161','CD152','CD223','KLRG1','CD27','CD107a','CD95','CD134','HLA-DR','CD1c','CD11b','CD64','CD141','CD1d',\n'CD314','CD35','CD57','CD272','CD278','CD58','CD39','CX3CR1','CD24','CD21','CD11a','CD79b','CD244','CD169','integrinB7','CD268','CD42b','CD54','CD62P',\n'CD119','TCR','Rat-IgG1','Rat-IgG2a','CD192','CD122','FceRIa','CD41','CD137','CD163','CD83','CD124','CD13','CD2','CD226','CD29','CD303','CD49b','CD81','IgD',\n'CD18','CD28','CD38','CD127','CD45','CD22','CD71','CD26','CD115','CD63','CD304','CD36','CD172a','CD72','CD158','CD93','CD49a','CD49d','CD73',\n'CD9','TCRVa7.2','TCRVd2','LOX-1','CD158b','CD158e1','CD142','CD319','CD352','CD94','CD162','CD85j','CD23','CD328','HLA-E','CD82','CD101','CD88','CD224']\n","metadata":{"execution":{"iopub.status.busy":"2022-12-30T01:15:21.072062Z","iopub.execute_input":"2022-12-30T01:15:21.074656Z","iopub.status.idle":"2022-12-30T01:15:21.091728Z","shell.execute_reply.started":"2022-12-30T01:15:21.074616Z","shell.execute_reply":"2022-12-30T01:15:21.090976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# X, y","metadata":{}},{"cell_type":"code","source":"X=df_m\nX","metadata":{"execution":{"iopub.status.busy":"2022-12-30T01:15:21.095195Z","iopub.execute_input":"2022-12-30T01:15:21.095621Z","iopub.status.idle":"2022-12-30T01:15:21.113557Z","shell.execute_reply.started":"2022-12-30T01:15:21.095586Z","shell.execute_reply":"2022-12-30T01:15:21.112469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y =df_targets\ny","metadata":{"execution":{"iopub.status.busy":"2022-12-30T01:15:21.115492Z","iopub.execute_input":"2022-12-30T01:15:21.116131Z","iopub.status.idle":"2022-12-30T01:15:21.165196Z","shell.execute_reply.started":"2022-12-30T01:15:21.116091Z","shell.execute_reply":"2022-12-30T01:15:21.163839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# X, y describe","metadata":{}},{"cell_type":"code","source":"X.describe()","metadata":{"execution":{"iopub.status.busy":"2022-12-30T01:19:35.709977Z","iopub.execute_input":"2022-12-30T01:19:35.710352Z","iopub.status.idle":"2022-12-30T01:19:35.749046Z","shell.execute_reply.started":"2022-12-30T01:19:35.710319Z","shell.execute_reply":"2022-12-30T01:19:35.748108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.describe()","metadata":{"execution":{"iopub.status.busy":"2022-12-30T01:15:21.166774Z","iopub.execute_input":"2022-12-30T01:15:21.167163Z","iopub.status.idle":"2022-12-30T01:15:21.806407Z","shell.execute_reply.started":"2022-12-30T01:15:21.167131Z","shell.execute_reply":"2022-12-30T01:15:21.805051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape, y.shape","metadata":{"execution":{"iopub.status.busy":"2022-12-30T01:15:21.808118Z","iopub.execute_input":"2022-12-30T01:15:21.808473Z","iopub.status.idle":"2022-12-30T01:15:21.816475Z","shell.execute_reply.started":"2022-12-30T01:15:21.808442Z","shell.execute_reply":"2022-12-30T01:15:21.815166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Models","metadata":{}},{"cell_type":"code","source":"model = LinearRegression(fit_intercept=True) \nkf = KFold(n_splits=n_splits_for_cross_valdition,  shuffle=True, random_state= 0 )\ny_pred = pd.DataFrame(cross_val_predict(model, X, y, cv=kf),index=index,columns=target_name)  \n\nfor i in target_name:\n    s = r2_score(y[i],y_pred[i])\n    df_scores.loc['LinearRegression R-Squared ',[i]]  = s\n    s = np.corrcoef(y[i],y_pred[i])[0,1]\n    df_scores.loc['LinearRegression Correlation Coeff',[i]]  = s\n    s = mean_squared_error(y[i],y_pred[i], squared=False) # squared=False -> RMSE, not MSE \n    df_scores.loc['LinearRegression RMSE',[i]]  = s\n\n\ndf_scores.style.background_gradient(cmap ='RdYlGn', low=0, high=0, axis=0, subset=None)\ndf_scores","metadata":{"execution":{"iopub.status.busy":"2022-12-30T01:15:21.818150Z","iopub.execute_input":"2022-12-30T01:15:21.818494Z","iopub.status.idle":"2022-12-30T01:15:23.342690Z","shell.execute_reply.started":"2022-12-30T01:15:21.818464Z","shell.execute_reply":"2022-12-30T01:15:23.341708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = RidgeCV(alphas=[1e-1, 1, 1e1,1e2,1e3,1e4,1e5]).fit(X, y)\nalpha_selected = model.alpha_\nprint('Optimal alpha found by cross-validation: ', alpha_selected )\n\nmodel = Ridge(alpha = alpha_selected )\nkf = KFold(n_splits=n_splits_for_cross_valdition,  shuffle=True, random_state= 0 )\ny_pred = pd.DataFrame(cross_val_predict(model, X, y, cv=kf),index=index,columns=target_name)  \n\nfor i in target_name:\n    s = r2_score(y[i],y_pred[i])\n    df_scores.loc['RidgeCV R-Squared ',[i]]  = s\n    s = np.corrcoef(y[i],y_pred[i])[0,1]\n    df_scores.loc['RidgeCV Correlation Coeff',[i]]  = s\n    s = mean_squared_error(y[i],y_pred[i], squared=False) # squared=False -> RMSE, not MSE \n    df_scores.loc['RidgeCV RMSE',[i]]  = s\n\n\ndf_scores.style.background_gradient(cmap ='RdYlGn', low=0, high=0, axis=0, subset=None)\ndf_scores","metadata":{"execution":{"iopub.status.busy":"2022-12-30T01:15:23.343828Z","iopub.execute_input":"2022-12-30T01:15:23.344645Z","iopub.status.idle":"2022-12-30T01:15:25.069998Z","shell.execute_reply.started":"2022-12-30T01:15:23.344615Z","shell.execute_reply":"2022-12-30T01:15:25.068748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import MultiTaskLasso\nfrom sklearn.linear_model import MultiTaskLassoCV\n\nmodel = MultiTaskLassoCV(alphas = None).fit(X, y)\n\n# best alpha parameter\nalpha = model.alpha_\nprint('Optimal alpha parameter: ', alpha )\n\nmodel = MultiTaskLasso(alpha = model.alpha_)\n\nkf = KFold(n_splits=n_splits_for_cross_valdition,  shuffle=True, random_state= 0 )\ny_pred = pd.DataFrame(cross_val_predict(model, X, y, cv=kf),index=index,columns=target_name)  \n\nfor i in target_name:\n    s = r2_score(y[i],y_pred[i])\n    df_scores.loc['MultiTaskLasso R-Squared ',[i]]  = s\n    s = np.corrcoef(y[i],y_pred[i])[0,1]\n    df_scores.loc['MultiTaskLasso Correlation Coeff',[i]]  = s\n    s = mean_squared_error(y[i],y_pred[i], squared=False) # squared=False -> RMSE, not MSE \n    df_scores.loc['MultiTaskLasso RMSE',[i]]  = s\n\ndf_scores.style.background_gradient(cmap ='RdYlGn', low=0, high=0, axis=0, subset=None)","metadata":{"execution":{"iopub.status.busy":"2022-12-30T01:15:25.071317Z","iopub.execute_input":"2022-12-30T01:15:25.071628Z","iopub.status.idle":"2022-12-30T01:19:34.305009Z","shell.execute_reply.started":"2022-12-30T01:15:25.071598Z","shell.execute_reply":"2022-12-30T01:19:34.304104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save resuls","metadata":{}},{"cell_type":"code","source":"df_scores.to_csv('Scores.csv')","metadata":{"execution":{"iopub.status.busy":"2022-12-30T01:19:34.306438Z","iopub.execute_input":"2022-12-30T01:19:34.306932Z","iopub.status.idle":"2022-12-30T01:19:34.315399Z","shell.execute_reply.started":"2022-12-30T01:19:34.306902Z","shell.execute_reply":"2022-12-30T01:19:34.314684Z"},"trusted":true},"execution_count":null,"outputs":[]}]}