{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about?","metadata":{}},{"cell_type":"markdown","source":"This notebook explores proteins by cell types.\n\n### Comparison of model results: \nLinearRegression, Ridge and MultiTaskLasso models give similar results, as shown in the scatterplot. For further data processing the Ridge model was taken, because this model on average showed the best parameters R-Squared, Correlation Coeff and RMSE relative to other models.\n\n### Ridge model:\n\nThere is a linear relationship (Correlation Coefficient>0.6) between:\n* CD mean and Ridge Correlation Coefficient;\n* CD mean predicted by Ridge and Ridge Correlation Coefficient.\n\nThere is a strong linear relationship (Correlation Coefficient>0.7) between:\n* Ridge Correlation Coefficient and Ridge RMSE;\n* CD mean predicted by Ridge and CD std predicted by Ridge;\n* CD mean and CD std.\n\n### Output:\nСD was sorted by value 'R-Squared' obtained by the Ridge model.","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# keeps the plots in one place. calls image as static pngs\n%matplotlib inline \nimport matplotlib.pyplot as plt # side-stepping mpl backend\nimport matplotlib.gridspec as gridspec # subplots\nimport mpld3 as mpl\n\n\n#Import models from scikit learn module:\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.linear_model import Ridge\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error\nfrom sklearn.linear_model import RidgeCV\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.model_selection import KFold \nfrom sklearn.metrics import r2_score\nfrom sklearn import metrics\nimport scipy.stats\nimport seaborn as sns\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-17T04:04:37.632822Z","iopub.execute_input":"2023-01-17T04:04:37.633270Z","iopub.status.idle":"2023-01-17T04:04:38.504774Z","shell.execute_reply.started":"2023-01-17T04:04:37.633182Z","shell.execute_reply":"2023-01-17T04:04:38.503834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load data","metadata":{}},{"cell_type":"code","source":"df_targets = pd.read_hdf('/kaggle/input/open-problems-multimodal/train_cite_targets.h5')\ndf_meta = pd.read_csv('/kaggle/input/feature-shop-for-multimodal-singlecell-competition/_citeseq_meta_all_text_also.csv')\ndf_meta = df_meta[df_meta['Train0OrTest1']==0]\n# df_meta = pd.read_csv('/kaggle/input/open-problems-multimodal/metadata.csv')","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:04:38.506462Z","iopub.execute_input":"2023-01-17T04:04:38.507449Z","iopub.status.idle":"2023-01-17T04:04:39.516021Z","shell.execute_reply.started":"2023-01-17T04:04:38.507411Z","shell.execute_reply":"2023-01-17T04:04:39.514477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare data","metadata":{}},{"cell_type":"code","source":"df_m= df_meta[['cell_id','HSC cell_type', 'EryP cell_type', 'NeuP cell_type', 'MasP cell_type', 'MkP cell_type','MoP cell_type','BP cell_type']].set_index('cell_id')\ndf_m\n","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:04:39.517943Z","iopub.execute_input":"2023-01-17T04:04:39.518395Z","iopub.status.idle":"2023-01-17T04:04:39.547311Z","shell.execute_reply.started":"2023-01-17T04:04:39.518360Z","shell.execute_reply":"2023-01-17T04:04:39.546088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_scores = pd.DataFrame()\n\nn_splits_for_cross_valdition = 2\n\nindex = df_meta['cell_id'].tolist() \n\ntarget_name =['CD86','CD274' ,'CD270','CD155' ,'CD112','CD47','CD48','CD40','CD154','CD52','CD3','CD8','CD56','CD19','CD33','CD11c','HLA-A-B-C','CD45RA','CD123','CD7','CD105','CD49f','CD194','CD4','CD44','CD14','CD16','CD25',\n'CD45RO','CD279','TIGIT','Mouse-IgG1','Mouse-IgG2a','Mouse-IgG2b','Rat-IgG2b','CD20','CD335','CD31','Podoplanin','CD146','IgM','CD5','CD195','CD32',\n'CD196','CD185','CD103','CD69','CD62L','CD161','CD152','CD223','KLRG1','CD27','CD107a','CD95','CD134','HLA-DR','CD1c','CD11b','CD64','CD141','CD1d',\n'CD314','CD35','CD57','CD272','CD278','CD58','CD39','CX3CR1','CD24','CD21','CD11a','CD79b','CD244','CD169','integrinB7','CD268','CD42b','CD54','CD62P',\n'CD119','TCR','Rat-IgG1','Rat-IgG2a','CD192','CD122','FceRIa','CD41','CD137','CD163','CD83','CD124','CD13','CD2','CD226','CD29','CD303','CD49b','CD81','IgD',\n'CD18','CD28','CD38','CD127','CD45','CD22','CD71','CD26','CD115','CD63','CD304','CD36','CD172a','CD72','CD158','CD93','CD49a','CD49d','CD73',\n'CD9','TCRVa7.2','TCRVd2','LOX-1','CD158b','CD158e1','CD142','CD319','CD352','CD94','CD162','CD85j','CD23','CD328','HLA-E','CD82','CD101','CD88','CD224']\n","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:04:39.550222Z","iopub.execute_input":"2023-01-17T04:04:39.550597Z","iopub.status.idle":"2023-01-17T04:04:39.560509Z","shell.execute_reply.started":"2023-01-17T04:04:39.550566Z","shell.execute_reply":"2023-01-17T04:04:39.559588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# X, y","metadata":{}},{"cell_type":"code","source":"X=df_m\nX","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:04:39.561692Z","iopub.execute_input":"2023-01-17T04:04:39.562131Z","iopub.status.idle":"2023-01-17T04:04:39.583443Z","shell.execute_reply.started":"2023-01-17T04:04:39.562095Z","shell.execute_reply":"2023-01-17T04:04:39.581818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"y =df_targets\ny","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:04:39.585156Z","iopub.execute_input":"2023-01-17T04:04:39.585582Z","iopub.status.idle":"2023-01-17T04:04:39.640083Z","shell.execute_reply.started":"2023-01-17T04:04:39.585542Z","shell.execute_reply":"2023-01-17T04:04:39.639302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# X, y describe","metadata":{}},{"cell_type":"code","source":"X.describe()","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:04:39.641252Z","iopub.execute_input":"2023-01-17T04:04:39.641657Z","iopub.status.idle":"2023-01-17T04:04:39.685018Z","shell.execute_reply.started":"2023-01-17T04:04:39.641630Z","shell.execute_reply":"2023-01-17T04:04:39.683124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.describe()","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:04:39.686997Z","iopub.execute_input":"2023-01-17T04:04:39.687442Z","iopub.status.idle":"2023-01-17T04:04:40.256735Z","shell.execute_reply.started":"2023-01-17T04:04:39.687409Z","shell.execute_reply":"2023-01-17T04:04:40.255839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr = y.corr()\ncorr.style.background_gradient(cmap='RdYlGn')","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:04:40.258095Z","iopub.execute_input":"2023-01-17T04:04:40.258365Z","iopub.status.idle":"2023-01-17T04:04:43.758882Z","shell.execute_reply.started":"2023-01-17T04:04:40.258342Z","shell.execute_reply":"2023-01-17T04:04:43.757962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape, y.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:04:43.762099Z","iopub.execute_input":"2023-01-17T04:04:43.762882Z","iopub.status.idle":"2023-01-17T04:04:43.769656Z","shell.execute_reply.started":"2023-01-17T04:04:43.762827Z","shell.execute_reply":"2023-01-17T04:04:43.768158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Models","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import MultiTaskLasso\nfrom sklearn.linear_model import MultiTaskLassoCV\n\nmodel = MultiTaskLassoCV(alphas = None).fit(X, y)\n\n# best alpha parameter\nalpha = model.alpha_\nprint('Optimal alpha parameter: ', alpha )\n\nmodel = MultiTaskLasso(alpha = model.alpha_)\n\nkf = KFold(n_splits=n_splits_for_cross_valdition,  shuffle=True, random_state= 0 )\ny_pred = pd.DataFrame(cross_val_predict(model, X, y, cv=kf),index=index,columns=target_name)  \n\nfor i in target_name:\n    s = r2_score(y[i],y_pred[i])\n    df_scores.loc['MultiTaskLasso R-Squared',[i]]  = s\n    s = np.corrcoef(y[i],y_pred[i])[0,1]\n    df_scores.loc['MultiTaskLasso Correlation Coeff',[i]]  = s\n    s = mean_squared_error(y[i],y_pred[i], squared=False) # squared=False -> RMSE, not MSE \n    df_scores.loc['MultiTaskLasso RMSE',[i]]  = s\n\ndf_scores","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:04:43.770725Z","iopub.execute_input":"2023-01-17T04:04:43.771077Z","iopub.status.idle":"2023-01-17T04:09:01.839624Z","shell.execute_reply.started":"2023-01-17T04:04:43.771018Z","shell.execute_reply":"2023-01-17T04:09:01.838367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = LinearRegression(fit_intercept=True) \nkf = KFold(n_splits=n_splits_for_cross_valdition,  shuffle=True, random_state= 0 )\ny_pred = pd.DataFrame(cross_val_predict(model, X, y, cv=kf),index=index,columns=target_name)  \n\nfor i in target_name:\n    s = r2_score(y[i],y_pred[i])\n    df_scores.loc['LinearRegression R-Squared',[i]]  = s\n    s = np.corrcoef(y[i],y_pred[i])[0,1]\n    df_scores.loc['LinearRegression Correlation Coeff',[i]]  = s\n    s = mean_squared_error(y[i],y_pred[i], squared=False) # squared=False -> RMSE, not MSE \n    df_scores.loc['LinearRegression RMSE',[i]]  = s\n\n\ndf_scores.style.background_gradient(cmap ='RdYlGn', low=0, high=0, axis=0, subset=None)\ndf_scores","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:09:01.840957Z","iopub.execute_input":"2023-01-17T04:09:01.841322Z","iopub.status.idle":"2023-01-17T04:09:02.800228Z","shell.execute_reply.started":"2023-01-17T04:09:01.841293Z","shell.execute_reply":"2023-01-17T04:09:02.798900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = RidgeCV(alphas=[1e-1, 1, 1e1,1e2,1e3,1e4,1e5]).fit(X, y)\nalpha_selected = model.alpha_\nprint('Optimal alpha found by cross-validation: ', alpha_selected )\n\nmodel = Ridge(alpha = alpha_selected )\nkf = KFold(n_splits=n_splits_for_cross_valdition,  shuffle=True, random_state= 0 )\ny_pred = pd.DataFrame(cross_val_predict(model, X, y, cv=kf),index=index,columns=target_name)  \n\nfor i in target_name:\n    s = r2_score(y[i],y_pred[i])\n    df_scores.loc['Ridge R-Squared',[i]]  = s\n    s = np.corrcoef(y[i],y_pred[i])[0,1]\n    df_scores.loc['Ridge Correlation Coeff',[i]]  = s\n    s = mean_squared_error(y[i],y_pred[i], squared=False) # squared=False -> RMSE, not MSE \n    df_scores.loc['Ridge RMSE',[i]]  = s\n\n\ndf_scores.style.background_gradient(cmap ='RdYlGn', low=0, high=0, axis=0, subset=None)\ndf_scores","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:09:02.801805Z","iopub.execute_input":"2023-01-17T04:09:02.802204Z","iopub.status.idle":"2023-01-17T04:09:04.432596Z","shell.execute_reply.started":"2023-01-17T04:09:02.802171Z","shell.execute_reply":"2023-01-17T04:09:04.431343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in target_name:\n    s = y[i].mean()\n    df_scores.loc['CD mean',[i]]  = s\n    s = y[i].std()\n    df_scores.loc['CD std',[i]]  = s\n    s = y_pred[i].mean()\n    df_scores.loc['CD mean pred',[i]]  = s\n    s = y_pred[i].std()\n    df_scores.loc['CD std pred',[i]]  = s\n\ndf_scores","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:09:04.433867Z","iopub.execute_input":"2023-01-17T04:09:04.434445Z","iopub.status.idle":"2023-01-17T04:09:05.075495Z","shell.execute_reply.started":"2023-01-17T04:09:04.434411Z","shell.execute_reply":"2023-01-17T04:09:05.074095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Comparison of models results","metadata":{}},{"cell_type":"code","source":"df_scor=df_scores.T","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:09:05.076859Z","iopub.execute_input":"2023-01-17T04:09:05.077792Z","iopub.status.idle":"2023-01-17T04:09:05.083809Z","shell.execute_reply.started":"2023-01-17T04:09:05.077733Z","shell.execute_reply":"2023-01-17T04:09:05.082083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.style.use('ggplot')\n\nfig, ax = plt.subplots(1,3)\n\nax[0].scatter(x=df_scor['LinearRegression R-Squared'], y=df_scor['Ridge R-Squared'])\nax[0].set(ylabel='LinearRegression R-Squared')\nax[0].set(xlabel='Ridge R-Squared')\n\nax[1].scatter(x=df_scor['LinearRegression R-Squared'], y=df_scor['MultiTaskLasso R-Squared'])\nax[1].set(ylabel='LinearRegression R-Squared')\nax[1].set(xlabel='MultiTaskLasso R-Squared')\n\nax[2].scatter(x=df_scor['Ridge R-Squared'], y=df_scor['MultiTaskLasso R-Squared'])\nax[2].set(ylabel='Ridge R-Squared')\nax[2].set(xlabel='MultiTaskLasso R-Squared')\n\nplt.title('')\nfor a in ax.flat:\n    a.set_facecolor('seashell')\n\nfig.set_figwidth(30)  \nfig.set_figheight(10) \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:09:05.085612Z","iopub.execute_input":"2023-01-17T04:09:05.086016Z","iopub.status.idle":"2023-01-17T04:09:05.525911Z","shell.execute_reply.started":"2023-01-17T04:09:05.085977Z","shell.execute_reply":"2023-01-17T04:09:05.525210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_scor.agg({'LinearRegression R-Squared': ['min', 'max', 'mean'],\n             'Ridge R-Squared': ['min', 'max', 'mean'],\n             'MultiTaskLasso R-Squared': ['min', 'max', 'mean']})","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:09:05.526941Z","iopub.execute_input":"2023-01-17T04:09:05.527438Z","iopub.status.idle":"2023-01-17T04:09:05.544250Z","shell.execute_reply.started":"2023-01-17T04:09:05.527410Z","shell.execute_reply":"2023-01-17T04:09:05.542743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,3)\n\nax[0].scatter(x=df_scor['LinearRegression Correlation Coeff'], y=df_scor['Ridge Correlation Coeff'])\nax[0].set(ylabel='LinearRegression Correlation Coeff')\nax[0].set(xlabel='Ridge Correlation Coeff')\n\nax[1].scatter(x=df_scor['LinearRegression Correlation Coeff'], y=df_scor['MultiTaskLasso Correlation Coeff'])\nax[1].set(ylabel='LinearRegression Correlation Coeff')\nax[1].set(xlabel='MultiTaskLasso Correlation Coeff')\n\nax[2].scatter(x=df_scor['Ridge Correlation Coeff'], y=df_scor['MultiTaskLasso Correlation Coeff'])\nax[2].set(ylabel='Ridge Correlation Coeff')\nax[2].set(xlabel='MultiTaskLassoCorrelation Coeff')\n\nplt.title('')\nfor a in ax.flat:\n    a.set_facecolor('seashell')\nplt.style.use('ggplot')\n\nfig.set_figwidth(30)  \nfig.set_figheight(10) \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:09:05.545855Z","iopub.execute_input":"2023-01-17T04:09:05.546441Z","iopub.status.idle":"2023-01-17T04:09:06.023366Z","shell.execute_reply.started":"2023-01-17T04:09:05.546412Z","shell.execute_reply":"2023-01-17T04:09:06.021889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_scor.agg({'LinearRegression Correlation Coeff': ['min', 'max', 'mean'],\n             'Ridge Correlation Coeff': ['min', 'max', 'mean'],\n             'MultiTaskLasso Correlation Coeff': ['min', 'max', 'mean']})","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:09:06.024700Z","iopub.execute_input":"2023-01-17T04:09:06.026116Z","iopub.status.idle":"2023-01-17T04:09:06.040670Z","shell.execute_reply.started":"2023-01-17T04:09:06.026028Z","shell.execute_reply":"2023-01-17T04:09:06.039687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,3)\n\n\nax[0].scatter(x=df_scor['LinearRegression RMSE'], y=df_scor['Ridge RMSE'])\nax[0].set(ylabel='LinearRegression RMSE')\nax[0].set(xlabel='Ridge RMSE')\n\nax[1].scatter(x=df_scor['LinearRegression RMSE'], y=df_scor['MultiTaskLasso RMSE'])\nax[1].set(ylabel='LinearRegression RMSE')\nax[1].set(xlabel='MultiTaskLasso RMSE')\n\nax[2].scatter(x=df_scor['Ridge RMSE'], y=df_scor['MultiTaskLasso RMSE'])\nax[2].set(ylabel='Ridge RMSE')\nax[2].set(xlabel='MultiTaskLasso RMSE')\n\nplt.title('')\nfor a in ax.flat:\n    a.set_facecolor('seashell')\nplt.style.use('ggplot')\n\nfig.set_figwidth(30)  \nfig.set_figheight(10) \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:09:06.041703Z","iopub.execute_input":"2023-01-17T04:09:06.042083Z","iopub.status.idle":"2023-01-17T04:09:06.458206Z","shell.execute_reply.started":"2023-01-17T04:09:06.042028Z","shell.execute_reply":"2023-01-17T04:09:06.456464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_scor.agg({'LinearRegression RMSE': ['min', 'max', 'mean'],\n             'Ridge RMSE': ['min', 'max', 'mean'],\n             'MultiTaskLasso RMSE': ['min', 'max', 'mean']})\n","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:09:06.460222Z","iopub.execute_input":"2023-01-17T04:09:06.460664Z","iopub.status.idle":"2023-01-17T04:09:06.477566Z","shell.execute_reply.started":"2023-01-17T04:09:06.460626Z","shell.execute_reply":"2023-01-17T04:09:06.476540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Scatter plot of scores","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2)\nax[0].scatter(x=df_scor['CD mean pred'], y=df_scor['CD mean'])\nax[0].set(ylabel='CD mean predicted by Ridge')\nax[0].set(xlabel='CD mean')\n# ax[0].plot(x0, fit0[0] * x0 + fit0[1], color='red')\nax[1].scatter(x=df_scor['CD std pred'], y=df_scor['CD std'])\nax[1].set(ylabel='CD std predicted by Ridge')\nax[1].set(xlabel='CD std')\nfig.set_figwidth(30)  \nfig.set_figheight(10) \nfor a in ax.flat:\n    a.set_facecolor('seashell') \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:23:01.435032Z","iopub.execute_input":"2023-01-17T04:23:01.435588Z","iopub.status.idle":"2023-01-17T04:23:01.775394Z","shell.execute_reply.started":"2023-01-17T04:23:01.435551Z","shell.execute_reply":"2023-01-17T04:23:01.773996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Scatter plot of scores","metadata":{}},{"cell_type":"code","source":"x0=df_scor['Ridge Correlation Coeff']\ny0=df_scor['Ridge RMSE']    \nfit0 = np.polyfit(x0, y0, 1)\n\nfig, ax = plt.subplots()\nplt.scatter(x=df_scor['Ridge Correlation Coeff'], y=df_scor['Ridge RMSE'])\nplt.plot(x0, fit0[0] * x0 + fit0[1], color='red')\n# ax[0].text(0.05, 5, 'y = ' + '{:.2f}'.format(fit0[1]) + ' + {:.2f}'.format(fit0[0]) + 'x', size=14)\nplt.text(0, 7,'Pearson correlation r='+ '{:.2f}'.format(scipy.stats.pearsonr(x0, y0)[0])+' pvalue='+ '{}'.format(scipy.stats.pearsonr(x0, y0)[1]))\nplt.ylabel('Ridge Correlation Coeff')\nplt.xlabel('Ridge RMSE')\nfig.set_figwidth(10)   \nfig.set_figheight(8)\nax.set_facecolor('seashell')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:09:06.822788Z","iopub.execute_input":"2023-01-17T04:09:06.823544Z","iopub.status.idle":"2023-01-17T04:09:07.005451Z","shell.execute_reply.started":"2023-01-17T04:09:06.823503Z","shell.execute_reply":"2023-01-17T04:09:07.004158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There is a strong linear relationship (Correlation Coefficient > 0.7) between Ridge Correlation Coefficient and Ridge RMSE.`","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(2,2)\n\nx1=df_scor['CD mean pred']\ny1=df_scor['Ridge Correlation Coeff']    \nfit1 = np.polyfit(x1, y1, 1)\n\nax[0,0].scatter(x=df_scor['CD mean pred'], y=df_scor['Ridge Correlation Coeff'])\nax[0,0].set(ylabel='CD mean predicted by Ridge')\nax[0,0].set(xlabel='Ridge Correlation Coeff')\nax[0,0].plot(x1, fit1[0] * x1 + fit1[1], color='red')\n# ax[1].text(0.05, 0.65, 'y = ' + '{:.2f}'.format(fit1[1]) + ' + {:.2f}'.format(fit1[0]) + 'x', size=14)\nax[0,0].text(0.02, 0.735,'Pearson correlation r='+ '{:.2f}'.format(scipy.stats.pearsonr(x1, y1)[0])+' pvalue='+ '{}'.format(scipy.stats.pearsonr(x1, y1)[1]))\n\n\nx2=df_scor['CD mean pred']\ny2=df_scor['CD std pred']    \nfit2 = np.polyfit(x2, y2, 1)\n\nax[0,1].scatter(x=df_scor['CD mean pred'], y=df_scor['CD std pred'])\nax[0,1].set(ylabel='CD mean predicted by Ridge')\nax[0,1].set(xlabel='CD std predicted by Ridge')\nax[0,1].plot(x2, fit2[0] * x2 + fit2[1], color='red')\n# ax[2].text(0.05, 5, 'y = ' + '{:.2f}'.format(fit2[1]) + ' + {:.2f}'.format(fit2[0]) + 'x', size=14)\nax[0,1].text(0.02, 4.7,'Pearson correlation r='+ '{:.2f}'.format(scipy.stats.pearsonr(x2, y2)[0])+' pvalue='+ '{}'.format(scipy.stats.pearsonr(x2, y2)[1]))\n\nx3=df_scor['CD mean']\ny3=df_scor['Ridge Correlation Coeff']    \nfit3 = np.polyfit(x3, y3, 1)\n\nax[1,0].scatter(x=df_scor['CD mean'], y=df_scor['Ridge Correlation Coeff'])\nax[1,0].set(ylabel='CD mean')\nax[1,0].set(xlabel='Ridge Correlation Coeff')\nax[1,0].plot(x3, fit3[0] * x3 + fit3[1], color='red')\n# ax[1].text(0.05, 0.65, 'y = ' + '{:.2f}'.format(fit1[1]) + ' + {:.2f}'.format(fit1[0]) + 'x', size=14)\nax[1,0].text(0.02, 0.735,'Pearson correlation r='+ '{:.2f}'.format(scipy.stats.pearsonr(x3, y3)[0])+' pvalue='+ '{}'.format(scipy.stats.pearsonr(x3, y3)[1]))\n\n\nx4=df_scor['CD mean']\ny4=df_scor['CD std']    \nfit4 = np.polyfit(x4, y4, 1)\n\nax[1,1].scatter(x=df_scor['CD mean'], y=df_scor['CD std'])\nax[1,1].set(ylabel='CD mean')\nax[1,1].set(xlabel='CD std')\nax[1,1].plot(x4, fit4[0] * x4 + fit4[1], color='red')\n# ax[2].text(0.05, 5, 'y = ' + '{:.2f}'.format(fit2[1]) + ' + {:.2f}'.format(fit2[0]) + 'x', size=14)\nax[1,1].text(0.02, 8.5,'Pearson correlation r='+ '{:.2f}'.format(scipy.stats.pearsonr(x4, y4)[0])+' pvalue='+ '{}'.format(scipy.stats.pearsonr(x4, y4)[1]))\n# plt.title('')\n\nfor a in ax.flat:\n    a.set_facecolor('seashell')\nplt.style.use('ggplot')\n\nfig.set_figwidth(30)  \nfig.set_figheight(20) \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:09:07.006699Z","iopub.execute_input":"2023-01-17T04:09:07.007008Z","iopub.status.idle":"2023-01-17T04:09:07.672142Z","shell.execute_reply.started":"2023-01-17T04:09:07.006982Z","shell.execute_reply":"2023-01-17T04:09:07.671305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* There is a linear relationship (Correlation Coefficient > 0.6) between CD mean predicted by Ridge and Ridge Correlation Coefficient; CD mean and Ridge Correlation Coefficient\n* There is a strong linear relationship (Correlation Coefficient > 0.7) between CD mean predicted by Ridge and CD std predicted by Ridge; CD mean and CD std. ","metadata":{}},{"cell_type":"markdown","source":"# Heat map of scores","metadata":{}},{"cell_type":"code","source":"df_scores_for_corr=df_scor[['Ridge R-Squared','Ridge Correlation Coeff','Ridge RMSE','CD mean','CD std','CD mean pred','CD std pred']].rename(columns={'CD mean pred': 'CD mean predicted by Ridge', 'CD std pred': 'CD std predicted by Ridge'})\n","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:09:07.673373Z","iopub.execute_input":"2023-01-17T04:09:07.673912Z","iopub.status.idle":"2023-01-17T04:09:07.680552Z","shell.execute_reply.started":"2023-01-17T04:09:07.673882Z","shell.execute_reply":"2023-01-17T04:09:07.678943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nplt.figure(figsize=(10,7))\n\nsns.heatmap(df_scores_for_corr.corr(method='pearson', min_periods=1), annot = True, fmt='.3g', center= 0, cmap= 'RdYlGn', cbar=False, square=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:09:07.682037Z","iopub.execute_input":"2023-01-17T04:09:07.682421Z","iopub.status.idle":"2023-01-17T04:09:08.037259Z","shell.execute_reply.started":"2023-01-17T04:09:07.682391Z","shell.execute_reply":"2023-01-17T04:09:08.035544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,7))\nsns.heatmap(df_scores_for_corr.corr(method='spearman', min_periods=1), annot = True, fmt='.3g', center= 0, cmap= 'RdYlGn', cbar=False, square=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:09:08.038718Z","iopub.execute_input":"2023-01-17T04:09:08.039116Z","iopub.status.idle":"2023-01-17T04:09:08.402061Z","shell.execute_reply.started":"2023-01-17T04:09:08.039078Z","shell.execute_reply":"2023-01-17T04:09:08.401131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Ridge","metadata":{}},{"cell_type":"code","source":"df_ridge = df_scor.loc[:,['Ridge R-Squared','Ridge Correlation Coeff','Ridge RMSE']]\ndf_ridge","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:09:08.406466Z","iopub.execute_input":"2023-01-17T04:09:08.406810Z","iopub.status.idle":"2023-01-17T04:09:08.421534Z","shell.execute_reply.started":"2023-01-17T04:09:08.406784Z","shell.execute_reply":"2023-01-17T04:09:08.420225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nplt.style.use('ggplot')\n\nfig, axes = plt.subplots(3, 1)\n\naxes[0].hist(df_ridge['Ridge R-Squared'])\naxes[0].set(xlabel='Ridge R-Squared')\n\naxes[1].hist(df_ridge['Ridge Correlation Coeff'])\naxes[1].set(xlabel='Ridge Correlation Coeff')\n\naxes[2].hist(df_ridge['Ridge RMSE'])\naxes[2].set(xlabel='Ridge RMSE')\n\n# axes[0].set_title('Ridge R-Squared')\n# axes[1].set_title('Ridge Correlation Coeff')\n# axes[2].set_title('Ridge RMSE')\n\nfor ax in axes.flat:\n    ax.set(ylabel='Frequency')\n\naxes[0].set_facecolor('seashell')\naxes[1].set_facecolor('seashell')\naxes[2].set_facecolor('seashell')\nfig.set_figwidth(20)  \nfig.set_figheight(30) \n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:09:08.423528Z","iopub.execute_input":"2023-01-17T04:09:08.423988Z","iopub.status.idle":"2023-01-17T04:09:09.102020Z","shell.execute_reply.started":"2023-01-17T04:09:08.423954Z","shell.execute_reply":"2023-01-17T04:09:09.101036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ridge.sort_values(by='Ridge R-Squared').tail(50)","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:09:09.103351Z","iopub.execute_input":"2023-01-17T04:09:09.103847Z","iopub.status.idle":"2023-01-17T04:09:09.120595Z","shell.execute_reply.started":"2023-01-17T04:09:09.103789Z","shell.execute_reply":"2023-01-17T04:09:09.119673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ridge.sort_values(by='Ridge R-Squared').head(50)","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:09:09.122092Z","iopub.execute_input":"2023-01-17T04:09:09.122486Z","iopub.status.idle":"2023-01-17T04:09:09.145431Z","shell.execute_reply.started":"2023-01-17T04:09:09.122458Z","shell.execute_reply":"2023-01-17T04:09:09.144103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save results","metadata":{}},{"cell_type":"code","source":"df_ridge.sort_values(by='Ridge R-Squared').to_csv('Scores Ridge.csv')","metadata":{"execution":{"iopub.status.busy":"2023-01-17T04:09:09.147714Z","iopub.execute_input":"2023-01-17T04:09:09.148130Z","iopub.status.idle":"2023-01-17T04:09:09.160094Z","shell.execute_reply.started":"2023-01-17T04:09:09.148095Z","shell.execute_reply":"2023-01-17T04:09:09.158618Z"},"trusted":true},"execution_count":null,"outputs":[]}]}