{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-20T08:31:43.521293Z","iopub.execute_input":"2023-01-20T08:31:43.521899Z","iopub.status.idle":"2023-01-20T08:31:43.551323Z","shell.execute_reply.started":"2023-01-20T08:31:43.521757Z","shell.execute_reply":"2023-01-20T08:31:43.550016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_splits_for_cross_valdition = 2 # if bigger, then for models on all rna (22K features) it may cause RAM crash and would be slower - \n\nrandom_state_cross_validation = 0 # random to be used in generating the folds - changing it we can study how stable are our results","metadata":{"execution":{"iopub.status.busy":"2023-01-20T08:31:43.553132Z","iopub.execute_input":"2023-01-20T08:31:43.554482Z","iopub.status.idle":"2023-01-20T08:31:43.559269Z","shell.execute_reply.started":"2023-01-20T08:31:43.554436Z","shell.execute_reply":"2023-01-20T08:31:43.558074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preparations","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.preprocessing import OneHotEncoder\n\nimport time\nt0start = time.time()\n\nimport os","metadata":{"execution":{"iopub.status.busy":"2023-01-20T08:31:43.560746Z","iopub.execute_input":"2023-01-20T08:31:43.561178Z","iopub.status.idle":"2023-01-20T08:31:44.978736Z","shell.execute_reply.started":"2023-01-20T08:31:43.561141Z","shell.execute_reply":"2023-01-20T08:31:44.977097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load data for target\n","metadata":{}},{"cell_type":"code","source":"%%time\ndf_y = pd.read_hdf('/kaggle/input/open-problems-multimodal/train_cite_targets.h5')\ndf_y\ndf_scores=pd.DataFrame()","metadata":{"execution":{"iopub.status.busy":"2023-01-20T08:31:44.981740Z","iopub.execute_input":"2023-01-20T08:31:44.982255Z","iopub.status.idle":"2023-01-20T08:31:46.008174Z","shell.execute_reply.started":"2023-01-20T08:31:44.982210Z","shell.execute_reply":"2023-01-20T08:31:46.006340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Key Params","metadata":{}},{"cell_type":"code","source":"target_names = df_y.columns\ntarget_names_cd=[]","metadata":{"execution":{"iopub.status.busy":"2023-01-20T08:31:46.010572Z","iopub.execute_input":"2023-01-20T08:31:46.011350Z","iopub.status.idle":"2023-01-20T08:31:46.020221Z","shell.execute_reply.started":"2023-01-20T08:31:46.011284Z","shell.execute_reply":"2023-01-20T08:31:46.018250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(target_names)):\n    if 'CD' in target_names[i]:\n        target_names_cd.append(target_names[i])","metadata":{"execution":{"iopub.status.busy":"2023-01-20T08:31:46.022799Z","iopub.execute_input":"2023-01-20T08:31:46.023616Z","iopub.status.idle":"2023-01-20T08:31:46.033885Z","shell.execute_reply.started":"2023-01-20T08:31:46.023554Z","shell.execute_reply":"2023-01-20T08:31:46.031909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_names = target_names_cd[64:72]","metadata":{"execution":{"iopub.status.busy":"2023-01-20T08:31:46.036586Z","iopub.execute_input":"2023-01-20T08:31:46.037639Z","iopub.status.idle":"2023-01-20T08:31:46.045589Z","shell.execute_reply.started":"2023-01-20T08:31:46.037572Z","shell.execute_reply":"2023-01-20T08:31:46.043852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_names","metadata":{"execution":{"iopub.status.busy":"2023-01-20T08:31:46.048581Z","iopub.execute_input":"2023-01-20T08:31:46.051415Z","iopub.status.idle":"2023-01-20T08:31:46.075991Z","shell.execute_reply.started":"2023-01-20T08:31:46.050982Z","shell.execute_reply":"2023-01-20T08:31:46.074300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Ridge on ALL RNA features","metadata":{}},{"cell_type":"code","source":"%%time\ndf_rna = pd.read_hdf('/kaggle/input/open-problems-multimodal/train_cite_inputs.h5')\ndisplay(df_rna) \n","metadata":{"execution":{"iopub.status.busy":"2023-01-20T08:31:46.078515Z","iopub.execute_input":"2023-01-20T08:31:46.079177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_rna.values\nX","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom sklearn.linear_model import RidgeCV\nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.model_selection import KFold \nfrom sklearn.metrics import r2_score\nfrom sklearn.metrics import mean_squared_error\n\nalpha_selected = 1e4\nfor t in df_rna.columns:\n    for i in range(len(target_names)):\n        if target_names[i] in t: \n            rna_name = t\n            for i in range(len(target_names)):\n                y = df_y[target_names[i]]\n                \n\n#print('Optimal alpha found by cross-validation: ', alpha_selected )\n\n\n                model = Ridge(alpha = alpha_selected )\n                kf = KFold(n_splits=n_splits_for_cross_valdition,  shuffle=True, random_state= random_state_cross_validation )\n                y_pred = cross_val_predict(model, X, y, cv=kf)\n\n                s = r2_score(y,y_pred)\n                df_scores.loc['r2 Ridge allRNA', target_names[i] ]  = s\n                s = np.corrcoef(y,y_pred)[0,1]\n                df_scores.loc['Corr Ridge allRNA', target_names[i] ]  = s\n                s = mean_squared_error(y,y_pred, squared=False) # squared=False -> RMSE, not MSE \n                df_scores.loc['RMSE Ridge allRNA', target_names[i] ]  = s\n#df_scores.to_csv('all_rna_features.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prediction based on both rna and proteins (except target one )","metadata":{}},{"cell_type":"code","source":"%%time\nfrom sklearn.linear_model import RidgeCV\nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.model_selection import KFold \nfrom sklearn.metrics import r2_score\nfrom sklearn.metrics import mean_squared_error\nalpha_selected = 1e4\nfor t in df_rna.columns:\n    for i in range(len(target_names)):\n        if target_names[i] in t: \n            rna_name = t\n            for i in range(len(target_names)):\n                y = df_y[target_names[i]]\n                X  = pd.concat( [ df_rna, df_y], axis = 1 )# .values\n                X = X.drop(target_names[i] , axis = 1)\n\n#print('Optimal alpha found by cross-validation: ', alpha_selected )\n\n\n                model = Ridge(alpha = alpha_selected)\n                kf = KFold(n_splits=n_splits_for_cross_valdition,  shuffle=True, random_state= random_state_cross_validation )\n                y_pred = cross_val_predict(model, X, y, cv=kf)\n\n                s = r2_score(y,y_pred)\n                df_scores.loc['r2 Ridge allRNA+Proteins', target_names[i] ]  = s\n                s = np.corrcoef(y,y_pred)[0,1]\n                df_scores.loc['Corr Ridge allRNA+Proteins', target_names[i] ]  = s\n                s = mean_squared_error(y,y_pred, squared=False) # squared=False -> RMSE, not MSE \n                df_scores.loc['RMSE Ridge allRNA+Proteins', target_names[i] ]  = s","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prediction based on RNA (except own RNA) and Proteins (except own proteins)","metadata":{}},{"cell_type":"code","source":"%%time\nfrom sklearn.linear_model import RidgeCV\nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.model_selection import KFold \nfrom sklearn.metrics import r2_score\nfrom sklearn.metrics import mean_squared_error\nalpha_selected = 1e4\nfor t in df_rna.columns:\n    for i in range(len(target_names)):\n        if target_names[i] in t: \n            rna_name = t\n            for i in range(len(target_names)):\n                y = df_y[target_names[i]]\n                X  = pd.concat( [ df_rna, df_y], axis = 1 )# .values\n                X = X.drop(target_names[i] , axis = 1)\n                X.drop(rna_name , axis = 1, inplace = True)\n                \n\n#print('Optimal alpha found by cross-validation: ', alpha_selected )\n\n\n                model = Ridge(alpha = alpha_selected)\n                kf = KFold(n_splits=n_splits_for_cross_valdition,  shuffle=True, random_state= random_state_cross_validation )\n                y_pred = cross_val_predict(model, X, y, cv=kf)\n\n                s = r2_score(y,y_pred)\n                df_scores.loc['r2 Ridge Proteins + RNA except own', target_names[i] ]  = s\n                s = np.corrcoef(y,y_pred)[0,1]\n                df_scores.loc['Corr Ridge Proteins + RNA except own', target_names[i] ]  = s\n                s = mean_squared_error(y,y_pred, squared=False) # squared=False -> RMSE, not MSE \n                df_scores.loc['RMSE Ridge Proteins + RNA except own', target_names[i] ]  = s\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  Prediction based on RNA (all except own RNA) ","metadata":{}},{"cell_type":"code","source":"%%time\nfrom sklearn.linear_model import RidgeCV\nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.model_selection import KFold \nfrom sklearn.metrics import r2_score\nfrom sklearn.metrics import mean_squared_error\nalpha_selected = 1e4\nfor t in df_rna.columns:\n    for i in range(len(target_names)):\n        if target_names[i] in t: \n            rna_name = t\n            for i in range(len(target_names)):\n                y = df_y[target_names[i]]\n                X  = pd.concat( [ df_rna, df_y], axis = 1 )# .values\n                X = X.drop(target_names[i] , axis = 1)\n                X.drop(rna_name , axis = 1, inplace = True)\n                l = [t for t in X.columns if 'ENSG' in t ]\n                X = X[l]\n                \n\n#print('Optimal alpha found by cross-validation: ', alpha_selected )\n\n\n                model = Ridge(alpha = alpha_selected)\n                kf = KFold(n_splits=n_splits_for_cross_valdition,  shuffle=True, random_state= random_state_cross_validation )\n                y_pred = cross_val_predict(model, X, y, cv=kf)\n\n                s = r2_score(y,y_pred)\n                df_scores.loc['r2 Ridge all RNA except own', target_names[i] ]  = s\n                s = np.corrcoef(y,y_pred)[0,1]\n                df_scores.loc['Corr Ridge all RNA except own', target_names[i] ]  = s\n                s = mean_squared_error(y,y_pred, squared=False) # squared=False -> RMSE, not MSE \n                df_scores.loc['RMSE Ridge all RNA except own', target_names[i] ]  = s\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_scores.to_csv('all_rna_features_9.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}