{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ?\n\nFor all proteins  we compare several simple Ridge based for predicting it:\n\nUse only one feature - own RNA for proteins,  we use protein rna - and predict by Ridge only based on that model\n\nUse cell type (one-hot encoded for prediction) \n\nUse cell type + own RNA\n\nUse top 100  PCA feautures as predictors \n","metadata":{}},{"cell_type":"markdown","source":"# Preparations","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\n# Main results will be stored here: \ndf_scores = pd.DataFrame()","metadata":{"execution":{"iopub.status.busy":"2023-01-17T11:39:04.360670Z","iopub.execute_input":"2023-01-17T11:39:04.361270Z","iopub.status.idle":"2023-01-17T11:39:04.384585Z","shell.execute_reply.started":"2023-01-17T11:39:04.361184Z","shell.execute_reply":"2023-01-17T11:39:04.383868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.preprocessing import OneHotEncoder\n\nimport time\nt0start = time.time()\n\nimport os","metadata":{"execution":{"iopub.status.busy":"2023-01-17T11:39:04.385620Z","iopub.execute_input":"2023-01-17T11:39:04.386011Z","iopub.status.idle":"2023-01-17T11:39:05.316596Z","shell.execute_reply.started":"2023-01-17T11:39:04.385988Z","shell.execute_reply":"2023-01-17T11:39:05.315208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load data for target","metadata":{}},{"cell_type":"code","source":"%%time\ndf_y = pd.read_hdf('/kaggle/input/open-problems-multimodal/train_cite_targets.h5')\ndf_y","metadata":{"execution":{"iopub.status.busy":"2023-01-17T11:39:05.319395Z","iopub.execute_input":"2023-01-17T11:39:05.319893Z","iopub.status.idle":"2023-01-17T11:39:06.075370Z","shell.execute_reply.started":"2023-01-17T11:39:05.319866Z","shell.execute_reply":"2023-01-17T11:39:06.074273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Key Params","metadata":{}},{"cell_type":"code","source":"target_names = df_y.columns","metadata":{"execution":{"iopub.status.busy":"2023-01-17T11:39:06.076944Z","iopub.execute_input":"2023-01-17T11:39:06.077587Z","iopub.status.idle":"2023-01-17T11:39:06.081897Z","shell.execute_reply.started":"2023-01-17T11:39:06.077551Z","shell.execute_reply":"2023-01-17T11:39:06.081211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(target_names)","metadata":{"execution":{"iopub.status.busy":"2023-01-17T11:39:06.082890Z","iopub.execute_input":"2023-01-17T11:39:06.083725Z","iopub.status.idle":"2023-01-17T11:39:06.097566Z","shell.execute_reply.started":"2023-01-17T11:39:06.083690Z","shell.execute_reply":"2023-01-17T11:39:06.096614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(target_names)):\n    target_name = target_names[i]\n\n    y = df_y[target_name]\n    df_scores.loc['Mean', target_name ]  = np.mean(y)\n    df_scores.loc['Std', target_name ]  = np.std(y)","metadata":{"execution":{"iopub.status.busy":"2023-01-17T11:39:06.098899Z","iopub.execute_input":"2023-01-17T11:39:06.100058Z","iopub.status.idle":"2023-01-17T11:39:06.323449Z","shell.execute_reply.started":"2023-01-17T11:39:06.100025Z","shell.execute_reply":"2023-01-17T11:39:06.321736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_scores\n","metadata":{"execution":{"iopub.status.busy":"2023-01-17T11:39:06.324724Z","iopub.execute_input":"2023-01-17T11:39:06.325067Z","iopub.status.idle":"2023-01-17T11:39:06.353452Z","shell.execute_reply.started":"2023-01-17T11:39:06.325035Z","shell.execute_reply":"2023-01-17T11:39:06.352433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  Predicting based on -  own RNA \n\ne.g. for all proteins we take their same predictor ","metadata":{}},{"cell_type":"code","source":"%%time\ndf_rna = pd.read_hdf('/kaggle/input/open-problems-multimodal/train_cite_inputs.h5')\ndisplay(df_rna) \n\nfor t in df_rna.columns:\n    for i in range(len(target_names)):\n        if target_names[i] in t: \n            rna_name = t","metadata":{"execution":{"iopub.status.busy":"2023-01-17T11:39:06.356977Z","iopub.execute_input":"2023-01-17T11:39:06.357261Z","iopub.status.idle":"2023-01-17T11:39:51.396689Z","shell.execute_reply.started":"2023-01-17T11:39:06.357235Z","shell.execute_reply":"2023-01-17T11:39:51.395539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for t in df_rna.columns:\n    for i in range(len(target_names)):\n        if target_names[i] in t: \n            rna_name = t\n            for i in range(len(target_names)):\n                target_name = target_names[i]\n                X = df_rna[[rna_name]]\n                y = df_y[target_name]\n                s = np.corrcoef(y,X.iloc[:,0])[0,1]\n                df_scores.loc['Corr ownRNA', target_name ]  = s","metadata":{"execution":{"iopub.status.busy":"2023-01-17T11:39:51.398130Z","iopub.execute_input":"2023-01-17T11:39:51.398456Z","iopub.status.idle":"2023-01-17T11:40:20.110405Z","shell.execute_reply.started":"2023-01-17T11:39:51.398427Z","shell.execute_reply":"2023-01-17T11:40:20.109063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_scores","metadata":{"execution":{"iopub.status.busy":"2023-01-17T11:40:20.112276Z","iopub.execute_input":"2023-01-17T11:40:20.112740Z","iopub.status.idle":"2023-01-17T11:40:20.138027Z","shell.execute_reply.started":"2023-01-17T11:40:20.112701Z","shell.execute_reply":"2023-01-17T11:40:20.136729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom sklearn.linear_model import RidgeCV\nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.model_selection import KFold \nfrom sklearn.metrics import r2_score\nfrom sklearn.metrics import mean_squared_error\n\n\nfor t in df_rna.columns:\n    for i in range(len(target_names)):\n        if target_names[i] in t: \n            rna_name = t\n            for i in range(len(target_names)):\n                X = df_rna[[rna_name]]\n                y = df_y[target_names[i]]\n\n                model = RidgeCV(alphas=[1e-1, 1, 1e1,1e2,1e3,1e4,1e5]).fit(X, y)\n                alpha_selected = model.alpha_\n#print('Optimal alpha found by cross-validation: ', alpha_selected )\n\n\n                model = Ridge(alpha = alpha_selected )\n                kf = KFold(n_splits=3,  shuffle=True, random_state= 0 )\n                y_pred = cross_val_predict(model, X, y, cv=kf)\n                s = r2_score(y,y_pred)\n                df_scores.loc['r2 Ridge ownRNA', target_names[i] ]  = s\n                s = np.corrcoef(y,y_pred)[0,1]\n                df_scores.loc['Corr Ridge ownRNA', target_names[i] ]  = s\n                s = mean_squared_error(y,y_pred, squared=False) # squared=False -> RMSE, not MSE \n                df_scores.loc['RMSE Ridge ownRNA', target_names[i] ]  = s\n","metadata":{"execution":{"iopub.status.busy":"2023-01-17T11:40:20.139391Z","iopub.execute_input":"2023-01-17T11:40:20.139716Z","iopub.status.idle":"2023-01-17T12:09:53.245173Z","shell.execute_reply.started":"2023-01-17T11:40:20.139681Z","shell.execute_reply":"2023-01-17T12:09:53.243525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_scores","metadata":{"execution":{"iopub.status.busy":"2023-01-17T12:09:53.246994Z","iopub.execute_input":"2023-01-17T12:09:53.247376Z","iopub.status.idle":"2023-01-17T12:09:53.273787Z","shell.execute_reply.started":"2023-01-17T12:09:53.247341Z","shell.execute_reply":"2023-01-17T12:09:53.272085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  Predicting based on - cell type information ","metadata":{}},{"cell_type":"code","source":"%%time\ndf_meta = pd.read_csv('/kaggle/input/feature-shop-for-multimodal-singlecell-competition/_citeseq_meta_all_text_also.csv', index_col = 0 )\ndf_meta = df_meta[df_meta['Train0OrTest1']==0]\ndf_meta","metadata":{"execution":{"iopub.status.busy":"2023-01-17T12:09:53.275256Z","iopub.execute_input":"2023-01-17T12:09:53.275591Z","iopub.status.idle":"2023-01-17T12:09:53.536320Z","shell.execute_reply.started":"2023-01-17T12:09:53.275559Z","shell.execute_reply":"2023-01-17T12:09:53.535612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_meta[['HSC cell_type', 'EryP cell_type', 'NeuP cell_type','MasP cell_type', 'MkP cell_type', 'BP cell_type', 'MoP cell_type' ]]\nX ","metadata":{"execution":{"iopub.status.busy":"2023-01-17T12:09:53.537375Z","iopub.execute_input":"2023-01-17T12:09:53.538123Z","iopub.status.idle":"2023-01-17T12:09:53.553236Z","shell.execute_reply.started":"2023-01-17T12:09:53.538095Z","shell.execute_reply":"2023-01-17T12:09:53.552192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom sklearn.linear_model import RidgeCV\nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.model_selection import KFold \nfrom sklearn.metrics import r2_score\nfrom sklearn.metrics import mean_squared_error\n\nfor t in df_rna.columns:\n        for i in range(len(target_names)):\n            if target_names[i] in t: \n                rna_name = t\n                for i in range(len(target_names)):\n                    y = df_y[target_names[i]]\n\n                    model = RidgeCV(alphas=[1e-1, 1, 1e1,1e2,1e3,1e4,1e5]).fit(X, y)\n                    alpha_selected = model.alpha_\n#print('Optimal alpha found by cross-validation: ', alpha_selected )\n\n\n                    model = Ridge(alpha = alpha_selected )\n                    kf = KFold(n_splits=3,  shuffle=True, random_state= 0 )\n                    y_pred = cross_val_predict(model, X, y, cv=kf)\n                    s = r2_score(y,y_pred)\n                    df_scores.loc['r2 Ridge cell types', target_names[i] ]  = s\n                    s = np.corrcoef(y,y_pred)[0,1]\n                    df_scores.loc['Corr Ridge cell types', target_names[i] ]  = s\n                    s = mean_squared_error(y,y_pred, squared=False) # squared=False -> RMSE, not MSE \n                    df_scores.loc['RMSE Ridge cell types', target_names[i] ]  = s\n","metadata":{"execution":{"iopub.status.busy":"2023-01-17T12:09:53.554592Z","iopub.execute_input":"2023-01-17T12:09:53.554993Z","iopub.status.idle":"2023-01-17T12:53:29.868428Z","shell.execute_reply.started":"2023-01-17T12:09:53.554956Z","shell.execute_reply":"2023-01-17T12:53:29.866918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predicting based on - cell type +  own RNA \n\n(own RNA -   RNA corresponding to the protein, e.g. for all proteins we take their RNA )","metadata":{}},{"cell_type":"code","source":"df_tmp = df_meta[['HSC cell_type', 'EryP cell_type', 'NeuP cell_type','MasP cell_type', 'MkP cell_type', 'BP cell_type', 'MoP cell_type' ]]\nX = pd.concat([ df_rna[[rna_name]],  df_tmp ], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2023-01-17T12:53:29.869891Z","iopub.execute_input":"2023-01-17T12:53:29.870218Z","iopub.status.idle":"2023-01-17T12:53:29.895923Z","shell.execute_reply.started":"2023-01-17T12:53:29.870190Z","shell.execute_reply":"2023-01-17T12:53:29.894343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom sklearn.linear_model import RidgeCV\nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.model_selection import KFold \nfrom sklearn.metrics import r2_score\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.metrics import mean_squared_error\n\nfor t in df_rna.columns:\n        for i in range(len(target_names)):\n            if target_names[i] in t: \n                rna_name = t\n                for i in range(len(target_names)):\n                    y = df_y[target_names[i]]\n\n                    model = RidgeCV(alphas=[1e-1, 1, 1e1,1e2,1e3,1e4,1e5]).fit(X, y)\n                    alpha_selected = model.alpha_\n#print('Optimal alpha found by cross-validation: ', alpha_selected )\n\n\n                    model = Ridge(alpha = alpha_selected )\n                    kf = KFold(n_splits=3,  shuffle=True, random_state= 0 )\n                    y_pred = cross_val_predict(model, X, y, cv=kf)\n                    s = r2_score(y,y_pred)\n                    df_scores.loc['r2 Ridge CT + ownRNA',  target_names[i] ]  = s\n                    s = np.corrcoef(y,y_pred)[0,1]\n                    df_scores.loc['Corr Ridge CT + ownRNA',  target_names[i] ]  = s\n                    s = mean_squared_error(y,y_pred, squared=False) # squared=False -> RMSE, not MSE \n                    df_scores.loc['RMSE Ridge CT + ownRNA',  target_names[i] ]  = s\n","metadata":{"execution":{"iopub.status.busy":"2023-01-17T12:53:29.897782Z","iopub.execute_input":"2023-01-17T12:53:29.898420Z","iopub.status.idle":"2023-01-17T13:36:54.183644Z","shell.execute_reply.started":"2023-01-17T12:53:29.898387Z","shell.execute_reply":"2023-01-17T13:36:54.182603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_scores","metadata":{"execution":{"iopub.status.busy":"2023-01-17T13:36:54.184734Z","iopub.execute_input":"2023-01-17T13:36:54.185756Z","iopub.status.idle":"2023-01-17T13:36:54.214421Z","shell.execute_reply.started":"2023-01-17T13:36:54.185723Z","shell.execute_reply":"2023-01-17T13:36:54.212993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_scores.to_csv('Scores_part1'+'.csv')","metadata":{"execution":{"iopub.status.busy":"2023-01-17T13:36:54.215891Z","iopub.execute_input":"2023-01-17T13:36:54.216242Z","iopub.status.idle":"2023-01-17T13:36:54.231128Z","shell.execute_reply.started":"2023-01-17T13:36:54.216208Z","shell.execute_reply":"2023-01-17T13:36:54.230068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t = time.time()-t0start\nprint('%.1f hours  = %.1f minutes = %.1f seconds passed total '%( t/3600,t/60, t ) )","metadata":{"execution":{"iopub.status.busy":"2023-01-17T13:36:54.232554Z","iopub.execute_input":"2023-01-17T13:36:54.233422Z","iopub.status.idle":"2023-01-17T13:36:54.239572Z","shell.execute_reply.started":"2023-01-17T13:36:54.233368Z","shell.execute_reply":"2023-01-17T13:36:54.238297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}