{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\n\n# Main results will be stored here: \ndf_scores = pd.DataFrame()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-04T09:44:17.704951Z","iopub.execute_input":"2023-01-04T09:44:17.705352Z","iopub.status.idle":"2023-01-04T09:44:17.711123Z","shell.execute_reply.started":"2023-01-04T09:44:17.705322Z","shell.execute_reply":"2023-01-04T09:44:17.709767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.preprocessing import OneHotEncoder\n\nimport time\nt0start = time.time()\n\nimport os","metadata":{"execution":{"iopub.status.busy":"2023-01-04T09:44:17.717434Z","iopub.execute_input":"2023-01-04T09:44:17.717851Z","iopub.status.idle":"2023-01-04T09:44:17.726083Z","shell.execute_reply.started":"2023-01-04T09:44:17.717795Z","shell.execute_reply":"2023-01-04T09:44:17.724820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load data for target","metadata":{}},{"cell_type":"code","source":"%%time\ndf_y = pd.read_hdf('/kaggle/input/open-problems-multimodal/train_cite_targets.h5')\ndf_y","metadata":{"execution":{"iopub.status.busy":"2023-01-04T09:44:17.728417Z","iopub.execute_input":"2023-01-04T09:44:17.728801Z","iopub.status.idle":"2023-01-04T09:44:18.578874Z","shell.execute_reply.started":"2023-01-04T09:44:17.728770Z","shell.execute_reply":"2023-01-04T09:44:18.577711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Key Params","metadata":{}},{"cell_type":"code","source":"target_names = df_y.columns","metadata":{"execution":{"iopub.status.busy":"2023-01-04T09:44:18.580404Z","iopub.execute_input":"2023-01-04T09:44:18.581540Z","iopub.status.idle":"2023-01-04T09:44:18.587304Z","shell.execute_reply.started":"2023-01-04T09:44:18.581494Z","shell.execute_reply":"2023-01-04T09:44:18.586098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(target_names)","metadata":{"execution":{"iopub.status.busy":"2023-01-04T09:44:18.588718Z","iopub.execute_input":"2023-01-04T09:44:18.589125Z","iopub.status.idle":"2023-01-04T09:44:18.599974Z","shell.execute_reply.started":"2023-01-04T09:44:18.589093Z","shell.execute_reply":"2023-01-04T09:44:18.598703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(target_names)):\n    target_name = target_names[i]\n\n    y = df_y[target_name]\n    df_scores.loc['Mean', target_name ]  = np.mean(y)\n    df_scores.loc['Std', target_name ]  = np.std(y)","metadata":{"execution":{"iopub.status.busy":"2023-01-04T09:44:18.605152Z","iopub.execute_input":"2023-01-04T09:44:18.605860Z","iopub.status.idle":"2023-01-04T09:44:18.909399Z","shell.execute_reply.started":"2023-01-04T09:44:18.605817Z","shell.execute_reply":"2023-01-04T09:44:18.908582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_scores\n","metadata":{"execution":{"iopub.status.busy":"2023-01-04T09:44:18.910606Z","iopub.execute_input":"2023-01-04T09:44:18.911108Z","iopub.status.idle":"2023-01-04T09:44:18.936409Z","shell.execute_reply.started":"2023-01-04T09:44:18.911076Z","shell.execute_reply":"2023-01-04T09:44:18.935259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  Predicting based on -  own RNA \n\ne.g. for all proteins we take their same predictor ","metadata":{}},{"cell_type":"code","source":"%%time\ndf_rna = pd.read_hdf('/kaggle/input/open-problems-multimodal/train_cite_inputs.h5')\ndisplay(df_rna) \n\nfor t in df_rna.columns:\n    for i in range(len(target_names)):\n        if target_names[i] in t: \n            rna_name = t","metadata":{"execution":{"iopub.status.busy":"2023-01-04T09:44:18.937728Z","iopub.execute_input":"2023-01-04T09:44:18.938090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for t in df_rna.columns:\n    for i in range(len(target_names)):\n        if target_names[i] in t: \n            rna_name = t\n            for i in range(len(target_names)):\n                target_name = target_names[i]\n                X = df_rna[[rna_name]]\n                y = df_y[target_name]\n                s = np.corrcoef(y,X.iloc[:,0])[0,1]\n                df_scores.loc['Corr ownRNA', target_name ]  = s","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_scores","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom sklearn.linear_model import RidgeCV\nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.model_selection import KFold \nfrom sklearn.metrics import r2_score\nfrom sklearn.metrics import mean_squared_error\n\n\nfor t in df_rna.columns:\n    for i in range(len(target_names)):\n        if target_names[i] in t: \n            rna_name = t\n            for i in range(len(target_names)):\n                X = df_rna[[rna_name]]\n                y = df_y[target_names[i]]\n\n                model = RidgeCV(alphas=[1e-1, 1, 1e1,1e2,1e3,1e4,1e5]).fit(X, y)\n                alpha_selected = model.alpha_\n#print('Optimal alpha found by cross-validation: ', alpha_selected )\n\n\n                model = Ridge(alpha = alpha_selected )\n                kf = KFold(n_splits=3,  shuffle=True, random_state= 0 )\n                y_pred = cross_val_predict(model, X, y, cv=kf)\n                s = r2_score(y,y_pred)\n                df_scores.loc['r2 Ridge ownRNA', target_names[i] ]  = s\n                s = np.corrcoef(y,y_pred)[0,1]\n                df_scores.loc['Corr Ridge ownRNA', target_names[i] ]  = s\n                s = mean_squared_error(y,y_pred, squared=False) # squared=False -> RMSE, not MSE \n                df_scores.loc['RMSE Ridge ownRNA', target_names[i] ]  = s\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_scores","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  Predicting based on - cell type information ","metadata":{}},{"cell_type":"code","source":"%%time\ndf_meta = pd.read_csv('/kaggle/input/feature-shop-for-multimodal-singlecell-competition/_citeseq_meta_all_text_also.csv', index_col = 0 )\ndf_meta = df_meta[df_meta['Train0OrTest1']==0]\ndf_meta","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_meta[['HSC cell_type', 'EryP cell_type', 'NeuP cell_type','MasP cell_type', 'MkP cell_type', 'BP cell_type', 'MoP cell_type' ]]\nX ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom sklearn.linear_model import RidgeCV\nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.model_selection import KFold \nfrom sklearn.metrics import r2_score\nfrom sklearn.metrics import mean_squared_error\n\nfor t in df_rna.columns:\n        for i in range(len(target_names)):\n            if target_names[i] in t: \n                rna_name = t\n                for i in range(len(target_names)):\n                    y = df_y[target_names[i]]\n\n                    model = RidgeCV(alphas=[1e-1, 1, 1e1,1e2,1e3,1e4,1e5]).fit(X, y)\n                    alpha_selected = model.alpha_\n#print('Optimal alpha found by cross-validation: ', alpha_selected )\n\n\n                    model = Ridge(alpha = alpha_selected )\n                    kf = KFold(n_splits=3,  shuffle=True, random_state= 0 )\n                    y_pred = cross_val_predict(model, X, y, cv=kf)\n                    s = r2_score(y,y_pred)\n                    df_scores.loc['r2 Ridge cell types', target_names[i] ]  = s\n                    s = np.corrcoef(y,y_pred)[0,1]\n                    df_scores.loc['Corr Ridge cell types', target_names[i] ]  = s\n                    s = mean_squared_error(y,y_pred, squared=False) # squared=False -> RMSE, not MSE \n                    df_scores.loc['RMSE Ridge cell types', target_names[i] ]  = s\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predicting based on - cell type +  own RNA \n\n(own RNA -   RNA corresponding to the protein, e.g. for all proteins we take their RNA )","metadata":{}},{"cell_type":"code","source":"df_tmp = df_meta[['HSC cell_type', 'EryP cell_type', 'NeuP cell_type','MasP cell_type', 'MkP cell_type', 'BP cell_type', 'MoP cell_type' ]]\nX = pd.concat([ df_rna[[rna_name]],  df_tmp ], axis = 1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom sklearn.linear_model import RidgeCV\nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.model_selection import KFold \nfrom sklearn.metrics import r2_score\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.metrics import mean_squared_error\n\nfor t in df_rna.columns:\n        for i in range(len(target_names)):\n            if target_names[i] in t: \n                rna_name = t\n                for i in range(len(target_names)):\n                    y = df_y[target_names[i]]\n\n                    model = RidgeCV(alphas=[1e-1, 1, 1e1,1e2,1e3,1e4,1e5]).fit(X, y)\n                    alpha_selected = model.alpha_\n#print('Optimal alpha found by cross-validation: ', alpha_selected )\n\n\n                    model = Ridge(alpha = alpha_selected )\n                    kf = KFold(n_splits=3,  shuffle=True, random_state= 0 )\n                    y_pred = cross_val_predict(model, X, y, cv=kf)\n                    s = r2_score(y,y_pred)\n                    df_scores.loc['r2 Ridge CT + ownRNA',  target_names[i] ]  = s\n                    s = np.corrcoef(y,y_pred)[0,1]\n                    df_scores.loc['Corr Ridge CT + ownRNA',  target_names[i] ]  = s\n                    s = mean_squared_error(y,y_pred, squared=False) # squared=False -> RMSE, not MSE \n                    df_scores.loc['RMSE Ridge CT + ownRNA',  target_names[i] ]  = s\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_scores","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_scores.to_csv('Scores_part1'+'.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t = time.time()-t0start\nprint('%.1f hours  = %.1f minutes = %.1f seconds passed total '%( t/3600,t/60, t ) )","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}