{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-31T13:19:58.638712Z","iopub.execute_input":"2022-12-31T13:19:58.639059Z","iopub.status.idle":"2022-12-31T13:19:58.715961Z","shell.execute_reply.started":"2022-12-31T13:19:58.639030Z","shell.execute_reply":"2022-12-31T13:19:58.714762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_y = pd.read_hdf('/kaggle/input/open-problems-multimodal/train_cite_targets.h5')\ntarget_names = df_y.columns\n","metadata":{"execution":{"iopub.status.busy":"2022-12-31T13:19:59.368210Z","iopub.execute_input":"2022-12-31T13:19:59.369614Z","iopub.status.idle":"2022-12-31T13:20:00.055147Z","shell.execute_reply.started":"2022-12-31T13:19:59.369558Z","shell.execute_reply":"2022-12-31T13:20:00.054469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_scores = pd.DataFrame()\ndf_pca = pd.read_csv('/kaggle/input/feature-shop-for-multimodal-singlecell-competition/citeseq_train_and_test_PCA200.csv', index_col = 0 )\ndf_rna = pd.read_hdf('/kaggle/input/open-problems-multimodal/train_cite_inputs.h5')\n \n","metadata":{"execution":{"iopub.status.busy":"2022-12-31T13:20:00.699037Z","iopub.execute_input":"2022-12-31T13:20:00.699660Z","iopub.status.idle":"2022-12-31T13:20:46.993006Z","shell.execute_reply.started":"2022-12-31T13:20:00.699628Z","shell.execute_reply":"2022-12-31T13:20:46.991872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom sklearn.linear_model import RidgeCV\nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.model_selection import KFold \nfrom sklearn.metrics import r2_score\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.metrics import mean_squared_error\n\nfor t in df_rna.columns:\n        for i in range(len(target_names)):\n            if target_names[i] in t: \n                rna_name = t\n                for i in range(len(target_names)):\n                    y = df_y[target_names[i]]\n                    X = df_pca.iloc[:len(y), :100]\n                    model = RidgeCV(alphas=[1e-1, 1, 1e1,1e2,1e3,1e4,1e5]).fit(X, y)\n                    alpha_selected = model.alpha_\n\n\n                    model = Ridge(alpha = alpha_selected )\n                    kf = KFold(n_splits=3,  shuffle=True, random_state= 0 )\n                    y_pred = cross_val_predict(model, X, y, cv=kf)\n                    s = r2_score(y,y_pred)\n                    df_scores.loc['r2 Ridge pca100', target_names[i] ]  = s\n                    s = np.corrcoef(y,y_pred)[0,1]\n                    df_scores.loc['Corr Ridge pca100', target_names[i] ]  = s\n                    s = mean_squared_error(y,y_pred, squared=False) # squared=False -> RMSE, not MSE \n                    df_scores.loc['RMSE Ridge pca100', target_names[i] ]  = s\n","metadata":{"execution":{"iopub.status.busy":"2022-12-31T13:20:46.994670Z","iopub.execute_input":"2022-12-31T13:20:46.994967Z","iopub.status.idle":"2022-12-31T18:52:19.652072Z","shell.execute_reply.started":"2022-12-31T13:20:46.994940Z","shell.execute_reply":"2022-12-31T18:52:19.649484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_scores","metadata":{"execution":{"iopub.status.busy":"2022-12-31T20:45:09.965077Z","iopub.execute_input":"2022-12-31T20:45:09.970101Z","iopub.status.idle":"2022-12-31T20:45:10.074384Z","shell.execute_reply.started":"2022-12-31T20:45:09.969998Z","shell.execute_reply":"2022-12-31T20:45:10.072853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_scores.to_csv('Scores_part2'+'.csv')","metadata":{"execution":{"iopub.status.busy":"2022-12-31T20:45:57.677169Z","iopub.execute_input":"2022-12-31T20:45:57.677707Z","iopub.status.idle":"2022-12-31T20:45:57.706282Z","shell.execute_reply.started":"2022-12-31T20:45:57.677669Z","shell.execute_reply":"2022-12-31T20:45:57.705054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}