{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport pickle as pkl\nfrom collections import defaultdict\nimport matplotlib.pyplot as plt\nimport catboost as cb\nimport lightgbm as lgb\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-08T22:15:43.411690Z","iopub.execute_input":"2023-02-08T22:15:43.412139Z","iopub.status.idle":"2023-02-08T22:15:43.447936Z","shell.execute_reply.started":"2023-02-08T22:15:43.412101Z","shell.execute_reply":"2023-02-08T22:15:43.446948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rna = pd.read_hdf(\"/kaggle/input/open-problems-multimodal/train_cite_inputs.h5\")\nrename=pd.read_csv('/kaggle/input/research-project-01-around-multimodal-singlecell/CD_info.csv')\nrna.columns=rna.columns.str.split('_').str[0]\nrna=rna.T\nrename_copy=rename.rename(columns={'ENSG ID':'gene_id'})\nrename_copy=rename_copy.merge(rna, on='gene_id', how='inner')\nrename_copy=rename_copy.drop(['Unnamed: 0','cd_names', 'names'], axis=1)\ntarget=rename_copy.T\ntarget.columns = target.iloc[0]\ntarget=target[1:]\nrna=rna.T\ntarget=target.T.drop_duplicates().T","metadata":{"execution":{"iopub.status.busy":"2023-02-08T22:15:49.504692Z","iopub.execute_input":"2023-02-08T22:15:49.505260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\nfrom tqdm import tqdm\nN = 100\nbest_features = defaultdict(list)\nall_best_features_dict = {}\n\nfor i in target.columns[30:40]:\n    model_path = f\"/kaggle/input/cd-rna-based-only-rna-30-40/models/{i}/model.cbm\"\n    if 'model_path' is locals():\n        model = cb.CatBoostRegressor()\n        model.load_model(model_path)\n\n        feature_importance_all = model.get_feature_importance()\n        max_indexes = np.flip(np.argsort(feature_importance_all))\n\n        feature_importance = feature_importance_all[max_indexes]\n        columns = target.columns.tolist()\n        feature_names = rna.columns[max_indexes].tolist()\n","metadata":{"execution":{"iopub.status.busy":"2023-02-08T22:02:30.170367Z","iopub.execute_input":"2023-02-08T22:02:30.170947Z","iopub.status.idle":"2023-02-08T22:02:30.179936Z","shell.execute_reply.started":"2023-02-08T22:02:30.170902Z","shell.execute_reply":"2023-02-08T22:02:30.179094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def correlation_score(y_true, y_pred):\n    \"\"\"Scores the predictions according to the competition rules. \n    \n    It is assumed that the predictions are not constant.\n    \n    Returns the average of each sample's Pearson correlation coefficient\"\"\"\n    \n    y2 = y_pred.copy()\n    y2 -= y2.mean(axis=0);    y2 /= y2.std(axis=0) \n    y1 = y_true.copy(); \n    y1 -= y1.mean(axis=0);    y1 /= y1.std(axis=0) \n        \n    c = (y1*y2).mean().mean()# Correlation for rescaled matrices is just matrix product and average \n        \n    return c","metadata":{"execution":{"iopub.status.busy":"2023-02-08T22:03:14.705660Z","iopub.execute_input":"2023-02-08T22:03:14.706161Z","iopub.status.idle":"2023-02-08T22:03:14.715180Z","shell.execute_reply.started":"2023-02-08T22:03:14.706121Z","shell.execute_reply":"2023-02-08T22:03:14.713688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nN = 22048\nbest_features = defaultdict(list)\n#all_best_features_dict = {}\nall_best_features_dict=pd.DataFrame()\n\nfor i in target.columns[30:40]:\n    model_path = f\"/kaggle/input/cd-rna-based-only-rna-30-40/models/{i}/model.cbm\"\n\n    model = cb.CatBoostRegressor()\n    model.load_model(model_path)\n\n    feature_importance_all = model.get_feature_importance()\n    max_indexes = np.flip(np.argsort(feature_importance_all))\n\n    feature_importance = feature_importance_all[max_indexes]\n    columns = target.columns.tolist()\n    feature_names = rna.columns[max_indexes].tolist()\n    for f_num in range(N):\n        best_features[f\"f_{f_num}\"].append(feature_names[f_num])\n        best_features[f\"f_{f_num}_score\"].append(feature_importance[f_num])        \n\n    y_true = target[i].values\n    y_pred = model.predict(\n        rna.loc[:,~rna.columns.isin([i])]\n    )\n\n    corr_score = correlation_score(y_true, y_pred)            \n    all_best_features_dict[f\"{i}_fi\"]=feature_names\n    all_best_features_dict[f\"{i}_score\"]=corr_score\n","metadata":{"execution":{"iopub.status.busy":"2023-02-08T22:09:32.379029Z","iopub.execute_input":"2023-02-08T22:09:32.379491Z","iopub.status.idle":"2023-02-08T22:09:32.435474Z","shell.execute_reply.started":"2023-02-08T22:09:32.379453Z","shell.execute_reply":"2023-02-08T22:09:32.433455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(rna.columns)","metadata":{"execution":{"iopub.status.busy":"2023-02-08T22:04:10.283821Z","iopub.execute_input":"2023-02-08T22:04:10.284272Z","iopub.status.idle":"2023-02-08T22:04:10.293177Z","shell.execute_reply.started":"2023-02-08T22:04:10.284239Z","shell.execute_reply":"2023-02-08T22:04:10.291462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_indexes = np.flip(np.argsort(feature_importance_all))#[:22050]\nmax_indexes","metadata":{"execution":{"iopub.status.busy":"2023-02-08T22:04:53.552357Z","iopub.execute_input":"2023-02-08T22:04:53.553421Z","iopub.status.idle":"2023-02-08T22:04:53.564431Z","shell.execute_reply.started":"2023-02-08T22:04:53.553371Z","shell.execute_reply":"2023-02-08T22:04:53.562959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_best_features_dict","metadata":{"execution":{"iopub.status.busy":"2023-02-08T22:04:58.786689Z","iopub.execute_input":"2023-02-08T22:04:58.787125Z","iopub.status.idle":"2023-02-08T22:04:58.797256Z","shell.execute_reply.started":"2023-02-08T22:04:58.787088Z","shell.execute_reply":"2023-02-08T22:04:58.795908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_best_features_dict.to_csv('all_best_features_dict_30-40.csv')","metadata":{"execution":{"iopub.status.busy":"2023-02-08T22:00:33.509126Z","iopub.status.idle":"2023-02-08T22:00:33.509542Z","shell.execute_reply.started":"2023-02-08T22:00:33.509328Z","shell.execute_reply":"2023-02-08T22:00:33.509347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}