{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2022-09-16T05:15:51.109411Z","iopub.status.busy":"2022-09-16T05:15:51.108693Z","iopub.status.idle":"2022-09-16T05:15:51.135322Z","shell.execute_reply":"2022-09-16T05:15:51.134208Z"},"papermill":{"duration":0.040981,"end_time":"2022-09-16T05:15:51.140588","exception":false,"start_time":"2022-09-16T05:15:51.099607","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 导入库\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm \nimport gc\nimport os\nimport sys\nfrom scipy import sparse\nimport torch\nimport torch.nn.functional as F\nfrom torch import Tensor\n!pip install tables","metadata":{"execution":{"iopub.execute_input":"2022-09-16T05:15:51.156695Z","iopub.status.busy":"2022-09-16T05:15:51.155736Z","iopub.status.idle":"2022-09-16T05:16:11.465576Z","shell.execute_reply":"2022-09-16T05:16:11.464056Z"},"papermill":{"duration":20.321781,"end_time":"2022-09-16T05:16:11.469339","exception":false,"start_time":"2022-09-16T05:15:51.147558","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cite 数据准备\ncite_x = pd.read_hdf('../input/open-problems-multimodal/train_cite_inputs.h5').values\nsparse.save_npz('./train_cite_inputs_sparse',sparse.csr_matrix(cite_x))\ndel cite_x\ngc.collect()\ncite_y = pd.read_hdf('../input/open-problems-multimodal/train_cite_targets.h5').values\nsparse.save_npz('./train_cite_targets_sparse',sparse.csr_matrix(cite_y))\ndel cite_y\ngc.collect()\ncite_x_test = pd.read_hdf('../input/open-problems-multimodal/test_cite_inputs.h5').values\nsparse.save_npz('./test_cite_inputs_sparse',sparse.csr_matrix(cite_x_test))\ndel cite_x_test\ngc.collect()","metadata":{"execution":{"iopub.execute_input":"2022-09-16T05:16:11.483906Z","iopub.status.busy":"2022-09-16T05:16:11.482203Z","iopub.status.idle":"2022-09-16T05:25:21.671077Z","shell.execute_reply":"2022-09-16T05:25:21.669671Z"},"papermill":{"duration":550.203272,"end_time":"2022-09-16T05:25:21.678574","exception":false,"start_time":"2022-09-16T05:16:11.475302","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# multi std and mean\nmulti_y = sparse.load_npz('../input/open-problems-msci-multiome-sparse-matrices/train_multi_targets_sparse.npz')\ny_mean = np.array(multi_y.mean(axis=1)).reshape(-1)\nmulti_y_c = multi_y.copy()\nmulti_y_c.data**=2\ny_std = np.sqrt(np.array(multi_y_c.mean(axis=1)).reshape(-1)-y_mean**2)\nnp.save(os.path.join('./','multi_y_mean'), y_mean, allow_pickle=True)\nnp.save(os.path.join('./','multi_y_std'), y_std, allow_pickle=True)\nindex = 456\ncheck_1 = multi_y[index]\ncheck_1_mstd = (check_1.toarray()-y_mean[index])/y_std[index]\nprint(np.mean(check_1_mstd))\nprint(np.std(check_1_mstd))\ndel multi_y,multi_y_c,check_1,check_1_mstd,index,y_std,y_mean\ngc.collect()","metadata":{"execution":{"iopub.execute_input":"2022-09-16T05:25:21.691883Z","iopub.status.busy":"2022-09-16T05:25:21.691463Z","iopub.status.idle":"2022-09-16T05:25:58.701944Z","shell.execute_reply":"2022-09-16T05:25:58.700714Z"},"papermill":{"duration":37.024786,"end_time":"2022-09-16T05:25:58.709580","exception":false,"start_time":"2022-09-16T05:25:21.684794","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submit multi\nmulti_target_col = pd.read_hdf('../input/open-problems-multimodal/train_multi_targets.h5',start=0,stop=1).columns\neval_ids = pd.read_csv('../input/open-problems-multimodal/evaluation_ids.csv', index_col='row_id')\n# Convert the string columns to more efficient categorical types\neval_ids.cell_id = eval_ids.cell_id.astype(pd.CategoricalDtype())\neval_ids.gene_id = eval_ids.gene_id.astype(pd.CategoricalDtype())\ncell_id_set = set(eval_ids.cell_id)\neval_ids_col = eval_ids.groupby('cell_id')['gene_id'].apply(lambda x:','.join(x))\neval_ids_index = eval_ids.cell_id\ndel eval_ids\ngc.collect()","metadata":{"execution":{"iopub.execute_input":"2022-09-16T05:25:58.722695Z","iopub.status.busy":"2022-09-16T05:25:58.721371Z","iopub.status.idle":"2022-09-16T05:28:55.372249Z","shell.execute_reply":"2022-09-16T05:28:55.370742Z"},"papermill":{"duration":176.671094,"end_time":"2022-09-16T05:28:55.385850","exception":false,"start_time":"2022-09-16T05:25:58.714756","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ret_index(x):\n    col_lst = x.split(',')\n    return np.where(multi_target_col.isin(col_lst))[0]\nstart = 0\nchunksize = 1000\ntotal_rows = 0\nrow_lst = []\nrow_name_lst = []\ncol_lst = []\nwhile True:\n    print(start)\n    multi_test_x = None # Free the memory if necessary\n    gc.collect()\n    multi_test_x = pd.read_hdf('../input/open-problems-multimodal/test_multi_inputs.h5', start=start, stop=start+chunksize)\n    rows_read = len(multi_test_x)\n    row_mask = multi_test_x.index.isin(cell_id_set)\n    row_lst+=row_mask.tolist()\n    multi_test_x_index = multi_test_x.loc[row_mask].index\n    row_name_lst+=multi_test_x_index.tolist()\n    if len(multi_test_x_index) > 0:\n        col_lst.append(np.stack(eval_ids_col[multi_test_x_index].apply(lambda x:ret_index(x)).values))\n    if rows_read < chunksize: \n        break\n    start += chunksize\nrow_lst = np.array(row_lst)\nrow_name_lst = np.array(row_name_lst)\ncol_lst = np.concatenate(col_lst)\ncol_name_lst = np.array(multi_target_col)\nassert col_lst.shape[0] == np.sum(row_lst)\nassert col_lst.shape[0] == len(row_name_lst)\nnp.save(os.path.join('./','multi_test_row'), row_lst, allow_pickle=True)\nnp.save(os.path.join('./','multi_test_row_name'), row_name_lst, allow_pickle=True)\nnp.save(os.path.join('./','multi_test_col'), col_lst, allow_pickle=True)\nnp.save(os.path.join('./','multi_test_col_name'), col_name_lst, allow_pickle=True)","metadata":{"execution":{"iopub.execute_input":"2022-09-16T05:28:55.400615Z","iopub.status.busy":"2022-09-16T05:28:55.399949Z","iopub.status.idle":"2022-09-16T05:34:15.562171Z","shell.execute_reply":"2022-09-16T05:34:15.560094Z"},"papermill":{"duration":320.174944,"end_time":"2022-09-16T05:34:15.566400","exception":false,"start_time":"2022-09-16T05:28:55.391456","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.008725,"end_time":"2022-09-16T05:34:15.586203","exception":false,"start_time":"2022-09-16T05:34:15.577478","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]}]}