{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ?\n\nSave  RNA seq data to sparse matrix. \n\n240090 cells\n\n21255 genes\n\nOriginal parquet file adata_train.parquet - 1.76Gig\n\n\nSaved sparse matrix - 856Mb (csr format) or  942Mb (coo format)\n\nVersions\n\n    2 csr sparse matrix - more easy to work with, also genes and obs_ids format changed\n    1 coo sparse matrix\n","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport time\nt0start = time.time() \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-28T19:17:49.414268Z","iopub.execute_input":"2023-09-28T19:17:49.414717Z","iopub.status.idle":"2023-09-28T19:17:49.895643Z","shell.execute_reply.started":"2023-09-28T19:17:49.414678Z","shell.execute_reply":"2023-09-28T19:17:49.893755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/open-problems-single-cell-perturbations/adata_train.parquet'\ndf = pd.read_parquet(fn)\nprint(df.shape)\ndf","metadata":{"execution":{"iopub.status.busy":"2023-09-28T19:23:05.462772Z","iopub.execute_input":"2023-09-28T19:23:05.463226Z","iopub.status.idle":"2023-09-28T19:24:18.174658Z","shell.execute_reply.started":"2023-09-28T19:23:05.463184Z","shell.execute_reply":"2023-09-28T19:24:18.173335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ns = set(df['obs_id'])\nlen(s)","metadata":{"execution":{"iopub.status.busy":"2023-09-28T19:31:04.443958Z","iopub.execute_input":"2023-09-28T19:31:04.444385Z","iopub.status.idle":"2023-09-28T19:31:55.067057Z","shell.execute_reply.started":"2023-09-28T19:31:04.444353Z","shell.execute_reply":"2023-09-28T19:31:55.065612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ns = set(df['gene'])\nlen(s)","metadata":{"execution":{"iopub.status.busy":"2023-09-28T19:32:20.216213Z","iopub.execute_input":"2023-09-28T19:32:20.216664Z","iopub.status.idle":"2023-09-28T19:33:31.978702Z","shell.execute_reply.started":"2023-09-28T19:32:20.216631Z","shell.execute_reply":"2023-09-28T19:33:31.977330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nimport numpy as np\nfrom scipy.sparse import coo_matrix\nfrom scipy.sparse import csr_matrix\n\n# Sample data with categorical labels\nrow_labels = df['obs_id'].values\ncol_labels = df['gene'].values# [\"X\", \"Y\", \"Z\", \"Z\"]\nvalues = df['count']\n\n# Create mappings from categorical labels to numerical indices\nrow_label_to_index = {label: index for index, label in enumerate(set(row_labels))}\ncol_label_to_index = {label: index for index, label in enumerate(set(col_labels))}\n\n# Map categorical labels to numerical indices and create triplets\nrow_indices = [row_label_to_index[label] for label in row_labels]\ncol_indices = [col_label_to_index[label] for label in col_labels]\n\n# Create a sparse matrix from the triplets\nsparse_matrix = coo_matrix((values, (row_indices, col_indices)))\n","metadata":{"execution":{"iopub.status.busy":"2023-09-28T19:35:59.855528Z","iopub.execute_input":"2023-09-28T19:35:59.855984Z","iopub.status.idle":"2023-09-28T19:44:36.319132Z","shell.execute_reply.started":"2023-09-28T19:35:59.855951Z","shell.execute_reply":"2023-09-28T19:44:36.318025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nsparse_matrix = sparse_matrix.tocsr()","metadata":{"execution":{"iopub.status.busy":"2023-09-28T20:40:08.413551Z","iopub.execute_input":"2023-09-28T20:40:08.413972Z","iopub.status.idle":"2023-09-28T20:40:44.048838Z","shell.execute_reply.started":"2023-09-28T20:40:08.413939Z","shell.execute_reply":"2023-09-28T20:40:44.047402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sparse_matrix.data.nbytes/1e6# , sparse_matrix.row.nbytes/1e6, sparse_matrix.col.nbytes/1e6","metadata":{"execution":{"iopub.status.busy":"2023-09-28T20:42:37.819287Z","iopub.execute_input":"2023-09-28T20:42:37.819688Z","iopub.status.idle":"2023-09-28T20:42:37.827222Z","shell.execute_reply.started":"2023-09-28T20:42:37.819657Z","shell.execute_reply":"2023-09-28T20:42:37.826077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom scipy.sparse import coo_matrix, save_npz\n\n# Define the filename where you want to save the matrix\nfilename = \"rna_seq_counts_sparse_matrix.npz\"\n\n# Save the sparse matrix to disk in npz format\nsave_npz(filename, sparse_matrix)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-09-28T20:42:43.101571Z","iopub.execute_input":"2023-09-28T20:42:43.101972Z","iopub.status.idle":"2023-09-28T20:46:51.637641Z","shell.execute_reply.started":"2023-09-28T20:42:43.101929Z","shell.execute_reply":"2023-09-28T20:46:51.636299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# row_index_to_label = {index: label  for index, label in enumerate(set(row_labels))}\n# col_index_to_label = {index: label  for index, label in enumerate(set(col_labels))}\n","metadata":{"execution":{"iopub.status.busy":"2023-09-28T20:00:07.956698Z","iopub.execute_input":"2023-09-28T20:00:07.957188Z","iopub.status.idle":"2023-09-28T20:00:46.614699Z","shell.execute_reply.started":"2023-09-28T20:00:07.957148Z","shell.execute_reply":"2023-09-28T20:00:46.613398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nsr = pd.Series(row_label_to_index)\nsr.to_csv('rna_seq_obs_id.csv')\nsr\n","metadata":{"execution":{"iopub.status.busy":"2023-09-28T20:34:54.011324Z","iopub.execute_input":"2023-09-28T20:34:54.012939Z","iopub.status.idle":"2023-09-28T20:34:54.891681Z","shell.execute_reply.started":"2023-09-28T20:34:54.012893Z","shell.execute_reply":"2023-09-28T20:34:54.890403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nsr = pd.Series(col_label_to_index)\nsr.to_csv('rna_seq_gene_symbols.csv')\nsr\n","metadata":{"execution":{"iopub.status.busy":"2023-09-28T20:35:14.384772Z","iopub.execute_input":"2023-09-28T20:35:14.385233Z","iopub.status.idle":"2023-09-28T20:35:14.455658Z","shell.execute_reply.started":"2023-09-28T20:35:14.385195Z","shell.execute_reply":"2023-09-28T20:35:14.454455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir()","metadata":{"execution":{"iopub.status.busy":"2023-09-28T20:06:45.719433Z","iopub.execute_input":"2023-09-28T20:06:45.720321Z","iopub.status.idle":"2023-09-28T20:06:45.728586Z","shell.execute_reply.started":"2023-09-28T20:06:45.720283Z","shell.execute_reply":"2023-09-28T20:06:45.726861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('%.1f seconds passed total '%(time.time()-t0start) )\nprint('%.1f minutes passed total '%( (time.time()-t0start)/60)  )\nprint('%.2f hours passed total '%( (time.time()-t0start)/3600)  )","metadata":{},"execution_count":null,"outputs":[]}]}