{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#### The following code illustrates how I have used sparse datasets to work with the MSCI data.  \n#### The Dataset class captures all the information I typically have needed.","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport scipy.sparse as sps\nfrom tqdm import tqdm as tqdm\n\nimport os\nimport gc\nfrom collections import defaultdict \n\nimport pickle as pkl\n\n# to enable reading .h5 in pandas\nif not os.path.exists('/opt/conda/lib/python3.7/site-packages/tables'):\n    !pip install --quiet tables","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","tags":[],"execution":{"iopub.status.busy":"2022-09-23T21:27:01.830995Z","iopub.execute_input":"2022-09-23T21:27:01.831407Z","iopub.status.idle":"2022-09-23T21:27:01.870024Z","shell.execute_reply.started":"2022-09-23T21:27:01.831374Z","shell.execute_reply":"2022-09-23T21:27:01.868671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define dataset locations\n\nDATA_DIR = \"/kaggle/input/open-problems-multimodal/\"\n\nSUBMISSON = os.path.join(DATA_DIR,\"sample_submission.csv\")\n\nEVALUATION_IDS = os.path.join(DATA_DIR,\"evaluation_ids.csv\")\nFP_CELL_METADATA = os.path.join(DATA_DIR,\"metadata.csv\")\n\nFP_CITE_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_cite_inputs.h5\")\nFP_CITE_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_cite_targets.h5\")\nFP_CITE_TEST_INPUTS = os.path.join(DATA_DIR,\"test_cite_inputs.h5\")\n\nFP_MULTIOME_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_multi_inputs.h5\")\nFP_MULTIOME_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_multi_targets.h5\")\nFP_MULTIOME_TEST_INPUTS = os.path.join(DATA_DIR,\"test_multi_inputs.h5\")\n\nFP_SUBMISSION = os.path.join(DATA_DIR,\"sample_submission.csv\")\nFP_EVALUATION_IDS = os.path.join(DATA_DIR,\"evaluation_ids.csv\")\n","metadata":{"tags":[],"execution":{"iopub.status.busy":"2022-09-23T21:27:05.790758Z","iopub.execute_input":"2022-09-23T21:27:05.791199Z","iopub.status.idle":"2022-09-23T21:27:05.799555Z","shell.execute_reply.started":"2022-09-23T21:27:05.791162Z","shell.execute_reply":"2022-09-23T21:27:05.798653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class dataset():\n    \"\"\"\n    Class used to contain various details of an .h5 formatted file\n    with data stored as a sparse matrix.  Other aspects (column names, row names),\n    are stored as properties of the class.\n    \"\"\"\n    \n    def __init__(self):\n        self.data = None\n        self.column_names = None\n        self.row_names = None\n        \n        self.low_var_column_indices = None\n        self.low_var_column_names = None\n        \n    def __del__(self):\n        del self.data\n        \n    def read_h5_as_sparse(self, filepath, chunksize = 20000, maxrows = None,\n                          min_unique_rate = 0, drop_column_names = None):\n\n        \"\"\"\n        Read an .h5 data file and return a sparse matrix\n\n        Parameters:\n\n            filepath - path to the target file, expected to be .h5 format\n            chunksize - how many rows to read per step\n            maxrows - maximum number of rows to return - always returns the first rows in the file\n            min_unique_rate - minimum for constancy of a column [0,1]; examples:\n                if min_unique_rate = 0, columns with constant content will be identified\n                if min_unique_rate = 0.25, columns with less than 25% unique values will be identified\n            drop_column_names - names of columns to drop when the file is read\n\n        Return:\n            sparse dataset as SciPy sparse.csr_matrix\n        \"\"\"\n\n        # general parameters for loading the file\n        start = 0\n\n        if maxrows is not None:\n            stop = min(chunksize, maxrows)\n        else:\n            stop = chunksize\n\n        # storage for intermediate sparse chunks\n        blocks = []\n\n        # read in file as blocks and convert blocks to sparse matrices\n        print('File: ', filepath)\n        while True:\n            print('Reading from ', start, ' to ', stop)\n            df = pd.read_hdf(filepath, start = start, stop = stop)\n            if drop_column_names is not None:\n                df.drop(labels = drop_column_names, axis = 1, inplace = True)\n\n            # keep track of low variance columns\n            if len(blocks) == 0:\n                # how many unique values (or less) constitue low variance\n                self.column_names = df.columns\n                low_var_count = max(1, int(df.shape[1] * min_unique_rate))\n                self.low_var_column_names = set(self.column_names[df.nunique(axis = 0) <= low_var_count])\n                self.row_names = df.index.to_list()\n            else:\n                self.low_var_column_names = self.low_var_column_names.intersection(\n                                                set(self.column_names[df.nunique(axis = 0) <= low_var_count]))\n                self.row_names.extend(df.index.to_list())\n\n            # read in the next block\n            blk = sps.csr_matrix(df, dtype = np.single)\n            blocks.append(blk)\n\n            # if we've found the end of the file, stop\n            if df.shape[0] < chunksize:\n                break\n\n            if maxrows is not None and maxrows <= chunksize * len(blocks):\n                break\n\n            # make adjustments for reading the next block\n            start = stop\n            if maxrows is not None:\n                stop = min(stop + chunksize, maxrows)\n            else:\n                stop = stop + chunksize\n\n        # combine blocks into one sparse matrix \n        self.data = sps.vstack(blocks, format = 'csr')\n\n        # find the column indices corresponding to the column names\n        self.low_var_column_indices = [self.column_names.tolist().index(col_name) for col_name in self.low_var_column_names]\n\n        # delete blocks to free memory\n        while len(blocks) > 0:\n            blk = blocks.pop()\n            del blk\n\n        _ = gc.collect()","metadata":{"tags":[],"execution":{"iopub.status.busy":"2022-09-23T21:27:08.548529Z","iopub.execute_input":"2022-09-23T21:27:08.548977Z","iopub.status.idle":"2022-09-23T21:27:08.57221Z","shell.execute_reply.started":"2022-09-23T21:27:08.548941Z","shell.execute_reply":"2022-09-23T21:27:08.570105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### If sparse datasets have been saved","metadata":{}},{"cell_type":"code","source":"f_in = open('multiome_train_sparse.pkl', 'rb')\nmulti_X_train = pkl.load(f_in)\nf_in.close()","metadata":{"execution":{"iopub.execute_input":"2022-09-22T23:02:26.673841Z","iopub.status.busy":"2022-09-22T23:02:26.673406Z","iopub.status.idle":"2022-09-22T23:03:01.9611Z","shell.execute_reply":"2022-09-22T23:03:01.960482Z","shell.execute_reply.started":"2022-09-22T23:02:26.673816Z"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### If low variance lists have been saved...","metadata":{}},{"cell_type":"code","source":"# if low variance lists have been saved...\nf_in = open('/kaggle/working/Multi_GT25perc_columns.pkl', 'rb')\nitems = pkl.load(f_in)\nf_in.close()\n\nlow_var_col_names = items[0]\nlow_var_col_indices = items[1]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# reload training data, dropping low variance columns\nmulti_X_train = dataset()\nmulti_X_train.read_h5_as_sparse(FP_MULTI_TRAIN_INPUTS, \n                                drop_column_names = low_var_col_names)\n\n# get metadata\nmeta = pd.read_csv(FP_CELL_METADATA)\n\n# trim down to only cell ids in training set\nmulti_meta = meta[meta.technology == 'multiome']\nmulti_train_meta = pd.DataFrame({'id': multi_X_train.row_names})\nmulti_train_meta = multi_train_meta.merge(meta, how = 'left', left_on = 'id', right_on = 'cell_id', )","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Look for low variance columns","metadata":{}},{"cell_type":"code","source":"# find low variance columns in training data\nmulti_X_train = dataset()\nmulti_X_train.read_h5_as_sparse(FP_MULTIOME_TRAIN_INPUTS, min_unique_rate = 0.25,\n                                chunksize = 20000)\nmulti_X_train_low_var_col_names = multi_X_train.low_var_column_names\nmulti_X_train_low_var_col_indices = multi_X_train.low_var_column_indices\n\nf_out = open('multiome_train_sparse.pkl', 'wb')\npkl.dump(multi_X_train, f_out)\nf_out.close()\n\ndel multi_X_train\n_ = gc.collect()\n\n# find low variance columns in test data\nmulti_X_test = dataset()\nmulti_X_test.read_h5_as_sparse(FP_MULTIOME_TEST_INPUTS, min_unique_rate = 0.25,\n                               chunksize = 20000)\nmulti_X_test_low_var_col_names = multi_X_test.low_var_column_names\nmulti_X_test_low_var_col_indices = multi_X_test.low_var_column_indices\n\nf_out = open('multiome_test_sparse.pkl', 'wb')\npkl.dump(multi_X_test, f_out)\nf_out.close()\n\ndel multi_X_test\n_ = gc.collect()","metadata":{"execution":{"iopub.execute_input":"2022-09-22T03:44:57.92791Z","iopub.status.busy":"2022-09-22T03:44:57.927535Z","iopub.status.idle":"2022-09-22T03:44:58.014368Z","shell.execute_reply":"2022-09-22T03:44:58.013703Z","shell.execute_reply.started":"2022-09-22T03:44:57.927886Z"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# find combined low variance columns\nlow_var_col_names = set(multi_X_train_low_var_col_names).union(set(multi_X_test_low_var_col_names))\nlow_var_col_indices = set(multi_X_train_low_var_col_indices).union(set(multi_X_test_low_var_col_indices))\n\nprint('Number of low variance columns to drop: ', len(low_var_col_names))\n\n# save names of low variance columns\nf_out = open('/kaggle/working/Multi_GT25perc_columns.pkl', 'wb')\npkl.dump([low_var_col_names, low_var_col_indices], f_out)\nf_out.close()","metadata":{"execution":{"iopub.execute_input":"2022-09-22T03:45:07.707909Z","iopub.status.busy":"2022-09-22T03:45:07.707536Z","iopub.status.idle":"2022-09-22T03:45:07.946856Z","shell.execute_reply":"2022-09-22T03:45:07.946286Z","shell.execute_reply.started":"2022-09-22T03:45:07.707886Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### EDA","metadata":{"execution":{"iopub.execute_input":"2022-09-22T23:00:00.876076Z","iopub.status.busy":"2022-09-22T23:00:00.875701Z","iopub.status.idle":"2022-09-22T23:00:00.878894Z","shell.execute_reply":"2022-09-22T23:00:00.878363Z","shell.execute_reply.started":"2022-09-22T23:00:00.876052Z"},"tags":[]}},{"cell_type":"code","source":"# load the training and test sparse matices\nf_in = open('../input/multiome-sparse-sets/multiome_train_sparse.pkl', 'rb')\nmulti_X_train = pkl.load(f_in)\nf_in.close()\n\nf_in = open('../input/multiome-sparse-sets/multiome_test_sparse.pkl', 'rb')\nmulti_X_test = pkl.load(f_in)\nf_in.close()\n\nmeta = pd.read_csv(FP_CELL_METADATA)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eda_train_df = pd.DataFrame({'cell_id': multi_X_train.row_names})\neda_train_df = eda_train_df.merge(meta, on = 'cell_id', how = 'left')","metadata":{"execution":{"iopub.execute_input":"2022-09-22T23:12:26.836757Z","iopub.status.busy":"2022-09-22T23:12:26.836358Z","iopub.status.idle":"2022-09-22T23:12:26.948435Z","shell.execute_reply":"2022-09-22T23:12:26.947842Z","shell.execute_reply.started":"2022-09-22T23:12:26.836713Z"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(eda_train_df['cell_type'].value_counts())\nprint(eda_train_df['day'].value_counts())","metadata":{"execution":{"iopub.execute_input":"2022-09-22T23:53:27.471296Z","iopub.status.busy":"2022-09-22T23:53:27.470905Z","iopub.status.idle":"2022-09-22T23:53:27.482058Z","shell.execute_reply":"2022-09-22T23:53:27.481446Z","shell.execute_reply.started":"2022-09-22T23:53:27.471272Z"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eda_test_df = pd.DataFrame({'cell_id': multi_X_test.row_names})\neda_test_df = eda_test_df.merge(meta, on = 'cell_id', how = 'left')\nprint(eda_test_df['cell_type'].value_counts())\nprint(eda_test_df['day'].value_counts())","metadata":{"execution":{"iopub.execute_input":"2022-09-22T23:53:50.242752Z","iopub.status.busy":"2022-09-22T23:53:50.242378Z","iopub.status.idle":"2022-09-22T23:53:50.352334Z","shell.execute_reply":"2022-09-22T23:53:50.351747Z","shell.execute_reply.started":"2022-09-22T23:53:50.242728Z"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# reduce dimensions with truncSVD\nfrom sklearn.decomposition import TruncatedSVD\n\nsvd = TruncatedSVD(n_components = 512, n_iter = 7, random_state = 42)\nsvd.fit(multi_X_train.data)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f_out = open('multi_svd_512.pkl', 'wb')\npkl.dump(svd, f_out)\nf_out.close()","metadata":{},"execution_count":null,"outputs":[]}]}