{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"The size of the dataset was one issue I ran into while analyzing the competition data.\nDue to a memory shortage, it was difficult to compute several of the dataset's important properties. This notebook's objective is to compute a few features chunkwise and store those results as its output. These properties can then be included into models or used for EDA.\n\nIf you feel like important features are missing, please let me know. I will add them in future versions.","metadata":{}},{"cell_type":"code","source":"!pip install --quiet hdf5plugin","metadata":{"execution":{"iopub.status.busy":"2022-09-06T10:20:54.045709Z","iopub.execute_input":"2022-09-06T10:20:54.046802Z","iopub.status.idle":"2022-09-06T10:21:08.010986Z","shell.execute_reply.started":"2022-09-06T10:20:54.046689Z","shell.execute_reply":"2022-09-06T10:21:08.009750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport torch\nimport pandas as pd\nfrom tqdm.auto import tqdm\nimport h5py\nimport hdf5plugin\nfrom pathlib import Path\nimport gc","metadata":{"execution":{"iopub.status.busy":"2022-09-06T10:21:08.015151Z","iopub.execute_input":"2022-09-06T10:21:08.015476Z","iopub.status.idle":"2022-09-06T10:21:08.775999Z","shell.execute_reply.started":"2022-09-06T10:21:08.015444Z","shell.execute_reply":"2022-09-06T10:21:08.774814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will use the GPU to speed up calculations by a factor of almost 5.","metadata":{}},{"cell_type":"code","source":"device = \"cuda\" if torch.cuda.is_available() else \"cpu\"","metadata":{"execution":{"iopub.status.busy":"2022-09-06T10:21:08.777383Z","iopub.execute_input":"2022-09-06T10:21:08.778642Z","iopub.status.idle":"2022-09-06T10:21:08.850315Z","shell.execute_reply.started":"2022-09-06T10:21:08.778611Z","shell.execute_reply":"2022-09-06T10:21:08.849105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Helper Functions\n\n# field should be a tensor only containing zeros and ones\n# this computes n tuples in the dataset\ndef get_naive_n_tuples(field, max_depth):\n    naive_num_tuples = []\n    #append one-tuples (=number of elements) to naive tuple count\n    naive_num_tuples.append(field.sum(dim=1))\n    \n    # iterativeley calculate a new field. 1 in the field means, beginning of n tuple\n    prev_field = field.clone().detach()\n    for i in range(max_depth):\n        prev_field = prev_field[:,:-1] * field[:,i+1:]\n        naive_num_tuples.append(prev_field.sum(dim=1))\n            \n    return naive_num_tuples\n\n# get_naive_n_tuples adds 2 2-tuple counts for every 3-tuple and so on\n# this function removes those duplicates\ndef clean_naive_n_tuples(n_tuples):\n    reversed_n_tuples = list(reversed(n_tuples))\n    reverse_clean_tuples = []\n    \n    for i in range(len(reversed_n_tuples)):\n        reverse_clean_tuples.append(reversed_n_tuples[i].clone().detach())\n        \n        # correct number\n        for j in range(i):\n            reverse_clean_tuples[i] -= (i+1-j) * reverse_clean_tuples[j] \n        \n    return list(reversed(reverse_clean_tuples))","metadata":{"execution":{"iopub.status.busy":"2022-09-06T10:21:08.853241Z","iopub.execute_input":"2022-09-06T10:21:08.854138Z","iopub.status.idle":"2022-09-06T10:21:08.864539Z","shell.execute_reply.started":"2022-09-06T10:21:08.854096Z","shell.execute_reply":"2022-09-06T10:21:08.863549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# list of files we want to analyse\nf_list = list(Path('../input/open-problems-multimodal/').glob('./*.h5'))\nf_list","metadata":{"execution":{"iopub.status.busy":"2022-09-06T10:21:08.867520Z","iopub.execute_input":"2022-09-06T10:21:08.867917Z","iopub.status.idle":"2022-09-06T10:21:08.881886Z","shell.execute_reply.started":"2022-09-06T10:21:08.867864Z","shell.execute_reply":"2022-09-06T10:21:08.880810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f_name = Path('../input/open-problems-multimodal/train_cite_targets.h5')\nwith h5py.File(f_name) as f:\n    print(len(f[f_name.stem + '/axis0']))","metadata":{"execution":{"iopub.status.busy":"2022-09-06T10:21:08.883566Z","iopub.execute_input":"2022-09-06T10:21:08.884444Z","iopub.status.idle":"2022-09-06T10:21:08.900673Z","shell.execute_reply.started":"2022-09-06T10:21:08.884409Z","shell.execute_reply":"2022-09-06T10:21:08.899073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for f_name in f_list:\n    print(f\"### OPERATING ON: {f_name} ###\")\n    # Get properties of dataset\n    with h5py.File(f_name) as f:\n        cell_names = [a.decode('utf-8') for a in f[f_name.stem + '/axis1']]\n        NUM_FEATURES = len(f[f_name.stem + '/axis0'])\n    NUM_CELLS = len(cell_names)\n    CHUNK_SIZE = int(200_000_000 / (NUM_FEATURES))\n\n    # the features we want to calculate for every cell\n    target_features = {\n        'count_non_zero': torch.zeros(NUM_CELLS, dtype=torch.int32, device=device),\n        'max_value': torch.zeros(NUM_CELLS, dtype=torch.float64, device=device),\n        'min_value': torch.zeros(NUM_CELLS, dtype=torch.float64, device=device),\n        'sum_values': torch.zeros(NUM_CELLS, dtype=torch.float64, device=device),\n        'mean_non_zero': torch.zeros(NUM_CELLS, dtype=torch.float64, device=device),\n\n        # We would like to calc std_dev but currently pytorch does not support it\n        #'std_dev_non_zero': torch.zeros(NUM_CELLS, dtype=torch.float64, device=device),\n\n        # number of consecutive non-zero elements / n_tupels\n        '1_tups': torch.zeros(NUM_CELLS, dtype=torch.int32, device=device),\n        '2_tups': torch.zeros(NUM_CELLS, dtype=torch.int32, device=device),\n        '3_tups': torch.zeros(NUM_CELLS, dtype=torch.int32, device=device),\n        '4_tups': torch.zeros(NUM_CELLS, dtype=torch.int32, device=device),\n        '5_tups': torch.zeros(NUM_CELLS, dtype=torch.int32, device=device),\n        '6_tups': torch.zeros(NUM_CELLS, dtype=torch.int32, device=device),\n        '7_tups': torch.zeros(NUM_CELLS, dtype=torch.int32, device=device),\n        '8_tups': torch.zeros(NUM_CELLS, dtype=torch.int32, device=device),\n        '9_tups': torch.zeros(NUM_CELLS, dtype=torch.int32, device=device),\n        '10_tups': torch.zeros(NUM_CELLS, dtype=torch.int32, device=device),\n        '11_tups': torch.zeros(NUM_CELLS, dtype=torch.int32, device=device),\n        '12_tups': torch.zeros(NUM_CELLS, dtype=torch.int32, device=device),\n        '13_tups': torch.zeros(NUM_CELLS, dtype=torch.int32, device=device),\n        '14_tups': torch.zeros(NUM_CELLS, dtype=torch.int32, device=device),\n        '15_tups': torch.zeros(NUM_CELLS, dtype=torch.int32, device=device),\n        '16_tups': torch.zeros(NUM_CELLS, dtype=torch.int32, device=device),\n    }\n    \n    # iterate over dataset calculating features\n    for i in tqdm(range(int(np.ceil(NUM_CELLS/CHUNK_SIZE)))):\n        ###### SPECIFICATION WHAT DATA TO LOAD AND LOADING OF DATA ONTO DEVICE (CPU OR CUDA) #########\n        S_INDEX = i * CHUNK_SIZE\n        E_INDEX = (i+1) * CHUNK_SIZE\n\n        # load data and send to torch device\n        with h5py.File(f_name) as f:\n            data = torch.tensor(f[f_name.stem + '/block0_values'][S_INDEX : E_INDEX], device=device)\n\n            \n        gc.collect()\n        ##### CALCULATION OF FEATURES #####\n        target_features['count_non_zero'][S_INDEX:E_INDEX] = data.gt(0).sum(dim=1)\n        target_features['max_value'][S_INDEX:E_INDEX] = data.max(dim=1)[0]\n        target_features['min_value'][S_INDEX:E_INDEX] = data.min(dim=1)[0]\n        target_features['sum_values'][S_INDEX:E_INDEX] = data.sum(dim=1)\n\n\n        # set zero values to nan (this helps in some computations \n        # e. g. when computation mean of non zero values)\n        data[torch.eq(data, 0)] = torch.nan\n\n        target_features['mean_non_zero'][S_INDEX:E_INDEX] = data.nanmean(dim=1)\n\n        # missing implementation in pytorch (is on their todo)\n        #target_features['std_dev_non_zero'][S_INDEX:E_INDEX] = data.nanstd(dim=1)\n\n        naive_n_tuples = get_naive_n_tuples(data.gt(0), 15)\n        clean_tuples = clean_naive_n_tuples(naive_n_tuples)\n\n        target_features['1_tups'][S_INDEX:E_INDEX] = clean_tuples[0]\n        target_features['2_tups'][S_INDEX:E_INDEX] = clean_tuples[1]\n        target_features['3_tups'][S_INDEX:E_INDEX] = clean_tuples[2]\n        target_features['4_tups'][S_INDEX:E_INDEX] = clean_tuples[3]\n        target_features['5_tups'][S_INDEX:E_INDEX] = clean_tuples[4]\n        target_features['6_tups'][S_INDEX:E_INDEX] = clean_tuples[5]\n        target_features['7_tups'][S_INDEX:E_INDEX] = clean_tuples[6]\n        target_features['8_tups'][S_INDEX:E_INDEX] = clean_tuples[7]\n        target_features['9_tups'][S_INDEX:E_INDEX] = clean_tuples[8]\n        target_features['10_tups'][S_INDEX:E_INDEX] = clean_tuples[9]\n        target_features['11_tups'][S_INDEX:E_INDEX] = clean_tuples[10]\n        target_features['12_tups'][S_INDEX:E_INDEX] = clean_tuples[11]\n        target_features['13_tups'][S_INDEX:E_INDEX] = clean_tuples[12]\n        target_features['14_tups'][S_INDEX:E_INDEX] = clean_tuples[13]\n        target_features['15_tups'][S_INDEX:E_INDEX] = clean_tuples[14]\n        target_features['16_tups'][S_INDEX:E_INDEX] = clean_tuples[15]\n        # target_features['feature'][S_INDEX:E_INDEX] = \n    \n    # calculations done, define index, build dataframe and safe as csv\n    target_features_cpu = {key:value.cpu() for key, value in target_features.items()}\n    df = pd.DataFrame(data=target_features_cpu, index=cell_names)\n    df.to_csv(f_name.stem + '_cell_features.csv')","metadata":{"execution":{"iopub.status.busy":"2022-09-06T10:21:08.902774Z","iopub.execute_input":"2022-09-06T10:21:08.903450Z","iopub.status.idle":"2022-09-06T10:31:19.169298Z","shell.execute_reply.started":"2022-09-06T10:21:08.903415Z","shell.execute_reply":"2022-09-06T10:31:19.168321Z"},"trusted":true},"execution_count":null,"outputs":[]}]}