{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Hello fellow Kagglers,\n\nThis notebook is a work-in-progress and will be updated frequently in the coming weeks.\n\nWhen this competition was launched I was rather puzzled by the data and objective, but after some research and help from my favourite community it all started to become clear.\n\nGenerally, this competition is about efficient compilation of neural network, which would reduce model runtime.\n\nThere are two main configurations\n\n**Tile**\n\nControls how a single tensor is split up and processed in parallell\n\n**Layout**\n\nControls how a single tensor is managed in memory\n\nAs far as I understand it correctly the data consists of:\n\n* Nodes with 140 features\n* Global tile configuration of 24 features applying to the whole graph\n* Per node layout configuration of 18 features for a selection of nodes\n\nMy main goal for now is to create the dataset which is advertised in the title, stay tuned!\n\nThe approach for the dataset will be to ordinally encode features, if needed with a log2 conversion.\n\nA neural network can then be constructed based on embeddings of those ordinal features.","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\n\nfrom tqdm.notebook import tqdm\nfrom numba import njit\n\nimport glob\nimport time\n\npd.options.display.max_colwidth = 999\npd.options.display.max_rows = 999","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Plotting Configuration","metadata":{}},{"cell_type":"code","source":"# MatplotLib Global Settings\nmpl.rcParams.update(mpl.rcParamsDefault)\nmpl.rcParams['xtick.labelsize'] = 16\nmpl.rcParams['ytick.labelsize'] = 16\nmpl.rcParams['axes.labelsize'] = 18\nmpl.rcParams['axes.titlesize'] = 24","metadata":{"execution":{"iopub.status.busy":"2023-09-04T09:32:11.171874Z","iopub.execute_input":"2023-09-04T09:32:11.172827Z","iopub.status.idle":"2023-09-04T09:32:11.181233Z","shell.execute_reply.started":"2023-09-04T09:32:11.172788Z","shell.execute_reply":"2023-09-04T09:32:11.179437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Field Names","metadata":{}},{"cell_type":"code","source":"NODE_FEATURE_NAMES = [\n    'is_root', # whether this node is the output\n    'element_size_in_bits', # deprecated, always 0\n    # 2–20: One hot vector of shape_element_type.\n    'shape_element_type_is_invalid_type',\n    'shape_element_type_is_pred',\n    'shape_element_type_is_s8',\n    'shape_element_type_is_s16',\n    'shape_element_type_is_s32',\n    'shape_element_type_is_s64',\n    'shape_element_type_is_u8',\n    'shape_element_type_is_u16',\n    'shape_element_type_is_u32',\n    'shape_element_type_is_u64',\n    'shape_element_type_is_f16',\n    'shape_element_type_is_f32',\n    'shape_element_type_is_f64',\n    'shape_element_type_is_bf16',\n    'shape_element_type_is_c64',\n    'shape_element_type_is_c128',\n    'shape_element_type_is_tuple',\n    'shape_element_type_is_opaque_type',\n    'shape_element_type_is_token',\n    # 21–28: Size (number of elements) for each dimension, or an upper bound on the size if the dimension is dynamic.\n    # In XLA, dimensions are numbered from 0 to N-1 for an N-dimensional array.\n    # The first element of 'shape_dimensions' is the size of dimension 0, the second element is the size of dimension 1, and so forth.\n    # Empty list indicates a scalar.\n    'shape_dimensions_0',\n    'shape_dimensions_1',\n    'shape_dimensions_2',\n    'shape_dimensions_3',\n    'shape_dimensions_4',\n    'shape_dimensions_5',\n    'shape_dimensions_sum',\n    'shape_dimensions_product',\n    'shape_tuple_shapes_size', # for tuples only, the shapes of constituent shapes in the tuple sequence\n    'parameter_number', # = K - indicating that is is the Kth parameter to the computation, only for Parameter operation\n    # 31–36: Dimensions present for some operations that require reshaping or broadcasting, including Reshape, Reduce, ReduceWindow, and Reverse.\n    'dimensions_0',\n    'dimensions_1',\n    'dimensions_2',\n    'dimensions_3',\n    'dimensions_4',\n    'dimensions_5',\n    # 37–92: Windowing information in an operation such as convolution.\n    # The window is moved across a base area and for each position of the window a computation is performed.\n    'window_size_0',\n    'window_size_1',\n    'window_size_2',\n    'window_size_3',\n    'window_size_4',\n    'window_size_5',\n    'window_size_sum',\n    'window_size_product',\n    'window_stride_0',\n    'window_stride_1',\n    'window_stride_2',\n    'window_stride_3',\n    'window_stride_4',\n    'window_stride_5',\n    'window_stride_sum',\n    'window_stride_product',\n    'window_padding_low_0',\n    'window_padding_low_1',\n    'window_padding_low_2',\n    'window_padding_low_3',\n    'window_padding_low_4',\n    'window_padding_low_5',\n    'window_padding_low_sum',\n    'window_padding_low_product',\n    'window_padding_high_0',\n    'window_padding_high_1',\n    'window_padding_high_2',\n    'window_padding_high_3',\n    'window_padding_high_4',\n    'window_padding_high_5',\n    'window_padding_high_sum',\n    'window_padding_high_product',\n    # 69–76: Dilation factor of the sliding window.\n    # A dilation factor of 1 means no dilation. window_dilation - 1 no-op entries (\"holes\") are implicitly placed between each kernel element.\n    'window_window_dilation_0',\n    'window_window_dilation_1',\n    'window_window_dilation_2',\n    'window_window_dilation_3',\n    'window_window_dilation_4',\n    'window_window_dilation_5',\n    'window_window_dilation_sum',\n    'window_window_dilation_product',\n    # 77-84: Dilation factor of the base area.\n    # A dilation factor of 1 means no dilation. base_dilation - 1 no-op entries (\"holes\") are implicitly placed between each base area element.\n    'window_base_dilation_0',\n    'window_base_dilation_1',\n    'window_base_dilation_2',\n    'window_base_dilation_3',\n    'window_base_dilation_4',\n    'window_base_dilation_5',\n    'window_base_dilation_sum',\n    'window_base_dilation_product',\n    # 85-92: Window reversal means that this dimension was logically reversed before the operation.\n    'window_window_reversal_0',\n    'window_window_reversal_1',\n    'window_window_reversal_2',\n    'window_window_reversal_3',\n    'window_window_reversal_4',\n    'window_window_reversal_5',\n    'window_window_reversal_true_count',\n    'window_window_reversal_false_count',\n    # 93–106: The dimension numbers used for a convolution.\n    'convolution_dim_numbers_input_batch_dim', # the dimension number that represents batch in the input\n    'convolution_dim_numbers_input_feature_dim', # the dimension number that represents features in the input\n    # 95–98: Dimension numbers for the spatial dimensions that the window moves through in the input.\n    'convolution_dim_numbers_input_spatial_dims_0',\n    'convolution_dim_numbers_input_spatial_dims_1',\n    'convolution_dim_numbers_input_spatial_dims_2',\n    'convolution_dim_numbers_input_spatial_dims_3',\n    'convolution_dim_numbers_kernel_input_feature_dim', # the dimension number that represents input features in the convolutional kernel (rhs)\n    'convolution_dim_numbers_kernel_output_feature_dim', # the dimension number that represents output features in the convolutional kernel (rhs)\n    # 101-104: Dimension numbers for the spatial dimensions that the window moves through in the kernel (rhs).\n    # window.strides(0) is the stride in the kernel_spatial_dimensions(0) dimension.\n    'convolution_dim_numbers_kernel_spatial_dims_0',\n    'convolution_dim_numbers_kernel_spatial_dims_1',\n    'convolution_dim_numbers_kernel_spatial_dims_2',\n    'convolution_dim_numbers_kernel_spatial_dims_3',\n    'convolution_dim_numbers_output_batch_dim', # the dimension number that represents batch in the output\n    'convolution_dim_numbers_output_feature_dim', # the dimension number that represents features in the output\n    'feature_group_count', # the number of feature groups, used for a convolution. Must be a divisor of the input feature dimension and output feature dimension. If not specified, it will use a default value of 1.\n    'batch_group_count', # the number of batch groups, used for a convolution.\n    # 109–120: [begin/start, end/limit) index range and stride for a slice operation.\n    'slice_dims_start_0',\n    'slice_dims_start_1',\n    'slice_dims_start_sum',\n    'slice_dims_start_product',\n    'slice_dims_stride_0',\n    'slice_dims_stride_1',\n    'slice_dims_stride_sum',\n    'slice_dims_stride_product',\n    'slice_dims_limit_0',\n    'slice_dims_limit_1',\n    'slice_dims_limit_sum',\n    'slice_dims_limit_product',\n    # 121 - 124: [start, start + size) range size for a dynamic slice ('start' is specified dynamically in the second operand of the operation).\n    'dynamic_slice_sizes_0',\n    'dynamic_slice_sizes_1',\n    'dynamic_slice_sizes_sum',\n    'dynamic_slice_sizes_product',\n    # 125–132: Padding configuration that describes the edge padding of a pad operation.\n    'padding_config_edge_padding_low_0',\n    'padding_config_edge_padding_low_1',\n    'padding_config_edge_padding_low_sum',\n    'padding_config_edge_padding_low_product',\n    'padding_config_edge_padding_high_0',\n    'padding_config_edge_padding_high_1',\n    'padding_config_edge_padding_high_sum',\n    'padding_config_edge_padding_high_product',\n    'is_stable - whether this Sort operation should be stable',\n    # 134–139: Physical layout used to pack the tensor shape.\n    'layout_minor_to_major_0',\n    'layout_minor_to_major_1',\n    'layout_minor_to_major_2',\n    'layout_minor_to_major_3',\n    'layout_minor_to_major_4',\n    'layout_minor_to_major_5',\n]","metadata":{"execution":{"iopub.status.busy":"2023-09-04T09:32:11.18297Z","iopub.execute_input":"2023-09-04T09:32:11.183385Z","iopub.status.idle":"2023-09-04T09:32:11.203362Z","shell.execute_reply.started":"2023-09-04T09:32:11.183347Z","shell.execute_reply":"2023-09-04T09:32:11.202363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TILE_FEATURE_NAMES = [\n    # 0–7: Tile sizes of the convolution kernel, only for a convolution operation.\n    'kernel_bounds_0',\n    'kernel_bounds_1',\n    'kernel_bounds_2',\n    'kernel_bounds_3',\n    'kernel_bounds_4',\n    'kernel_bounds_5',\n    'kernel_bounds_sum',\n    'kernel_bounds_product',\n    # 8–15: Output tile sizes.\n    'output_bounds_0',\n    'output_bounds_1',\n    'output_bounds_2',\n    'output_bounds_3',\n    'output_bounds_4',\n    'output_bounds_5',\n    'output_bounds_sum',\n    'output_bounds_product',\n    # 16-23: Input tile sizes.\n    'input_bounds_0',\n    'input_bounds_1',\n    'input_bounds_2',\n    'input_bounds_3',\n    'input_bounds_4',\n    'input_bounds_5',\n    'input_bounds_sum',\n    'input_bounds_product',\n]","metadata":{"execution":{"iopub.status.busy":"2023-09-04T09:32:11.206318Z","iopub.execute_input":"2023-09-04T09:32:11.207581Z","iopub.status.idle":"2023-09-04T09:32:11.223936Z","shell.execute_reply.started":"2023-09-04T09:32:11.207536Z","shell.execute_reply":"2023-09-04T09:32:11.222705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LAYOUT_FEATURE_NAMES = [\n    # 0–5: Physical layout of the output tensor\n    'output_layout_0',\n    'output_layout_1',\n    'output_layout_2',\n    'output_layout_3',\n    'output_layout_4',\n    'output_layout_5',\n    # 6-11: Physical layout of the input  tensor\n    'intput_layout_0',\n    'intput_layout_1',\n    'intput_layout_2',\n    'intput_layout_3',\n    'intput_layout_4',\n    'intput_layout_5',\n    # 12-17: Physical layout of the kernel tensor, only for a convolution operation\n    'kernel_layout_0',\n    'kernel_layout_1',\n    'kernel_layout_2',\n    'kernel_layout_3',\n    'kernel_layout_4',\n    'kernel_layout_5',\n]","metadata":{"execution":{"iopub.status.busy":"2023-09-04T09:32:11.225401Z","iopub.execute_input":"2023-09-04T09:32:11.225802Z","iopub.status.idle":"2023-09-04T09:32:11.242709Z","shell.execute_reply.started":"2023-09-04T09:32:11.225772Z","shell.execute_reply":"2023-09-04T09:32:11.241585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Config","metadata":{}},{"cell_type":"code","source":"# Feature Depth\nN_NODE_FEATURES = 140\nN_LAYOUT_FEATURES = 18\nN_TILE_FEATURES = 24\n# Describe Statistics Percentiles\nPERCENTILES = [0.01, 0.10, 0.05, 0.25, 0.50, 0.75, 0.90, 0.95, 0.99, 0.999]","metadata":{"execution":{"iopub.status.busy":"2023-09-04T09:32:11.244027Z","iopub.execute_input":"2023-09-04T09:32:11.244339Z","iopub.status.idle":"2023-09-04T09:32:11.256109Z","shell.execute_reply.started":"2023-09-04T09:32:11.244313Z","shell.execute_reply":"2023-09-04T09:32:11.255006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Sample Submission","metadata":{}},{"cell_type":"code","source":"sample_submission = pd.read_csv('/kaggle/input/predict-ai-model-runtime/sample_submission.csv')\n\ndisplay(sample_submission.sample(10))\ndisplay(sample_submission.info())","metadata":{"execution":{"iopub.status.busy":"2023-09-04T09:32:11.25754Z","iopub.execute_input":"2023-09-04T09:32:11.2579Z","iopub.status.idle":"2023-09-04T09:32:11.364871Z","shell.execute_reply.started":"2023-09-04T09:32:11.25787Z","shell.execute_reply":"2023-09-04T09:32:11.363721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# File Paths","metadata":{}},{"cell_type":"code","source":"FILE_PATHS_ALL = glob.glob('/kaggle/input/predict-ai-model-runtime/npz_all/npz/**/*.npz', recursive=True)\nFILE_PATHS_TRAIN = glob.glob('/kaggle/input/predict-ai-model-runtime/npz_all/npz/**/train/*', recursive=True)\nFILE_PATHS_VAL = glob.glob('/kaggle/input/predict-ai-model-runtime/npz_all/npz/**/valid/*', recursive=True)\nFILE_PATHS_TEST = glob.glob('/kaggle/input/predict-ai-model-runtime/npz_all/npz/**/test/*', recursive=True)\n\nprint(f'Found {len(FILE_PATHS_ALL)} Total Samples')\nprint(f'Found {len(FILE_PATHS_TRAIN)} Training Samples')\nprint(f'Found {len(FILE_PATHS_VAL)} Validation Samples')\nprint(f'Found {len(FILE_PATHS_TEST)} Test Samples')","metadata":{"execution":{"iopub.status.busy":"2023-09-04T09:32:11.366581Z","iopub.execute_input":"2023-09-04T09:32:11.367032Z","iopub.status.idle":"2023-09-04T09:32:15.006034Z","shell.execute_reply.started":"2023-09-04T09:32:11.366991Z","shell.execute_reply":"2023-09-04T09:32:15.004876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tile Example File","metadata":{}},{"cell_type":"code","source":"for fp in FILE_PATHS_TRAIN:\n    if 'tile' in fp:\n        print('='*10, 'TILE', '='*10)\n        data = np.load(fp)\n        for key in data.files:\n            print(key, 'shape', data[key].shape)\n        break","metadata":{"execution":{"iopub.status.busy":"2023-09-04T09:32:15.007909Z","iopub.execute_input":"2023-09-04T09:32:15.008323Z","iopub.status.idle":"2023-09-04T09:32:15.02441Z","shell.execute_reply.started":"2023-09-04T09:32:15.008285Z","shell.execute_reply":"2023-09-04T09:32:15.023495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Layout Example File","metadata":{}},{"cell_type":"code","source":"for fp in FILE_PATHS_TRAIN:\n    if 'layout' in fp:\n        print('='*16, 'LAYOUT', '='*16)\n        data = np.load(fp)\n        for key in data.files:\n            print(key.ljust(16), '\\tshape', data[key].shape)\n        break","metadata":{"execution":{"iopub.status.busy":"2023-09-04T09:32:15.029604Z","iopub.execute_input":"2023-09-04T09:32:15.03016Z","iopub.status.idle":"2023-09-04T09:32:15.035875Z","shell.execute_reply.started":"2023-09-04T09:32:15.030126Z","shell.execute_reply":"2023-09-04T09:32:15.034808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# General Statistics","metadata":{}},{"cell_type":"code","source":"META_DATA = {\n    'N_NODES': [],\n    'OPCODES': [],\n    'MIN_MAX_RUNTIME_RATIO': [],\n    'SUBSET': []\n}\n# Iterate Over Files\nfor fp_idx, fp in enumerate(tqdm(FILE_PATHS_ALL)):\n    data = np.load(fp)\n    # Tile Uses All Nodes\n    if 'tile' in fp:\n        META_DATA['N_NODES'].append(data['node_opcode'].size)\n        META_DATA['OPCODES'] += data['node_opcode'].tolist()\n    # Layout Uses Node Subset\n    else:\n        node_config_ids = data['node_config_ids']\n        META_DATA['N_NODES'].append(node_config_ids.size)\n        META_DATA['OPCODES'] += data['node_opcode'][node_config_ids].tolist()\n        \n    # Test Files Do Not Have Runtime Information\n    if 'test' in fp:\n        META_DATA['MIN_MAX_RUNTIME_RATIO'].append(np.nan)\n    else:\n        META_DATA['MIN_MAX_RUNTIME_RATIO'].append(data['config_runtime'].min() / data['config_runtime'].max())\n    # Subset\n    META_DATA['SUBSET'].append(fp.split('/')[-2])\n    \n# Make DataFrame to GroupBy Subset\nMETA_DATA_DF = pd.DataFrame({\n    'N_NODES': META_DATA['N_NODES'],\n    'MIN_MAX_RUNTIME_RATIO': META_DATA['MIN_MAX_RUNTIME_RATIO'],\n    'SUBSET': META_DATA['SUBSET'],\n})\n\n# Filter To Acquire Train/Val\nFILTER_TEST = np.array(META_DATA['SUBSET']) != 'test'","metadata":{"execution":{"iopub.status.busy":"2023-09-04T09:32:15.037717Z","iopub.execute_input":"2023-09-04T09:32:15.038148Z","iopub.status.idle":"2023-09-04T09:32:15.700328Z","shell.execute_reply.started":"2023-09-04T09:32:15.038116Z","shell.execute_reply":"2023-09-04T09:32:15.69908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(pd.Series(META_DATA['N_NODES']).describe(percentiles=PERCENTILES).to_frame('Value').astype(int))\n\nplt.figure(figsize=(15,8))\nplt.title('Number of Nodes Distribution')\nMETA_DATA_DF.groupby('SUBSET')['N_NODES'].plot(kind='hist',alpha=0.50, bins=32)\nplt.xlim(0)\nplt.grid()\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-04T09:32:15.701994Z","iopub.execute_input":"2023-09-04T09:32:15.702441Z","iopub.status.idle":"2023-09-04T09:32:16.232948Z","shell.execute_reply.started":"2023-09-04T09:32:15.7024Z","shell.execute_reply":"2023-09-04T09:32:16.231613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"opcodes, counts = np.unique(META_DATA['OPCODES'], return_counts=True)\n\nidxs = np.arange(opcodes.size)\nplt.figure(figsize=(20,8))\nplt.title('Opcode Distribution')\nplt.bar(idxs, counts)\nplt.xticks(idxs, opcodes, rotation=15, size=12)\nplt.xlim(-0.50, idxs.max()+0.50)\nplt.grid(axis='y')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-04T09:32:16.23454Z","iopub.execute_input":"2023-09-04T09:32:16.2349Z","iopub.status.idle":"2023-09-04T09:32:16.766951Z","shell.execute_reply.started":"2023-09-04T09:32:16.23487Z","shell.execute_reply":"2023-09-04T09:32:16.765556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(pd.Series(META_DATA['MIN_MAX_RUNTIME_RATIO']).describe(percentiles=PERCENTILES).to_frame('Value').round(2))\n\nplt.figure(figsize=(15,8))\nplt.title('Minimum/Maximum Runtime Ratio')\nplt.hist(META_DATA['MIN_MAX_RUNTIME_RATIO'], bins=32)\nplt.xlim(0,1)\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-04T09:32:16.768753Z","iopub.execute_input":"2023-09-04T09:32:16.769095Z","iopub.status.idle":"2023-09-04T09:32:17.18613Z","shell.execute_reply.started":"2023-09-04T09:32:16.769064Z","shell.execute_reply":"2023-09-04T09:32:17.184949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Node Features","metadata":{}},{"cell_type":"code","source":"NODE_FEATURES = [{} for _ in range(N_NODE_FEATURES)]\n\nfor fp in tqdm(FILE_PATHS_ALL):\n    data = np.load(fp)\n    node_feat = data['node_feat']\n    # Layout Uses Node Subset\n    if 'layout' in fp:\n        node_config_ids = data['node_config_ids']\n        node_feat = node_feat[node_config_ids]\n    # Iterate Over Node Features\n    for vv_idx, vv in enumerate(data['node_feat'].T):\n        # Log2 Transforms\n        if vv_idx in [21, 22, 23, 24, 25, 27, 28, 29, 37, 38, 43, 44, 107, 108, 109, 110, 111, 112, 117, 118, 119, 120, 123, 124, 129, 130, 131]:\n            v = 0 if v == 0 else np.round(np.log2(v))\n        # Value Counts\n        values, counts = np.unique(vv, return_counts=True)\n        for v, c in zip(values, counts):\n            if v in NODE_FEATURES[vv_idx]:\n                NODE_FEATURES[vv_idx][v] += c\n            else:\n                NODE_FEATURES[vv_idx][v] = c","metadata":{"execution":{"iopub.status.busy":"2023-09-04T09:32:17.187625Z","iopub.execute_input":"2023-09-04T09:32:17.18809Z","iopub.status.idle":"2023-09-04T09:32:18.072206Z","shell.execute_reply.started":"2023-09-04T09:32:17.188048Z","shell.execute_reply":"2023-09-04T09:32:18.070964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NODE_FEATURES_STATS = pd.DataFrame()\n# Number of Unique Values\nNODE_FEATURES_STATS['n_unique'] = [len(d) for d in NODE_FEATURES]\n# Maximum Value\nNODE_FEATURES_STATS['max'] = [int(max(d.keys())) for d in NODE_FEATURES]\n# Unique Values\nNODE_FEATURES_STATS['values'] = [d.keys() for d in NODE_FEATURES]\n# Feature Name\nNODE_FEATURES_STATS['feature_name'] = NODE_FEATURE_NAMES","metadata":{"execution":{"iopub.status.busy":"2023-09-04T09:32:18.073521Z","iopub.execute_input":"2023-09-04T09:32:18.073893Z","iopub.status.idle":"2023-09-04T09:32:18.08584Z","shell.execute_reply.started":"2023-09-04T09:32:18.073864Z","shell.execute_reply":"2023-09-04T09:32:18.0845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Max: {NODE_FEATURES_STATS[\"max\"].max()}')\n\ndisplay(NODE_FEATURES_STATS.sort_values('n_unique', ascending=False))","metadata":{"execution":{"iopub.status.busy":"2023-09-04T09:32:18.087974Z","iopub.execute_input":"2023-09-04T09:32:18.088427Z","iopub.status.idle":"2023-09-04T09:32:18.205075Z","shell.execute_reply.started":"2023-09-04T09:32:18.088385Z","shell.execute_reply":"2023-09-04T09:32:18.203904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Layout Features","metadata":{}},{"cell_type":"code","source":"LAYOUT_FEATURES = [{} for _ in range(N_LAYOUT_FEATURES)]\n\nfor fp in tqdm(FILE_PATHS_ALL):\n    if 'layout' in fp:\n        # Read Data\n        data = np.load(fp, mmap_mode='r')\n        # Load Node Configurations\n        node_config_feat = data['node_config_feat']\n        # Reshape\n        C, N, F = node_config_feat.shape\n        node_config_feat = node_config_feat.T.reshape(F, N*C)\n        # Iterate Over Node Configuration Features\n        for vv_idx, vv in enumerate(node_config_feat):\n            # Log2 Transforms\n            if vv_idx in []:\n                v = 0 if v == 0 else np.round(np.log2(v))\n            # Value Counts\n            values, counts = np.unique(vv, return_counts=True)\n            for v, c in enumerate(counts):\n                if v in LAYOUT_FEATURES[vv_idx]:\n                    LAYOUT_FEATURES[vv_idx][v] += c\n                else:\n                    LAYOUT_FEATURES[vv_idx][v] = c","metadata":{"execution":{"iopub.status.busy":"2023-09-04T09:32:18.206487Z","iopub.execute_input":"2023-09-04T09:32:18.207621Z","iopub.status.idle":"2023-09-04T09:32:18.236505Z","shell.execute_reply.started":"2023-09-04T09:32:18.207566Z","shell.execute_reply":"2023-09-04T09:32:18.235677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LAYOUT_FEATURES_STATS = pd.DataFrame()\n# Number of Unique Values\nLAYOUT_FEATURES_STATS['n_unique'] = [len(d) for d in LAYOUT_FEATURES]\n# Maximum Value\nLAYOUT_FEATURES_STATS['max'] = [int(max(d.keys())) for d in LAYOUT_FEATURES]\n# Unique Values\nLAYOUT_FEATURES_STATS['values'] = [d.keys() for d in LAYOUT_FEATURES]\n# Feature Name\nLAYOUT_FEATURES_STATS['feature_name'] = LAYOUT_FEATURE_NAMES","metadata":{"execution":{"iopub.status.busy":"2023-09-04T09:32:18.237545Z","iopub.execute_input":"2023-09-04T09:32:18.238277Z","iopub.status.idle":"2023-09-04T09:32:19.007675Z","shell.execute_reply.started":"2023-09-04T09:32:18.238243Z","shell.execute_reply":"2023-09-04T09:32:19.005927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Max: {LAYOUT_FEATURES_STATS[\"max\"].max()}')\n\ndisplay(LAYOUT_FEATURES_STATS.sort_values('n_unique', ascending=False))","metadata":{"execution":{"iopub.status.busy":"2023-09-04T09:32:19.008907Z","iopub.status.idle":"2023-09-04T09:32:19.009455Z","shell.execute_reply.started":"2023-09-04T09:32:19.009246Z","shell.execute_reply":"2023-09-04T09:32:19.009268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tile Features","metadata":{}},{"cell_type":"code","source":"TILE_FEATURES = [{} for _ in range(N_TILE_FEATURES)]\n\nfor fp in tqdm(FILE_PATHS_ALL):\n    if 'tile' in fp:\n        # Read Data\n        data = np.load(fp, mmap_mode='r')\n        # Load Node Configurations\n        node_config_feat = data['config_feat'].T\n        # Iterate Over Node Configuration Features\n        for vv_idx, vv in enumerate(node_config_feat):\n            # Log2 Transforms\n            if vv_idx in []:\n                v = 0 if v == 0 else np.round(np.log2(v))\n            # Value Counts\n            values, counts = np.unique(vv, return_counts=True)\n            for v, c in enumerate(counts):\n                if v in TILE_FEATURES[vv_idx]:\n                    TILE_FEATURES[vv_idx][v] += c\n                else:\n                    TILE_FEATURES[vv_idx][v] = c","metadata":{"execution":{"iopub.status.busy":"2023-09-04T09:32:19.011287Z","iopub.status.idle":"2023-09-04T09:32:19.012394Z","shell.execute_reply.started":"2023-09-04T09:32:19.012159Z","shell.execute_reply":"2023-09-04T09:32:19.012184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TILE_FEATURES_STATS = pd.DataFrame()\n# Number of Unique Values\nTILE_FEATURES_STATS['n_unique'] = [len(d) for d in TILE_FEATURES]\n# Maximum Value\nTILE_FEATURES_STATS['max'] = [int(max(d.keys())) for d in TILE_FEATURES]\n# Unique Values\nTILE_FEATURES_STATS['values'] = [d.keys() for d in TILE_FEATURES]\n# Feature Name\nTILE_FEATURES_STATS['feature_name'] = TILE_FEATURE_NAMES","metadata":{"execution":{"iopub.status.busy":"2023-09-04T09:32:19.01359Z","iopub.status.idle":"2023-09-04T09:32:19.014123Z","shell.execute_reply.started":"2023-09-04T09:32:19.013831Z","shell.execute_reply":"2023-09-04T09:32:19.01385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Max: {TILE_FEATURES_STATS[\"max\"].max()}')\n\ndisplay(TILE_FEATURES_STATS.sort_values('n_unique', ascending=False))","metadata":{"execution":{"iopub.status.busy":"2023-09-04T09:32:19.015257Z","iopub.status.idle":"2023-09-04T09:32:19.015694Z","shell.execute_reply.started":"2023-09-04T09:32:19.015487Z","shell.execute_reply":"2023-09-04T09:32:19.015506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataset","metadata":{}},{"cell_type":"code","source":"# TODO","metadata":{"execution":{"iopub.status.busy":"2023-09-04T09:32:19.020514Z","iopub.status.idle":"2023-09-04T09:32:19.02174Z","shell.execute_reply.started":"2023-09-04T09:32:19.021426Z","shell.execute_reply":"2023-09-04T09:32:19.021454Z"},"trusted":true},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}