{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport tqdm","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:39:51.614856Z","iopub.execute_input":"2023-09-02T17:39:51.615268Z","iopub.status.idle":"2023-09-02T17:39:53.809056Z","shell.execute_reply.started":"2023-09-02T17:39:51.615237Z","shell.execute_reply":"2023-09-02T17:39:53.808136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to perform EDA on a dataset\ndef perform_eda(dataset, dataset_name, colors):\n    print(f\"Exploratory Data Analysis for {dataset_name}\")\n    \n    num_plots = len(dataset.files)\n    num_rows = 6\n    num_columns = 5\n    num_figures = (num_plots + num_columns - 1) // num_columns\n    \n    for figure in range(num_figures):\n        plt.figure(figsize=(20, 12))\n        \n        for idx in range(num_columns):\n            subplot_idx = figure * num_columns + idx + 1\n            \n            if subplot_idx <= num_plots:\n                array_name = dataset.files[subplot_idx - 1]\n                array_data = dataset[array_name]\n                \n                # Check if the array is 1D, if not, flatten it\n                if array_data.ndim > 1:\n                    array_data = array_data.flatten()\n                \n                # Compute summary statistics\n                mean = np.mean(array_data)\n                median = np.median(array_data)\n                std_dev = np.std(array_data)\n                \n                # Reuse colors for different arrays in the same dataset\n                color = colors[subplot_idx % len(colors)]\n                \n                plt.subplot(num_rows, num_columns, idx + 1)\n                plt.hist(array_data, bins=20, color=color, alpha=0.7)\n                plt.xlabel('Values')\n                plt.ylabel('Frequency')\n                plt.title(f'Histogram of {array_name} in {dataset_name}')\n                \n                # Print summary statistics\n                print(f\"Summary Statistics for {array_name} in {dataset_name}:\")\n                print(f\"Mean: {mean}\")\n                print(f\"Median: {median}\")\n                print(f\"Standard Deviation: {std_dev}\")\n        \n        plt.tight_layout()\n        plt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:39:53.811112Z","iopub.execute_input":"2023-09-02T17:39:53.811946Z","iopub.status.idle":"2023-09-02T17:39:53.824783Z","shell.execute_reply.started":"2023-09-02T17:39:53.811908Z","shell.execute_reply":"2023-09-02T17:39:53.824036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the list of distinct colors for histograms\ndistinct_colors = ['blue', 'green', 'red', 'purple', 'orange']\n\n# Load and perform EDA on each dataset\ndatasets = [\n    ('/kaggle/input/predict-ai-model-runtime/npz_all/npz/layout/nlp/default/train/albert_en_base_batch_size_16_test.npz', 'Dataset 1'),\n    ('/kaggle/input/predict-ai-model-runtime/npz_all/npz/layout/nlp/random/train/albert_en_base_batch_size_16_train.npz', 'Dataset 2'),\n    ('/kaggle/input/predict-ai-model-runtime/npz_all/npz/layout/xla/default/train/alexnet_train_batch_32.npz', 'Dataset 3'),\n    ('/kaggle/input/predict-ai-model-runtime/npz_all/npz/layout/xla/random/train/alexnet_train_batch_32.npz', 'Dataset 4'),\n    ('/kaggle/input/predict-ai-model-runtime/npz_all/npz/tile/xla/train/alexnet_train_batch_32_-1bae27a41d70f4dc.npz', 'Dataset 5')\n]\n\nfor dataset_path, dataset_name in datasets:\n    dataset = np.load(dataset_path)\n    perform_eda(dataset, dataset_name, distinct_colors)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:39:53.826275Z","iopub.execute_input":"2023-09-02T17:39:53.828251Z","iopub.status.idle":"2023-09-02T17:40:22.135918Z","shell.execute_reply.started":"2023-09-02T17:39:53.828212Z","shell.execute_reply":"2023-09-02T17:40:22.134874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to create Box Plots\ndef create_box_plots(array_data, array_name, dataset_name):\n    plt.figure(figsize=(10, 6))\n    plt.boxplot(array_data)\n    plt.xticks([1], [array_name], rotation=45)\n    plt.title(f'Box Plot of {array_name} in {dataset_name}')\n    plt.show()\n\n# Function to create Scatter Plots\ndef create_scatter_plots(array_data1, array_data2, dataset_name):\n    plt.figure(figsize=(10, 6))\n    plt.scatter(array_data1, array_data2, alpha=0.5)\n    plt.xlabel('X-axis Label')\n    plt.ylabel('Y-axis Label')\n    plt.title(f'Scatter Plot in {dataset_name}')\n    plt.show()\n\n# Function to create KDE Plots\ndef create_kde_plots(array_data, array_name, dataset_name):\n    plt.figure(figsize=(10, 6))\n    sns.kdeplot(array_data, shade=True)\n    plt.xlabel('Values')\n    plt.ylabel('Density')\n    plt.title(f'KDE Plot of {array_name} in {dataset_name}')\n    plt.show()\n\n# Function to create Pair Plots\ndef create_pair_plots(df, dataset_name):\n    sns.pairplot(df)\n    plt.title(f'Pair Plot in {dataset_name}')\n    plt.show()\n\n# Function to create Heatmaps\ndef create_heatmaps(df, dataset_name):\n    correlation_matrix = df.corr()\n    plt.figure(figsize=(10, 8))\n    sns.heatmap(correlation_matrix, annot=True, cmap='coolwarm')\n    plt.title(f'Correlation Heatmap in {dataset_name}')\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:40:22.138349Z","iopub.execute_input":"2023-09-02T17:40:22.138760Z","iopub.status.idle":"2023-09-02T17:40:22.150212Z","shell.execute_reply.started":"2023-09-02T17:40:22.138732Z","shell.execute_reply":"2023-09-02T17:40:22.149213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the .npz file\narchive = np.load('/kaggle/input/predict-ai-model-runtime/npz_all/npz/layout/nlp/random/train/albert_en_base_batch_size_16_train.npz')\n\n# List the keys within the archive\nprint(archive.files)","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:40:22.151745Z","iopub.execute_input":"2023-09-02T17:40:22.152051Z","iopub.status.idle":"2023-09-02T17:40:22.173564Z","shell.execute_reply.started":"2023-09-02T17:40:22.152025Z","shell.execute_reply":"2023-09-02T17:40:22.172307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset_path, dataset_name in datasets:\n    dataset = np.load(dataset_path)\n#create_scatter_plots(dataset['node_feat'], dataset['config_runtime'], dataset_name)\n    create_kde_plots(dataset['config_runtime'], 'Configuration Runtime', dataset_name)","metadata":{"execution":{"iopub.status.busy":"2023-09-02T17:40:22.175238Z","iopub.execute_input":"2023-09-02T17:40:22.175730Z","iopub.status.idle":"2023-09-02T17:40:25.113408Z","shell.execute_reply.started":"2023-09-02T17:40:22.175696Z","shell.execute_reply":"2023-09-02T17:40:25.112174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Work in progress","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}