{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":58266,"databundleVersionId":6641124,"sourceType":"competition"}],"dockerImageVersionId":30615,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom tqdm import tqdm\nimport pandas as pd\npd.options.mode.chained_assignment = None  # default='warn'","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-13T20:48:16.131589Z","iopub.execute_input":"2023-12-13T20:48:16.132033Z","iopub.status.idle":"2023-12-13T20:48:16.138061Z","shell.execute_reply.started":"2023-12-13T20:48:16.132001Z","shell.execute_reply":"2023-12-13T20:48:16.136886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Load the layout data","metadata":{}},{"cell_type":"code","source":"class LayoutDataProcessor:\n    def __init__(self, directory, split):\n        self.directory = os.path.join(directory, split)\n        self.data = []\n        self.feature_occurrences = {}\n        self.initialized = False\n\n    def load_data(self):\n        for filename in tqdm(os.listdir(self.directory)):\n            filepath = os.path.join(self.directory, filename)\n            self.process_file(filepath, filename)\n        self.calculate_occurrence_rates()\n\n    def process_file(self, filepath, filename):\n        data = np.load(filepath)\n        node_config_ids = data['node_config_ids']\n        node_config_feat = data['node_config_feat']\n        config_runtime = data['config_runtime']\n        node_feat = data['node_feat']\n        node_opcode = data['node_opcode']\n        node_feat_avg = np.mean(node_feat, axis=0)  # Calculate average node features\n\n        # Initialize feature occurrence tracking if not done yet\n        if not self.initialized:\n            self.initialize_feature_occurrences(node_config_feat.shape[2])\n            self.initialized = True\n\n        # Process each configuration\n        for i in range(len(config_runtime)):\n            # Configuration feature array for the current configuration\n            current_config_features = node_config_feat[i, :, :]\n\n            # Append features to the data dictionary\n            row = {\n                'config_id': f\"{filename}\",\n                'runtime': config_runtime[i],\n                'node_feat_avg': node_feat_avg.tolist(),  # Add the average node features\n            }\n\n            # Add node_config_feat features and update feature occurrences\n            self.add_config_features(row, current_config_features)\n\n            self.data.append(row)\n\n    def initialize_feature_occurrences(self, num_features):\n        for i in range(num_features):\n            self.feature_occurrences[f\"feature_{i}\"] = {}\n\n    def add_config_features(self, row, config_features):\n        for feature_index in range(config_features.shape[1]):\n            feature_name = f\"feature_{feature_index}\"\n            feature_value = config_features[0, feature_index]\n            row[feature_name] = feature_value\n\n            # Update occurrence counts for each feature\n            self.feature_occurrences[feature_name].setdefault(feature_value, 0)\n            self.feature_occurrences[feature_name][feature_value] += 1\n\n\n    def calculate_occurrence_rates(self):\n        # Calculate occurrence rates for each feature\n        for row in self.data:\n            for feature_name, occurrences in self.feature_occurrences.items():\n                feature_value = row.get(feature_name)\n                if feature_value is not None:  # Ensure the feature was recorded for this config\n                    total_occurrences = sum(occurrences.values())\n                    row[feature_name + '_rate'] = occurrences[feature_value] / total_occurrences\n\n    def get_dataframe(self):\n        df = pd.DataFrame(self.data)\n        # Unpack 'node_feat_avg' into separate columns\n        node_feat_avg_df = df.pop('node_feat_avg').apply(pd.Series)\n        node_feat_avg_df.columns = [f'node_feat_avg_{i}' for i in range(node_feat_avg_df.shape[1])]\n        # Concatenate with the original DataFrame\n        df = pd.concat([df, node_feat_avg_df], axis=1)\n        return df","metadata":{"execution":{"iopub.status.busy":"2023-12-13T20:33:16.532343Z","iopub.execute_input":"2023-12-13T20:33:16.532834Z","iopub.status.idle":"2023-12-13T20:33:16.550933Z","shell.execute_reply.started":"2023-12-13T20:33:16.532801Z","shell.execute_reply":"2023-12-13T20:33:16.549797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Working with validation set of xla:default collection","metadata":{}},{"cell_type":"code","source":"processor = LayoutDataProcessor('/kaggle/input/predict-ai-model-runtime/npz_all/npz/layout/xla/default', 'valid')\nprocessor.load_data()\ndf_valid = processor.get_dataframe()","metadata":{"execution":{"iopub.status.busy":"2023-12-13T20:33:16.552756Z","iopub.execute_input":"2023-12-13T20:33:16.553131Z","iopub.status.idle":"2023-12-13T20:33:45.806893Z","shell.execute_reply.started":"2023-12-13T20:33:16.553101Z","shell.execute_reply":"2023-12-13T20:33:45.805915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_valid.columns","metadata":{"execution":{"iopub.status.busy":"2023-12-13T20:33:45.809044Z","iopub.execute_input":"2023-12-13T20:33:45.809386Z","iopub.status.idle":"2023-12-13T20:33:45.819843Z","shell.execute_reply.started":"2023-12-13T20:33:45.809356Z","shell.execute_reply":"2023-12-13T20:33:45.818708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"minmax scale target by config_id","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\nfrom sklearn.preprocessing import MinMaxScaler\n\nscaler = MinMaxScaler()\n\n# Iterate over each config_id and scale the target column within each group\nfor config_id in tqdm(df_valid['config_id'].unique()):\n    # Selecting the rows corresponding to the current config_id\n    idx = df_valid['config_id'] == config_id\n    # Scaling the target column for the current group\n    df_valid.loc[idx, 'runtime'] = scaler.fit_transform(df_valid.loc[idx, ['runtime']])","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-12-13T20:33:45.821328Z","iopub.execute_input":"2023-12-13T20:33:45.821779Z","iopub.status.idle":"2023-12-13T20:33:46.486680Z","shell.execute_reply.started":"2023-12-13T20:33:45.821736Z","shell.execute_reply":"2023-12-13T20:33:46.485563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_valid.to_csv('processed_layout_data.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-12-13T20:33:46.488435Z","iopub.execute_input":"2023-12-13T20:33:46.489155Z","iopub.status.idle":"2023-12-13T20:34:01.307068Z","shell.execute_reply.started":"2023-12-13T20:33:46.489112Z","shell.execute_reply":"2023-12-13T20:34:01.306014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Load data from here","metadata":{}},{"cell_type":"code","source":"df_valid = pd.read_csv('processed_layout_data.csv')","metadata":{"execution":{"iopub.status.busy":"2023-12-13T20:34:19.616486Z","iopub.execute_input":"2023-12-13T20:34:19.616958Z","iopub.status.idle":"2023-12-13T20:34:21.642963Z","shell.execute_reply.started":"2023-12-13T20:34:19.616920Z","shell.execute_reply":"2023-12-13T20:34:21.641806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_valid.info()","metadata":{"execution":{"iopub.status.busy":"2023-12-13T20:34:21.677544Z","iopub.execute_input":"2023-12-13T20:34:21.677933Z","iopub.status.idle":"2023-12-13T20:34:21.706422Z","shell.execute_reply.started":"2023-12-13T20:34:21.677901Z","shell.execute_reply":"2023-12-13T20:34:21.704351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LinearRegression, Lasso, Ridge\nfrom sklearn.metrics import r2_score\nfrom sklearn.model_selection import GridSearchCV","metadata":{"execution":{"iopub.status.busy":"2023-12-13T20:34:21.954262Z","iopub.execute_input":"2023-12-13T20:34:21.955112Z","iopub.status.idle":"2023-12-13T20:34:22.143841Z","shell.execute_reply.started":"2023-12-13T20:34:21.955065Z","shell.execute_reply":"2023-12-13T20:34:22.142223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"train test split, ensure config_ids are not split across train and test splits","metadata":{}},{"cell_type":"code","source":"unique_config_ids = df_valid['config_id'].unique()\ntrain_config_ids, test_config_ids = train_test_split(unique_config_ids, test_size=0.2, random_state=42)\n\n# Creating train and test dataframes based on config_id\ntrain_df = df_valid[df_valid['config_id'].isin(train_config_ids)]\ntest_df = df_valid[df_valid['config_id'].isin(test_config_ids)]\n\n# Separating features and target variable\nX_train = train_df.drop(['config_id', 'runtime',], axis=1)\ny_train = train_df['runtime']\nX_test = test_df.drop(['config_id', 'runtime',], axis=1)\ny_test = test_df['runtime']","metadata":{"execution":{"iopub.status.busy":"2023-12-13T20:34:23.662980Z","iopub.execute_input":"2023-12-13T20:34:23.664157Z","iopub.status.idle":"2023-12-13T20:34:23.761664Z","shell.execute_reply.started":"2023-12-13T20:34:23.664112Z","shell.execute_reply":"2023-12-13T20:34:23.760836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install LightGBM\nfrom lightgbm import LGBMRegressor","metadata":{"execution":{"iopub.status.busy":"2023-12-13T20:34:38.251953Z","iopub.execute_input":"2023-12-13T20:34:38.252339Z","iopub.status.idle":"2023-12-13T20:34:55.048085Z","shell.execute_reply.started":"2023-12-13T20:34:38.252308Z","shell.execute_reply":"2023-12-13T20:34:55.047081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_valid['runtime'].describe()","metadata":{"execution":{"iopub.status.busy":"2023-12-13T20:34:55.050070Z","iopub.execute_input":"2023-12-13T20:34:55.051135Z","iopub.status.idle":"2023-12-13T20:34:55.072056Z","shell.execute_reply.started":"2023-12-13T20:34:55.051096Z","shell.execute_reply":"2023-12-13T20:34:55.070987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"best params from hyperparameter tuning","metadata":{}},{"cell_type":"code","source":"lasso_alpha=100\nridge_alpha=100\nlgbm_params = {'colsample_bytree': 0.9101057142919446,\n 'learning_rate': 0.1,\n 'min_child_samples': 340,\n 'min_child_weight': 0.07452212998940164,\n 'n_estimators': 500,\n 'num_leaves': 15,\n 'reg_alpha': 0.1,\n 'reg_lambda': 10,\n 'subsample': 0.861899732308057}\n\n\nlr = LinearRegression()\nlasso = Lasso(alpha = lasso_alpha)\nridge = Ridge(alpha = ridge_alpha)\nlgbm = LGBMRegressor(**lgbm_params)\n\nmodels = [lr, lasso, ridge, lgbm]\nfor model in tqdm(models):\n    model.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-12-13T20:42:45.158258Z","iopub.execute_input":"2023-12-13T20:42:45.158764Z","iopub.status.idle":"2023-12-13T20:42:50.919178Z","shell.execute_reply.started":"2023-12-13T20:42:45.158728Z","shell.execute_reply":"2023-12-13T20:42:50.917854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Making predictions\ntrain_predictions_lin = lin_reg.predict(X_train)\ntrain_predictions_lasso = lasso_reg.predict(X_train)\ntrain_predictions_ridge = ridge_reg.predict(X_train)\ntrain_predictions_lgbm = lgbm.predict(X_train)\n\npredictions_lin = lin_reg.predict(X_test)\npredictions_lasso = lasso_reg.predict(X_test)\npredictions_ridge = ridge_reg.predict(X_test)\npredictions_lgbm = lgbm.predict(X_test)\n\n# Calculating R² scores\nr2_lin = r2_score(y_test, predictions_lin)\nr2_lasso = r2_score(y_test, predictions_lasso)\nr2_ridge = r2_score(y_test, predictions_ridge)\nr2_lgbm = r2_score(y_test, predictions_lgbm)\n\nr2_train_lin = r2_score(y_train, train_predictions_lin)\nr2_train_lasso = r2_score(y_train, train_predictions_lasso)\nr2_train_ridge = r2_score(y_train, train_predictions_ridge)\nr2_train_lgbm = r2_score(y_train, train_predictions_lgbm)\n\nprint(\"R² Scores Train:\")\nprint(f\"Linear Regression: {r2_train_lin}\")\nprint(f\"Lasso Regression: {r2_train_lasso}\")\nprint(f\"Ridge Regression: {r2_train_ridge}\")\nprint(f\"LGBM Regression: {r2_train_lgbm}\")\n\nprint(\"R² Scores Test:\")\nprint(f\"Linear Regression: {r2_lin}\")\nprint(f\"Lasso Regression: {r2_lasso}\")\nprint(f\"Ridge Regression: {r2_ridge}\")\nprint(f\"LGBM Regression: {r2_lgbm}\")","metadata":{"execution":{"iopub.status.busy":"2023-12-13T20:45:33.713441Z","iopub.execute_input":"2023-12-13T20:45:33.713872Z","iopub.status.idle":"2023-12-13T20:45:34.416801Z","shell.execute_reply.started":"2023-12-13T20:45:33.713839Z","shell.execute_reply":"2023-12-13T20:45:34.415145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"calculate rankings and kendall tau correlation","metadata":{}},{"cell_type":"code","source":"def rank_configurations(predictions, full_df):\n    ranked_configurations = []\n\n    # Create a mapping of DataFrame indices to the range of indices in predictions\n    index_mapping = {idx: i for i, idx in enumerate(full_df.index)}\n\n    # Group data by 'config_id' and process each group\n    for config_id, group in full_df.groupby('config_id'):\n        # Get the corresponding prediction indices for the current group\n        prediction_indices = [index_mapping[idx] for idx in group.index]\n\n        # Rank configurations by predicted runtime\n        ranked_indices = group.index[np.argsort(predictions[prediction_indices])]\n\n        # Store the original indices of the ranked configurations\n        ranked_configurations.append(list(ranked_indices))\n\n    return ranked_configurations","metadata":{"execution":{"iopub.status.busy":"2023-12-13T20:46:53.276511Z","iopub.execute_input":"2023-12-13T20:46:53.277424Z","iopub.status.idle":"2023-12-13T20:46:53.289451Z","shell.execute_reply.started":"2023-12-13T20:46:53.277388Z","shell.execute_reply":"2023-12-13T20:46:53.288534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.stats import kendalltau\n\ndef calculate_kendall_tau(predicted_rankings, true_rankings):\n    kendall_tau_scores = []\n\n    for predicted, true in zip(predicted_rankings, true_rankings):\n        tau, _ = kendalltau(predicted, true)\n        kendall_tau_scores.append(tau)\n\n    # Calculate the average Kendall tau correlation\n    average_kendall_tau = sum(kendall_tau_scores) / len(kendall_tau_scores)\n    return average_kendall_tau\n\n","metadata":{"execution":{"iopub.status.busy":"2023-12-13T20:46:23.610422Z","iopub.execute_input":"2023-12-13T20:46:23.610927Z","iopub.status.idle":"2023-12-13T20:46:23.619451Z","shell.execute_reply.started":"2023-12-13T20:46:23.610886Z","shell.execute_reply":"2023-12-13T20:46:23.617859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epsilon = 1e-7  # Small value to add to zero targets\n\ntrain_df['runtime'] = train_df['runtime'].apply(lambda x: x if x != 0 else x + epsilon)\ntest_df['runtime'] = test_df['runtime'].apply(lambda x: x if x != 0 else x + epsilon)","metadata":{"execution":{"iopub.status.busy":"2023-12-13T20:48:22.361715Z","iopub.execute_input":"2023-12-13T20:48:22.362090Z","iopub.status.idle":"2023-12-13T20:48:22.398858Z","shell.execute_reply.started":"2023-12-13T20:48:22.362062Z","shell.execute_reply":"2023-12-13T20:48:22.397813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use the existing predictions to rank configurations\nranked_train_lr = rank_configurations(train_predictions_lin, train_df)\nranked_test_lr = rank_configurations(predictions_lin, test_df)\n\nranked_train_lasso = rank_configurations(train_predictions_lasso, train_df)\nranked_test_lasso = rank_configurations(predictions_lasso, test_df)\n\nranked_train_ridge = rank_configurations(train_predictions_ridge, train_df)\nranked_test_ridge = rank_configurations(predictions_ridge, test_df)\n\nranked_train_lgbm = rank_configurations(train_predictions_lgbm, train_df)\nranked_test_lgbm = rank_configurations(predictions_lgbm, test_df)\n\ntrue_ranked_train = rank_configurations(y_train.to_numpy(), train_df)\ntrue_ranked_test = rank_configurations(y_test.to_numpy(), test_df)","metadata":{"execution":{"iopub.status.busy":"2023-12-13T20:48:51.162013Z","iopub.execute_input":"2023-12-13T20:48:51.162420Z","iopub.status.idle":"2023-12-13T20:48:51.736897Z","shell.execute_reply.started":"2023-12-13T20:48:51.162391Z","shell.execute_reply":"2023-12-13T20:48:51.735823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kendall_tau_lr_train = calculate_kendall_tau(ranked_train_lr, true_ranked_train)\nkendall_tau_lr_test = calculate_kendall_tau(ranked_test_lr, true_ranked_test)\n\nkendall_tau_lasso_train = calculate_kendall_tau(ranked_train_lasso, true_ranked_train)\nkendall_tau_lasso_test = calculate_kendall_tau(ranked_test_lasso, true_ranked_test)\n\nkendall_tau_ridge_train = calculate_kendall_tau(ranked_train_ridge, true_ranked_train)\nkendall_tau_ridge_test = calculate_kendall_tau(ranked_test_ridge, true_ranked_test)\n\nkendall_tau_lgbm_train = calculate_kendall_tau(ranked_train_lgbm, true_ranked_train)\nkendall_tau_lgbm_test = calculate_kendall_tau(ranked_test_lgbm, true_ranked_test)\n\n# Print Kendall tau correlations\nprint(\"Kendall tau (LR, Train):\", kendall_tau_lr_train)\nprint(\"Kendall tau (LR, Test):\", kendall_tau_lr_test)\n\nprint(\"Kendall tau (Lasso, Train):\", kendall_tau_lasso_train)\nprint(\"Kendall tau (Lasso, Test):\", kendall_tau_lasso_test)\n\nprint(\"Kendall tau (Ridge, Train):\", kendall_tau_ridge_train)\nprint(\"Kendall tau (Ridge, Test):\", kendall_tau_ridge_test)\n\nprint(\"Kendall tau (LGBM, Train):\", kendall_tau_lgbm_train)\nprint(\"Kendall tau (LGBM, Test):\", kendall_tau_lgbm_test)","metadata":{"execution":{"iopub.status.busy":"2023-12-13T20:50:33.603187Z","iopub.execute_input":"2023-12-13T20:50:33.603961Z","iopub.status.idle":"2023-12-13T20:50:33.744005Z","shell.execute_reply.started":"2023-12-13T20:50:33.603923Z","shell.execute_reply":"2023-12-13T20:50:33.742930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Code below is old, used for hyperparameter search.","metadata":{}},{"cell_type":"code","source":"#X_train = X_train.drop(columns=['node_opcode'])\n#X_test = X_test.drop('node_opcode', axis=1)\n# Training models\nlin_reg = LinearRegression().fit(X_train, y_train)\n\nalpha_grid = {'alpha': [0.001, 0.01, 0.1, 1, 10, 100]}\n\n# Setting up GridSearchCV for Lasso Regression\nlasso = Lasso()\ngrid_search_lasso = GridSearchCV(estimator=lasso, param_grid=alpha_grid, cv=3, scoring='neg_mean_squared_error',verbose=4)\ngrid_search_lasso.fit(X_train, y_train)\nlasso_reg = grid_search_lasso.best_estimator_\nprint(\"Lasso Alpha\")\nprint(grid_search_lasso.best_params_['alpha'])\n\nridge = Ridge()\ngrid_search_ridge = GridSearchCV(estimator=ridge, param_grid=alpha_grid, cv=3, scoring='neg_mean_squared_error',verbose=4)\ngrid_search_ridge.fit(X_train, y_train)\nridge_reg = grid_search_ridge.best_estimator_\nprint(\"Ridge Alpha\")\nprint(grid_search_ridge.best_params_['alpha'])\n\n# Making predictions\ntrain_predictions_lin = lin_reg.predict(X_train)\ntrain_predictions_lasso = lasso_reg.predict(X_train)\ntrain_predictions_ridge = ridge_reg.predict(X_train)\n\npredictions_lin = lin_reg.predict(X_test)\npredictions_lasso = lasso_reg.predict(X_test)\npredictions_ridge = ridge_reg.predict(X_test)\n\n# Calculating R² scores\nr2_lin = r2_score(y_test, predictions_lin)\nr2_lasso = r2_score(y_test, predictions_lasso)\nr2_ridge = r2_score(y_test, predictions_ridge)\n\nr2_train_lin = r2_score(y_train, train_predictions_lin)\nr2_train_lasso = r2_score(y_train, train_predictions_lasso)\nr2_train_ridge = r2_score(y_train, train_predictions_ridge)\n\nprint(\"R² Scores Train:\")\nprint(f\"Linear Regression: {r2_train_lin}\")\nprint(f\"Lasso Regression: {r2_train_lasso}\")\nprint(f\"Ridge Regression: {r2_train_ridge}\")\n\nprint(\"R² Scores Test:\")\nprint(f\"Linear Regression: {r2_lin}\")\nprint(f\"Lasso Regression: {r2_lasso}\")\nprint(f\"Ridge Regression: {r2_ridge}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import RandomizedSearchCV\nfrom scipy.stats import randint as sp_randint\nfrom scipy.stats import uniform as sp_uniform","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:06:57.188198Z","iopub.execute_input":"2023-12-11T22:06:57.189397Z","iopub.status.idle":"2023-12-11T22:06:57.194704Z","shell.execute_reply.started":"2023-12-11T22:06:57.189352Z","shell.execute_reply":"2023-12-11T22:06:57.193635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"param_dist = {\n    'num_leaves': sp_randint(3, 50), \n    'min_child_samples': sp_randint(5, 500), \n    'min_child_weight': sp_uniform(0.01, 0.1),\n    'subsample': sp_uniform(0.8, 0.2),\n    'colsample_bytree': sp_uniform(0.8, 0.2),\n    'reg_alpha': [0, 1e-1, 1, 2, 5, 7, 10],\n    'reg_lambda': [0, 1e-1, 1, 5, 10, 20, 50],\n    'learning_rate': [0.001, 0.005, 0.01, 0.05, 0.1, 0.2],\n    'n_estimators': [100, 250, 500, 1000, 1500]\n}","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:06:57.197708Z","iopub.execute_input":"2023-12-11T22:06:57.199110Z","iopub.status.idle":"2023-12-11T22:06:57.213910Z","shell.execute_reply.started":"2023-12-11T22:06:57.199063Z","shell.execute_reply":"2023-12-11T22:06:57.212428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgbm = LGBMRegressor()","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:06:57.216726Z","iopub.execute_input":"2023-12-11T22:06:57.218056Z","iopub.status.idle":"2023-12-11T22:06:57.227671Z","shell.execute_reply.started":"2023-12-11T22:06:57.218008Z","shell.execute_reply":"2023-12-11T22:06:57.226811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_search = RandomizedSearchCV(lgbm, param_distributions=param_dist, n_iter=25, cv=4, scoring='neg_mean_squared_error', verbose=4)\nrandom_search.fit(X_train, y_train)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-12-11T22:06:57.229800Z","iopub.execute_input":"2023-12-11T22:06:57.232118Z","iopub.status.idle":"2023-12-11T22:18:31.554521Z","shell.execute_reply.started":"2023-12-11T22:06:57.232080Z","shell.execute_reply":"2023-12-11T22:18:31.553384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_lgbm = random_search.best_estimator_","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:18:31.556274Z","iopub.execute_input":"2023-12-11T22:18:31.556716Z","iopub.status.idle":"2023-12-11T22:18:31.561858Z","shell.execute_reply.started":"2023-12-11T22:18:31.556672Z","shell.execute_reply":"2023-12-11T22:18:31.560774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_search.best_params_","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:18:31.563562Z","iopub.execute_input":"2023-12-11T22:18:31.563998Z","iopub.status.idle":"2023-12-11T22:18:31.575463Z","shell.execute_reply.started":"2023-12-11T22:18:31.563959Z","shell.execute_reply":"2023-12-11T22:18:31.574302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_preds_lgb = best_lgbm.predict(X_train)\n\npreds_lgb = best_lgbm.predict(X_test)\n\n# Calculating R² scores\nr2_lgb = r2_score(y_test, preds_lgb)\n\nr2_train_lgb=r2_score(y_train, train_preds_lgb)\n\nprint(\"R² Scores Train:\")\nprint(f\"LGBM: {r2_train_lgb}\")\n\nprint(\"R² Scores Test:\")\nprint(f\"LGBM: {r2_lgb}\")","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:18:31.576632Z","iopub.execute_input":"2023-12-11T22:18:31.576978Z","iopub.status.idle":"2023-12-11T22:18:32.165224Z","shell.execute_reply.started":"2023-12-11T22:18:31.576947Z","shell.execute_reply":"2023-12-11T22:18:32.164112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rank_configurations(predictions, full_df):\n    ranked_configurations = []\n\n    # Create a mapping of DataFrame indices to the range of indices in predictions\n    index_mapping = {idx: i for i, idx in enumerate(full_df.index)}\n\n    # Group data by 'config_id' and process each group\n    for config_id, group in full_df.groupby('config_id'):\n        # Get the corresponding prediction indices for the current group\n        prediction_indices = [index_mapping[idx] for idx in group.index]\n\n        # Rank configurations by predicted runtime\n        ranked_indices = group.index[np.argsort(predictions[prediction_indices])]\n\n        # Store the original indices of the ranked configurations\n        ranked_configurations.append(list(ranked_indices))\n\n    return ranked_configurations\ndef calculate_top_k_slowdown(predicted_rankings, full_df, runtime_column='runtime', k=5):\n    total_slowdown = 0\n\n    for predicted in tqdm(predicted_rankings):\n        # Extract the top-k predicted configurations\n        top_k_predicted = predicted[:k]\n\n        # Best runtime among top-k predicted configurations\n        best_runtime_top_k = full_df.loc[top_k_predicted, runtime_column].max()\n\n        # Best runtime among all configurations in the model group\n        config_id = full_df.loc[top_k_predicted[0], 'config_id']\n        best_runtime_all = full_df[full_df['config_id'] == config_id][runtime_column].max()\n\n        # Calculate the speedup instead of slowdown\n        speedup = best_runtime_all / best_runtime_top_k\n\n        # Speedup should always be >= 1; if it's not, cap it at 1 to avoid negative slowdown\n        speedup = max(speedup, 1)\n\n        # Calculate the slowdown as the inverse of speedup, minus 1 to get the additional time taken\n        slowdown = (1 / speedup) - 1\n        total_slowdown += slowdown\n\n    # Average slowdown across all models\n    average_slowdown = total_slowdown / len(predicted_rankings)\n    return average_slowdown","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:18:32.169041Z","iopub.execute_input":"2023-12-11T22:18:32.169403Z","iopub.status.idle":"2023-12-11T22:18:32.180986Z","shell.execute_reply.started":"2023-12-11T22:18:32.169368Z","shell.execute_reply":"2023-12-11T22:18:32.179715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epsilon = 1e-7  # Small value to add to zero runtimes\n\n# Adjusting the 'runtime' column in train_df and test_df\ntrain_df['runtime'] = train_df['runtime'].apply(lambda x: x if x != 0 else x + epsilon)\ntest_df['runtime'] = test_df['runtime'].apply(lambda x: x if x != 0 else x + epsilon)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:18:32.182280Z","iopub.execute_input":"2023-12-11T22:18:32.182612Z","iopub.status.idle":"2023-12-11T22:18:32.228157Z","shell.execute_reply.started":"2023-12-11T22:18:32.182582Z","shell.execute_reply":"2023-12-11T22:18:32.226875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.stats import kendalltau\n\ndef calculate_kendall_tau(predicted_rankings, true_rankings):\n    kendall_tau_scores = []\n\n    for predicted, true in zip(predicted_rankings, true_rankings):\n        tau, _ = kendalltau(predicted, true)\n        kendall_tau_scores.append(tau)\n\n    # Calculate the average Kendall tau correlation\n    average_kendall_tau = sum(kendall_tau_scores) / len(kendall_tau_scores)\n    return average_kendall_tau\n\n","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:18:32.229455Z","iopub.execute_input":"2023-12-11T22:18:32.229803Z","iopub.status.idle":"2023-12-11T22:18:32.236852Z","shell.execute_reply.started":"2023-12-11T22:18:32.229772Z","shell.execute_reply":"2023-12-11T22:18:32.235689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use the existing predictions to rank configurations\nranked_train_lr = rank_configurations(train_predictions_lin, train_df)\nranked_test_lr = rank_configurations(predictions_lin, test_df)\n\nranked_train_lasso = rank_configurations(train_predictions_lasso, train_df)\nranked_test_lasso = rank_configurations(predictions_lasso, test_df)\n\nranked_train_ridge = rank_configurations(train_predictions_ridge, train_df)\nranked_test_ridge = rank_configurations(predictions_ridge, test_df)\n\nranked_train_lgbm = rank_configurations(train_preds_lgb, train_df)\nranked_test_lgbm = rank_configurations(preds_lgb, test_df)\n\ntrue_ranked_train = rank_configurations(y_train.to_numpy(), train_df)\ntrue_ranked_test = rank_configurations(y_test.to_numpy(), test_df)\n\n# Calculate and print the average top-k slowdown for the train predictions\naverage_slowdown_lr_train = calculate_top_k_slowdown(ranked_train_lr, train_df)\naverage_slowdown_lasso_train = calculate_top_k_slowdown(ranked_train_lasso, train_df)\naverage_slowdown_ridge_train = calculate_top_k_slowdown(ranked_train_ridge, train_df)\naverage_slowdown_lgbm_train = calculate_top_k_slowdown(ranked_train_lgbm, train_df)\n\nprint(\"Average Top-k Slowdown (LR, Train):\", average_slowdown_lr_train)\nprint(\"Average Top-k Slowdown (Lasso, Train):\", average_slowdown_lasso_train)\nprint(\"Average Top-k Slowdown (Ridge, Train):\", average_slowdown_ridge_train)\nprint(\"Average Top-k Slowdown (LGBM, Train):\", average_slowdown_lgbm_train)\n\n# Calculate and print the average top-k slowdown for the test predictions\naverage_slowdown_lr_test = calculate_top_k_slowdown(ranked_test_lr, test_df)\naverage_slowdown_lasso_test = calculate_top_k_slowdown(ranked_test_lasso, test_df)\naverage_slowdown_ridge_test = calculate_top_k_slowdown(ranked_test_ridge, test_df)\naverage_slowdown_lgbm_test = calculate_top_k_slowdown(ranked_test_lgbm, test_df)\n\nprint(\"Average Top-k Slowdown (LR, Test):\", average_slowdown_lr_test)\nprint(\"Average Top-k Slowdown (Lasso, Test):\", average_slowdown_lasso_test)\nprint(\"Average Top-k Slowdown (Ridge, Test):\", average_slowdown_ridge_test)\nprint(\"Average Top-k Slowdown (LGBM, Test):\", average_slowdown_lgbm_test)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:18:32.238521Z","iopub.execute_input":"2023-12-11T22:18:32.239052Z","iopub.status.idle":"2023-12-11T22:18:33.122739Z","shell.execute_reply.started":"2023-12-11T22:18:32.239009Z","shell.execute_reply":"2023-12-11T22:18:33.121555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Example usage:\nkendall_tau_lr_train = calculate_kendall_tau(ranked_train_lr, true_ranked_train)\nkendall_tau_lr_test = calculate_kendall_tau(ranked_test_lr, true_ranked_test)\n\nkendall_tau_lasso_train = calculate_kendall_tau(ranked_train_lasso, true_ranked_train)\nkendall_tau_lasso_test = calculate_kendall_tau(ranked_test_lasso, true_ranked_test)\n\nkendall_tau_ridge_train = calculate_kendall_tau(ranked_train_ridge, true_ranked_train)\nkendall_tau_ridge_test = calculate_kendall_tau(ranked_test_ridge, true_ranked_test)\n\nkendall_tau_lgbm_train = calculate_kendall_tau(ranked_train_lgbm, true_ranked_train)\nkendall_tau_lgbm_test = calculate_kendall_tau(ranked_test_lgbm, true_ranked_test)\n\n# Print Kendall tau correlations\nprint(\"Kendall tau (LR, Train):\", kendall_tau_lr_train)\nprint(\"Kendall tau (LR, Test):\", kendall_tau_lr_test)\n\nprint(\"Kendall tau (Lasso, Train):\", kendall_tau_lasso_train)\nprint(\"Kendall tau (Lasso, Test):\", kendall_tau_lasso_test)\n\nprint(\"Kendall tau (Ridge, Train):\", kendall_tau_ridge_train)\nprint(\"Kendall tau (Ridge, Test):\", kendall_tau_ridge_test)\n\nprint(\"Kendall tau (LGBM, Train):\", kendall_tau_lgbm_train)\nprint(\"Kendall tau (LGBM, Test):\", kendall_tau_lgbm_test)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T22:18:33.124466Z","iopub.execute_input":"2023-12-11T22:18:33.124799Z","iopub.status.idle":"2023-12-11T22:18:33.268310Z","shell.execute_reply.started":"2023-12-11T22:18:33.124769Z","shell.execute_reply":"2023-12-11T22:18:33.267167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}