{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":10384,"databundleVersionId":120379,"sourceType":"competition"},{"sourceId":11887405,"sourceType":"datasetVersion","datasetId":7471552},{"sourceId":11911329,"sourceType":"datasetVersion","datasetId":7488281},{"sourceId":11953587,"sourceType":"datasetVersion","datasetId":7417814}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Preprocessing Training Data**\n","metadata":{}},{"cell_type":"markdown","source":"This project uses methodologies from **Paying Attention to Astronomical Transients: Introducing the Time-series Transformer for Photometric Classification** (Allam & McEwen, 2023), which proposes a transformer-based architecture for classifying light curves. ","metadata":{}},{"cell_type":"code","source":"# !pip install tensorflow-model-optimization>=0.7.2 \n# !pip install astropy>=1.1.2 \n# !pip install colorama>=0.4.4 \n# !pip install imbalanced-learn==0.7.0\n# !pip install joblib>=1.2.0\n# !pip install optuna>=2.10.0\n# !pip install --no-build-isolation git+https://github.com/astrolabsoftware/fink-client\n# !pip install --no-build-isolation git+https://github.com/astrolabsoftware/fink-utils\n# !pip install george","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-20T19:37:37.020874Z","iopub.execute_input":"2025-05-20T19:37:37.021562Z","iopub.status.idle":"2025-05-20T19:38:11.378005Z","shell.execute_reply.started":"2025-05-20T19:37:37.021533Z","shell.execute_reply":"2025-05-20T19:38:11.377154Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import os\n# import numpy as np\n# import pandas as pd\n# from sklearn.preprocessing import RobustScaler, OneHotEncoder, StandardScaler\n# from sklearn.model_selection import train_test_split\n# from tensorflow.keras.utils import to_categorical\n# from scipy import stats, optimize as op\n# import george\n# from george import kernels\n# import warnings\n# from functools import partial\n# from typing import Dict, List, Union\n# from astropy.table import Table, vstack\n\n# # Suppress warnings\n# warnings.filterwarnings(\"ignore\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-20T19:38:11.379695Z","iopub.execute_input":"2025-05-20T19:38:11.379949Z","iopub.status.idle":"2025-05-20T19:38:11.674986Z","shell.execute_reply.started":"2025-05-20T19:38:11.379927Z","shell.execute_reply":"2025-05-20T19:38:11.674423Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Constants\n# NUM_CLASSES = 15  # Including 'others' class\n# TIME_STEPS = 30   # Sequence length for time series\n# FILTERS = ['lsstu', 'lsstg', 'lsstr', 'lssti', 'lsstz', 'lssty']\n# METADATA_FEATURES = ['hostgal_photoz', 'mwebv', 'ddf']\n# ALL_FEATURES = FILTERS + METADATA_FEATURES\n\n# # PLAsTiCC class mapping to sequential indices\n# PLASTICC_CLASS_MAPPING = {\n#     90: 0,   # SNIa\n#     67: 1,   # SNIa-91bg\n#     52: 2,   # SNIax\n#     42: 3,   # SNII\n#     62: 4,   # SNIbc\n#     95: 5,   # SLSN-I\n#     15: 6,   # TDE\n#     64: 7,   # KN\n#     88: 8,   # AGN\n#     92: 9,   # RRL\n#     65: 10,  # M-dwarf\n#     16: 11,  # EB\n#     53: 12,  # Mira\n#     6: 13,   # µ-Lens-Single\n#     99: 14   # Others (not in training set)\n# }\n\n# # Class names for reference\n# CLASS_NAMES = {\n#     0: 'SNIa',\n#     1: 'SNIa-91bg',\n#     2: 'SNIax',\n#     3: 'SNII',\n#     4: 'SNIbc',\n#     5: 'SLSN-I',\n#     6: 'TDE',\n#     7: 'KN',\n#     8: 'AGN',\n#     9: 'RRL',\n#     10: 'M-dwarf',\n#     11: 'EB',\n#     12: 'Mira',\n#     13: 'µ-Lens-Single',\n#     14: 'Others'\n# }\n\n# # LSST passband wavelengths\n# LSST_PB_WAVELENGTHS = {\n#     'lsstu': 3671.0,\n#     'lsstg': 4827.0,\n#     'lsstr': 6223.0,\n#     'lssti': 7546.0,\n#     'lsstz': 8691.0,\n#     'lssty': 9710.0\n# }\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-20T19:38:11.675692Z","iopub.execute_input":"2025-05-20T19:38:11.675892Z","iopub.status.idle":"2025-05-20T19:38:11.682202Z","shell.execute_reply.started":"2025-05-20T19:38:11.675878Z","shell.execute_reply":"2025-05-20T19:38:11.681452Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def load_and_merge_data(data_path, metadata_path):\n#     \"\"\"Load and merge light curve data with metadata\"\"\"\n#     data = pd.read_csv(data_path)\n#     metadata = pd.read_csv(metadata_path)\n    \n#     # Merge and clean data\n#     merged = pd.merge(data, metadata, on='object_id', how='left')\n    \n#     # Handle missing values\n#     merged = merged.dropna(subset=['flux', 'flux_error'] + METADATA_FEATURES)\n    \n#     return merged\n\n# def filter_dataframe_only_supernova(object_list_filename: str, dataframe: pd.DataFrame) -> pd.DataFrame:\n#     \"\"\"Filter dataframe that contains many classes to only Supernovae types.\"\"\"\n#     plasticc_object_list = np.genfromtxt(object_list_filename, dtype=\"U\")\n#     filtered_dataframe = dataframe[dataframe[\"object_id\"].isin(plasticc_object_list)]\n#     return filtered_dataframe\n\n# def transient_trim(object_list: List[str], df: pd.DataFrame) -> tuple:\n#     \"\"\"Trim off light-curve plateau to leave only the transient part +/- 50 time-steps\"\"\"\n#     adf = pd.DataFrame(data=[], columns=df.columns)\n#     good_object_list = []\n#     for obj in object_list:\n#         obs = df[df[\"object_id\"] == obj]\n#         obs_time = obs[\"mjd\"]\n#         obs_detected_time = obs_time[obs[\"detected\"] == 1]\n#         if len(obs_detected_time) == 0:\n#             print(f\"Zero detected points for object:{object_list.index(obj)}\")\n#             continue\n#         is_obs_transient = (obs_time > obs_detected_time.iat[0] - 50) & (\n#             obs_time < obs_detected_time.iat[-1] + 50\n#         )\n#         obs_transient = obs[is_obs_transient]\n#         if len(obs_transient[\"mjd\"]) == 0:\n#             is_obs_transient = (obs_time > obs_detected_time.iat[0] - 1000) & (\n#                 obs_time < obs_detected_time.iat[-1] + 1000\n#             )\n#             obs_transient = obs[is_obs_transient]\n#         obs_transient[\"mjd\"] -= min(obs_transient[\"mjd\"])  # so all transients start at time 0\n#         good_object_list.append(object_list.index(obj))\n#         adf = np.vstack((adf, obs_transient))\n\n#     obs_transient = pd.DataFrame(data=adf, columns=obs_transient.columns)\n#     filter_indices = good_object_list\n#     new_filtered_object_list = np.take(np.array(object_list), filter_indices, axis=0)\n#     return obs_transient, list(new_filtered_object_list)\n\n# def fit_2d_gp(obj_data: pd.DataFrame, return_kernel: bool = False, pb_wavelengths: Dict = LSST_PB_WAVELENGTHS, **kwargs):\n#     \"\"\"Fit a 2D Gaussian process.\"\"\"\n#     guess_length_scale = 20.0\n\n#     obj_times = obj_data.mjd.astype(float)\n#     obj_flux = obj_data.flux.astype(float)\n#     obj_flux_error = obj_data.flux_error.astype(float)\n#     obj_wavelengths = obj_data[\"filter\"].map(pb_wavelengths)\n\n#     def neg_log_like(p):\n#         gp.set_parameter_vector(p)\n#         loglike = gp.log_likelihood(obj_flux, quiet=True)\n#         return -loglike if np.isfinite(loglike) else 1e25\n\n#     def grad_neg_log_like(p):\n#         gp.set_parameter_vector(p)\n#         return -gp.grad_log_likelihood(obj_flux, quiet=True)\n\n#     signal_to_noises = np.abs(obj_flux) / np.sqrt(\n#         obj_flux_error**2 + (1e-2 * np.max(obj_flux)) ** 2\n#     )\n#     scale = np.abs(obj_flux[signal_to_noises.idxmax()])\n\n#     kernel = (0.5 * scale) ** 2 * george.kernels.Matern32Kernel(\n#         [guess_length_scale**2, 6000**2], ndim=2\n#     )\n#     kernel.freeze_parameter(\"k2:metric:log_M_1_1\")\n\n#     gp = george.GP(kernel)\n#     default_gp_param = gp.get_parameter_vector()\n#     x_data = np.vstack([obj_times, obj_wavelengths]).T\n#     gp.compute(x_data, obj_flux_error)\n\n#     bounds = [(0, np.log(1000**2))]\n#     bounds = [(default_gp_param[0] - 10, default_gp_param[0] + 10)] + bounds\n#     results = op.minimize(\n#         neg_log_like,\n#         gp.get_parameter_vector(),\n#         jac=grad_neg_log_like,\n#         method=\"L-BFGS-B\",\n#         bounds=bounds,\n#         tol=1e-6,\n#     )\n\n#     if results.success:\n#         gp.set_parameter_vector(results.x)\n#     else:\n#         obj = obj_data[\"object_id\"][0]\n#         print(f\"GP fit failed for {obj}! Using guessed GP parameters.\")\n#         gp.set_parameter_vector(default_gp_param)\n\n#     gp_predict = partial(gp.predict, obj_flux)\n\n#     if return_kernel:\n#         return kernel, gp_predict\n#     return gp_predict\n\n# def predict_2d_gp(gp_predict, gp_times, gp_wavelengths):\n#     \"\"\"Outputs the predictions of a Gaussian Process.\"\"\"\n#     unique_wavelengths = np.unique(gp_wavelengths)\n#     number_gp = len(gp_times)\n#     obj_gps = []\n#     for wavelength in unique_wavelengths:\n#         gp_wavelengths = np.ones(number_gp) * wavelength\n#         pred_x_data = np.vstack([gp_times, gp_wavelengths]).T\n#         pb_pred, pb_pred_var = gp_predict(pred_x_data, return_var=True)\n#         obj_gp_pb_array = np.column_stack((gp_times, pb_pred, np.sqrt(pb_pred_var)))\n#         obj_gp_pb = Table(\n#             [\n#                 obj_gp_pb_array[:, 0],\n#                 obj_gp_pb_array[:, 1],\n#                 obj_gp_pb_array[:, 2],\n#                 [wavelength] * number_gp,\n#             ],\n#             names=[\"mjd\", \"flux\", \"flux_error\", \"filter\"],\n#         )\n#         if len(obj_gps) == 0:\n#             obj_gps = obj_gp_pb\n#         else:\n#             obj_gps = vstack((obj_gps, obj_gp_pb))\n\n#     return obj_gps.to_pandas()\n\n# def generate_gp_single_event(df: pd.DataFrame, timesteps: int = 100, pb_wavelengths: Dict = LSST_PB_WAVELENGTHS) -> pd.DataFrame:\n#     \"\"\"Generate GP interpolation for a single event.\"\"\"\n#     filters = list(np.unique(df[\"filter\"]))\n#     gp_wavelengths = np.vectorize(pb_wavelengths.get)(filters)\n#     inverse_pb_wavelengths = {v: k for k, v in pb_wavelengths.items()}\n\n#     gp_predict = fit_2d_gp(df, pb_wavelengths=pb_wavelengths)\n#     gp_times = np.linspace(min(df[\"mjd\"]), max(df[\"mjd\"]), timesteps)\n#     obj_gps = predict_2d_gp(gp_predict, gp_times, gp_wavelengths)\n#     obj_gps[\"filter\"] = obj_gps[\"filter\"].map(inverse_pb_wavelengths)\n\n#     return obj_gps\n\n# def generate_gp_all_objects(object_list: List[str], obs_transient: pd.DataFrame, timesteps: int = 100, pb_wavelengths: Dict = LSST_PB_WAVELENGTHS) -> pd.DataFrame:\n#     \"\"\"Generate Gaussian Process interpolation for all objects.\"\"\"\n#     filters = list(np.unique(obs_transient[\"filter\"]))\n#     columns = [\"mjd\"] + filters + [\"object_id\"]\n#     adf = pd.DataFrame(data=[], columns=columns)\n\n#     for object_id in object_list:\n#         print(f\"OBJECT ID:{object_id} at INDEX:{object_list.index(object_id)}\")\n#         df = obs_transient[obs_transient[\"object_id\"] == object_id]\n#         obj_gps = generate_gp_single_event(df, timesteps, pb_wavelengths)\n#         obj_gps = pd.pivot_table(obj_gps, index=\"mjd\", columns=\"filter\", values=\"flux\")\n#         obj_gps = obj_gps.reset_index()\n#         obj_gps[\"object_id\"] = object_id\n#         adf = np.vstack((adf, obj_gps))\n    \n#     return pd.DataFrame(data=adf, columns=obj_gps.columns)\n\n# def remap_filters(df: pd.DataFrame, filter_map: Dict) -> pd.DataFrame:\n#     \"\"\"Remap integer filters to the corresponding filters.\"\"\"\n#     df.rename({\"passband\": \"filter\"}, axis=\"columns\", inplace=True)\n#     df[\"filter\"].replace(to_replace=filter_map, inplace=True)\n#     return df\n\n# def robust_scale(dataframe: pd.DataFrame, scale_columns: List[Union[str, int]]) -> pd.DataFrame:\n#     \"\"\"Standardize a dataset along axis=0 (rows)\"\"\"\n#     scaler = RobustScaler()\n#     scaler = scaler.fit(dataframe[scale_columns])\n#     dataframe.loc[:, scale_columns] = scaler.transform(dataframe[scale_columns].to_numpy())\n#     return dataframe\n\n# def z_score_normalize(dataframe: pd.DataFrame, scale_columns: List[Union[str, int]]) -> pd.DataFrame:\n#     \"\"\"Apply z-score normalization to specified columns.\n    \n#     Parameters\n#     ----------\n#     dataframe: pd.DataFrame\n#         Dataframe containing the data to normalize\n#     scale_columns: List[Union[str, int]]\n#         Columns to apply z-score normalization to\n        \n#     Returns\n#     -------\n#     pd.DataFrame\n#         Dataframe with normalized columns\n#     \"\"\"\n#     scaler = StandardScaler()\n#     scaler = scaler.fit(dataframe[scale_columns])\n#     dataframe.loc[:, scale_columns] = scaler.transform(dataframe[scale_columns].to_numpy())\n#     return dataframe\n\n# def map_classes(df: pd.DataFrame) -> pd.DataFrame:\n#     \"\"\"Map original target codes to 0-14 indices.\n    \n#     Parameters\n#     ----------\n#     df: pd.DataFrame\n#         Dataframe containing the 'target' column with original class codes\n        \n#     Returns\n#     -------\n#     pd.DataFrame\n#         Dataframe with added 'mapped_target' column and invalid classes removed\n#     \"\"\"\n#     df[\"mapped_target\"] = df[\"target\"].map(PLASTICC_CLASS_MAPPING)\n#     df = df[df[\"mapped_target\"].notna()]  # Drop invalid/unmapped classes\n#     return df\n\n# def create_dataset(features, labels, time_steps, step):\n#     \"\"\"Create time series sequences with specified time steps and step size.\n    \n#     Parameters\n#     ----------\n#     features : pd.DataFrame\n#         DataFrame containing the features\n#     labels : pd.Series\n#         Series containing the labels\n#     time_steps : int\n#         Number of time steps in each sequence\n#     step : int\n#         Step size between sequences\n        \n#     Returns\n#     -------\n#     tuple\n#         (sequences, labels) where sequences is a numpy array of shape (n_sequences, time_steps, n_features)\n#         and labels is a numpy array of shape (n_sequences,)\n#     \"\"\"\n#     sequences, sequence_labels = [], []\n    \n#     for i in range(0, len(features) - time_steps + 1, step):\n#         sequences.append(features[i:i + time_steps].values)\n#         sequence_labels.append(labels.iloc[i])\n    \n#     return np.array(sequences), np.array(sequence_labels)\n\n# def extract_redshift(df):\n#     \"\"\"Extract photometric redshift for each object (static value).\"\"\"\n#     # Group by object_id and take the first occurrence (same for all rows)\n#     redshift_df = df.groupby(\"object_id\")[\"hostgal_photoz\"].first().reset_index()\n#     return redshift_df[\"hostgal_photoz\"].values.reshape(-1, 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-20T19:38:11.683731Z","iopub.execute_input":"2025-05-20T19:38:11.683942Z","iopub.status.idle":"2025-05-20T19:38:11.712685Z","shell.execute_reply.started":"2025-05-20T19:38:11.683927Z","shell.execute_reply":"2025-05-20T19:38:11.711922Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# # Path configuration\n# DATA_PATH = '/kaggle/input/PLAsTiCC-2018/training_set.csv'\n# METADATA_PATH ='/kaggle/input/PLAsTiCC-2018/training_set_metadata.csv'\n# SAVE_DIR = \"/kaggle/input/processed\"\n# os.makedirs(SAVE_DIR, exist_ok=True)\n\n# # Load and preprocess data\n# print(\"Loading data...\")\n# data = pd.read_csv(DATA_PATH, sep=',')\n\n# # Remap filters\n# print(\"Remapping filters...\")\n# data = remap_filters(data, {0: 'lsstu', 1: 'lsstg', 2: 'lsstr', \n#                            3: 'lssti', 4: 'lsstz', 5: 'lssty'})\n# data.rename({'flux_err': 'flux_error'}, axis='columns', inplace=True)\n\n# # Get unique filters and objects\n# filters = list(np.unique(data['filter']))\n# object_list = list(np.unique(data['object_id']))\n\n# # Trim to transient period\n# print(\"Trimming transients...\")\n# obs_transient, object_list = transient_trim(object_list, data)\n\n# # Generate GP predictions\n# print(\"Generating GP predictions...\")\n# generated_gp_dataset = generate_gp_all_objects(object_list, obs_transient, filters)\n\n# # Load and merge metadata\n# print(\"Merging metadata...\")\n# metadata_pd = pd.read_csv(METADATA_PATH, sep=',', index_col='object_id')\n# metadata_pd = metadata_pd.reset_index()\n# metadata_pd['object_id'] = metadata_pd['object_id'].astype(np.int32)\n# metadata_pd['target'] = metadata_pd['target'].astype(np.int32)\n\n# # Merge and clean data\n# df_combi = generated_gp_dataset.merge(metadata_pd, on='object_id', how='left')\n# columns_to_drop = [\n# \"ra\",         # Right Ascension (sky coordinate; not predictive)\n# \"decl\",       # Declination (sky coordinate; not predictive)\n# \"gal_l\",      # Galactic longitude (redundant with ra/decl)\n# \"gal_b\",      # Galactic latitude (redundant with ra/decl)\n# \"hostgal_specz\",  # Spectroscopic redshift (not available in test set)\n# \"distmod\",    # Distance modulus (derived from hostgal_photoz; redundant)\n# \"hostgal_photoz_err\",  # Photoz error (optional: keep if modeling uncertainty)\n# ]\n# df = df_combi.drop(columns=columns_to_drop)\n\n# # Map classes\n# print(\"Mapping classes...\")\n# df = map_classes(df)\n\n# # Split data\n# print(\"Splitting data...\")\n# obj_ids = df['object_id'].unique()\n# train_ids, test_ids = train_test_split(obj_ids, test_size=0.15, random_state=42)\n# train_ids, val_ids = train_test_split(train_ids, test_size=0.176, random_state=42)  # 85-15 split\n\n# train_data = df[df['object_id'].isin(train_ids)]\n# val_data = df[df['object_id'].isin(val_ids)]\n# test_data = df[df['object_id'].isin(test_ids)]\n\n# # Apply z-score normalization to features\n# print(\"Scaling features...\")\n# z_scaler = StandardScaler()\n# z_scaler.fit(train_data[filters])\n# train_data[filters] = z_scaler.transform(train_data[filters].copy())\n# val_data[filters] = z_scaler.transform(val_data[filters].copy())\n# test_data[filters] = z_scaler.transform(test_data[filters].copy())\n\n# # Create sequences\n# print(\"Creating sequences...\")\n# TIME_STEPS = 20\n# STEP = 20\n\n# X_train, y_train = create_dataset(\n#     train_data[filters],\n#     train_data['mapped_target'],\n#     TIME_STEPS,\n#     STEP\n# )\n\n# X_val, y_val = create_dataset(\n#     val_data[filters],\n#     val_data['mapped_target'],\n#     TIME_STEPS,\n#     STEP\n# )\n\n# X_test, y_test = create_dataset(\n#     test_data[filters],\n#     test_data['mapped_target'],\n#     TIME_STEPS,\n#     STEP\n# )\n\n# # One-hot encode labels\n# y_train = to_categorical(y_train, num_classes=NUM_CLASSES)\n# y_val = to_categorical(y_val, num_classes=NUM_CLASSES)\n# y_test = to_categorical(y_test, num_classes=NUM_CLASSES)\n\n# # Save data\n# print(\"Saving processed data...\")\n# # Save features\n# np.save(os.path.join(SAVE_DIR, \"X_train.npy\"), X_train)\n# np.save(os.path.join(SAVE_DIR, \"X_val.npy\"), X_val)\n# np.save(os.path.join(SAVE_DIR, \"X_test.npy\"), X_test)\n\n# # Save labels\n# np.save(os.path.join(SAVE_DIR, \"y_train.npy\"), y_train)\n# np.save(os.path.join(SAVE_DIR, \"y_val.npy\"), y_val)\n# np.save(os.path.join(SAVE_DIR, \"y_test.npy\"), y_test)\n\n\n# # Save class mapping and names\n# np.save(os.path.join(SAVE_DIR, \"class_mapping.npy\"), PLASTICC_CLASS_MAPPING)\n# np.save(os.path.join(SAVE_DIR, \"class_names.npy\"), CLASS_NAMES)\n\n# print(f\"Processing complete! Final shapes:\")\n# print(f\"Train: {X_train.shape}, {y_train.shape}\")\n# print(f\"Validation: {X_val.shape}, {y_val.shape}\")\n# print(f\"Test: {X_test.shape}, {y_test.shape}\")\n\n# # Print class distribution\n# print(\"\\nClass distribution in training set:\")\n# class_dist = train_data['mapped_target'].value_counts().sort_index()\n# for idx, count in class_dist.items():\n#     print(f\"{CLASS_NAMES[idx]}: {count}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-20T19:38:11.713357Z","iopub.execute_input":"2025-05-20T19:38:11.713614Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Phase 1: Training model and evaluating using Train data\n","metadata":{}},{"cell_type":"markdown","source":"Training","metadata":{}},{"cell_type":"code","source":"# import argparse\n# import json\n# import logging\n# import os\n# import shutil\n# import subprocess\n# import sys\n# import time\n# import warnings\n# from pathlib import Path\n# import platform\n# import numpy as np\n# import psutil\n# import io\n# import imageio\n# import matplotlib.pyplot as plt\n# import tensorflow as tf\n# from PIL import Image\n# from datetime import datetime\n# from sklearn.metrics import precision_score, recall_score\n# from tensorflow.keras import optimizers\n# from tensorflow.keras.callbacks import (\n#     CSVLogger,\n#     EarlyStopping,\n#     ModelCheckpoint,\n#     ReduceLROnPlateau,\n# )\n\n# # Set up logging\n# class CustomFormatter(logging.Formatter):\n#     grey = \"\\x1b[38;20m\"\n#     yellow = \"\\x1b[33;20m\"\n#     red = \"\\x1b[31;20m\"\n#     bold_red = \"\\x1b[31;1m\"\n#     reset = \"\\x1b[0m\"\n#     white = \"\\x1b[37;20m\"\n\n#     FORMAT = \"[%(asctime)s] \"\n#     FORMAT += \"{%(filename)s:%(lineno)d} \"\n#     FORMAT += \"%(levelname)s \"\n#     FORMAT += \"- %(message)s\"\n\n#     FORMATS = {\n#         logging.DEBUG: grey + FORMAT + reset,\n#         logging.INFO: white + FORMAT + reset,\n#         logging.WARNING: yellow + FORMAT + reset,\n#         logging.ERROR: red + FORMAT + reset,\n#         logging.CRITICAL: bold_red + FORMAT + reset,\n#     }\n\n#     def format(self, record):\n#         DATEFORMAT = \"%y-%m-%d %H:%M:%S\"\n#         log_fmt = self.FORMATS.get(record.levelno)\n#         formatter = logging.Formatter(log_fmt, datefmt=DATEFORMAT)\n#         return formatter.format(record)\n\n# def powerpuffgirls_logger(name=\"kaggle_notebook\", log_dir=\"/kaggle/working/logs\"):\n#     # Ensure the log directory exists\n#     os.makedirs(log_dir, exist_ok=True)\n    \n#     # Create log file with timestamp\n#     log_filename = f\"{name}_{datetime.now().strftime('%Y%m%d_%H%M%S')}.log\"\n#     log_path = os.path.join(log_dir, log_filename)\n    \n#     # Create logger\n#     logger = logging.getLogger(name)\n#     logger.setLevel(logging.DEBUG)\n    \n#     # Avoid duplicate handlers\n#     if not logger.handlers:\n#         # File handler\n#         file_handler = logging.FileHandler(log_path)\n#         file_handler.setLevel(logging.DEBUG)\n#         file_formatter = logging.Formatter('%(asctime)s - %(levelname)s - %(message)s')\n#         file_handler.setFormatter(file_formatter)\n#         logger.addHandler(file_handler)\n\n#         # Console handler\n#         console_handler = logging.StreamHandler()\n#         console_handler.setLevel(logging.INFO)\n#         console_formatter = logging.Formatter('%(levelname)s - %(message)s')\n#         console_handler.setFormatter(console_formatter)\n#         logger.addHandler(console_handler)\n\n#     return logger\n\n# log = powerpuffgirls_logger()\n\n# log.info(\"Logger initialized successfully!\")\n# log.debug(\"This is a debug message for deeper inspection.\")\n# log.warning(\"Watch out for potential issues.\")\n\n# # Visualization utilities\n# plt.rc(\"font\", size=20)\n# plt.rc(\"figure\", figsize=(15, 3))\n\n# RANDOM_SEED = 42\n# np.random.seed(RANDOM_SEED)\n# tf.random.set_seed(RANDOM_SEED)\n\n# def check_and_clean_nan_data(X, y):\n#     \"\"\"Check for NaN values in the data and remove corresponding samples if found.\"\"\"\n#     # Check for NaN in X\n#     nan_mask_X = np.isnan(X).any(axis=(1, 2))\n#     # Check for NaN in y\n#     nan_mask_y = np.isnan(y).any(axis=1)\n    \n#     # Combine masks\n#     nan_mask = nan_mask_X | nan_mask_y\n    \n#     if np.any(nan_mask):\n#         num_nans = np.sum(nan_mask)\n#         total_samples = X.shape[0]\n#         log.warning(f\"Found {num_nans} samples with NaN values out of {total_samples} ({(num_nans/total_samples)*100:.2f}%)\")\n        \n#         # Keep only non-NaN samples\n#         X_clean = X[~nan_mask]\n#         y_clean = y[~nan_mask]\n        \n#         log.info(f\"Data shape after removing NaN samples: X={X_clean.shape}, y={y_clean.shape}\")\n#         return X_clean, y_clean\n    \n#     log.info(\"No NaN values found in the data\")\n#     return X, y\n\n# def find_optimal_batch_size(training_set_length: int) -> int:\n#     \"\"\"Determine optimal batch size to use. Ideally leave a large remainder such that the GPU is\n#     full for most of the time.\n#     \"\"\"\n#     if training_set_length < 10000:\n#         batch_size_list = [16, 32, 64]\n#     else:\n#         batch_size_list = [2048, 4096]\n#     ratios = []\n#     for batch_size in batch_size_list:\n#         remainder = training_set_length % batch_size\n#         if remainder == 0:\n#             batch_size = remainder\n#         else:\n#             ratios.append(batch_size / remainder)\n\n#     index, ratio = min(enumerate(ratios), key=lambda x: abs(x[1] - 1))\n#     return batch_size_list[index]\n\n# def lazy_load_plasticc_noZ(X, y):\n#     \"\"\"Create a TensorFlow dataset from numpy arrays without redshift information.\"\"\"\n#     def generator():\n#         for x, L in zip(X, y):\n#             yield (x, L)\n\n#     dataset = tf.data.Dataset.from_generator(\n#         generator=generator,\n#         output_signature=(\n#             tf.type_spec_from_value(X[0]),\n#             tf.type_spec_from_value(y[0]),\n#         ),\n#     )\n#     return dataset.repeat()\n\n# class WeightedLogLoss(tf.keras.losses.Loss):\n#     # initialize instance attributes\n#     def __init__(self, name=\"weighted_log_loss\"):\n#         super().__init__(name=name)\n\n#     # compute loss\n#     def call(self, y_true, y_pred):\n#         # Cast inputs to float32 first\n#         y_true = tf.cast(y_true, tf.float32)\n#         y_pred = tf.cast(y_pred, tf.float32)\n        \n#         # Calculate weights\n#         wtable = tf.reduce_sum(y_true, axis=0) / tf.cast(tf.shape(y_true)[0], tf.float32)\n        \n#         # Clip predictions to avoid numerical instability\n#         yc = tf.clip_by_value(y_pred, 1e-15, 1 - 1e-15)\n        \n#         # Calculate loss using same dtype throughout\n#         loss = -(\n#             tf.reduce_mean(\n#                 tf.math.divide_no_nan(\n#                     tf.reduce_mean(y_true * tf.math.log(yc), axis=0), wtable\n#                 )\n#             )\n#         )\n        \n#         return loss\n\n# class DistributedWeightedLogLoss(tf.keras.losses.Loss):\n#     # initialize instance attributes\n#     def __init__(\n#         self,\n#         reduction=tf.keras.losses.Reduction.AUTO,\n#         name=\"weighted_log_loss\",\n#     ):\n#         super().__init__(reduction=reduction, name=name)\n\n#     # compute loss\n#     def call(self, y_true, y_pred):\n#         # Cast inputs to float32 first\n#         y_true = tf.cast(y_true, tf.float32)\n#         y_pred = tf.cast(y_pred, tf.float32)\n        \n#         # Calculate weights using TensorFlow ops for better distributed support\n#         wtable = tf.reduce_sum(y_true, axis=0) / tf.cast(tf.shape(y_true)[0], tf.float32)\n        \n#         # Clip predictions to avoid numerical instability\n#         yc = tf.clip_by_value(y_pred, 1e-15, 1 - 1e-15)\n        \n#         # Calculate loss using same dtype throughout\n#         loss = -(\n#             tf.reduce_mean(\n#                 tf.math.divide_no_nan(\n#                     tf.reduce_mean(y_true * tf.math.log(yc), axis=0), wtable\n#                 )\n#             )\n#         )\n        \n#         return loss\n\n# class SGEBreakoutCallback(tf.keras.callbacks.Callback):\n#     \"\"\"Callback to stop training if job runs too long.\"\"\"\n#     def __init__(self, threshold=24):\n#         super(SGEBreakoutCallback, self).__init__()\n#         self.threshold = threshold\n\n#     def on_epoch_end(self, epoch, logs={}):\n#         try:\n#             # Check if we're in an SGE environment\n#             if not os.environ.get('JOB_ID'):\n#                 return\n\n#             hrs = subprocess.run(\n#                 f\"qstat -j {os.environ.get('JOB_ID')} | grep 'cpu' | awk '{{print $3}}' | awk -F ':' '{{print $1}}' | awk -F  '=' '{{print $2}}'\",\n#                 check=True,\n#                 capture_output=True,\n#                 shell=True,\n#                 text=True,\n#             ).stdout.strip()\n\n#             if hrs and int(hrs) > self.threshold:\n#                 log.info(\"Stopping training...\")\n#                 self.model.stop_training = True\n#         except Exception as e:\n#             log.warning(f\"Error in SGEBreakoutCallback: {e}. Continuing training.\")\n\n# class Training(object):\n#     def __init__(\n#         self,\n#         architecture,\n#         dataset,\n#         fink=None,\n#         avocado=None,\n#         testset=None,\n#         redshift=False,\n#     ):\n#         self.architecture = architecture\n#         self.dataset = dataset\n#         self.fink = fink\n#         self.avocado = avocado\n#         self.testset = testset\n#         self.redshift = redshift\n\n#     def __call__(self):\n#         \"\"\"Train a given architecture with, or without redshift, on either UGRIZY or GR passbands\"\"\"\n#         def build_label():\n#             UNIXTIMESTAMP = int(time.time())\n#             try:\n#                 VERSION = (\n#                     subprocess.check_output([\"git\", \"describe\", \"--always\"])\n#                     .strip()\n#                     .decode()\n#                 )\n#             except Exception:\n#                 VERSION = \"unknown\"\n#             JOB_ID = os.environ.get(\"JOB_ID\")\n#             LABEL = f\"{UNIXTIMESTAMP}-{JOB_ID}-{VERSION}\"\n#             return LABEL\n\n#         LABEL = build_label()\n#         checkpoint_path = Path(self.architecture) / \"models\" / self.dataset / \"checkpoints\" / f\"checkpoint-{LABEL}.keras\"\n#         csv_logger_file = Path(\"logs\") / self.architecture / f\"training-{LABEL}.log\"\n\n#         # Create necessary directories\n#         checkpoint_path.parent.mkdir(parents=True, exist_ok=True)\n#         csv_logger_file.parent.mkdir(parents=True, exist_ok=True)\n\n#         # Also create the final model save directory\n#         model_save_dir = Path(self.architecture) / \"models\" / self.dataset\n#         model_save_dir.mkdir(parents=True, exist_ok=True)\n#         weights_save_dir = model_save_dir / \"weights\"\n#         weights_save_dir.mkdir(parents=True, exist_ok=True)\n\n#         # Lazy load data\n#         data_dir = Path(\"/kaggle/input/processed\")\n#         X_train = np.load(data_dir / \"X_train.npy\", mmap_mode=\"r\")\n#         y_train = np.load(data_dir / \"y_train.npy\", mmap_mode=\"r\")\n\n#         X_test = np.load(data_dir / \"X_test.npy\", mmap_mode=\"r\")\n#         y_test = np.load(data_dir / \"y_test.npy\", mmap_mode=\"r\")\n\n#         # Check and clean NaN values\n#         X_train, y_train = check_and_clean_nan_data(X_train, y_train)\n#         X_test, y_test = check_and_clean_nan_data(X_test, y_test)\n\n#         num_classes = y_train.shape[1]\n\n#         if self.fink is not None:\n#             # Take only G, R bands\n#             X_train = X_train[:, :, 0:3:2]\n#             X_test = X_test[:, :, 0:3:2]\n\n#         log.info(f\"{X_train.shape, y_train.shape}\")\n\n#         num_samples, timesteps, num_features = X_train.shape\n\n#         BATCH_SIZE = find_optimal_batch_size(num_samples)\n#         log.info(f\"BATCH_SIZE:{BATCH_SIZE}\")\n\n#         input_shape = (BATCH_SIZE, timesteps, num_features)\n#         log.info(f\"input_shape:{input_shape}\")\n\n#         drop_remainder = False\n\n#         def get_compiled_model_and_data(loss, drop_remainder):\n#             hyper_results_file = \"/kaggle/input/hyperparameters/hyperparameter/results.json\"\n            \n#             train_ds = (\n#                 lazy_load_plasticc_noZ(X_train, y_train)\n#                 .shuffle(1000, seed=RANDOM_SEED)\n#                 .batch(BATCH_SIZE, drop_remainder=drop_remainder)\n#                 .prefetch(tf.data.AUTOTUNE)\n#                 .cache()\n#             )\n#             test_ds = (\n#                 lazy_load_plasticc_noZ(X_test, y_test)\n#                 .batch(BATCH_SIZE, drop_remainder=drop_remainder)\n#                 .prefetch(tf.data.AUTOTUNE)\n#                 .cache()\n#             )\n\n#             # Load hyperparameters\n#             try:\n#                 with open(hyper_results_file, \"r\") as f:\n#                     hyper_results = json.load(f)\n#                     if not isinstance(hyper_results, list):\n#                         raise ValueError(\"Hyperparameter results must be a list\")\n#                     if not hyper_results:\n#                         raise ValueError(\"Hyperparameter results list is empty\")\n#             except Exception as e:\n#                 log.warning(f\"Error loading hyperparameters: {e}. Using default values.\")\n#                 hyper_results = [{\n#                     \"name\": \"default_config\",\n#                     \"value\": 0.0,\n#                     \"hyperparameters\": {\n#                         \"units\": 128,\n#                         \"dropout\": 0.2,\n#                         \"learning_rate\": 0.001\n#                     }\n#                 }]\n\n#             # Get best hyperparameters\n#             try:\n#                 best_trial = min(hyper_results, key=lambda x: x[\"value\"])\n#                 hyperparameters = best_trial[\"hyperparameters\"]\n#             except Exception as e:\n#                 log.warning(f\"Error finding best trial: {e}. Using first trial.\")\n#                 hyperparameters = hyper_results[0][\"hyperparameters\"]\n\n#             # Build model\n#             model = tf.keras.Sequential([\n#                 tf.keras.layers.Input(shape=(timesteps, num_features)),\n#                 tf.keras.layers.LSTM(hyperparameters[\"units\"], return_sequences=True),\n#                 tf.keras.layers.Dropout(hyperparameters[\"dropout\"]),\n#                 tf.keras.layers.LSTM(hyperparameters[\"units\"]),\n#                 tf.keras.layers.Dropout(hyperparameters[\"dropout\"]),\n#                 tf.keras.layers.Dense(num_classes, activation=\"softmax\")\n#             ])\n\n#             # Compile model\n#             model.compile(\n#                 optimizer=tf.keras.optimizers.Adam(learning_rate=hyperparameters[\"learning_rate\"]),\n#                 loss=loss,\n#                 metrics=[\"accuracy\"]\n#             )\n\n#             # Initialize event dictionary\n#             event = {\n#                 \"name\": \"training_run\",\n#                 \"value\": 0.0,\n#                 \"hyperparameters\": hyperparameters\n#             }\n\n#             return model, train_ds, test_ds, event, hyper_results_file\n\n#         # Set up distributed training if multiple GPUs available\n#         if len(tf.config.list_physical_devices(\"GPU\")) > 1:\n#             strategy = tf.distribute.MirroredStrategy()\n#             log.info(\"Number of devices: {}\".format(strategy.num_replicas_in_sync))\n#             BATCH_SIZE = BATCH_SIZE * strategy.num_replicas_in_sync\n#             VALIDATION_BATCH_SIZE = BATCH_SIZE * strategy.num_replicas_in_sync\n\n#             with strategy.scope():\n#                 loss = WeightedLogLoss()\n#                 model, train_ds, test_ds, event, hyper_results_file = get_compiled_model_and_data(loss, drop_remainder)\n#         else:\n#             # For single GPU or CPU, use the same batch size for validation\n#             VALIDATION_BATCH_SIZE = BATCH_SIZE\n#             loss = WeightedLogLoss()\n#             model, train_ds, test_ds, event, hyper_results_file = get_compiled_model_and_data(loss, drop_remainder)\n\n#         # Set up callbacks\n#         callbacks = [\n#             CSVLogger(csv_logger_file),\n#             EarlyStopping(\n#                 monitor=\"val_loss\",\n#                 patience=10,\n#                 restore_best_weights=True\n#             ),\n#             ModelCheckpoint(\n#                 checkpoint_path,\n#                 monitor=\"val_loss\",\n#                 save_best_only=True\n#             ),\n#             ReduceLROnPlateau(\n#                 monitor=\"val_loss\",\n#                 factor=0.5,\n#                 patience=5,\n#                 min_lr=1e-6\n#             ),\n#             SGEBreakoutCallback()\n#         ]\n\n#         # Calculate steps per epoch\n#         steps_per_epoch = 1100\n#         validation_steps = 350\n        \n#         log.info(f\"Steps per epoch: {steps_per_epoch}\")\n#         log.info(f\"Validation steps: {validation_steps}\")\n#         log.info(f\"Batch size: {BATCH_SIZE}\")\n#         log.info(f\"Validation batch size: {VALIDATION_BATCH_SIZE}\")\n\n#         # Train model from scratch\n#         history = model.fit(\n#             train_ds,\n#             validation_data=test_ds,\n#             epochs=10,\n#             steps_per_epoch=steps_per_epoch,\n#             validation_steps=validation_steps,\n#             callbacks=callbacks,\n#             verbose=1\n#         )\n\n#         # Evaluate model\n#         log.info(f\"PERCENT OF RAM USED: {psutil.virtual_memory().percent}\")\n#         log.info(f\"RAM USED: {psutil.virtual_memory().active / (1024*1024*1024)}\")\n\n#         # # Evaluate with different batch sizes\n#         # log.info(f\"LL-BATCHED-32 Model Evaluate: {model.evaluate(test_ds, verbose=0)[0]}\")\n#         # log.info(f\"LL-BATCHED-OP Model Evaluate: {model.evaluate(test_ds, verbose=0, batch_size=VALIDATION_BATCH_SIZE)[0]}\")\n\n#         # if drop_remainder:\n#         #     ind = np.array([x for x in range((y_test.shape[0] // BATCH_SIZE) * BATCH_SIZE)])\n#         #     y_test = np.take(y_test, ind, axis=0)\n            \n#         # y_preds = model.predict(test_ds, step=1000)\n#         # log.info(f\"{y_preds.shape}, {type(y_preds)}\")\n\n#         # WLOSS = loss(y_test, y_preds).numpy()\n#         # log.info(f\"LL-Test Model Predictions: {WLOSS:.8f}\")\n#         # if \"pytest\" in sys.modules:\n#         #     return WLOSS\n\n#         # Save model\n#         LABEL = \"GR-\" + LABEL if self.fink else \"UGRIZY-\" + LABEL\n#         # LABEL += f\"-LL{WLOSS:.3f}\"\n\n#         if platform.system() != \"Darwin\":\n#             # Create directories if they don't exist\n#             model_dir = Path(self.architecture) / \"models\" / self.dataset\n#             weights_dir = model_dir / \"weights\"\n#             model_dir.mkdir(parents=True, exist_ok=True)\n#             weights_dir.mkdir(parents=True, exist_ok=True)\n\n#             # Save model and weights with proper extensions\n#             model_path = model_dir / f\"model-{LABEL}.keras\"\n#             weights_path = weights_dir / f\"weights-{LABEL}.weights.h5\"\n            \n#             log.info(f\"Saving model to: {model_path}\")\n#             log.info(f\"Saving weights to: {weights_path}\")\n            \n#             model.save(str(model_path))\n#             model.save_weights(str(weights_path))\n\n#         return event","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T16:28:40.826359Z","iopub.execute_input":"2025-05-24T16:28:40.826970Z","iopub.status.idle":"2025-05-24T16:28:40.894684Z","shell.execute_reply.started":"2025-05-24T16:28:40.826945Z","shell.execute_reply":"2025-05-24T16:28:40.893113Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# if __name__ == \"__main__\":\n#     import sys\n#     sys.argv = ['script.py', '--architecture', 't2', '--dataset', 'plasticc', '--fink']\n#     parser = argparse.ArgumentParser()\n#     parser.add_argument(\"--architecture\", type=str, default=\"t2\", required=True)\n#     parser.add_argument(\"--dataset\", type=str,default= \"plasticc\", required=True)\n#     parser.add_argument(\"--fink\", action=\"store_true\")\n#     parser.add_argument(\"--avocado\", action=\"store_true\")\n#     parser.add_argument(\"--testset\", action=\"store_true\")\n#     args = parser.parse_args()\n\n#     training = Training(\n#     architecture=args.architecture,\n#     dataset=args.dataset,\n#     fink=args.fink,\n#     avocado=args.avocado,\n#     testset=args.testset,\n# )\n#     training()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T16:28:41.969160Z","iopub.execute_input":"2025-05-24T16:28:41.969419Z","iopub.status.idle":"2025-05-24T16:29:49.140198Z","shell.execute_reply.started":"2025-05-24T16:28:41.969399Z","shell.execute_reply":"2025-05-24T16:29:49.139590Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Evaluation","metadata":{}},{"cell_type":"code","source":"# import numpy as np\n# import tensorflow as tf\n# import matplotlib.pyplot as plt\n# import seaborn as sns\n# import pandas as pd\n# from sklearn.metrics import confusion_matrix\n# # from astronet.metrics import WeightedLogLoss\n# class WeightedLogLoss(tf.keras.losses.Loss):\n#     def __init__(self, reduction=tf.keras.losses.Reduction.NONE, name=\"weighted_log_loss\"):\n#         super().__init__(reduction=reduction, name=name)\n\n#     def call(self, y_true, y_pred):\n#         y_true = tf.cast(y_true, tf.float32)\n#         y_pred = tf.cast(y_pred, tf.float32)\n        \n#         wtable = tf.reduce_sum(y_true, axis=0) / tf.cast(tf.shape(y_true)[0], tf.float32)\n#         yc = tf.clip_by_value(y_pred, 1e-15, 1 - 1e-15)\n        \n#         loss = -(\n#             tf.reduce_mean(\n#                 tf.math.divide_no_nan(\n#                     tf.reduce_mean(y_true * tf.math.log(yc), axis=0), wtable\n#                 )\n#             )\n#         )\n#         return loss\n\n#     def get_config(self):\n#         config = super().get_config()\n#         return config\n\n# class DistributedWeightedLogLoss(tf.keras.losses.Loss):\n#     # initialize instance attributes\n#     def __init__(\n#         self,\n#         reduction=tf.keras.losses.Reduction.AUTO,\n#         name=\"weighted_log_loss\",\n#     ):\n#         super().__init__(reduction=reduction, name=name)\n\n#     # compute loss\n#     def call(self, y_true, y_pred):\n#         # Cast inputs to float32 first\n#         y_true = tf.cast(y_true, tf.float32)\n#         y_pred = tf.cast(y_pred, tf.float32)\n        \n#         # Calculate weights using TensorFlow ops for better distributed support\n#         wtable = tf.reduce_sum(y_true, axis=0) / tf.cast(tf.shape(y_true)[0], tf.float32)\n        \n#         # Clip predictions to avoid numerical instability\n#         yc = tf.clip_by_value(y_pred, 1e-15, 1 - 1e-15)\n        \n#         # Calculate loss using same dtype throughout\n#         loss = -(\n#             tf.reduce_mean(\n#                 tf.math.divide_no_nan(\n#                     tf.reduce_mean(y_true * tf.math.log(yc), axis=0), wtable\n#                 )\n#             )\n#         )\n        \n#         return loss\n# def check_and_clean_nan_data(X, y):\n#     \"\"\"Check for NaN values in the data and remove corresponding samples if found.\"\"\"\n#     # Check for NaN in X (across all dimensions except batch)\n#     nan_mask_X = np.isnan(X).any(axis=(1, 2))\n#     # Check for NaN in y\n#     nan_mask_y = np.isnan(y).any(axis=1)\n    \n#     # Combine masks\n#     nan_mask = nan_mask_X | nan_mask_y\n    \n#     if np.any(nan_mask):\n#         num_nans = np.sum(nan_mask)\n#         total_samples = X.shape[0]\n#         print(f\"Found {num_nans} samples with NaN values out of {total_samples} ({(num_nans/total_samples)*100:.2f}%)\")\n        \n#         # Keep only non-NaN samples\n#         X_clean = X[~nan_mask]\n#         y_clean = y[~nan_mask]\n        \n#         print(f\"Data shape after removing NaN samples: X={X_clean.shape}, y={y_clean.shape}\")\n#         return X_clean, y_clean\n    \n#     print(\"No NaN values found in the data\")\n#     return X, y\n\n# # Load test data\n# X_test = np.load('/kaggle/input/processed/X_test.npy')\n# y_test = np.load('/kaggle/input/processed/y_test.npy')\n\n# print(\"\\nOriginal data shapes:\")\n# print(f\"X_test shape: {X_test.shape}\")\n# print(f\"y_test shape: {y_test.shape}\")\n\n# # Make sure all arrays have the same number of samples\n# n_samples = y_test.shape[0]  # Use y_test as reference\n# X_test = X_test[:n_samples]\n\n# print(\"\\nAdjusted data shapes:\")\n# print(f\"X_test shape: {X_test.shape}\")\n# print(f\"y_test shape: {y_test.shape}\")\n\n# # Check and clean NaN values in test data\n# print(\"\\nChecking for NaN values in test data...\")\n# X_test, y_test = check_and_clean_nan_data(X_test, y_test)\n\n# # Load the model with custom objects\n# custom_objects = {'WeightedLogLoss': WeightedLogLoss}\n# model = tf.keras.models.load_model('/kaggle/input/hyperparameters/model-UGRIZY-1748131995-None-unknown.keras', custom_objects=custom_objects)\n\n# # Print model information\n# print(\"\\nModel Configuration:\")\n# model.summary()\n\n# # Make predictions\n# print(\"\\nMaking predictions...\")\n# y_pred = model.predict(X_test, verbose=1)\n\n# # Ensure probabilities sum to 1 (normalize if needed)\n# y_pred_normalized = y_pred / y_pred.sum(axis=1, keepdims=True)\n\n# # Calculate weighted log loss using normalized predictions\n# print(\"\\nCalculating metrics...\")\n# wloss = WeightedLogLoss()\n# loss_value = wloss(y_test, y_pred_normalized).numpy()\n# print(f\"\\nWeighted Log Loss: {loss_value:.3f}\")\n\n# # Convert predictions to class indices\n# y_true = np.argmax(y_test, axis=1)\n# y_pred_classes = np.argmax(y_pred_normalized, axis=1)\n\n# # Create confusion matrix\n# print(\"\\nGenerating confusion matrix...\")\n# cm = confusion_matrix(y_true, y_pred_classes)\n# cm_normalized = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n\n# # Plot confusion matrix\n# plt.figure(figsize=(12, 10))\n# sns.heatmap(cm_normalized, annot=True, fmt='.2f', cmap='Blues')\n# plt.title(f'Confusion Matrix (Normalized)\\nWeighted Log Loss = {loss_value:.3f}')\n# plt.ylabel('True Label')\n# plt.xlabel('Predicted Label')\n# plt.tight_layout()\n# plt.savefig('confusion_matrix.png')\n# plt.close()\n\n# print(\"\\nConfusion Matrix saved as 'confusion_matrix.png'\")\n\n# # Create DataFrame with class probabilities\n# print(\"\\nCreating CSV with class probabilities...\")\n# class_columns = [f'class_{i}_prob' for i in range(y_pred_normalized.shape[1])]\n# prob_df = pd.DataFrame(y_pred_normalized, columns=class_columns)\n\n# # Add true class and predicted class columns\n# prob_df['true_class'] = y_true\n# prob_df['predicted_class'] = y_pred_classes\n\n# # Verify probabilities sum to 1\n# prob_df['probability_sum'] = prob_df[class_columns].sum(axis=1)\n# print(\"\\nProbability sum verification:\")\n# print(prob_df['probability_sum'].describe())\n\n# # Save to CSV\n# csv_filename = 'class_probabilities.csv'\n# prob_df.to_csv(csv_filename, index=False)\n# print(f\"\\nSaved class probabilities to {csv_filename}\")\n\n# # Print sample of the CSV file\n# print(\"\\nSample of the CSV file:\")\n# print(prob_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T06:25:38.024826Z","iopub.execute_input":"2025-05-25T06:25:38.025123Z","iopub.status.idle":"2025-05-25T06:25:39.784631Z","shell.execute_reply.started":"2025-05-25T06:25:38.025100Z","shell.execute_reply":"2025-05-25T06:25:39.783957Z"}},"outputs":[],"execution_count":null},{"cell_type":"raw","source":"preprocessing of test data:","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import os\n# import numpy as np\n# import pandas as pd\n# from sklearn.preprocessing import RobustScaler, OneHotEncoder, StandardScaler\n# from sklearn.model_selection import train_test_split\n# from tensorflow.keras.utils import to_categorical\n# from scipy import stats, optimize as op\n# import george\n# from george import kernels\n# import warnings\n# from functools import partial\n# from typing import Dict, List, Union\n# from astropy.table import Table, vstack\n\n# # Suppress warnings\n# warnings.filterwarnings(\"ignore\")\n\n\n# # Constants\n# NUM_CLASSES = 15  # Including 'others' class\n# TIME_STEPS = 30   # Sequence length for time series\n# FILTERS = ['lsstu', 'lsstg', 'lsstr', 'lssti', 'lsstz', 'lssty']\n# METADATA_FEATURES = ['hostgal_photoz', 'mwebv', 'ddf']\n# ALL_FEATURES = FILTERS + METADATA_FEATURES\n# BATCH_SIZE = 50000  # Process 50,000 samples at a time\n\n# # PLAsTiCC class mapping to sequential indices\n# PLASTICC_CLASS_MAPPING = {\n#     90: 0,   # SNIa\n#     67: 1,   # SNIa-91bg\n#     52: 2,   # SNIax\n#     42: 3,   # SNII\n#     62: 4,   # SNIbc\n#     95: 5,   # SLSN-I\n#     15: 6,   # TDE\n#     64: 7,   # KN\n#     88: 8,   # AGN\n#     92: 9,   # RRL\n#     65: 10,  # M-dwarf\n#     16: 11,  # EB\n#     53: 12,  # Mira\n#     6: 13,   # µ-Lens-Single\n#     99: 14   # Others (not in training set)\n# }\n\n# # Class names for reference\n# CLASS_NAMES = {\n#     0: 'SNIa',\n#     1: 'SNIa-91bg',\n#     2: 'SNIax',\n#     3: 'SNII',\n#     4: 'SNIbc',\n#     5: 'SLSN-I',\n#     6: 'TDE',\n#     7: 'KN',\n#     8: 'AGN',\n#     9: 'RRL',\n#     10: 'M-dwarf',\n#     11: 'EB',\n#     12: 'Mira',\n#     13: 'µ-Lens-Single',\n#     14: 'Others'\n# }\n\n# # LSST passband wavelengths\n# LSST_PB_WAVELENGTHS = {\n#     'lsstu': 3671.0,\n#     'lsstg': 4827.0,\n#     'lsstr': 6223.0,\n#     'lssti': 7546.0,\n#     'lsstz': 8691.0,\n#     'lssty': 9710.0\n# }\n\n# def load_and_merge_data(data_path, metadata_path):\n#     \"\"\"Load and merge light curve data with metadata\"\"\"\n#     data = pd.read_csv(data_path)\n#     metadata = pd.read_csv(metadata_path)\n    \n#     # Merge and clean data\n#     merged = pd.merge(data, metadata, on='object_id', how='left')\n    \n#     # Handle missing values\n#     merged = merged.dropna(subset=['flux', 'flux_error'] + METADATA_FEATURES)\n    \n#     return merged\n\n# def filter_dataframe_only_supernova(object_list_filename: str, dataframe: pd.DataFrame) -> pd.DataFrame:\n#     \"\"\"Filter dataframe that contains many classes to only Supernovae types.\"\"\"\n#     plasticc_object_list = np.genfromtxt(object_list_filename, dtype=\"U\")\n#     filtered_dataframe = dataframe[dataframe[\"object_id\"].isin(plasticc_object_list)]\n#     return filtered_dataframe\n\n# def transient_trim(object_list: List[str], df: pd.DataFrame) -> tuple:\n#     \"\"\"Trim off light-curve plateau to leave only the transient part +/- 50 time-steps\"\"\"\n#     adf = pd.DataFrame(data=[], columns=df.columns)\n#     good_object_list = []\n#     for obj in object_list:\n#         obs = df[df[\"object_id\"] == obj]\n#         obs_time = obs[\"mjd\"]\n#         obs_detected_time = obs_time[obs[\"detected\"] == 1]\n#         if len(obs_detected_time) == 0:\n#             print(f\"Zero detected points for object:{object_list.index(obj)}\")\n#             continue\n#         is_obs_transient = (obs_time > obs_detected_time.iat[0] - 50) & (\n#             obs_time < obs_detected_time.iat[-1] + 50\n#         )\n#         obs_transient = obs[is_obs_transient]\n#         if len(obs_transient[\"mjd\"]) == 0:\n#             is_obs_transient = (obs_time > obs_detected_time.iat[0] - 1000) & (\n#                 obs_time < obs_detected_time.iat[-1] + 1000\n#             )\n#             obs_transient = obs[is_obs_transient]\n#         obs_transient[\"mjd\"] -= min(obs_transient[\"mjd\"])  # so all transients start at time 0\n#         good_object_list.append(object_list.index(obj))\n#         adf = np.vstack((adf, obs_transient))\n\n#     obs_transient = pd.DataFrame(data=adf, columns=obs_transient.columns)\n#     filter_indices = good_object_list\n#     new_filtered_object_list = np.take(np.array(object_list), filter_indices, axis=0)\n#     return obs_transient, list(new_filtered_object_list)\n\n# def fit_2d_gp(obj_data: pd.DataFrame, return_kernel: bool = False, pb_wavelengths: Dict = LSST_PB_WAVELENGTHS, **kwargs):\n#     \"\"\"Fit a 2D Gaussian process.\"\"\"\n#     guess_length_scale = 20.0\n\n#     obj_times = obj_data.mjd.astype(float)\n#     obj_flux = obj_data.flux.astype(float)\n#     obj_flux_error = obj_data.flux_error.astype(float)\n#     obj_wavelengths = obj_data[\"filter\"].map(pb_wavelengths)\n\n#     def neg_log_like(p):\n#         gp.set_parameter_vector(p)\n#         loglike = gp.log_likelihood(obj_flux, quiet=True)\n#         return -loglike if np.isfinite(loglike) else 1e25\n\n#     def grad_neg_log_like(p):\n#         gp.set_parameter_vector(p)\n#         return -gp.grad_log_likelihood(obj_flux, quiet=True)\n\n#     signal_to_noises = np.abs(obj_flux) / np.sqrt(\n#         obj_flux_error**2 + (1e-2 * np.max(obj_flux)) ** 2\n#     )\n#     scale = np.abs(obj_flux[signal_to_noises.idxmax()])\n\n#     kernel = (0.5 * scale) ** 2 * george.kernels.Matern32Kernel(\n#         [guess_length_scale*2, 6000*2], ndim=2\n#     )\n#     kernel.freeze_parameter(\"k2:metric:log_M_1_1\")\n\n#     gp = george.GP(kernel)\n#     default_gp_param = gp.get_parameter_vector()\n#     x_data = np.vstack([obj_times, obj_wavelengths]).T\n#     gp.compute(x_data, obj_flux_error)\n\n#     bounds = [(0, np.log(1000**2))]\n#     bounds = [(default_gp_param[0] - 10, default_gp_param[0] + 10)] + bounds\n#     results = op.minimize(\n#         neg_log_like,\n#         gp.get_parameter_vector(),\n#         jac=grad_neg_log_like,\n#         method=\"L-BFGS-B\",\n#         bounds=bounds,\n#         tol=1e-6,\n#     )\n\n#     if results.success:\n#         gp.set_parameter_vector(results.x)\n#     else:\n#         obj = obj_data[\"object_id\"][0]\n#         print(f\"GP fit failed for {obj}! Using guessed GP parameters.\")\n#         gp.set_parameter_vector(default_gp_param)\n\n#     gp_predict = partial(gp.predict, obj_flux)\n\n#     if return_kernel:\n#         return kernel, gp_predict\n#     return gp_predict\n\n# def predict_2d_gp(gp_predict, gp_times, gp_wavelengths):\n#     \"\"\"Outputs the predictions of a Gaussian Process.\"\"\"\n#     unique_wavelengths = np.unique(gp_wavelengths)\n#     number_gp = len(gp_times)\n#     obj_gps = []\n#     for wavelength in unique_wavelengths:\n#         gp_wavelengths = np.ones(number_gp) * wavelength\n#         pred_x_data = np.vstack([gp_times, gp_wavelengths]).T\n#         pb_pred, pb_pred_var = gp_predict(pred_x_data, return_var=True)\n#         obj_gp_pb_array = np.column_stack((gp_times, pb_pred, np.sqrt(pb_pred_var)))\n#         obj_gp_pb = Table(\n#             [\n#                 obj_gp_pb_array[:, 0],\n#                 obj_gp_pb_array[:, 1],\n#                 obj_gp_pb_array[:, 2],\n#                 [wavelength] * number_gp,\n#             ],\n#             names=[\"mjd\", \"flux\", \"flux_error\", \"filter\"],\n#         )\n#         if len(obj_gps) == 0:\n#             obj_gps = obj_gp_pb\n#         else:\n#             obj_gps = vstack((obj_gps, obj_gp_pb))\n\n#     return obj_gps.to_pandas()\n\n# def generate_gp_single_event(df: pd.DataFrame, timesteps: int = 100, pb_wavelengths: Dict = LSST_PB_WAVELENGTHS) -> pd.DataFrame:\n#     \"\"\"Generate GP interpolation for a single event.\"\"\"\n#     filters = list(np.unique(df[\"filter\"]))\n#     gp_wavelengths = np.vectorize(pb_wavelengths.get)(filters)\n#     inverse_pb_wavelengths = {v: k for k, v in pb_wavelengths.items()}\n\n#     gp_predict = fit_2d_gp(df, pb_wavelengths=pb_wavelengths)\n#     gp_times = np.linspace(min(df[\"mjd\"]), max(df[\"mjd\"]), timesteps)\n#     obj_gps = predict_2d_gp(gp_predict, gp_times, gp_wavelengths)\n#     obj_gps[\"filter\"] = obj_gps[\"filter\"].map(inverse_pb_wavelengths)\n\n#     return obj_gps\n\n# def generate_gp_all_objects(object_list: List[str], obs_transient: pd.DataFrame, timesteps: int = 100, pb_wavelengths: Dict = LSST_PB_WAVELENGTHS) -> pd.DataFrame:\n#     \"\"\"Generate Gaussian Process interpolation for all objects.\"\"\"\n#     # Initialize with all possible columns\n#     columns = ['mjd'] + list(pb_wavelengths.keys()) + ['object_id']\n#     adf = pd.DataFrame(columns=columns)\n    \n#     for object_id in object_list:\n#         print(f\"OBJECT ID:{object_id} at INDEX:{object_list.index(object_id)}\")\n#         df = obs_transient[obs_transient['object_id'] == object_id]\n        \n#         try:\n#             # Generate GP predictions for this object\n#             obj_gps = generate_gp_single_event(df, timesteps, pb_wavelengths)\n            \n#             # Pivot to get one row per time step\n#             obj_gps = pd.pivot_table(obj_gps, index='mjd', columns='filter', values='flux')\n#             obj_gps = obj_gps.reset_index()\n            \n#             # Ensure all filter columns exist\n#             for filter_name in pb_wavelengths.keys():\n#                 if filter_name not in obj_gps.columns:\n#                     obj_gps[filter_name] = np.nan\n            \n#             # Add object_id\n#             obj_gps['object_id'] = object_id\n            \n#             # Reorder columns to match the expected order\n#             obj_gps = obj_gps[columns]\n            \n#             # Append to the main DataFrame\n#             adf = pd.concat([adf, obj_gps], ignore_index=True)\n            \n#         except Exception as e:\n#             print(f\"Error processing object {object_id}: {str(e)}\")\n#             continue\n    \n#     return adf\n\n# def remap_filters(df: pd.DataFrame, filter_map: Dict) -> pd.DataFrame:\n#     \"\"\"Remap integer filters to the corresponding filters.\"\"\"\n#     df.rename({\"passband\": \"filter\"}, axis=\"columns\", inplace=True)\n#     df[\"filter\"].replace(to_replace=filter_map, inplace=True)\n#     return df\n\n# def robust_scale(dataframe: pd.DataFrame, scale_columns: List[Union[str, int]]) -> pd.DataFrame:\n#     \"\"\"Standardize a dataset along axis=0 (rows)\"\"\"\n#     scaler = RobustScaler()\n#     scaler = scaler.fit(dataframe[scale_columns])\n#     dataframe.loc[:, scale_columns] = scaler.transform(dataframe[scale_columns].to_numpy())\n#     return dataframe\n\n# def z_score_normalize(dataframe: pd.DataFrame, scale_columns: List[Union[str, int]]) -> pd.DataFrame:\n#     \"\"\"Apply z-score normalization to specified columns.\n    \n#     Parameters\n#     ----------\n#     dataframe: pd.DataFrame\n#         Dataframe containing the data to normalize\n#     scale_columns: List[Union[str, int]]\n#         Columns to apply z-score normalization to\n        \n#     Returns\n#     -------\n#     pd.DataFrame\n#         Dataframe with normalized columns\n#     \"\"\"\n#     scaler = StandardScaler()\n#     scaler = scaler.fit(dataframe[scale_columns])\n#     dataframe.loc[:, scale_columns] = scaler.transform(dataframe[scale_columns].to_numpy())\n#     return dataframe\n\n# def map_classes(df: pd.DataFrame) -> pd.DataFrame:\n#     \"\"\"Map original target codes to 0-14 indices.\n    \n#     Parameters\n#     ----------\n#     df: pd.DataFrame\n#         Dataframe containing the 'target' column with original class codes\n        \n#     Returns\n#     -------\n#     pd.DataFrame\n#         Dataframe with added 'mapped_target' column and invalid classes removed\n#     \"\"\"\n#     df[\"mapped_target\"] = df[\"target\"].map(PLASTICC_CLASS_MAPPING)\n#     df = df[df[\"mapped_target\"].notna()]  # Drop invalid/unmapped classes\n#     return df\n\n# def create_dataset(features, labels, time_steps, step):\n#     \"\"\"Create time series sequences with specified time steps and step size.\n    \n#     Parameters\n#     ----------\n#     features : pd.DataFrame\n#         DataFrame containing the features\n#     labels : pd.Series\n#         Series containing the labels\n#     time_steps : int\n#         Number of time steps in each sequence\n#     step : int\n#         Step size between sequences\n        \n#     Returns\n#     -------\n#     tuple\n#         (sequences, labels) where sequences is a numpy array of shape (n_sequences, time_steps, n_features)\n#         and labels is a numpy array of shape (n_sequences,)\n#     \"\"\"\n#     sequences, sequence_labels = [], []\n    \n#     for i in range(0, len(features) - time_steps + 1, step):\n#         sequences.append(features[i:i + time_steps].values)\n#         sequence_labels.append(labels.iloc[i])\n    \n#     return np.array(sequences), np.array(sequence_labels)\n\n# def extract_redshift(df):\n#     \"\"\"Extract photometric redshift for each object (static value).\"\"\"\n#     # Group by object_id and take the first occurrence (same for all rows)\n#     redshift_df = df.groupby(\"object_id\")[\"hostgal_photoz\"].first().reset_index()\n#     return redshift_df[\"hostgal_photoz\"].values.reshape(-1, 1)\n\n# def process_batch(batch_data, metadata_pd, z_scaler, batch_num, save_dir):\n#     \"\"\"Process a batch of test data.\"\"\"\n#     print(f\"\\nProcessing batch {batch_num}...\")\n    \n#     # Remap filters\n#     print(f\"Remapping filters for batch {batch_num}...\")\n#     batch_data = remap_filters(batch_data, {0: 'lsstu', 1: 'lsstg', 2: 'lsstr', \n#                                           3: 'lssti', 4: 'lsstz', 5: 'lssty'})\n#     batch_data.rename({'flux_err': 'flux_error'}, axis='columns', inplace=True)\n\n#     # Get unique objects in this batch\n#     object_list = list(np.unique(batch_data['object_id']))\n#     print(f\"Processing {len(object_list)} objects in batch {batch_num}\")\n\n#     # Trim to transient period\n#     print(f\"Trimming transients for batch {batch_num}...\")\n#     obs_transient, object_list = transient_trim(object_list, batch_data)\n\n#     if len(object_list) == 0:\n#         print(f\"No valid objects in batch {batch_num}, skipping...\")\n#         return\n\n#     # Generate GP predictions\n#     print(f\"Generating GP predictions for batch {batch_num}...\")\n#     generated_gp_dataset = generate_gp_all_objects(object_list, obs_transient, timesteps=100, pb_wavelengths=LSST_PB_WAVELENGTHS)\n    \n#     if generated_gp_dataset.empty:\n#         print(f\"No GP data generated for batch {batch_num}, skipping...\")\n#         return\n        \n#     # Merge with metadata for objects in this batch\n#     print(f\"Merging metadata for batch {batch_num}...\")\n#     batch_metadata = metadata_pd[metadata_pd['object_id'].isin(object_list)]\n#     df_combi = generated_gp_dataset.merge(batch_metadata, on='object_id', how='left')\n    \n#     # Drop unnecessary columns\n#     columns_to_drop = [\n#         \"ra\", \"decl\", \"gal_l\", \"gal_b\", \"hostgal_specz\", \"distmod\", \"hostgal_photoz_err\"\n#     ]\n#     df = df_combi.drop(columns=[col for col in columns_to_drop if col in df_combi.columns])\n\n#     # Apply z-score normalization using pre-fitted scaler\n#     print(f\"Scaling features for batch {batch_num}...\")\n#     filters = ['lsstu', 'lsstg', 'lsstr', 'lssti', 'lsstz', 'lssty']\n    \n#     # Ensure we have the exact same columns in the same order as training\n#     if z_scaler is not None:\n#         # Check if scaler has feature names\n#         if hasattr(z_scaler, 'feature_names_in_'):\n#             scaler_features = list(z_scaler.feature_names_in_)\n#             print(f\"Scaler expects features: {scaler_features}\")\n#         else:\n#             scaler_features = filters\n            \n#         # Ensure all required features exist in df\n#         missing_features = [f for f in scaler_features if f not in df.columns]\n#         if missing_features:\n#             print(f\"Warning: Missing features {missing_features}, filling with zeros\")\n#             for feature in missing_features:\n#                 df[feature] = 0.0\n        \n#         # Select features in the exact same order as scaler expects\n#         features_to_scale = df[scaler_features].copy()\n        \n#         # Apply scaling\n#         df[scaler_features] = z_scaler.transform(features_to_scale)\n        \n#         # Use the scaler's feature order for downstream processing\n#         filters = scaler_features\n#     else:\n#         print(\"No pre-fitted scaler available, using raw features\")\n#         # If no scaler, ensure all filter columns exist\n#         for f in filters:\n#             if f not in df.columns:\n#                 df[f] = 0.0\n\n#     # Create sequences  \n#     print(f\"Creating sequences for batch {batch_num}...\")\n#     TIME_STEPS = 20\n#     STEP = 20\n\n#     X_batch, object_ids = create_dataset_with_ids(\n#         df[filters],\n#         df['object_id'],\n#         TIME_STEPS,\n#         STEP\n#     )\n\n#     # Save batch\n#     print(f\"Saving batch {batch_num}...\")\n#     np.save(os.path.join(save_dir, f\"X_test_batch_{batch_num}.npy\"), X_batch)\n#     np.save(os.path.join(save_dir, f\"object_ids_batch_{batch_num}.npy\"), object_ids)\n    \n#     print(f\"Batch {batch_num} complete! Shape: {X_batch.shape}\")\n\n# def create_dataset_with_ids(features, object_ids, time_steps, step):\n#     \"\"\"Create time series sequences with object IDs for test data.\n    \n#     Parameters\n#     ----------\n#     features : pd.DataFrame\n#         DataFrame containing the features\n#     object_ids : pd.Series\n#         Series containing the object IDs\n#     time_steps : int\n#         Number of time steps in each sequence\n#     step : int\n#         Step size between sequences\n        \n#     Returns\n#     -------\n#     tuple\n#         (sequences, object_ids) where sequences is a numpy array of shape (n_sequences, time_steps, n_features)\n#         and object_ids is a numpy array of shape (n_sequences,)\n#     \"\"\"\n#     sequences, sequence_object_ids = [], []\n    \n#     for i in range(0, len(features) - time_steps + 1, step):\n#         sequences.append(features[i:i + time_steps].values)\n#         sequence_object_ids.append(object_ids.iloc[i])\n    \n#     return np.array(sequences), np.array(sequence_object_ids)\n\n# # Main execution\n# TEST_DATA_PATH = '/kaggle/input/PLAsTiCC-2018/test_set.csv'\n# TEST_METADATA_PATH = '/kaggle/input/PLAsTiCC-2018/test_set_metadata.csv'\n# SAVE_DIR = \"/kaggle/working/output\"\n# Z_SCALER_PATH = \"/kaggle/input/hyperparameters/z_scaler.npy\"  # Path to pre-fitted scaler\n\n# os.makedirs(SAVE_DIR, exist_ok=True)\n\n# # Load pre-fitted z-scaler from training\n# print(\"Loading pre-fitted z-scaler...\")\n# try:\n#     z_scaler = np.load(Z_SCALER_PATH, allow_pickle=True).item()\n#     print(f\"Loaded z-scaler with feature names: {z_scaler.feature_names_in_ if hasattr(z_scaler, 'feature_names_in_') else 'Not available'}\")\n# except:\n#     print(\"Could not load pre-fitted z-scaler. Will create a new one from first batch.\")\n#     z_scaler = None\n\n# # Load test metadata once\n# print(\"Loading test metadata...\")\n# metadata_pd = pd.read_csv(TEST_METADATA_PATH, sep=',', index_col='object_id')\n# metadata_pd = metadata_pd.reset_index()\n# metadata_pd['object_id'] = metadata_pd['object_id'].astype(np.int32)\n\n# # Process test data in batches\n# print(\"Starting batch processing of test data...\")\n# batch_num = 0\n# chunk_size = BATCH_SIZE\n# z_scaler_global = z_scaler  # Store the global scaler\n\n# for chunk in pd.read_csv(TEST_DATA_PATH, sep=',', chunksize=chunk_size):\n#     batch_num += 1\n#     print(f\"\\nLoaded batch {batch_num} with {len(chunk)} rows\")\n    \n#     try:\n#         process_batch(chunk, metadata_pd, z_scaler_global, batch_num, SAVE_DIR)\n#     except Exception as e:\n#         print(f\"Error processing batch {batch_num}: {str(e)}\")\n#         continue\n\n# print(\"\\nAll batches processed!\")\n\n# # Save additional info\n# np.save(os.path.join(SAVE_DIR, \"class_mapping.npy\"), PLASTICC_CLASS_MAPPING)\n# np.save(os.path.join(SAVE_DIR, \"class_names.npy\"), CLASS_NAMES)\n# print(f\"Saved class mapping and names to {SAVE_DIR}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import os\n# import numpy as np\n# import glob\n\n# # Configuration\n# SAVE_DIR = \"/kaggle/working/output\"\n# OUTPUT_FILE_X = \"X_test_complete.npy\"\n# OUTPUT_FILE_IDS = \"object_ids_complete.npy\"\n\n# def merge_all_batches(save_dir, output_file_x, output_file_ids):\n#     \"\"\"\n#     Merge all batch files into single numpy arrays.\n    \n#     Parameters\n#     ----------\n#     save_dir : str\n#         Directory containing the batch files\n#     output_file_x : str\n#         Name of output file for features\n#     output_file_ids : str\n#         Name of output file for object IDs\n#     \"\"\"\n    \n#     # Find all batch files\n#     x_batch_files = sorted(glob.glob(os.path.join(save_dir, \"X_test_batch_*.npy\")))\n#     id_batch_files = sorted(glob.glob(os.path.join(save_dir, \"object_ids_batch_*.npy\")))\n    \n#     print(f\"Found {len(x_batch_files)} X batch files\")\n#     print(f\"Found {len(id_batch_files)} ID batch files\")\n    \n#     if len(x_batch_files) == 0:\n#         print(\"No batch files found!\")\n#         return\n    \n#     if len(x_batch_files) != len(id_batch_files):\n#         print(\"Warning: Mismatch between number of X and ID batch files!\")\n    \n#     # Initialize lists to store all batches\n#     all_x_batches = []\n#     all_id_batches = []\n    \n#     total_samples = 0\n    \n#     # Load and combine all X batches\n#     print(\"\\nLoading X batches...\")\n#     for i, batch_file in enumerate(x_batch_files):\n#         print(f\"Loading {batch_file}...\")\n#         try:\n#             batch = np.load(batch_file)\n#             all_x_batches.append(batch)\n#             total_samples += batch.shape[0]\n#             print(f\"  - Shape: {batch.shape}\")\n#         except Exception as e:\n#             print(f\"  - Error loading {batch_file}: {str(e)}\")\n#             continue\n    \n#     # Load and combine all ID batches\n#     print(\"\\nLoading ID batches...\")\n#     for i, batch_file in enumerate(id_batch_files):\n#         print(f\"Loading {batch_file}...\")\n#         try:\n#             batch = np.load(batch_file)\n#             all_id_batches.append(batch)\n#             print(f\"  - Shape: {batch.shape}\")\n#         except Exception as e:\n#             print(f\"  - Error loading {batch_file}: {str(e)}\")\n#             continue\n    \n#     if len(all_x_batches) == 0:\n#         print(\"No valid X batches loaded!\")\n#         return\n    \n#     # Concatenate all batches\n#     print(f\"\\nConcatenating {len(all_x_batches)} X batches...\")\n#     X_complete = np.concatenate(all_x_batches, axis=0)\n#     print(f\"Complete X shape: {X_complete.shape}\")\n    \n#     if len(all_id_batches) > 0:\n#         print(f\"Concatenating {len(all_id_batches)} ID batches...\")\n#         ids_complete = np.concatenate(all_id_batches, axis=0)\n#         print(f\"Complete IDs shape: {ids_complete.shape}\")\n#     else:\n#         print(\"No ID batches to concatenate\")\n#         ids_complete = None\n    \n#     # Save merged files\n#     print(f\"\\nSaving complete X array to {output_file_x}...\")\n#     np.save(os.path.join(save_dir, output_file_x), X_complete)\n    \n#     if ids_complete is not None:\n#         print(f\"Saving complete IDs array to {output_file_ids}...\")\n#         np.save(os.path.join(save_dir, output_file_ids), ids_complete)\n    \n#     # Print summary\n#     print(f\"\\n{'='*50}\")\n#     print(\"MERGE SUMMARY\")\n#     print(f\"{'='*50}\")\n#     print(f\"Total batches processed: {len(all_x_batches)}\")\n#     print(f\"Total samples: {X_complete.shape[0]:,}\")\n#     print(f\"Feature dimensions: {X_complete.shape[1:]}\")\n#     print(f\"Final X file: {os.path.join(save_dir, output_file_x)}\")\n#     if ids_complete is not None:\n#         print(f\"Final IDs file: {os.path.join(save_dir, output_file_ids)}\")\n#     print(f\"{'='*50}\")\n    \n#     # Calculate file sizes\n#     x_size_mb = os.path.getsize(os.path.join(save_dir, output_file_x)) / (1024 * 1024)\n#     print(f\"X file size: {x_size_mb:.2f} MB\")\n    \n#     if ids_complete is not None:\n#         ids_size_mb = os.path.getsize(os.path.join(save_dir, output_file_ids)) / (1024 * 1024)\n#         print(f\"IDs file size: {ids_size_mb:.2f} MB\")\n    \n#     return X_complete, ids_complete\n\n# def cleanup_batch_files(save_dir, keep_batches=False):\n#     \"\"\"\n#     Optionally remove individual batch files after merging.\n    \n#     Parameters\n#     ----------\n#     save_dir : str\n#         Directory containing the batch files\n#     keep_batches : bool\n#         If False, delete individual batch files after merging\n#     \"\"\"\n#     if not keep_batches:\n#         print(\"\\nCleaning up individual batch files...\")\n        \n#         # Remove X batch files\n#         x_batch_files = glob.glob(os.path.join(save_dir, \"X_test_batch_*.npy\"))\n#         for file in x_batch_files:\n#             try:\n#                 os.remove(file)\n#                 print(f\"Removed {file}\")\n#             except Exception as e:\n#                 print(f\"Error removing {file}: {str(e)}\")\n        \n#         # Remove ID batch files\n#         id_batch_files = glob.glob(os.path.join(save_dir, \"object_ids_batch_*.npy\"))\n#         for file in id_batch_files:\n#             try:\n#                 os.remove(file)\n#                 print(f\"Removed {file}\")\n#             except Exception as e:\n#                 print(f\"Error removing {file}: {str(e)}\")\n        \n#         print(\"Cleanup complete!\")\n#     else:\n#         print(\"Keeping individual batch files as requested.\")\n\n# # Execute the merging\n# print(\"Starting batch merge process...\")\n\n# try:\n#     X_complete, ids_complete = merge_all_batches(SAVE_DIR, OUTPUT_FILE_X, OUTPUT_FILE_IDS)\n    \n#     # Optionally cleanup batch files (set to True to remove individual batches)\n#     cleanup_batch_files(SAVE_DIR, keep_batches=True)  # Change to False to delete batch files\n    \n#     print(\"\\nMerge process completed successfully!\")\n    \n# except Exception as e:\n#     print(f\"Error during merge process: {str(e)}\")\n#     import traceback\n#     traceback.print_exc()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T06:32:00.258796Z","iopub.execute_input":"2025-05-25T06:32:00.259076Z","iopub.status.idle":"2025-05-25T06:32:04.578624Z","shell.execute_reply.started":"2025-05-25T06:32:00.259056Z","shell.execute_reply":"2025-05-25T06:32:04.577857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T13:02:52.134622Z","iopub.execute_input":"2025-05-26T13:02:52.134861Z","iopub.status.idle":"2025-05-26T13:44:52.081848Z","shell.execute_reply.started":"2025-05-26T13:02:52.134841Z","shell.execute_reply":"2025-05-26T13:44:52.080557Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# import shutil\n# import os\n\n# def copy_and_move_file(input_path, copy_path, move_path):\n#     try:\n#         # Check if the file exists\n#         if not os.path.isfile(input_path):\n#             print(f\"❌ File not found at: {input_path}\")\n#             return\n\n#         # Create directories if they don't exist\n#         os.makedirs(os.path.dirname(copy_path), exist_ok=True)\n#         os.makedirs(os.path.dirname(move_path), exist_ok=True)\n\n#         # Copy file\n#         shutil.copy2(input_path, copy_path)\n#         print(f\"📋 File copied from {input_path} to {copy_path}\")\n\n#         # Move file\n#         shutil.move(input_path, move_path)\n#         print(f\"🚚 File moved from {input_path} to {move_path}\")\n\n#     except Exception as e:\n#         print(f\"⚠️ Error occurred: {e}\")\n\n# # 🧪 Example usage\n# if __name__ == \"__main__\":\n#     input_file = \"/kaggle/input/hyperparameters/submission.csv\"\n#     copy_file = \"/kaggle/working/submission.csv\"\n#     move_file = \"/kaggle/working/submission.csv\"\n\n#     copy_and_move_file(input_file, copy_file, move_file)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T05:50:26.578903Z","iopub.execute_input":"2025-05-26T05:50:26.579193Z","iopub.status.idle":"2025-05-26T05:50:26.647198Z","shell.execute_reply.started":"2025-05-26T05:50:26.579169Z","shell.execute_reply":"2025-05-26T05:50:26.646524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}