{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":10384,"databundleVersionId":120379,"sourceType":"competition"},{"sourceId":390908,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":321927,"modelId":342544}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\n\n# === PATHS ===\ndataset_dir = \"PLASTICC_dataset\"\nos.makedirs(f\"{dataset_dir}/train\", exist_ok=True)\nos.makedirs(f\"{dataset_dir}/test\", exist_ok=True)\n\n# === TRAIN PROCESSING ===\ntrain_lc_df = pd.read_csv(\"/kaggle/input/PLAsTiCC-2018/training_set.csv\")\ntrain_meta_df = pd.read_csv(\"/kaggle/input/PLAsTiCC-2018/training_set_metadata.csv\")\n\ntrain_label_rows = []\n\nfor object_id, group in train_lc_df.groupby(\"object_id\"):\n    lc_array = group[[\"mjd\", \"passband\", \"flux\", \"flux_err\", \"detected\"]].values.astype(np.float32)\n    np.save(f\"{dataset_dir}/train/{object_id}.npy\", lc_array)\n\n    row = train_meta_df[train_meta_df.object_id == object_id].iloc[0]\n    label = int(row[\"target\"])\n    params = row.drop([\"object_id\", \"target\"]).values.astype(np.float32)\n    \n    train_label_rows.append([object_id, label] + params.tolist())\n\ntrain_columns = [\"id\", \"label\"] + train_meta_df.columns.drop([\"object_id\", \"target\"]).tolist()\ntrain_label_df = pd.DataFrame(train_label_rows, columns=train_columns)\ntrain_label_df.to_csv(f\"{dataset_dir}/train_labels.csv\", index=False)\n\n# === TEST PROCESSING ===\ntest_lc_df = pd.read_csv(\"/kaggle/input/PLAsTiCC-2018/test_set_sample.csv\")\ntest_meta_df = pd.read_csv(\"/kaggle/input/PLAsTiCC-2018/test_set_metadata.csv\")\n\ntest_label_rows = []\n\nfor object_id, group in test_lc_df.groupby(\"object_id\"):\n    lc_array = group[[\"mjd\", \"passband\", \"flux\", \"flux_err\", \"detected\"]].values.astype(np.float32)\n    np.save(f\"{dataset_dir}/test/{object_id}.npy\", lc_array)\n\n    row = test_meta_df[test_meta_df.object_id == object_id].iloc[0]\n    \n    # Assign dummy label (e.g., -1) for test samples\n    label = -1\n    params = row.drop([\"object_id\"]).values.astype(np.float32)\n    \n    test_label_rows.append([object_id, label] + params.tolist())\n\ntest_columns = [\"id\", \"label\"] + test_meta_df.columns.drop([\"object_id\"]).tolist()\ntest_label_df = pd.DataFrame(test_label_rows, columns=test_columns)\ntest_label_df.to_csv(f\"{dataset_dir}/test_labels.csv\", index=False)\n\nprint(\"✅ PLAsTiCC data converted to Deep-LC format (train + test)\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T18:24:26.147291Z","iopub.execute_input":"2025-05-13T18:24:26.148324Z","iopub.status.idle":"2025-05-13T18:24:59.27097Z","shell.execute_reply.started":"2025-05-13T18:24:26.148291Z","shell.execute_reply":"2025-05-13T18:24:59.270052Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Clone the repo\n!git clone https://github.com/ckm3/Deep-LC.git\n\n# Go into the folder\n%cd Deep-LC\n\n# Install the package\n!pip install -e .\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T19:51:15.273695Z","iopub.execute_input":"2025-05-13T19:51:15.274044Z","iopub.status.idle":"2025-05-13T19:53:11.918799Z","shell.execute_reply.started":"2025-05-13T19:51:15.274017Z","shell.execute_reply":"2025-05-13T19:53:11.917552Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%cd /kaggle/working/Deep-LC/src\n\nimport numpy as np\nfrom deep_lc import LocalDataset\nimport os\nimport pandas as pd\n# === Load train_labels.csv (contains labels and params) ===\nimport pandas as pd\n\ntrain_df = pd.read_csv(\"/kaggle/working/PLASTICC_dataset/train_labels.csv\")\n\ntrain_lcs = []\ntrain_params = []\ntrain_labels = []\nprint(train_df.columns.tolist())\n\nfor i in range(len(train_df)):\n    lc_id = int(train_df.iloc[i][\"id\"])  # was 'object_id'\n    train_lcs.append(np.load(f\"/kaggle/working/PLASTICC_dataset/train/{lc_id}.npy\"))\n    train_params.append(train_df.iloc[i].drop([\"id\", \"label\"]).values.astype(np.float32))  # was 'object_id', 'target'\n    train_labels.append(train_df.iloc[i][\"label\"])  # was 'target'\n\n\n# === Load test ===\n# We assign dummy label -1 to each test entry\ntest_filenames = sorted(os.listdir(\"/kaggle/working/PLASTICC_dataset/test\"))\ntest_lcs, test_params, test_labels = [], [], []\n\nfor fname in test_filenames:\n    if not fname.endswith(\".npy\"):\n        continue\n    lc_id = int(fname.replace(\".npy\", \"\"))\n    test_lcs.append(np.load(f\"/kaggle/working/PLASTICC_dataset/test/{fname}\"))\n    # Use dummy parameters (e.g., all zeros of same shape as train_params[0])\n    test_params.append(np.zeros_like(train_params[0]))  # or customize if needed\n    test_labels.append(-1)  # dummy label\n\n# === Create LocalDataset objects for Deep-LC ===\ntraining_set = LocalDataset(train_lcs, train_params, train_labels)\ntest_set = LocalDataset(test_lcs, test_params, test_labels)\n\nprint(\"✅ Dataset loaded:\")\nprint(f\"Training samples: {len(training_set)}\")\nprint(f\"Testing samples: {len(test_set)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T18:46:30.556189Z","iopub.execute_input":"2025-05-13T18:46:30.55732Z","iopub.status.idle":"2025-05-13T18:46:36.803394Z","shell.execute_reply.started":"2025-05-13T18:46:30.557278Z","shell.execute_reply":"2025-05-13T18:46:36.802435Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from deep_lc.config import PROPOSAL_NUM, BATCH_SIZE, LR, LABELS\n\nhyper_parameters = {\n    \"batch_size\": BATCH_SIZE,\n    \"lr\": LR / 10,  # smaller LR for fine-tuning\n    \"weight_decay\": 0,\n    \"labels\": LABELS,  # Update if your classes differ\n    \"proposal_num\": PROPOSAL_NUM,\n}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T18:48:14.584665Z","iopub.execute_input":"2025-05-13T18:48:14.585129Z","iopub.status.idle":"2025-05-13T18:48:14.590738Z","shell.execute_reply.started":"2025-05-13T18:48:14.585099Z","shell.execute_reply":"2025-05-13T18:48:14.589769Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel_dict = torch.load(\"/kaggle/input/combined_17_conformal.ckpt/other/default/1/combined_17_conformal_calibrated.ckpt\", \n                        map_location=device)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"WITH TRANING 85:15 SPLIT","metadata":{}},{"cell_type":"code","source":"# Clone the repo\n!git clone https://github.com/ckm3/Deep-LC.git\n\n# Go into the folder\n%cd Deep-LC\n\n# Install the package\n!pip install -e .\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T20:11:27.972335Z","iopub.execute_input":"2025-05-13T20:11:27.972548Z","iopub.status.idle":"2025-05-13T20:13:06.662749Z","shell.execute_reply.started":"2025-05-13T20:11:27.972532Z","shell.execute_reply":"2025-05-13T20:13:06.661863Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\n\n# Paths to your CSV files\ntrain_data_path = '/kaggle/input/PLAsTiCC-2018/training_set.csv'\nmetadata_path = '/kaggle/input/PLAsTiCC-2018/training_set_metadata.csv'\n\n# Output directory\noutput_base_dir = '/kaggle/working/'\n\n# Load the CSV files\ntrain_df = pd.read_csv(train_data_path)\nmetadata_df = pd.read_csv(metadata_path)\n\n# Merge observation data with metadata on object_id\nmerged_df = train_df.merge(metadata_df, on='object_id')\n\n# Create output directories\ntrain_dir = os.path.join(output_base_dir, 'train')\ntest_dir = os.path.join(output_base_dir, 'test')\nos.makedirs(train_dir, exist_ok=True)\nos.makedirs(test_dir, exist_ok=True)\n\n# Get all unique object_ids\nunique_ids = metadata_df['object_id'].unique()\n\n# Split object_ids into train and test\ntrain_ids, test_ids = train_test_split(unique_ids, test_size=0.15, random_state=42)\n\n# Filter the full merged data accordingly\ntrain_merged = merged_df[merged_df['object_id'].isin(train_ids)]\ntest_merged = merged_df[merged_df['object_id'].isin(test_ids)]\n\n# Save metadata rows for train and test\ntrain_metadata = metadata_df[metadata_df['object_id'].isin(train_ids)]\ntest_metadata = metadata_df[metadata_df['object_id'].isin(test_ids)]\n\ntrain_metadata.to_csv(os.path.join(output_base_dir, 'train_labels.csv'), index=False)\ntest_metadata.to_csv(os.path.join(output_base_dir, 'test_labels.csv'), index=False)\n\n# Function to save light curve data\ndef save_light_curve_data(df, output_dir):\n    object_ids = df['object_id'].unique()\n    for obj_id in object_ids:\n        object_data = df[df['object_id'] == obj_id]\n        mjd_values = object_data['mjd'].values\n        flux_values = object_data['flux'].values\n        object_dir = os.path.join(output_dir, str(obj_id))\n        os.makedirs(object_dir, exist_ok=True)\n        lc_data = np.stack((mjd_values, flux_values), axis=-1)\n        np.save(os.path.join(object_dir, f\"{obj_id}.npy\"), lc_data)\n\n# Save light curve data\nsave_light_curve_data(train_merged, train_dir)\nsave_light_curve_data(test_merged, test_dir)\n\nprint(f\"Dataset has been organized. Training and testing data saved in {output_base_dir}.\")\n\n# Count and verify .npy files\ndef count_npy_files(directory):\n    count = 0\n    for root, dirs, files in os.walk(directory):\n        count += sum(1 for file in files if file.endswith('.npy'))\n    return count\n\ntrain_count = count_npy_files(train_dir)\ntest_count = count_npy_files(test_dir)\n\nprint(f\"Number of samples in train directory: {train_count}\")\nprint(f\"Number of samples in test directory: {test_count}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T20:13:32.478748Z","iopub.execute_input":"2025-05-13T20:13:32.479041Z","iopub.status.idle":"2025-05-13T20:13:47.482436Z","shell.execute_reply.started":"2025-05-13T20:13:32.47901Z","shell.execute_reply":"2025-05-13T20:13:47.481626Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%cd /kaggle/working/Deep-LC/src\nfrom deep_lc import LocalDataset\n# Paths to the directories containing .npy files\ntrain_lcs = '/kaggle/working/train'\ntest_lcs = '/kaggle/working/test'\n\n# Paths to your CSV label files\ntrain_labels = '/kaggle/working/train_labels.csv'\ntest_labels = '/kaggle/working/test_labels.csv'\n\n# You can also use metadata for params if needed, but LC-only is fine for now\ntrain_params = None\ntest_params = None\n# === Create LocalDataset objects for Deep-LC ===\ntraining_set = LocalDataset(train_lcs, train_params, train_labels)\ntest_set = LocalDataset(test_lcs, test_params, test_labels)\n\nprint(\"✅ Dataset loaded:\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T20:14:15.679738Z","iopub.execute_input":"2025-05-13T20:14:15.680476Z","iopub.status.idle":"2025-05-13T20:14:23.48186Z","shell.execute_reply.started":"2025-05-13T20:14:15.680446Z","shell.execute_reply":"2025-05-13T20:14:23.481233Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Load your training labels CSV\ntrain_labels_path = '/kaggle/working/train_labels.csv'\ntrain_df = pd.read_csv(train_labels_path)\n\n# Get sorted unique class IDs\nunique_class_ids = sorted(train_df['target'].unique())\n\n# Create label names like \"Class_6\", \"Class_15\", ...\nLABELS = [f\"Class_{cls_id}\" for cls_id in unique_class_ids]\n\n# Display the labels for confirmation\nprint(f\"LABELS ({len(LABELS)} classes):\")\nprint(LABELS)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T20:14:32.554626Z","iopub.execute_input":"2025-05-13T20:14:32.555205Z","iopub.status.idle":"2025-05-13T20:14:32.57032Z","shell.execute_reply.started":"2025-05-13T20:14:32.555181Z","shell.execute_reply":"2025-05-13T20:14:32.569596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from deep_lc.config import PROPOSAL_NUM, BATCH_SIZE, LR, LABELS\n\nhyper_parameters = {\n    \"batch_size\": BATCH_SIZE,\n    \"lr\": LR / 10,  # smaller LR for fine-tuning\n    \"weight_decay\": 0,\n    \"labels\": LABELS,  # Update if your classes differ\n    \"proposal_num\": PROPOSAL_NUM,\n}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T21:03:30.858Z","iopub.execute_input":"2025-05-13T21:03:30.858726Z","iopub.status.idle":"2025-05-13T21:03:30.863607Z","shell.execute_reply.started":"2025-05-13T21:03:30.858696Z","shell.execute_reply":"2025-05-13T21:03:30.862708Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# Explicitly set weights_only=False\nmodel_dict = torch.load(\n    \"/kaggle/input/combined/other/default/1/combined_17_conformal_calibrated.ckpt\",\n    map_location=device,\n    weights_only=False\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T21:05:56.701589Z","iopub.execute_input":"2025-05-13T21:05:56.702081Z","iopub.status.idle":"2025-05-13T21:05:56.842849Z","shell.execute_reply.started":"2025-05-13T21:05:56.702057Z","shell.execute_reply":"2025-05-13T21:05:56.842287Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from deep_lc import finetune\n\nfinetune(\n    training_set=training_set,\n    test_set=test_set,\n    hyper_params=hyper_parameters,\n    base_model=model_dict,\n    save_dir=\".\",\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T21:06:33.928761Z","iopub.execute_input":"2025-05-13T21:06:33.92947Z","iopub.status.idle":"2025-05-13T21:06:33.943035Z","shell.execute_reply.started":"2025-05-13T21:06:33.929447Z","shell.execute_reply":"2025-05-13T21:06:33.942124Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ckpt_path = \"/kaggle/input/combined/other/default/1/combined_17_conformal_calibrated.ckpt\"\nckpt = torch.load(ckpt_path, map_location=\"cpu\",weights_only=False)\n\nprint(ckpt.keys())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T20:55:59.553586Z","iopub.execute_input":"2025-05-13T20:55:59.553853Z","iopub.status.idle":"2025-05-13T20:55:59.6656Z","shell.execute_reply.started":"2025-05-13T20:55:59.553833Z","shell.execute_reply":"2025-05-13T20:55:59.664986Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\nseed = 42\n\nrandom.seed(seed)  # python random generator\nnp.random.seed(seed)  # numpy random generator\n\ntorch.manual_seed(seed)\ntorch.cuda.manual_seed_all(seed)\ntorch.backends.cudnn.deterministic = True\n\n\ndef seed_worker(worker_id):\n    worker_seed = torch.initial_seed() % 2**32\n    np.random.seed(worker_seed)\n    import random\n\n    random.seed(worker_seed)\n\n\ng = torch.Generator()\ng.manual_seed(seed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T21:02:09.045381Z","iopub.execute_input":"2025-05-13T21:02:09.045648Z","iopub.status.idle":"2025-05-13T21:02:09.053106Z","shell.execute_reply.started":"2025-05-13T21:02:09.045626Z","shell.execute_reply":"2025-05-13T21:02:09.052569Z"}},"outputs":[],"execution_count":null}]}