{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.17","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":10384,"databundleVersionId":120379,"sourceType":"competition"}],"dockerImageVersionId":31042,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:01:29.693054Z","iopub.execute_input":"2025-05-25T17:01:29.693413Z","iopub.status.idle":"2025-05-25T17:01:29.708021Z","shell.execute_reply.started":"2025-05-25T17:01:29.693389Z","shell.execute_reply":"2025-05-25T17:01:29.703653Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# PLAsTiCC Astronomical Classification - Phase 1\n# EC452 Machine Learning Semester Project\n!pip install lightgbm\n\n!pip install xgboost\n\n# Import libraries\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import log_loss, classification_report, confusion_matrix, ConfusionMatrixDisplay\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler\nfrom sklearn.ensemble import RandomForestClassifier\nfrom lightgbm import LGBMClassifier\nfrom xgboost import XGBClassifier\nfrom sklearn.impute import SimpleImputer\nimport joblib\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Load data from Kaggle dataset path\nmeta_path = '/kaggle/input/PLAsTiCC-2018/training_set_metadata.csv'\nlc_path = '/kaggle/input/PLAsTiCC-2018/training_set.csv'\n\ntrain_meta = pd.read_csv(meta_path)\ntrain_lc = pd.read_csv(lc_path)\n\n# Data Exploration\nprint(\"\\nMetadata shape:\", train_meta.shape)\nprint(\"Light curves shape:\", train_lc.shape)\n\n# Enhanced Feature Engineering\ndef create_features(df):\n    grouped = df.groupby('object_id')\n\n    # Basic flux features\n    features = grouped['flux'].agg(['mean', 'std', 'min', 'max', 'skew', 'median']).add_prefix('flux_')\n    features['flux_range'] = features['flux_max'] - features['flux_min']\n    features['flux_ratio'] = (features['flux_max'] - features['flux_mean']) / (features['flux_mean'] - features['flux_min'] + 1e-6)\n\n    # Passband-specific features\n    passbands = sorted(df['passband'].unique())\n    for band in passbands:\n        band_group = df[df['passband'] == band].groupby('object_id')\n        features[f'flux_mean_band_{band}'] = band_group['flux'].mean()\n        features[f'flux_std_band_{band}'] = band_group['flux'].std()\n        features[f'flux_amp_band_{band}'] = band_group['flux'].max() - band_group['flux'].min()\n\n    # Time features\n    features['mjd_span'] = grouped['mjd'].max() - grouped['mjd'].min()\n    features['mjd_density'] = grouped.size() / (features['mjd_span'] + 1e-6)\n\n    # Detection features\n    features['detection_rate'] = grouped['detected'].mean()\n    features['detected_count'] = grouped['detected'].sum()\n\n    # Flux error features\n    features['flux_err_ratio'] = grouped['flux_err'].mean() / (grouped['flux'].std() + 1e-6)\n\n    return features\n\nprint(\"\\nCreating features...\")\ntrain_features = create_features(train_lc)\n\n# Merge data\ntrain = train_meta.merge(train_features, left_on='object_id', right_index=True, how='left')\n\n# Handle missing values\nnum_imputer = SimpleImputer(strategy='median')\nnumerical_cols = train.select_dtypes(include=np.number).columns\ntrain[numerical_cols] = num_imputer.fit_transform(train[numerical_cols])\n\n# Feature selection\ntrain = train.drop(['object_id'], axis=1)\n\n# New features\ntrain['hostgal_photoz_uncertainty'] = train['hostgal_photoz_err'] / (train['hostgal_photoz'] + 1e-6)\ntrain['specz_photoz_diff'] = np.abs(train['hostgal_specz'] - train['hostgal_photoz'])\n\n# Prepare data\nX = train.drop('target', axis=1)\ny = train['target']\n\n# Encode target\nle = LabelEncoder()\ny_encoded = le.fit_transform(y)\nn_classes = len(le.classes_)\n\n# Class weights\nclass_counts = np.bincount(y_encoded)\nclass_weights = {i: sum(class_counts)/class_counts[i] for i in range(n_classes)}\nsample_weights = np.array([class_weights[y] for y in y_encoded])\n\n# Train-test split\nX_train, X_val, y_train, y_val, weights_train, _ = train_test_split(\n    X, y_encoded, sample_weights, test_size=0.15,\n    stratify=y_encoded, random_state=42\n)\n\n# Feature scaling\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_val_scaled = scaler.transform(X_val)\n\n# Model configuration\nmodels = {\n    'Random Forest': RandomForestClassifier(\n        n_estimators=300,\n        class_weight='balanced',\n        n_jobs=-1,\n        random_state=42\n    ),\n\n    'XGBoost': XGBClassifier(\n        n_estimators=1000,\n        learning_rate=0.05,\n        objective='multi:softprob',\n        num_class=n_classes,\n        random_state=42,\n        n_jobs=-1\n    )\n}\n\n# Train models\nresults = {}\nfor name, model in models.items():\n    print(f\"\\nTraining {name}...\")\n    if name == 'XGBoost':\n        model.fit(X_train_scaled, y_train,\n                  sample_weight=weights_train,\n                  eval_set=[(X_val_scaled, y_val)],\n                  verbose=False)\n    else:\n        model.fit(X_train_scaled, y_train,\n                  sample_weight=weights_train)\n\n    probs = model.predict_proba(X_val_scaled)\n    loss = log_loss(y_val, probs)\n    results[name] = loss\n    print(f\"{name} Validation Log Loss: {loss:.4f}\")\n\n# Results\nprint(\"\\nModel Performance:\")\nfor name, loss in sorted(results.items(), key=lambda x: x[1]):\n    print(f\"{name}: {loss:.4f}\")\n\n# Feature Importance\nbest_model_name = min(results, key=results.get)\nbest_model = models[best_model_name]\n\nif hasattr(best_model, 'feature_importances_'):\n    importances = pd.Series(best_model.feature_importances_, index=X.columns)\n    plt.figure(figsize=(12, 8))\n    importances.sort_values().tail(20).plot.barh()\n    plt.title(f'{best_model_name} Feature Importance')\n    plt.show()\n\n# Save artifacts (optional in Kaggle environment, paths can be adjusted)\njoblib.dump(best_model, f'best_{best_model_name.lower()}_model.pkl')\njoblib.dump(scaler, 'feature_scaler.pkl')\njoblib.dump(le, 'label_encoder.pkl')\n\n# Confusion Matrix\ny_pred = best_model.predict(X_val_scaled)\ncm = confusion_matrix(y_val, y_pred)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=le.classes_.astype(str))\nfig, ax = plt.subplots(figsize=(15, 15))\ndisp.plot(ax=ax, xticks_rotation=90)\nplt.title('Confusion Matrix')\nplt.show()\n\nprint(\"\\nClassification Report:\")\nprint(classification_report(y_val, y_pred, target_names=le.classes_.astype(str)))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:01:29.710278Z","iopub.execute_input":"2025-05-25T17:01:29.710668Z","iopub.status.idle":"2025-05-25T17:03:01.312424Z","shell.execute_reply.started":"2025-05-25T17:01:29.710645Z","shell.execute_reply":"2025-05-25T17:03:01.306714Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport glob\nimport gc\nimport pandas as pd\nimport numpy as np\n\n# Load test metadata\ntest_meta = pd.read_csv('/kaggle/input/PLAsTiCC-2018/test_set_metadata.csv')\n\n# Collect ONLY the 11 test batch files (exclude test_set.csv)\nbatch_files = sorted(glob.glob('/kaggle/input/PLAsTiCC-2018/test_set_batch*.csv'))\n\n# List to hold predictions\nsubmission_parts = []\n\n# Features used in training\ntrained_features = X.columns.tolist()\n\n# Prediction function\ndef process_test_batch(batch_df):\n    # Create features for this batch's light curves\n    test_features = create_features(batch_df)\n\n    # Select metadata rows for these object_ids\n    meta = test_meta[test_meta['object_id'].isin(test_features.index)].copy()\n    meta['hostgal_photoz_uncertainty'] = meta['hostgal_photoz_err'] / (meta['hostgal_photoz'] + 1e-6)\n    meta['specz_photoz_diff'] = np.abs(meta['hostgal_specz'] - meta['hostgal_photoz'])\n\n    # Merge features and metadata\n    test_full = meta.merge(test_features, left_on='object_id', right_index=True, how='left')\n\n    # Fill missing values\n    test_full = test_full.fillna(0)\n\n    # Add missing columns to match training features\n    for col in trained_features:\n        if col not in test_full.columns:\n            test_full[col] = 0\n\n    # Reorder columns to match training features\n    test_full = test_full[trained_features]\n\n    object_ids = meta['object_id']\n    test_scaled = scaler.transform(test_full)\n\n    # Predict probabilities\n    preds = best_model.predict_proba(test_scaled)\n\n    # Build predictions DataFrame\n    df_preds = pd.DataFrame(preds, columns=le.classes_.astype(str))\n    df_preds.insert(0, 'object_id', object_ids.values)\n\n    # Add the required '99' class with zero probabilities\n    df_preds['99'] = 0.0\n\n    return df_preds\n\nprint(\"\\n🚀 Starting prediction on the 11 test batches only...\\n\")\n\nfor i, batch_file in enumerate(batch_files):\n    print(f\"Processing: {os.path.basename(batch_file)} ({i+1}/{len(batch_files)})\")\n    batch_df = pd.read_csv(batch_file)\n    batch_preds = process_test_batch(batch_df)\n    submission_parts.append(batch_preds)\n    del batch_df, batch_preds\n    gc.collect()\n\n# Combine predictions from all batches\nfinal_submission = pd.concat(submission_parts, ignore_index=True)\n\n# Save submission file\nfinal_submission.to_csv('submission.csv', index=False)\nprint(\"\\n✅ Saved submission.csv with shape:\", final_submission.shape)\n\n# Fix column names to match sample_submission format\n# Convert '6.0' → 'class_6'\nfinal_submission.columns = ['object_id'] + [f'class_{int(float(c))}' for c in final_submission.columns[1:]]\n\n# Reorder columns to exactly match sample_submission.csv\nsample_sub_path = '/kaggle/input/PLAsTiCC-2018/sample_submission.csv'\nsample_cols = pd.read_csv(sample_sub_path, nrows=1).columns.tolist()\nfinal_submission = final_submission[sample_cols]\n\n# Save corrected submission\nfinal_submission.to_csv('submission.csv', index=False)\nprint(\"\\n✅ Final submission.csv saved with correct column names and order.\")\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:03:01.315891Z","iopub.execute_input":"2025-05-25T17:03:01.316181Z","iopub.status.idle":"2025-05-25T17:14:26.757857Z","shell.execute_reply.started":"2025-05-25T17:03:01.316156Z","shell.execute_reply":"2025-05-25T17:14:26.750791Z"}},"outputs":[],"execution_count":null}]}