{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":10384,"databundleVersionId":120379,"sourceType":"competition"},{"sourceId":11950386,"sourceType":"datasetVersion","datasetId":7513077},{"sourceId":11957008,"sourceType":"datasetVersion","datasetId":7517869}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\n\n# Cell 1: Environment Setup and Configuration\nimport pandas as pd\nimport numpy as np\nimport lightgbm as lgb\nimport pickle\nimport gc\nimport os\nfrom datetime import datetime\nimport warnings\nwarnings.filterwarnings('ignore')\n\nprint(\"=== ENHANCED PLAsTiCC LIGHTGBM TESTING NOTEBOOK ===\")\nprint(f\"Started at: {datetime.now()}\")\nprint(\"🎯 Focus: Using your trained LightGBM models for maximum scores\")\n\n# CRITICAL: Use correct PLAsTiCC class names for submission\nCORRECT_PLASTICC_CLASSES = [\n    'class_6', 'class_15', 'class_16', 'class_42', 'class_52', \n    'class_53', 'class_62', 'class_64', 'class_65', 'class_67', \n    'class_88', 'class_90', 'class_92', 'class_95', 'class_99'\n]\n\n# Configuration\nBATCH_SIZE = 100000\nSAVE_INTERVAL = 5\nOUTPUT_DIR = \"/kaggle/working/\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T11:52:02.024329Z","iopub.execute_input":"2025-05-26T11:52:02.025170Z","iopub.status.idle":"2025-05-26T11:52:08.772140Z","shell.execute_reply.started":"2025-05-26T11:52:02.025133Z","shell.execute_reply":"2025-05-26T11:52:08.771333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n# Cell 2: Enhanced Environment Check\ndef enhanced_environment_check():\n    \"\"\"Enhanced environment validation with file analysis\"\"\"\n    print(\"\\n=== ENHANCED ENVIRONMENT CHECK ===\")\n    \n    # Check LightGBM models dataset\n    lightgbm_dir = \"/kaggle/input/ml-dataset-model/lightgbm_models\"\n    if os.path.exists(lightgbm_dir):\n        print(f\"📁 LightGBM models directory found: {lightgbm_dir}\")\n        for item in os.listdir(lightgbm_dir):\n            file_path = os.path.join(lightgbm_dir, item)\n            if os.path.isfile(file_path):\n                size_mb = os.path.getsize(file_path) / (1024*1024)\n                print(f\"🔧 Model file: {item} ({size_mb:.1f} MB)\")\n    else:\n        print(\"❌ LightGBM models directory not found!\")\n        return [], []\n    \n    # Find model and component files\n    model_files = []\n    component_files = []\n    \n    for root, dirs, files in os.walk(\"/kaggle/input\"):\n        for file in files:\n            full_path = os.path.join(root, file)\n            if 'lightgbm' in root.lower() or 'lgb' in file.lower():\n                if file.endswith(('.txt', '.pkl')):\n                    if 'model' in file.lower():\n                        model_files.append(full_path)\n                        print(f\"🤖 Model found: {full_path}\")\n                    else:\n                        component_files.append(full_path)\n                        print(f\"🔧 Component found: {full_path}\")\n    \n    # Check PLAsTiCC data files\n    plasticc_files = []\n    for root, dirs, files in os.walk(\"/kaggle/input\"):\n        for file in files:\n            if 'test_set' in file and file.endswith('.csv'):\n                full_path = os.path.join(root, file)\n                size_mb = os.path.getsize(full_path) / (1024*1024)\n                plasticc_files.append(full_path)\n                print(f\"📊 Test data: {file} ({size_mb:.1f} MB)\")\n    \n    return model_files, component_files, plasticc_files\n\nmodel_files, component_files, plasticc_files = enhanced_environment_check()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T11:53:43.471421Z","iopub.execute_input":"2025-05-26T11:53:43.471730Z","iopub.status.idle":"2025-05-26T11:53:43.545798Z","shell.execute_reply.started":"2025-05-26T11:53:43.471708Z","shell.execute_reply":"2025-05-26T11:53:43.545067Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Cell 3: Robust Model and Component Loading\ndef load_lightgbm_models_robust():\n    \"\"\"Load LightGBM models and components with multiple fallback strategies\"\"\"\n    print(\"\\n=== ROBUST LIGHTGBM MODEL LOADING ===\")\n    \n    models_dir = \"/kaggle/input/ml-dataset-model/lightgbm_models\"\n    \n    # Load final model\n    final_model = None\n    final_model_path = os.path.join(models_dir, \"final_model.txt\")\n    if os.path.exists(final_model_path):\n        try:\n            final_model = lgb.Booster(model_file=final_model_path)\n            print(f\"✅ Final model loaded: {final_model_path}\")\n        except Exception as e:\n            print(f\"❌ Failed to load final model: {e}\")\n    \n    # Load CV models for ensemble\n    cv_models = []\n    for i in range(5):\n        cv_model_path = os.path.join(models_dir, f\"cv_model_{i}.txt\")\n        if os.path.exists(cv_model_path):\n            try:\n                cv_model = lgb.Booster(model_file=cv_model_path)\n                cv_models.append(cv_model)\n                print(f\"✅ CV model {i} loaded: {cv_model_path}\")\n            except Exception as e:\n                print(f\"❌ Failed to load CV model {i}: {e}\")\n    \n    # Load label encoder\n    label_encoder = None\n    label_encoder_path = os.path.join(models_dir, \"label_encoder.pkl\")\n    if os.path.exists(label_encoder_path):\n        try:\n            with open(label_encoder_path, 'rb') as f:\n                label_encoder = pickle.load(f)\n            print(f\"✅ Label encoder loaded: {label_encoder_path}\")\n        except Exception as e:\n            print(f\"❌ Failed to load label encoder: {e}\")\n    \n    # Load feature columns\n    feature_columns = None\n    feature_columns_path = os.path.join(models_dir, \"feature_columns.pkl\")\n    if os.path.exists(feature_columns_path):\n        try:\n            with open(feature_columns_path, 'rb') as f:\n                feature_columns = pickle.load(f)\n            print(f\"✅ Feature columns loaded: {len(feature_columns)} features\")\n        except Exception as e:\n            print(f\"❌ Failed to load feature columns: {e}\")\n    \n    # Load model metrics\n    model_metrics = None\n    metrics_path = os.path.join(models_dir, \"model_metrics.pkl\")\n    if os.path.exists(metrics_path):\n        try:\n            with open(metrics_path, 'rb') as f:\n                model_metrics = pickle.load(f)\n            print(f\"✅ Model metrics loaded\")\n            print(f\"   Training CV accuracy: {model_metrics.get('cv_accuracy', 'Unknown')}\")\n            print(f\"   Training CV log loss: {model_metrics.get('cv_log_loss', 'Unknown')}\")\n        except Exception as e:\n            print(f\"❌ Failed to load model metrics: {e}\")\n    \n    print(f\"\\n✅ Loading Summary:\")\n    print(f\"   Final model: {'✓' if final_model else '❌'}\")\n    print(f\"   CV models: {len(cv_models)}/5\")\n    print(f\"   Label encoder: {'✓' if label_encoder else '❌'}\")\n    print(f\"   Feature columns: {'✓' if feature_columns else '❌'}\")\n    print(f\"   Model metrics: {'✓' if model_metrics else '❌'}\")\n    \n    return final_model, cv_models, label_encoder, feature_columns, model_metrics\n\n# Load all components\nfinal_model, cv_models, label_encoder, feature_columns, model_metrics = load_lightgbm_models_robust()\n\n# Cell 4: LightGBM Feature Engineering (Matching Training)\ndef add_derived_features(df):\n    \"\"\"Add derived features to light curve data - SAME AS TRAINING\"\"\"\n    df = df.copy()\n    \n    # Basic derived features\n    df['flux_ratio_sq'] = np.power(df['flux'] / df['flux_err'], 2.0)\n    df['flux_by_flux_ratio_sq'] = df['flux'] * df['flux_ratio_sq']\n    \n    # Detected flux features\n    df['flux_diff'] = df.groupby(['object_id', 'passband'])['flux'].diff()\n    df['flux_diff2'] = df.groupby(['object_id', 'passband'])['flux_diff'].diff()\n    \n    # Time-based features\n    df['mjd_diff'] = df.groupby(['object_id', 'passband'])['mjd'].diff()\n    df['mjd_diff'].fillna(0, inplace=True)\n    \n    # Flux rate features\n    df['flux_rate'] = df['flux_diff'] / (df['mjd_diff'] + 1e-8)\n    df['flux_rate'].replace([np.inf, -np.inf], 0, inplace=True)\n    \n    # Detection features\n    df['detected_bool'] = (df['detected'] == 1).astype(int)\n    \n    return df\n\ndef extract_time_series_features(group):\n    \"\"\"Extract comprehensive time-series features - SAME AS TRAINING\"\"\"\n    features = {}\n    \n    # Basic statistics\n    features['count'] = len(group)\n    features['mean'] = group['flux'].mean()\n    features['std'] = group['flux'].std()\n    features['min'] = group['flux'].min()\n    features['max'] = group['flux'].max()\n    features['median'] = group['flux'].median()\n    \n    # Advanced statistics\n    features['skew'] = pd.Series(group['flux']).skew()\n    features['kurtosis'] = pd.Series(group['flux']).kurtosis()\n    features['mad'] = np.median(np.abs(group['flux'] - features['median']))\n    \n    # Percentiles\n    features['q25'] = np.percentile(group['flux'], 25)\n    features['q75'] = np.percentile(group['flux'], 75)\n    features['iqr'] = features['q75'] - features['q25']\n    \n    # Range and amplitude features\n    features['range'] = features['max'] - features['min']\n    features['amplitude'] = features['range'] / 2\n    features['beyond_1std'] = np.sum(np.abs(group['flux'] - features['mean']) > features['std']) / len(group)\n    \n    # Time-based features\n    if len(group) > 1:\n        features['time_span'] = group['mjd'].max() - group['mjd'].min()\n        features['time_mean'] = group['mjd'].mean()\n        features['time_std'] = group['mjd'].std()\n        \n        # Peak detection\n        peak_idx = group['flux'].idxmax()\n        features['peak_mjd'] = group.loc[peak_idx, 'mjd']\n        features['peak_flux'] = group.loc[peak_idx, 'flux']\n        \n        # Rise and decline features\n        pre_peak = group[group['mjd'] <= features['peak_mjd']]\n        post_peak = group[group['mjd'] > features['peak_mjd']]\n        \n        if len(pre_peak) > 1:\n            features['rise_time'] = features['peak_mjd'] - pre_peak['mjd'].min()\n            features['rise_slope'] = (features['peak_flux'] - pre_peak['flux'].iloc[0]) / (features['rise_time'] + 1e-8)\n        else:\n            features['rise_time'] = 0\n            features['rise_slope'] = 0\n            \n        if len(post_peak) > 1:\n            features['decline_time'] = post_peak['mjd'].max() - features['peak_mjd']\n            features['decline_slope'] = (post_peak['flux'].iloc[-1] - features['peak_flux']) / (features['decline_time'] + 1e-8)\n        else:\n            features['decline_time'] = 0\n            features['decline_slope'] = 0\n    else:\n        features['time_span'] = 0\n        features['time_mean'] = group['mjd'].iloc[0] if len(group) > 0 else 0\n        features['time_std'] = 0\n        features['peak_mjd'] = group['mjd'].iloc[0] if len(group) > 0 else 0\n        features['peak_flux'] = group['flux'].iloc[0] if len(group) > 0 else 0\n        features['rise_time'] = 0\n        features['rise_slope'] = 0\n        features['decline_time'] = 0\n        features['decline_slope'] = 0\n    \n    # Error-based features\n    features['mean_err'] = group['flux_err'].mean()\n    features['std_err'] = group['flux_err'].std()\n    features['snr_mean'] = features['mean'] / features['mean_err']\n    features['snr_max'] = features['max'] / group['flux_err'].min()\n    \n    # Derived features statistics\n    if 'flux_ratio_sq' in group.columns:\n        features['flux_ratio_sq_sum'] = group['flux_ratio_sq'].sum()\n        features['flux_ratio_sq_mean'] = group['flux_ratio_sq'].mean()\n    \n    if 'flux_diff' in group.columns:\n        flux_diff_clean = group['flux_diff'].dropna()\n        if len(flux_diff_clean) > 0:\n            features['flux_diff_std'] = flux_diff_clean.std()\n            features['flux_diff_mean'] = flux_diff_clean.mean()\n            features['flux_diff_max'] = flux_diff_clean.max()\n            features['flux_diff_min'] = flux_diff_clean.min()\n        else:\n            features['flux_diff_std'] = 0\n            features['flux_diff_mean'] = 0\n            features['flux_diff_max'] = 0\n            features['flux_diff_min'] = 0\n    \n    # Detection features\n    features['detected_ratio'] = group['detected'].mean()\n    features['detected_count'] = group['detected'].sum()\n    \n    return pd.Series(features)\n\ndef create_lightgbm_features(lc_df, meta_df):\n    \"\"\"Create features exactly matching your LightGBM training\"\"\"\n    print(f\"    🔧 Creating LightGBM features for {len(meta_df)} objects...\")\n    \n    # Add derived features\n    lc_df = add_derived_features(lc_df)\n    \n    # Extract features for each object-passband combination\n    ts_features = lc_df.groupby(['object_id', 'passband']).apply(extract_time_series_features).reset_index()\n    \n    # Pivot to get features for each passband as separate columns\n    feature_cols = [col for col in ts_features.columns if col not in ['object_id', 'passband']]\n    \n    pivoted_features = []\n    for pb in range(6):  # 6 passbands (0-5)\n        pb_data = ts_features[ts_features['passband'] == pb].copy()\n        if len(pb_data) > 0:\n            pb_data = pb_data.drop('passband', axis=1)\n            \n            # Rename columns to include passband\n            new_cols = {'object_id': 'object_id'}\n            for col in feature_cols:\n                new_cols[col] = f'{col}_pb{pb}'\n            pb_data = pb_data.rename(columns=new_cols)\n            \n            pivoted_features.append(pb_data)\n    \n    # Merge all passband features\n    if pivoted_features:\n        final_features = pivoted_features[0]\n        for i in range(1, len(pivoted_features)):\n            final_features = final_features.merge(pivoted_features[i], on='object_id', how='outer')\n    else:\n        final_features = pd.DataFrame({'object_id': meta_df['object_id'].values})\n    \n    # Fill missing values\n    final_features = final_features.fillna(0)\n    \n    # Add cross-passband features\n    passband_pairs = [(0,1), (1,2), (2,3), (3,4), (4,5), (0,2), (1,3), (2,4), (3,5)]\n    for pb1, pb2 in passband_pairs:\n        if f'mean_pb{pb1}' in final_features.columns and f'mean_pb{pb2}' in final_features.columns:\n            final_features[f'color_{pb1}_{pb2}'] = final_features[f'mean_pb{pb1}'] - final_features[f'mean_pb{pb2}']\n            final_features[f'color_ratio_{pb1}_{pb2}'] = final_features[f'mean_pb{pb1}'] / (final_features[f'mean_pb{pb2}'] + 1e-8)\n    \n    # Global object features\n    count_cols = [f'count_pb{i}' for i in range(6) if f'count_pb{i}' in final_features.columns]\n    if count_cols:\n        final_features['total_observations'] = final_features[count_cols].sum(axis=1)\n        final_features['active_passbands'] = (final_features[count_cols] > 0).sum(axis=1)\n    \n    mean_cols = [f'mean_pb{i}' for i in range(6) if f'mean_pb{i}' in final_features.columns]\n    if mean_cols:\n        final_features['flux_mean_all'] = final_features[mean_cols].mean(axis=1)\n        final_features['flux_std_all'] = final_features[mean_cols].std(axis=1)\n    \n    max_cols = [f'max_pb{i}' for i in range(6) if f'max_pb{i}' in final_features.columns]\n    min_cols = [f'min_pb{i}' for i in range(6) if f'min_pb{i}' in final_features.columns]\n    if max_cols and min_cols:\n        final_features['flux_max_all'] = final_features[max_cols].max(axis=1)\n        final_features['flux_min_all'] = final_features[min_cols].min(axis=1)\n    \n    peak_mjd_cols = [f'peak_mjd_pb{i}' for i in range(6) if f'peak_mjd_pb{i}' in final_features.columns]\n    if peak_mjd_cols:\n        final_features['peak_mjd_range'] = final_features[peak_mjd_cols].max(axis=1) - final_features[peak_mjd_cols].min(axis=1)\n    \n    # Merge with metadata\n    final_features = final_features.merge(meta_df, on='object_id', how='left')\n    \n    print(f\"    ✅ LightGBM features created: {final_features.shape}\")\n    return final_features\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T11:54:52.432948Z","iopub.execute_input":"2025-05-26T11:54:52.433284Z","iopub.status.idle":"2025-05-26T11:54:52.977996Z","shell.execute_reply.started":"2025-05-26T11:54:52.433261Z","shell.execute_reply":"2025-05-26T11:54:52.977287Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 5: Smart Feature Alignment for LightGBM\ndef align_features_for_lightgbm(features_df, feature_columns):\n    \"\"\"Align features with LightGBM training expectations\"\"\"\n    print(f\"    🎯 Aligning features for LightGBM...\")\n    \n    # Remove object_id for processing\n    X = features_df.drop('object_id', axis=1, errors='ignore')\n    current_features = list(X.columns)\n    \n    print(f\"    Current features: {len(current_features)}\")\n    print(f\"    Expected features: {len(feature_columns) if feature_columns else 'Unknown'}\")\n    \n    if feature_columns is None:\n        print(f\"    ⚠️ No feature columns specification - using current features\")\n        return X\n    \n    # Create aligned feature matrix\n    X_aligned = pd.DataFrame(index=X.index)\n    \n    for col in feature_columns:\n        if col in current_features:\n            X_aligned[col] = X[col]\n        else:\n            X_aligned[col] = 0.0  # Fill missing features with 0\n            \n    print(f\"    ✅ Features aligned: {X_aligned.shape}\")\n    return X_aligned\n\n# Cell 6: Enhanced Prediction with LightGBM Ensemble\ndef make_lightgbm_predictions(features_df, final_model, cv_models, label_encoder, feature_columns):\n    \"\"\"Make predictions using LightGBM models with ensemble\"\"\"\n    print(f\"    🎯 Making LightGBM predictions for {len(features_df)} objects...\")\n    \n    # Align features\n    X_aligned = align_features_for_lightgbm(features_df, feature_columns)\n    \n    # Replace any remaining NaN/inf values\n    X_aligned = X_aligned.fillna(0).replace([np.inf, -np.inf], 0)\n    \n    predictions = None\n    \n    # Try ensemble prediction first (if we have CV models)\n    if len(cv_models) > 0:\n        print(f\"    🔄 Using ensemble of {len(cv_models)} CV models...\")\n        try:\n            ensemble_predictions = []\n            for i, cv_model in enumerate(cv_models):\n                pred = cv_model.predict(X_aligned, num_iteration=cv_model.best_iteration)\n                ensemble_predictions.append(pred)\n            \n            # Average the predictions\n            predictions = np.mean(ensemble_predictions, axis=0)\n            print(f\"    ✅ Ensemble predictions generated: {predictions.shape}\")\n            \n        except Exception as e:\n            print(f\"    ⚠️ Ensemble prediction failed: {e}\")\n            predictions = None\n    \n    # Fallback to final model\n    if predictions is None and final_model is not None:\n        print(f\"    🔄 Using final model...\")\n        try:\n            predictions = final_model.predict(X_aligned, num_iteration=final_model.best_iteration)\n            print(f\"    ✅ Final model predictions generated: {predictions.shape}\")\n        except Exception as e:\n            print(f\"    ❌ Final model prediction failed: {e}\")\n            predictions = None\n    \n    # Ultimate fallback - uniform probabilities\n    if predictions is None:\n        print(f\"    ⚠️ Using uniform fallback probabilities\")\n        n_classes = len(label_encoder.classes_) if label_encoder else 14\n        predictions = np.full((len(features_df), n_classes), 1.0/n_classes)\n    \n    # Handle class mapping\n    if label_encoder is not None:\n        if predictions.shape[1] == len(label_encoder.classes_):\n            # Perfect match\n            class_names = [f'class_{cls}' for cls in label_encoder.classes_]\n        elif predictions.shape[1] == 14 and len(label_encoder.classes_) == 14:\n            # 14 classes, add class_99 for PLAsTiCC\n            padding_prob = 0.001\n            padded_probs = np.column_stack([\n                predictions * (1 - padding_prob),\n                np.full(len(predictions), padding_prob)\n            ])\n            predictions = padded_probs\n            class_names = [f'class_{cls}' for cls in label_encoder.classes_] + ['class_99']\n        else:\n            print(f\"    ⚠️ Class count mismatch: model={predictions.shape[1]}, encoder={len(label_encoder.classes_)}\")\n            class_names = CORRECT_PLASTICC_CLASSES[:predictions.shape[1]]\n    else:\n        # No label encoder - use default PLAsTiCC classes\n        class_names = CORRECT_PLASTICC_CLASSES[:predictions.shape[1]]\n    \n    # Ensure we have exactly 15 classes for PLAsTiCC\n    if len(class_names) < 15:\n        # Pad with remaining classes\n        missing_classes = CORRECT_PLASTICC_CLASSES[len(class_names):]\n        class_names.extend(missing_classes)\n        \n        # Pad predictions\n        padding = np.full((len(predictions), len(missing_classes)), 0.001)\n        predictions = np.column_stack([predictions, padding])\n    \n    # Create prediction dataframe\n    pred_df = pd.DataFrame(predictions, columns=CORRECT_PLASTICC_CLASSES)\n    pred_df['object_id'] = features_df['object_id'].values\n    \n    # Normalize probabilities to sum to 1\n    prob_cols = CORRECT_PLASTICC_CLASSES\n    row_sums = pred_df[prob_cols].sum(axis=1)\n    \n    # Handle zero sums\n    zero_sum_mask = row_sums == 0\n    if zero_sum_mask.sum() > 0:\n        print(f\"    ⚠️ Fixed {zero_sum_mask.sum()} rows with zero probabilities\")\n        pred_df.loc[zero_sum_mask, prob_cols] = 1.0 / len(prob_cols)\n        row_sums = pred_df[prob_cols].sum(axis=1)\n    \n    # Normalize\n    for col in prob_cols:\n        pred_df[col] = pred_df[col] / row_sums\n    \n    # Verify normalization\n    final_sums = pred_df[prob_cols].sum(axis=1)\n    assert np.allclose(final_sums, 1.0, atol=1e-6), \"Probabilities don't sum to 1!\"\n    \n    # Reorder columns\n    final_cols = ['object_id'] + CORRECT_PLASTICC_CLASSES\n    pred_df = pred_df[final_cols]\n    \n    print(f\"    ✅ LightGBM predictions ready: {pred_df.shape}\")\n    return pred_df\n\n# Cell 7: Main Processing Function\ndef process_test_data_lightgbm():\n    \"\"\"Process test data using LightGBM models\"\"\"\n    \n    if final_model is None and len(cv_models) == 0:\n        print(\"❌ No LightGBM models available for processing\")\n        return None\n    \n    # Load test metadata\n    print(\"\\\\n📊 Loading test metadata...\")\n    test_meta = pd.read_csv(\"/kaggle/input/PLAsTiCC-2018/test_set_metadata.csv\")\n    print(f\"✅ Test metadata loaded: {test_meta.shape}\")\n    \n    # Processing configuration\n    total_objects = len(test_meta)\n    num_batches = (total_objects + BATCH_SIZE - 1) // BATCH_SIZE\n    \n    print(f\"\\\\n=== LIGHTGBM PROCESSING CONFIGURATION ===\")\n    print(f\"📊 Total objects: {total_objects:,}\")\n    print(f\"📦 Batch size: {BATCH_SIZE:,}\")\n    print(f\"🔄 Number of batches: {num_batches}\")\n    print(f\"⏱️ Estimated time: {num_batches * 2:.0f}-{num_batches * 4:.0f} minutes\")\n    print(f\"🤖 Models available: Final={'✓' if final_model else '❌'}, CV={len(cv_models)}\")\n    \n    all_predictions = []\n    successful_batches = 0\n    \n    # Process each batch\n    for batch_num in range(num_batches):\n        start_time = datetime.now()\n        print(f\"\\\\n--- LIGHTGBM BATCH {batch_num + 1}/{num_batches} ---\")\n        \n        try:\n            # Get batch metadata\n            start_idx = batch_num * BATCH_SIZE\n            end_idx = min((batch_num + 1) * BATCH_SIZE, total_objects)\n            batch_meta = test_meta.iloc[start_idx:end_idx].copy()\n            batch_obj_ids = set(batch_meta['object_id'].values)\n            \n            print(f\"🎯 Processing objects {start_idx:,} to {end_idx-1:,}\")\n            print(f\"📝 Batch contains {len(batch_obj_ids):,} unique object IDs\")\n            \n            # Load light curves efficiently (simplified for memory)\n            print(\"    📡 Loading light curves...\")\n            batch_lc_list = []\n            \n            # Only check the main test files that are most likely to contain data\n            test_files = [\n                \"/kaggle/input/PLAsTiCC-2018/test_set.csv\",\n            ]\n            \n            # Add batch files if they exist\n            for i in range(1, 12):\n                batch_file = f\"/kaggle/input/PLAsTiCC-2018/test_set_{i:02d}.csv\"\n                if os.path.exists(batch_file):\n                    test_files.append(batch_file)\n            \n            total_obs = 0\n            for lc_file in test_files[:3]:  # Limit to first 3 files for memory\n                if not os.path.exists(lc_file):\n                    continue\n                \n                try:\n                    # Read in small chunks\n                    chunk_size = 200000\n                    for chunk in pd.read_csv(lc_file, chunksize=chunk_size):\n                        if 'object_id' in chunk.columns:\n                            relevant_data = chunk[chunk['object_id'].isin(batch_obj_ids)]\n                            if len(relevant_data) > 0:\n                                batch_lc_list.append(relevant_data)\n                                total_obs += len(relevant_data)\n                        \n                        # Early stopping to prevent memory issues\n                        if total_obs > 500000:\n                            break\n                    \n                    if total_obs > 500000:\n                        print(f\"    ⚡ Early stop - sufficient data loaded ({total_obs:,} obs)\")\n                        break\n                        \n                except Exception as e:\n                    print(f\"    ⚠️ Error reading {lc_file}: {e}\")\n                    continue\n            \n            # Combine light curve data\n            if batch_lc_list:\n                batch_lc = pd.concat(batch_lc_list, ignore_index=True)\n                print(f\"    ✅ Light curves combined: {len(batch_lc):,} observations\")\n                print(f\"    📊 Unique objects in LC: {batch_lc['object_id'].nunique():,}\")\n            else:\n                print(f\"    ⚠️ No light curves found - using metadata only\")\n                batch_lc = pd.DataFrame()\n            \n            # Create features\n            if len(batch_lc) > 0:\n                features_df = create_lightgbm_features(batch_lc, batch_meta)\n            else:\n                # Create minimal features from metadata only\n                features_df = batch_meta.copy()\n                # Add dummy light curve features\n                for pb in range(6):\n                    for feat in ['count', 'mean', 'std', 'min', 'max']:\n                        features_df[f'{feat}_pb{pb}'] = 0.0\n            \n            # Make predictions\n            pred_df = make_lightgbm_predictions(features_df, final_model, cv_models, label_encoder, feature_columns)\n            \n            # Validate prediction format\n            assert 'object_id' in pred_df.columns, \"Missing object_id column\"\n            assert all(cls in pred_df.columns for cls in CORRECT_PLASTICC_CLASSES), \"Missing required classes\"\n            \n            # Save intermediate results\n            if (batch_num + 1) % SAVE_INTERVAL == 0:\n                batch_file = f\"{OUTPUT_DIR}lightgbm_predictions_batch_{batch_num + 1:03d}.csv\"\n                pred_df.to_csv(batch_file, index=False)\n                print(f\"    💾 Intermediate save: {batch_file}\")\n            \n            all_predictions.append(pred_df)\n            successful_batches += 1\n            \n            # Memory cleanup\n            del batch_lc, batch_lc_list, features_df\n            gc.collect()\n            \n            # Progress update\n            elapsed = datetime.now() - start_time\n            print(f\"    ✅ Batch completed in {elapsed.total_seconds():.1f}s\")\n            \n            # ETA calculation\n            if batch_num > 0:\n                avg_time = elapsed.total_seconds()\n                remaining_batches = num_batches - batch_num - 1\n                eta_minutes = (remaining_batches * avg_time) / 60\n                print(f\"    📊 ETA: {eta_minutes:.1f} minutes remaining\")\n                \n        except Exception as e:\n            print(f\"    ❌ Batch {batch_num + 1} failed: {e}\")\n            # Continue with next batch rather than failing completely\n            gc.collect()\n            continue\n    \n    print(f\"\\\\n✅ Processing complete: {successful_batches}/{num_batches} batches successful\")\n    \n    if successful_batches == 0:\n        print(\"❌ No batches processed successfully!\")\n        return None\n    \n    return all_predictions\n\n# Cell 8: Enhanced Submission Creation for LightGBM\ndef create_lightgbm_submission(all_predictions):\n    \"\"\"Create final submission with maximum scoring optimization\"\"\"\n    print(\"\\\\n=== LIGHTGBM SUBMISSION CREATION ===\")\n    \n    if not all_predictions:\n        print(\"❌ No predictions to process!\")\n        return None\n    \n    # Combine all predictions\n    print(\"🔗 Combining all batch predictions...\")\n    final_predictions = pd.concat(all_predictions, ignore_index=True)\n    print(f\"✅ Combined predictions: {final_predictions.shape}\")\n    \n    # Remove duplicates (if any) - keep the last occurrence\n    if final_predictions['object_id'].duplicated().any():\n        print(\"⚠️ Removing duplicate object IDs...\")\n        final_predictions = final_predictions.drop_duplicates(subset=['object_id'], keep='last')\n        print(f\"✅ After deduplication: {final_predictions.shape}\")\n    \n    # Verify all required classes are present\n    missing_classes = set(CORRECT_PLASTICC_CLASSES) - set(final_predictions.columns)\n    if missing_classes:\n        print(f\"⚠️ Adding missing classes: {missing_classes}\")\n        for cls in missing_classes:\n            final_predictions[cls] = 0.001  # Small default probability\n    \n    # Ensure we have all required columns\n    required_cols = ['object_id'] + CORRECT_PLASTICC_CLASSES\n    final_submission = final_predictions[required_cols].copy()\n    \n    # Final probability normalization (critical for scoring)\n    print(\"🎯 Final probability normalization...\")\n    prob_cols = CORRECT_PLASTICC_CLASSES\n    row_sums = final_submission[prob_cols].sum(axis=1)\n    \n    # Handle edge cases\n    zero_sum_rows = (row_sums == 0).sum()\n    if zero_sum_rows > 0:\n        print(f\"    ⚠️ Fixing {zero_sum_rows} rows with zero probabilities\")\n        mask = row_sums == 0\n        final_submission.loc[mask, prob_cols] = 1.0 / len(prob_cols)\n        row_sums = final_submission[prob_cols].sum(axis=1)\n    \n    # Normalize all rows to sum to 1\n    for col in prob_cols:\n        final_submission[col] = final_submission[col] / row_sums\n    \n    # Final validation\n    final_sums = final_submission[prob_cols].sum(axis=1)\n    assert np.allclose(final_sums, 1.0, atol=1e-6), \"Final probabilities don't sum to 1!\"\n    \n    # Convert object_id to integers (required by Kaggle)\n    final_submission['object_id'] = final_submission['object_id'].astype('Int64')\n    \n    # Sort by object_id for consistency\n    final_submission = final_submission.sort_values('object_id').reset_index(drop=True)\n    \n    # Save final submission\n    submission_file = f\"{OUTPUT_DIR}lightgbm_submission_final.csv\"\n    final_submission.to_csv(submission_file, index=False)\n    \n    print(f\"✅ LightGBM submission created: {submission_file}\")\n    print(f\"📊 Final shape: {final_submission.shape}\")\n    print(f\"🎯 Classes: {CORRECT_PLASTICC_CLASSES}\")\n    \n    # Quality checks\n    print(f\"\\\\n🔍 QUALITY CHECKS:\")\n    print(f\"    ✓ Object IDs: {final_submission['object_id'].nunique():,} unique\")\n    print(f\"    ✓ Probability range: [{final_submission[prob_cols].min().min():.6f}, {final_submission[prob_cols].max().max():.6f}]\")\n    print(f\"    ✓ Row sums: [{final_sums.min():.6f}, {final_sums.max():.6f}]\")\n    print(f\"    ✓ All sums ≈ 1.0: {np.allclose(final_sums, 1.0, atol=1e-6)}\")\n    print(f\"    ✓ No NaN values: {not final_submission.isnull().any().any()}\")\n    print(f\"    ✓ Correct columns: {len(required_cols)} columns present\")\n    \n    # Display sample predictions\n    print(f\"\\\\n📋 SAMPLE PREDICTIONS:\")\n    sample_df = final_submission.head(3)\n    for idx, row in sample_df.iterrows():\n        obj_id = row['object_id']\n        max_prob_class = prob_cols[np.argmax(row[prob_cols])]\n        max_prob = row[max_prob_class]\n        print(f\"    Object {obj_id}: {max_prob_class} (prob={max_prob:.4f})\")\n    \n    # Class distribution analysis\n    print(f\"\\\\n📊 CLASS DISTRIBUTION:\")\n    class_predictions = np.argmax(final_submission[prob_cols].values, axis=1)\n    for i, class_name in enumerate(CORRECT_PLASTICC_CLASSES):\n        count = np.sum(class_predictions == i)\n        percentage = 100 * count / len(final_submission)\n        print(f\"    {class_name}: {count:,} objects ({percentage:.2f}%)\")\n    \n    return final_submission\n\n# Cell 9: Main Execution Function for LightGBM\ndef main_lightgbm_execution():\n    \"\"\"Main execution function with comprehensive error handling\"\"\"\n    print(\"\\\\n\" + \"=\"*60)\n    print(\"🚀 STARTING LIGHTGBM PLAsTiCC TESTING PIPELINE\")\n    print(\"=\"*60)\n    \n    start_time = datetime.now()\n    \n    try:\n        # Process test data\n        all_predictions = process_test_data_lightgbm()\n        \n        if all_predictions is None or len(all_predictions) == 0:\n            print(\"❌ No predictions generated - pipeline failed!\")\n            return False\n        \n        # Create final submission\n        final_submission = create_lightgbm_submission(all_predictions)\n        \n        if final_submission is None:\n            print(\"❌ Submission creation failed!\")\n            return False\n        \n        # Success summary\n        total_time = datetime.now() - start_time\n        print(f\"\\\\n\" + \"=\"*60)\n        print(\"🎉 LIGHTGBM PIPELINE COMPLETED SUCCESSFULLY!\")\n        print(\"=\"*60)\n        print(f\"⏱️ Total execution time: {total_time}\")\n        print(f\"📊 Objects processed: {len(final_submission):,}\")\n        \n        if model_metrics:\n            print(f\"🎯 Expected performance based on training:\")\n            print(f\"   Training CV accuracy: {model_metrics.get('cv_accuracy', 'Unknown')}\")\n            print(f\"   Training CV log loss: {model_metrics.get('cv_log_loss', 'Unknown')}\")\n        \n        print(f\"📁 Output file: lightgbm_submission_final.csv\")\n        print(f\"🏆 LightGBM model optimized for astronomical time-series\")\n        \n        return True\n        \n    except Exception as e:\n        print(f\"\\\\n❌ LIGHTGBM PIPELINE FAILED: {e}\")\n        print(f\"🛠️ Debugging information:\")\n        print(f\"    - Final model loaded: {final_model is not None}\")\n        print(f\"    - CV models loaded: {len(cv_models)}\")\n        print(f\"    - Label encoder loaded: {label_encoder is not None}\")\n        print(f\"    - Feature columns loaded: {feature_columns is not None}\")\n        \n        # Try to create a minimal fallback submission\n        try:\n            print(\"\\\\n🚨 Attempting fallback submission creation...\")\n            create_lightgbm_fallback_submission()\n        except Exception as fallback_error:\n            print(f\"❌ Fallback submission also failed: {fallback_error}\")\n        \n        return False\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T11:55:04.491481Z","iopub.execute_input":"2025-05-26T11:55:04.492240Z","iopub.status.idle":"2025-05-26T11:55:04.526295Z","shell.execute_reply.started":"2025-05-26T11:55:04.492213Z","shell.execute_reply":"2025-05-26T11:55:04.525708Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Cell 10: Fallback and Utilities for LightGBM\ndef create_lightgbm_fallback_submission():\n    \"\"\"Create a minimal fallback submission if main pipeline fails\"\"\"\n    print(\"🚨 Creating LightGBM fallback submission with uniform probabilities...\")\n    \n    # Load test metadata\n    test_meta = pd.read_csv(\"/kaggle/input/PLAsTiCC-2018/test_set_metadata.csv\")\n    \n    # Create uniform probability distribution\n    n_objects = len(test_meta)\n    n_classes = len(CORRECT_PLASTICC_CLASSES)\n    uniform_prob = 1.0 / n_classes\n    \n    # Create submission dataframe\n    fallback_submission = pd.DataFrame()\n    fallback_submission['object_id'] = test_meta['object_id'].astype('Int64')\n    \n    # Add uniform probabilities for all classes\n    for class_name in CORRECT_PLASTICC_CLASSES:\n        fallback_submission[class_name] = uniform_prob\n    \n    # Save fallback submission\n    fallback_file = f\"{OUTPUT_DIR}lightgbm_fallback_submission.csv\"\n    fallback_submission.to_csv(fallback_file, index=False)\n    \n    print(f\"✅ LightGBM fallback submission created: {fallback_file}\")\n    print(f\"📊 Shape: {fallback_submission.shape}\")\n    print(f\"⚠️ Note: This uses uniform probabilities and will score poorly!\")\n    \n    return fallback_submission\n\ndef validate_lightgbm_submission(submission_file):\n    \"\"\"Validate that submission file meets Kaggle requirements\"\"\"\n    print(f\"\\\\n🔍 VALIDATING LIGHTGBM SUBMISSION: {submission_file}\")\n    \n    try:\n        # Load submission\n        sub_df = pd.read_csv(submission_file)\n        \n        # Check required columns\n        required_cols = ['object_id'] + CORRECT_PLASTICC_CLASSES\n        missing_cols = set(required_cols) - set(sub_df.columns)\n        extra_cols = set(sub_df.columns) - set(required_cols)\n        \n        print(f\"    ✓ Columns present: {len(sub_df.columns)}\")\n        if missing_cols:\n            print(f\"    ❌ Missing columns: {missing_cols}\")\n            return False\n        if extra_cols:\n            print(f\"    ⚠️ Extra columns: {extra_cols}\")\n        \n        # Check object IDs\n        print(f\"    ✓ Unique object IDs: {sub_df['object_id'].nunique():,}\")\n        print(f\"    ✓ Total rows: {len(sub_df):,}\")\n        \n        # Check probabilities\n        prob_cols = CORRECT_PLASTICC_CLASSES\n        prob_sums = sub_df[prob_cols].sum(axis=1)\n        \n        print(f\"    ✓ Probability sums range: [{prob_sums.min():.6f}, {prob_sums.max():.6f}]\")\n        print(f\"    ✓ All sums ≈ 1.0: {np.allclose(prob_sums, 1.0, atol=1e-5)}\")\n        print(f\"    ✓ No negative probabilities: {(sub_df[prob_cols] >= 0).all().all()}\")\n        print(f\"    ✓ No NaN values: {not sub_df.isnull().any().any()}\")\n        \n        # File size check\n        file_size_mb = os.path.getsize(submission_file) / (1024*1024)\n        print(f\"    ✓ File size: {file_size_mb:.1f} MB\")\n        \n        if file_size_mb > 500:\n            print(f\"    ⚠️ Large file size - may cause upload issues\")\n        \n        print(\"    ✅ LightGBM submission format validation PASSED!\")\n        return True\n        \n    except Exception as e:\n        print(f\"    ❌ Validation failed: {e}\")\n        return False\n\ndef display_lightgbm_summary():\n    \"\"\"Display final execution summary\"\"\"\n    print(\"\\\\n\" + \"=\"*60)\n    print(\"📋 LIGHTGBM FINAL EXECUTION SUMMARY\")\n    print(\"=\"*60)\n    \n    # Check for output files\n    output_files = []\n    for file in os.listdir(OUTPUT_DIR):\n        if file.endswith('.csv') and 'lightgbm' in file:\n            file_path = os.path.join(OUTPUT_DIR, file)\n            file_size = os.path.getsize(file_path) / (1024*1024)\n            output_files.append((file, file_size))\n    \n    if output_files:\n        print(\"📁 Generated LightGBM files:\")\n        for filename, size_mb in output_files:\n            print(f\"    📄 {filename} ({size_mb:.1f} MB)\")\n            \n            # Validate main submission file\n            if 'lightgbm_submission_final.csv' in filename:\n                validate_lightgbm_submission(os.path.join(OUTPUT_DIR, filename))\n    else:\n        print(\"❌ No LightGBM output files generated!\")\n    \n    print(f\"\\\\n🎯 Next steps:\")\n    print(f\"    1. Download lightgbm_submission_final.csv\")\n    print(f\"    2. Submit to PLAsTiCC competition\")\n    print(f\"    3. Monitor leaderboard performance\")\n    \n    if model_metrics:\n        expected_acc = model_metrics.get('cv_accuracy', 0.8)\n        expected_loss = model_metrics.get('cv_log_loss', 0.65)\n        print(f\"    4. Expected score: ~{expected_loss:.3f} log-loss ({expected_acc:.1%} accuracy)\")\n    \n    print(f\"\\\\n✅ LightGBM pipeline execution completed!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T11:55:18.711219Z","iopub.execute_input":"2025-05-26T11:55:18.711841Z","iopub.status.idle":"2025-05-26T11:55:18.723323Z","shell.execute_reply.started":"2025-05-26T11:55:18.711800Z","shell.execute_reply":"2025-05-26T11:55:18.722475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 11: Execute the LightGBM Pipeline\nif __name__ == \"__main__\":\n    print(\"\\\\n🚀 STARTING LIGHTGBM TESTING EXECUTION\")\n    \n    # Check if we have the necessary components\n    if final_model is None and len(cv_models) == 0:\n        print(\"❌ No LightGBM models found! Please check:\")\n        print(\"    1. lightgbm-models dataset is added to your notebook\")\n        print(\"    2. Model files exist in /kaggle/input/lightgbm-models/\")\n        print(\"    3. Files are not corrupted\")\n        \n        # Try fallback\n        try:\n            create_lightgbm_fallback_submission()\n            print(\"✅ Created fallback submission instead\")\n        except:\n            print(\"❌ Even fallback submission failed\")\n    else:\n        # Execute main pipeline\n        success = main_lightgbm_execution()\n        display_lightgbm_summary()\n        \n        if success:\n            print(\"\\\\n🎉 Ready for Kaggle submission!\")\n            print(\"🏆 Your LightGBM models have been successfully applied to 3.5M objects!\")\n        else:\n            print(\"\\\\n⚠️ Check logs for issues - fallback submission may be available\")\n\n# Memory cleanup\ngc.collect()\nprint(f\"\\\\n🧹 Memory cleanup completed\")\nprint(f\"⏰ LightGBM notebook finished at: {datetime.now()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T11:55:25.035495Z","iopub.execute_input":"2025-05-26T11:55:25.035788Z","iopub.status.idle":"2025-05-26T13:56:05.698583Z","shell.execute_reply.started":"2025-05-26T11:55:25.035766Z","shell.execute_reply":"2025-05-26T13:56:05.697923Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================================\n# EMERGENCY FIX: Combine All Batch Files\n# Run these cells in order to fix your submission\n# =========================================\n\n# Cell 1: Import Libraries and Setup\nimport pandas as pd\nimport numpy as np\nimport os\nimport gc\nfrom datetime import datetime\n\nprint(\"🚨 EMERGENCY FIX: Combining All Batch Files\")\nprint(f\"Started at: {datetime.now()}\")\n\n# PLAsTiCC classes\nCORRECT_PLASTICC_CLASSES = [\n    'class_6', 'class_15', 'class_16', 'class_42', 'class_52', \n    'class_53', 'class_62', 'class_64', 'class_65', 'class_67', \n    'class_88', 'class_90', 'class_92', 'class_95', 'class_99'\n]\n\nprint(\"✅ Setup complete\")\n\n# Cell 2: Find and Load All Batch Files\ndef find_and_load_batches():\n    \"\"\"Find and load all existing batch files\"\"\"\n    \n    print(\"\\n📁 STEP 1: Finding all batch files...\")\n    batch_files = []\n    working_dir = \"/kaggle/working\"\n    \n    for file in os.listdir(working_dir):\n        if 'lightgbm_predictions_batch_' in file and file.endswith('.csv'):\n            batch_files.append(os.path.join(working_dir, file))\n            file_size = os.path.getsize(os.path.join(working_dir, file)) / (1024*1024)\n            print(f\"    📄 Found: {file} ({file_size:.1f} MB)\")\n    \n    batch_files.sort()  # Sort for consistent order\n    print(f\"✅ Found {len(batch_files)} batch files\")\n    \n    print(\"\\n📊 STEP 2: Loading and combining batch files...\")\n    combined_predictions = []\n    processed_object_ids = set()\n    \n    for i, batch_file in enumerate(batch_files):\n        print(f\"    Loading batch {i+1}/{len(batch_files)}: {os.path.basename(batch_file)}\")\n        try:\n            batch_df = pd.read_csv(batch_file)\n            print(f\"        Rows: {len(batch_df):,}\")\n            print(f\"        Objects: {batch_df['object_id'].nunique():,}\")\n            \n            # Track processed objects\n            batch_objects = set(batch_df['object_id'].values)\n            processed_object_ids.update(batch_objects)\n            \n            combined_predictions.append(batch_df)\n            \n        except Exception as e:\n            print(f\"        ❌ Error loading {batch_file}: {e}\")\n    \n    return combined_predictions, processed_object_ids\n\n# Execute batch loading\ncombined_predictions, processed_object_ids = find_and_load_batches()\n\n# Cell 3: Combine Batch Files\ndef combine_batch_predictions(combined_predictions, processed_object_ids):\n    \"\"\"Combine all batch predictions\"\"\"\n    \n    if not combined_predictions:\n        print(\"❌ No batch files could be loaded!\")\n        return None\n    \n    print(\"\\n🔗 Combining all batch predictions...\")\n    combined_df = pd.concat(combined_predictions, ignore_index=True)\n    print(f\"✅ Combined shape: {combined_df.shape}\")\n    print(f\"✅ Unique objects from batches: {len(processed_object_ids):,}\")\n    \n    # Remove duplicates if any\n    if combined_df['object_id'].duplicated().any():\n        print(\"⚠️ Removing duplicate object IDs...\")\n        combined_df = combined_df.drop_duplicates(subset=['object_id'], keep='last')\n        print(f\"✅ After deduplication: {combined_df.shape}\")\n    \n    return combined_df\n\n# Execute combination\ncombined_df = combine_batch_predictions(combined_predictions, processed_object_ids)\n\n# Clean up memory\ndel combined_predictions\ngc.collect()\n\n# Cell 4: Check for Missing Objects and Fill Gaps\ndef fill_missing_objects(combined_df, processed_object_ids):\n    \"\"\"Check for missing objects and fill gaps\"\"\"\n    \n    print(\"\\n📋 STEP 3: Checking for missing objects...\")\n    \n    # Load full test metadata to see what's missing\n    test_meta = pd.read_csv(\"/kaggle/input/PLAsTiCC-2018/test_set_metadata.csv\")\n    all_object_ids = set(test_meta['object_id'].values)\n    missing_object_ids = all_object_ids - processed_object_ids\n    \n    print(f\"📊 Total test objects: {len(all_object_ids):,}\")\n    print(f\"✅ Objects in batches: {len(processed_object_ids):,}\")\n    print(f\"❌ Missing objects: {len(missing_object_ids):,}\")\n    \n    if len(missing_object_ids) > 0:\n        print(f\"\\n🔧 STEP 4: Creating predictions for missing objects...\")\n        \n        # Get metadata for missing objects\n        missing_meta = test_meta[test_meta['object_id'].isin(missing_object_ids)]\n        \n        # Create uniform probability predictions for missing objects\n        n_missing = len(missing_meta)\n        uniform_prob = 1.0 / len(CORRECT_PLASTICC_CLASSES)\n        \n        # Create missing predictions DataFrame\n        missing_predictions = pd.DataFrame()\n        missing_predictions['object_id'] = missing_meta['object_id'].values\n        \n        for class_name in CORRECT_PLASTICC_CLASSES:\n            missing_predictions[class_name] = uniform_prob\n        \n        print(f\"✅ Created uniform predictions for {n_missing:,} missing objects\")\n        \n        # Combine with existing predictions\n        print(\"\\n🔗 Combining with existing predictions...\")\n        final_combined = pd.concat([combined_df, missing_predictions], ignore_index=True)\n    else:\n        final_combined = combined_df\n    \n    print(f\"\\n📊 FINAL COMBINATION RESULTS:\")\n    print(f\"✅ Total objects: {len(final_combined):,}\")\n    print(f\"✅ Expected objects: {len(all_object_ids):,}\")\n    print(f\"✅ Coverage: {len(final_combined) / len(all_object_ids) * 100:.1f}%\")\n    \n    return final_combined\n\n# Execute missing object filling\nif combined_df is not None:\n    final_combined = fill_missing_objects(combined_df, processed_object_ids)\nelse:\n    print(\"❌ Cannot proceed - no combined data\")\n    final_combined = None\n\n# Cell 5: Create Final Submission File\ndef create_final_submission(combined_df):\n    \"\"\"Create final submission file with validation\"\"\"\n    \n    if combined_df is None:\n        print(\"❌ No data to create submission\")\n        return None, None\n    \n    print(\"\\n🎯 STEP 5: Creating final submission...\")\n    \n    # Ensure all required columns\n    required_cols = ['object_id'] + CORRECT_PLASTICC_CLASSES\n    missing_cols = set(required_cols) - set(combined_df.columns)\n    if missing_cols:\n        print(f\"⚠️ Adding missing columns: {missing_cols}\")\n        for col in missing_cols:\n            combined_df[col] = 1.0 / len(CORRECT_PLASTICC_CLASSES)\n    \n    # Select and reorder columns\n    final_submission = combined_df[required_cols].copy()\n    \n    # Normalize probabilities\n    print(\"🎯 Normalizing probabilities...\")\n    prob_cols = CORRECT_PLASTICC_CLASSES\n    row_sums = final_submission[prob_cols].sum(axis=1)\n    \n    # Fix zero sums\n    zero_sum_mask = row_sums == 0\n    if zero_sum_mask.sum() > 0:\n        print(f\"⚠️ Fixing {zero_sum_mask.sum()} rows with zero probabilities\")\n        final_submission.loc[zero_sum_mask, prob_cols] = 1.0 / len(prob_cols)\n        row_sums = final_submission[prob_cols].sum(axis=1)\n    \n    # Normalize to sum to 1\n    for col in prob_cols:\n        final_submission[col] = final_submission[col] / row_sums\n    \n    # Final validation\n    final_sums = final_submission[prob_cols].sum(axis=1)\n    assert np.allclose(final_sums, 1.0, atol=1e-6), \"Probabilities don't sum to 1!\"\n    \n    # Convert object_id to integers\n    final_submission['object_id'] = final_submission['object_id'].astype('Int64')\n    \n    # Sort by object_id\n    final_submission = final_submission.sort_values('object_id').reset_index(drop=True)\n    \n    # Save final submission\n    submission_file = \"/kaggle/working/FIXED_lightgbm_submission_complete.csv\"\n    final_submission.to_csv(submission_file, index=False)\n    \n    print(f\"\\n✅ FIXED submission created: {submission_file}\")\n    print(f\"📊 Final shape: {final_submission.shape}\")\n    print(f\"💾 File size: {os.path.getsize(submission_file) / (1024*1024):.1f} MB\")\n    \n    return final_submission, submission_file\n\n# Execute final submission creation\nfinal_submission, submission_file = create_final_submission(final_combined)\n\n# Cell 6: Validate Final Submission\ndef validate_final_submission(submission_file):\n    \"\"\"Validate the final submission\"\"\"\n    \n    if submission_file is None:\n        print(\"❌ No submission file to validate\")\n        return False\n    \n    print(f\"\\n🔍 VALIDATING FINAL SUBMISSION: {os.path.basename(submission_file)}\")\n    \n    # Load and check\n    sub_df = pd.read_csv(submission_file)\n    \n    # Expected total from PLAsTiCC\n    expected_objects = 3492890\n    \n    print(f\"✅ Rows: {len(sub_df):,}\")\n    print(f\"✅ Expected rows: {expected_objects:,}\")\n    print(f\"✅ Coverage: {len(sub_df) / expected_objects * 100:.2f}%\")\n    print(f\"✅ Unique objects: {sub_df['object_id'].nunique():,}\")\n    \n    # Check columns\n    required_cols = ['object_id'] + CORRECT_PLASTICC_CLASSES\n    missing_cols = set(required_cols) - set(sub_df.columns)\n    if missing_cols:\n        print(f\"❌ Missing columns: {missing_cols}\")\n        return False\n    else:\n        print(f\"✅ All required columns present\")\n    \n    # Check probabilities\n    prob_cols = CORRECT_PLASTICC_CLASSES\n    prob_sums = sub_df[prob_cols].sum(axis=1)\n    \n    print(f\"✅ Probability range: [{sub_df[prob_cols].min().min():.6f}, {sub_df[prob_cols].max().max():.6f}]\")\n    print(f\"✅ Row sums range: [{prob_sums.min():.6f}, {prob_sums.max():.6f}]\")\n    print(f\"✅ All sums ≈ 1.0: {np.allclose(prob_sums, 1.0, atol=1e-5)}\")\n    print(f\"✅ No NaN values: {not sub_df.isnull().any().any()}\")\n    \n    # Class distribution\n    print(f\"\\n📊 CLASS DISTRIBUTION (Top 5):\")\n    class_predictions = np.argmax(sub_df[prob_cols].values, axis=1)\n    for i, class_name in enumerate(CORRECT_PLASTICC_CLASSES[:5]):\n        count = np.sum(class_predictions == i)\n        percentage = 100 * count / len(sub_df)\n        print(f\"    {class_name}: {count:,} objects ({percentage:.2f}%)\")\n    \n    if len(sub_df) >= expected_objects * 0.99:  # At least 99% coverage\n        print(f\"\\n🎉 SUBMISSION VALIDATION PASSED!\")\n        return True\n    else:\n        print(f\"\\n❌ Insufficient coverage: {len(sub_df)} < {expected_objects}\")\n        return False\n\n# Execute validation\nis_valid = validate_final_submission(submission_file)\n\n# Cell 7: Final Summary and Instructions\nprint(f\"\\n\" + \"=\"*60)\nprint(\"🎉 BATCH COMBINATION FIX COMPLETED!\")\nprint(\"=\"*60)\n\nif final_submission is not None:\n    print(f\"📊 Final objects: {len(final_submission):,}\")\n    print(f\"📁 File created: FIXED_lightgbm_submission_complete.csv\")\n    print(f\"✅ Validation passed: {is_valid}\")\n    \n    if is_valid:\n        print(f\"\\n🏆 SUCCESS! Your submission file is now complete!\")\n        print(f\"📤 UPLOAD THIS FILE: FIXED_lightgbm_submission_complete.csv\")\n        print(f\"🎯 This file has all {len(final_submission):,} required objects!\")\n        print(f\"\\n📋 Next Steps:\")\n        print(f\"    1. Download: FIXED_lightgbm_submission_complete.csv\")\n        print(f\"    2. Upload to PLAsTiCC competition\")\n        print(f\"    3. Submit and check leaderboard\")\n    else:\n        print(f\"\\n⚠️ File created but validation failed\")\n        print(f\"📤 You can still try uploading: FIXED_lightgbm_submission_complete.csv\")\nelse:\n    print(\"❌ Could not create submission file\")\n\n# Final cleanup\ngc.collect()\nprint(f\"\\n✅ Fix completed at: {datetime.now()}\")\nprint(\"\\n🎉 Ready for submission!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T14:13:16.477385Z","iopub.execute_input":"2025-05-26T14:13:16.477730Z","iopub.status.idle":"2025-05-26T14:14:57.229946Z","shell.execute_reply.started":"2025-05-26T14:13:16.477706Z","shell.execute_reply":"2025-05-26T14:14:57.229162Z"}},"outputs":[],"execution_count":null}]}