{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# https://www.kaggle.com/code/digixintelligence/euphoria-1-0?scriptVersionId=247586720\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n\"\"\"\nCredit to the following authors and notebooks/discussions:\n\nhttps://www.kaggle.com/code/bakuer30/drw-remix-vi - For tuning and improving XGBoost parameters and features\nhttps://www.kaggle.com/competitions/drw-crypto-market-prediction/discussion/581193 - For the idea of temporal weights / time slices\nhttps://www.kaggle.com/competitions/drw-crypto-market-prediction/discussion/584475 - For pointing out that the data at the very beginning may be more valuable too\n\nEnhanced version with:\n1. Multiple noise removal percentages (0.01% to 1%)\n2. Global clean slice that removes records identified as outliers across multiple time slices\n3. Intelligent ensembling based on validation scores\n4. Comprehensive analysis of generalization impact\n\"\"\"\n\n# Multi-Configuration Runner with Multiple Noise Removal Percentages\n\n# ========== CONFIGURATION ==========\nEARLY_PERCENTAGE = 0.35  # Change this to 0.20, 0.25, 0.30, 0.35, 0.40, or 0.45\nNOISE_REMOVAL_PERCENTAGES = [0.0001, 0.0005, 0.001, 0.002, 0.005, 0.01]  # 0.01%, 0.05%, 0.1%, 0.2%, 0.5%, 1%\nMIN_SCORE_THRESHOLD = 0.08  # Minimum score for a slice to be included in intelligent ensemble\nMIN_OUTLIER_SLICES = 2  # Minimum number of slices a record must be an outlier in to be globally removed\n# ===================================\n\n# Imports\nimport sys\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import KFold\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\nfrom scipy.stats import pearsonr\nimport json\n\n# Feature Engineering\ndef feature_engineering(df):\n    df['volume_weighted_sell'] = df['sell_qty'] * df['volume']\n    df['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-8)\n    df['selling_pressure'] = df['sell_qty'] / (df['volume'] + 1e-8)\n    df['effective_spread_proxy'] = np.abs(df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-8)\n    \n    df = df.replace([np.inf, -np.inf], np.nan)\n    df = df.fillna(0)\n    return df \n\n# Configuration\nclass Config:\n    TRAIN_PATH = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    TEST_PATH = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    SUBMISSION_PATH = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n\n    FEATURES = [\n        \"X863\", \"X856\", \"X598\", \"X862\", \"X385\", \"X852\", \"X603\", \"X860\", \"X674\",\n        \"X415\", \"X345\", \"X855\", \"X174\", \"X302\", \"X178\", \"X168\", \"X612\",\n        \"buy_qty\", \"sell_qty\", \"volume\", \"X888\", \"X421\", \"X333\", \"X292\",\n    ]\n\n    LABEL_COLUMN = \"label\"\n    N_FOLDS = 3\n    RANDOM_STATE = 42\n\nXGB_PARAMS = {\n    'tree_method': 'hist', \n    'device': 'gpu',\n    'n_jobs': -1,\n    'colsample_bytree': 0.4111224922845363, \n    'colsample_bynode': 0.28869302181383194,\n    'gamma': 1.4665430311056709, \n    'learning_rate': 0.014053505540364681, \n    'max_depth': 7, \n    'max_leaves': 40, \n    'n_estimators': 500,\n    'reg_alpha': 27.791606770656145, \n    'reg_lambda': 84.90603428439086,\n    'subsample': 0.06567,\n    'verbosity': 0,\n    'random_state': Config.RANDOM_STATE\n}\n\nLEARNERS = [\n    {\"name\": \"xgb\", \"Estimator\": XGBRegressor, \"params\": XGB_PARAMS},\n]\n\n# Loading Data\ndef create_time_decay_weights(n: int, decay: float = 0.9, reverse: bool = False) -> np.ndarray:\n    \"\"\"Create time decay weights. If reverse=True, older data gets higher weight.\"\"\"\n    positions = np.arange(n)\n    if reverse:\n        normalized = 1.0 - (positions / (n - 1))\n    else:\n        normalized = positions / (n - 1)\n    weights = decay ** (1.0 - normalized)\n    return weights * n / weights.sum()\n\ndef load_data():\n    train_df = pd.read_parquet(Config.TRAIN_PATH, columns=Config.FEATURES + [Config.LABEL_COLUMN])\n    test_df = pd.read_parquet(Config.TEST_PATH, columns=Config.FEATURES)\n    submission_df = pd.read_csv(Config.SUBMISSION_PATH)\n\n    train_df = feature_engineering(train_df)\n    test_df = feature_engineering(test_df)\n    \n    print(f\"Loaded data - Train: {train_df.shape}, Test: {test_df.shape}, Submission: {submission_df.shape}\")\n    return train_df.reset_index(drop=True), test_df.reset_index(drop=True), submission_df\n\nConfig.FEATURES += [\"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\"]\nConfig.FEATURES = list(set(Config.FEATURES))  # remove duplicates\n\n# Training and Evaluation\ndef get_model_slices(n_samples: int, include_clean: bool = True, noise_pct: float = 0.001, include_global_clean: bool = True):\n    base_slices = [\n        {\"name\": \"full_data\", \"type\": \"full\", \"cutoff\": 0, \"clean\": False, \"noise_pct\": 0},\n        {\"name\": \"last_75pct\", \"type\": \"recent\", \"cutoff\": int(0.25 * n_samples), \"clean\": False, \"noise_pct\": 0},\n        {\"name\": \"last_50pct\", \"type\": \"recent\", \"cutoff\": int(0.50 * n_samples), \"clean\": False, \"noise_pct\": 0},\n        {\"name\": f\"first_{int(EARLY_PERCENTAGE*100)}pct\", \"type\": \"early\", \"cutoff\": int(EARLY_PERCENTAGE * n_samples), \"clean\": False, \"noise_pct\": 0},\n    ]\n    \n    if not include_clean:\n        return base_slices\n    \n    # Add clean versions of each slice\n    clean_slices = []\n    for s in base_slices:\n        clean_slice = s.copy()\n        clean_slice[\"name\"] = s[\"name\"] + f\"_clean{int(noise_pct*10000)/100}pct\"\n        clean_slice[\"clean\"] = True\n        clean_slice[\"noise_pct\"] = noise_pct\n        clean_slices.append(clean_slice)\n    \n    # Add global clean slice (removes outliers found across multiple slices)\n    if include_global_clean:\n        global_clean_slice = {\n            \"name\": f\"global_clean{int(noise_pct*10000)/100}pct\",\n            \"type\": \"global_clean\",\n            \"cutoff\": 0,\n            \"clean\": True,\n            \"noise_pct\": noise_pct,\n            \"global_clean\": True\n        }\n        clean_slices.append(global_clean_slice)\n    \n    return base_slices + clean_slices\n\ndef identify_slice_noise(X_train, y_train, sample_weights, noise_pct=0.001):\n    \"\"\"Use full XGB model to identify noisiest records in a slice\"\"\"\n    # Ensure sample weights match the training data size\n    if len(sample_weights) != len(X_train):\n        print(f\"Warning: Weight size mismatch. Weights: {len(sample_weights)}, Data: {len(X_train)}\")\n        sample_weights = sample_weights[:len(X_train)]\n    \n    # Use the exact same parameters as the main model\n    model = XGBRegressor(**XGB_PARAMS)\n    model.fit(X_train, y_train, sample_weight=sample_weights, verbose=False)\n    \n    # Get predictions and errors\n    preds = model.predict(X_train)\n    errors = np.abs(y_train - preds)\n    \n    # Identify top noise_pct as noise\n    threshold = np.percentile(errors, (1 - noise_pct) * 100)\n    clean_mask = errors <= threshold\n    \n    # Calculate metrics for analysis\n    train_score_before = pearsonr(y_train, preds)[0]\n    train_score_after = pearsonr(y_train[clean_mask], preds[clean_mask])[0] if clean_mask.sum() > 0 else 0\n    \n    return clean_mask, {\n        'train_score_before': train_score_before,\n        'train_score_after': train_score_after,\n        'removed_count': (~clean_mask).sum(),\n        'mean_error_removed': errors[~clean_mask].mean() if (~clean_mask).sum() > 0 else 0\n    }\n\ndef identify_global_outliers(train_df, base_slices, noise_pct, min_outlier_count=2):\n    \"\"\"\n    Identify records that are outliers in multiple slices.\n    \n    This function:\n    1. Trains separate models on each base slice (full_data, last_75pct, last_50pct, first_Xpct)\n    2. Identifies outliers in each slice based on prediction errors\n    3. Counts how many slices each record appears as an outlier\n    4. Marks records as global outliers if they appear in min_outlier_count or more slices\n    \n    The intuition is that records consistently identified as outliers across different\n    time windows are likely truly problematic and should be removed globally.\n    \"\"\"\n    n_samples = len(train_df)\n    outlier_counts = np.zeros(n_samples)\n    outlier_details = {}\n    \n    full_weights = create_time_decay_weights(n_samples)\n    \n    print(f\"\\n  Identifying global outliers (noise_pct={noise_pct*100:.2f}%)...\")\n    \n    for s in base_slices:\n        slice_type = s[\"type\"]\n        cutoff = s[\"cutoff\"]\n        \n        if slice_type == \"full\":\n            subset = train_df.reset_index(drop=True)\n            indices = np.arange(n_samples)\n            sw = full_weights\n        elif slice_type == \"recent\":\n            subset = train_df.iloc[cutoff:].reset_index(drop=True)\n            indices = np.arange(cutoff, n_samples)\n            sw = create_time_decay_weights(len(subset))\n        elif slice_type == \"early\":\n            subset = train_df.iloc[:cutoff].reset_index(drop=True)\n            indices = np.arange(cutoff)\n            sw = create_time_decay_weights(len(subset))\n        \n        if len(subset) == 0:\n            continue\n            \n        X = subset[Config.FEATURES].values\n        y = subset[Config.LABEL_COLUMN].values\n        \n        # Identify outliers in this slice\n        model = XGBRegressor(**XGB_PARAMS)\n        model.fit(X, y, sample_weight=sw, verbose=False)\n        preds = model.predict(X)\n        errors = np.abs(y - preds)\n        \n        threshold = np.percentile(errors, (1 - noise_pct) * 100)\n        outlier_mask = errors > threshold\n        \n        # Map back to original indices\n        slice_outliers = indices[outlier_mask]\n        outlier_counts[slice_outliers] += 1\n        \n        outlier_details[s[\"name\"]] = {\n            \"n_outliers\": outlier_mask.sum(),\n            \"outlier_indices\": slice_outliers.tolist()\n        }\n        \n        print(f\"    {s['name']}: {outlier_mask.sum()} outliers found\")\n    \n    # Create global clean mask (keep records that are outliers in fewer than min_outlier_count slices)\n    global_clean_mask = outlier_counts < min_outlier_count\n    n_global_outliers = (outlier_counts >= min_outlier_count).sum()\n    \n    print(f\"  Total global outliers (in {min_outlier_count}+ slices): {n_global_outliers}\")\n    \n    return global_clean_mask, outlier_details\n\ndef train_and_evaluate(train_df, test_df, model_slices):\n    n_samples = len(train_df)\n    \n    oof_preds = {\n        learner[\"name\"]: {s[\"name\"]: np.zeros(n_samples) for s in model_slices}\n        for learner in LEARNERS\n    }\n    test_preds = {\n        learner[\"name\"]: {s[\"name\"]: np.zeros(len(test_df)) for s in model_slices}\n        for learner in LEARNERS\n    }\n    \n    # Store detailed metrics\n    slice_metrics = {s[\"name\"]: {\"folds\": [], \"noise_analysis\": []} for s in model_slices}\n    \n    # For global clean slice, first identify global outliers\n    global_clean_masks = {}\n    for s in model_slices:\n        if s.get(\"global_clean\", False):\n            base_slices = [sl for sl in model_slices if not sl[\"clean\"]]\n            global_clean_mask, outlier_details = identify_global_outliers(\n                train_df, base_slices, s[\"noise_pct\"], MIN_OUTLIER_SLICES\n            )\n            global_clean_masks[s[\"noise_pct\"]] = global_clean_mask\n            slice_metrics[s[\"name\"]][\"outlier_details\"] = outlier_details\n\n    full_weights = create_time_decay_weights(n_samples)\n    kf = KFold(n_splits=Config.N_FOLDS, shuffle=False)\n\n    for fold, (train_idx, valid_idx) in enumerate(kf.split(train_df), start=1):\n        print(f\"\\n--- Fold {fold}/{Config.N_FOLDS} ---\")\n        X_valid = train_df.iloc[valid_idx][Config.FEATURES]\n        y_valid = train_df.iloc[valid_idx][Config.LABEL_COLUMN]\n\n        for s in model_slices:\n            cutoff = s[\"cutoff\"]\n            slice_name = s[\"name\"]\n            slice_type = s[\"type\"]\n            is_clean = s[\"clean\"]\n            noise_pct = s.get(\"noise_pct\", 0)\n            is_global_clean = s.get(\"global_clean\", False)\n            \n            if is_global_clean:\n                # Use global clean mask\n                global_mask = global_clean_masks[noise_pct]\n                \n                # Get training indices that pass the global clean mask\n                train_mask = global_mask[train_idx]\n                clean_train_idx = train_idx[train_mask]\n                \n                if len(clean_train_idx) == 0:\n                    print(f\"  Skipping slice: {slice_name} (no training data after global cleaning)\")\n                    continue\n                \n                X_train = train_df.iloc[clean_train_idx][Config.FEATURES]\n                y_train = train_df.iloc[clean_train_idx][Config.LABEL_COLUMN]\n                \n                # Get weights for the clean indices\n                sw = full_weights[clean_train_idx]\n                # Renormalize weights\n                sw = sw * len(sw) / sw.sum()\n                \n                X_train_np = X_train.values\n                y_train_np = y_train.values\n                \n                print(f\"  Training slice: {slice_name}, samples: {len(X_train)} (global outlier removal)\")\n                \n            elif slice_type == \"full\":\n                X_train = train_df.iloc[train_idx][Config.FEATURES]\n                y_train = train_df.iloc[train_idx][Config.LABEL_COLUMN]\n                sw = full_weights[train_idx]\n                \n            elif slice_type == \"recent\":\n                mask = train_idx >= cutoff\n                filtered_idx = train_idx[mask]\n                if len(filtered_idx) == 0:\n                    print(f\"  Skipping slice: {slice_name} (no training data in fold)\")\n                    continue\n                    \n                X_train = train_df.iloc[filtered_idx][Config.FEATURES]\n                y_train = train_df.iloc[filtered_idx][Config.LABEL_COLUMN]\n                \n                # Create weights for the subset\n                subset_positions = filtered_idx - cutoff\n                subset_weights = create_time_decay_weights(len(train_df) - cutoff)\n                sw = subset_weights[subset_positions]\n                    \n            elif slice_type == \"early\":\n                mask = train_idx < cutoff\n                filtered_idx = train_idx[mask]\n                if len(filtered_idx) == 0:\n                    print(f\"  Skipping slice: {slice_name} (no training data in fold)\")\n                    continue\n                    \n                X_train = train_df.iloc[filtered_idx][Config.FEATURES]\n                y_train = train_df.iloc[filtered_idx][Config.LABEL_COLUMN]\n                \n                # Create weights for the subset\n                subset_weights = create_time_decay_weights(cutoff)\n                sw = subset_weights[filtered_idx]\n\n            if not is_global_clean:\n                X_train_np = X_train.values\n                y_train_np = y_train.values\n            \n            X_valid_np = X_valid.values\n            y_valid_np = y_valid.values\n            \n            noise_analysis = None\n            \n            # For clean slices (but not global clean), identify and remove noise\n            if is_clean and not is_global_clean:\n                original_size = len(X_train)\n                clean_mask, noise_analysis = identify_slice_noise(X_train_np, y_train_np, sw, noise_pct)\n                \n                X_train_np = X_train_np[clean_mask]\n                y_train_np = y_train_np[clean_mask]\n                sw = sw[clean_mask]\n                \n                print(f\"  Training slice: {slice_name}, samples: {len(X_train_np)} (removed {original_size - len(X_train_np)} noisy records)\")\n                slice_metrics[slice_name][\"noise_analysis\"].append(noise_analysis)\n            elif not is_global_clean:\n                print(f\"  Training slice: {slice_name}, samples: {len(X_train)}\")\n\n            for learner in LEARNERS:\n                model = learner[\"Estimator\"](**learner[\"params\"])\n                model.fit(X_train_np, y_train_np, sample_weight=sw, \n                          eval_set=[(X_valid_np, y_valid_np)], verbose=False)\n                \n                # Get validation predictions\n                if is_global_clean or slice_type == \"full\":\n                    # For global clean and full data, predict on all validation samples\n                    preds = model.predict(X_valid)\n                    oof_preds[learner[\"name\"]][slice_name][valid_idx] = preds\n                    \n                    valid_score = pearsonr(y_valid, preds)[0]\n                    slice_metrics[slice_name][\"folds\"].append({\n                        \"fold\": fold,\n                        \"valid_score\": valid_score,\n                        \"n_valid\": len(valid_idx)\n                    })\n                    \n                elif slice_type == \"early\":\n                    mask = valid_idx < cutoff\n                    if mask.any():\n                        idxs = valid_idx[mask]\n                        preds = model.predict(train_df.iloc[idxs][Config.FEATURES])\n                        oof_preds[learner[\"name\"]][slice_name][idxs] = preds\n                        \n                        valid_score = pearsonr(train_df.iloc[idxs][Config.LABEL_COLUMN], preds)[0]\n                        slice_metrics[slice_name][\"folds\"].append({\n                            \"fold\": fold,\n                            \"valid_score\": valid_score,\n                            \"n_valid\": len(idxs)\n                        })\n                    \n                    if (~mask).any():\n                        base_name = \"full_data\"\n                        oof_preds[learner[\"name\"]][slice_name][valid_idx[~mask]] = oof_preds[learner[\"name\"]][base_name][valid_idx[~mask]]\n                else:\n                    mask = valid_idx >= cutoff if slice_type == \"recent\" else np.ones(len(valid_idx), dtype=bool)\n                    if mask.any():\n                        idxs = valid_idx[mask]\n                        preds = model.predict(train_df.iloc[idxs][Config.FEATURES])\n                        oof_preds[learner[\"name\"]][slice_name][idxs] = preds\n                        \n                        valid_score = pearsonr(train_df.iloc[idxs][Config.LABEL_COLUMN], preds)[0]\n                        slice_metrics[slice_name][\"folds\"].append({\n                            \"fold\": fold,\n                            \"valid_score\": valid_score,\n                            \"n_valid\": len(idxs)\n                        })\n                    \n                    if slice_type == \"recent\" and cutoff > 0 and (~mask).any():\n                        base_name = \"full_data\"\n                        oof_preds[learner[\"name\"]][slice_name][valid_idx[~mask]] = oof_preds[learner[\"name\"]][base_name][valid_idx[~mask]]\n\n                # Test predictions\n                test_preds[learner[\"name\"]][slice_name] += model.predict(test_df[Config.FEATURES])\n\n    # Normalize test predictions\n    for learner_name in test_preds:\n        for slice_name in test_preds[learner_name]:\n            test_preds[learner_name][slice_name] /= Config.N_FOLDS\n\n    return oof_preds, test_preds, slice_metrics\n\n# Analysis and Submission\ndef analyze_generalization(slice_metrics):\n    \"\"\"Analyze how noise removal affects generalization\"\"\"\n    print(\"\\n\" + \"=\"*80)\n    print(\"GENERALIZATION ANALYSIS\")\n    print(\"=\"*80)\n    \n    for slice_name, metrics in slice_metrics.items():\n        if metrics[\"folds\"]:\n            valid_scores = [f[\"valid_score\"] for f in metrics[\"folds\"]]\n            avg_valid = np.mean(valid_scores)\n            std_valid = np.std(valid_scores)\n            \n            print(f\"\\n{slice_name}:\")\n            print(f\"  Average validation score: {avg_valid:.4f} (±{std_valid:.4f})\")\n            \n            if metrics[\"noise_analysis\"]:\n                train_before = np.mean([n[\"train_score_before\"] for n in metrics[\"noise_analysis\"]])\n                train_after = np.mean([n[\"train_score_after\"] for n in metrics[\"noise_analysis\"]])\n                avg_removed = np.mean([n[\"removed_count\"] for n in metrics[\"noise_analysis\"]])\n                \n                print(f\"  Training score before cleaning: {train_before:.4f}\")\n                print(f\"  Training score after cleaning: {train_after:.4f}\")\n                print(f\"  Average samples removed: {avg_removed:.0f}\")\n                print(f\"  Generalization gap (train-valid): {train_after - avg_valid:.4f}\")\n            \n            # Special handling for global clean slice\n            if \"outlier_details\" in metrics:\n                print(f\"  Global outlier analysis:\")\n                for slice_name_detail, details in metrics[\"outlier_details\"].items():\n                    print(f\"    - {slice_name_detail}: {details['n_outliers']} outliers\")\n\ndef create_intelligent_ensemble(train_df, oof_preds, test_preds, slice_scores, min_score=0.08):\n    \"\"\"Create ensemble using only slices above threshold\"\"\"\n    print(f\"\\nCreating intelligent ensemble (min score: {min_score})\")\n    \n    learner_ensembles = {}\n    \n    for learner_name in oof_preds:\n        # Filter slices by score\n        good_slices = {s: score for s, score in slice_scores[learner_name].items() if score >= min_score}\n        \n        if not good_slices:\n            print(f\"Warning: No slices above threshold for {learner_name}, using all slices\")\n            good_slices = slice_scores[learner_name]\n        \n        print(f\"\\n{learner_name}: Using {len(good_slices)}/{len(slice_scores[learner_name])} slices\")\n        for s, score in good_slices.items():\n            print(f\"  - {s}: {score:.4f}\")\n        \n        total_score = sum(good_slices.values())\n        \n        # Weighted ensemble of good slices\n        oof_intelligent = sum(good_slices[s] / total_score * oof_preds[learner_name][s] for s in good_slices)\n        test_intelligent = sum(good_slices[s] / total_score * test_preds[learner_name][s] for s in good_slices)\n        \n        score_intelligent = pearsonr(train_df[Config.LABEL_COLUMN], oof_intelligent)[0]\n        \n        learner_ensembles[learner_name] = {\n            \"oof\": oof_intelligent,\n            \"test\": test_intelligent,\n            \"score\": score_intelligent,\n            \"n_slices\": len(good_slices)\n        }\n    \n    return learner_ensembles\n\ndef save_submission(submission_df, predictions, filename_suffix, description):\n    \"\"\"Save submission with clear naming\"\"\"\n    filename = f\"submission_{filename_suffix}.csv\"\n    submission_copy = submission_df.copy()\n    submission_copy[\"prediction\"] = predictions\n    submission_copy.to_csv(filename, index=False)\n    print(f\"\\nSaved: {filename}\")\n    print(f\"Description: {description}\")\n    return filename\n\n# Main execution\ndef run_all_experiments(train_df, test_df, submission_df):\n    \"\"\"Run all experiments with different noise removal percentages\"\"\"\n    \n    all_results = {}\n    \n    # 1. First create baseline (original 4 slices only)\n    print(\"\\n\" + \"=\"*80)\n    print(\"CREATING BASELINE (Original 4 slices only)\")\n    print(\"=\"*80)\n    \n    baseline_slices = get_model_slices(len(train_df), include_clean=False)\n    oof_baseline, test_baseline, metrics_baseline = train_and_evaluate(train_df, test_df, baseline_slices)\n    \n    # Calculate baseline scores\n    baseline_scores = {}\n    for learner_name in oof_baseline:\n        baseline_scores[learner_name] = {}\n        for s in oof_baseline[learner_name]:\n            score = pearsonr(train_df[Config.LABEL_COLUMN], oof_baseline[learner_name][s])[0]\n            baseline_scores[learner_name][s] = score\n    \n    # Create baseline ensemble\n    final_baseline = np.mean([\n        np.mean(list(test_baseline[learner_name].values()), axis=0) \n        for learner_name in test_baseline\n    ], axis=0)\n    \n    save_submission(submission_df, final_baseline, \n                   f\"baseline_4slices_early{int(EARLY_PERCENTAGE*100)}pct\",\n                   \"Baseline ensemble of 4 original time slices (no cleaning)\")\n    \n    all_results[\"baseline\"] = {\n        \"scores\": baseline_scores,\n        \"final_score\": pearsonr(train_df[Config.LABEL_COLUMN], \n                               np.mean([np.mean(list(oof_baseline[learner_name].values()), axis=0) \n                                       for learner_name in oof_baseline], axis=0))[0]\n    }\n    \n    # 2. Run experiments with different noise removal percentages\n    for noise_pct in NOISE_REMOVAL_PERCENTAGES:\n        print(f\"\\n\" + \"=\"*80)\n        print(f\"EXPERIMENT: Noise Removal = {noise_pct*100:.2f}%\")\n        print(\"=\"*80)\n        \n        # Get slices with this noise removal percentage\n        model_slices = get_model_slices(len(train_df), include_clean=True, noise_pct=noise_pct)\n        \n        # Train and evaluate\n        oof_preds, test_preds, slice_metrics = train_and_evaluate(train_df, test_df, model_slices)\n        \n        # Analyze generalization\n        analyze_generalization(slice_metrics)\n        \n        # Calculate all slice scores\n        slice_scores = {}\n        for learner_name in oof_preds:\n            slice_scores[learner_name] = {}\n            for s in oof_preds[learner_name]:\n                score = pearsonr(train_df[Config.LABEL_COLUMN], oof_preds[learner_name][s])[0]\n                slice_scores[learner_name][s] = score\n        \n        # Save results\n        experiment_key = f\"noise_{int(noise_pct*10000)/100}pct\"\n        all_results[experiment_key] = {\n            \"noise_pct\": noise_pct,\n            \"scores\": slice_scores,\n            \"metrics\": slice_metrics\n        }\n        \n        # Create different ensemble combinations\n        # Each ensemble represents a different hypothesis about which slices are most valuable\n        \n        # a) All slices - simple average (hypothesis: all perspectives equally valuable)\n        n_total_slices = len(model_slices)\n        final_all_simple = np.mean([\n            np.mean(list(test_preds[learner_name].values()), axis=0) \n            for learner_name in test_preds\n        ], axis=0)\n        \n        save_submission(submission_df, final_all_simple,\n                       f\"all{n_total_slices}slices_simple_noise{int(noise_pct*10000)/100}pct_early{int(EARLY_PERCENTAGE*100)}pct\",\n                       f\"Simple average of all {n_total_slices} slices with {noise_pct*100:.2f}% noise removal\")\n        \n        # b) All slices - weighted by score\n        final_all_weighted = np.zeros(len(test_df))\n        total_weight = 0\n        for learner_name in test_preds:\n            for slice_name, preds in test_preds[learner_name].items():\n                weight = slice_scores[learner_name][slice_name]\n                final_all_weighted += weight * preds\n                total_weight += weight\n        final_all_weighted /= total_weight\n        \n        save_submission(submission_df, final_all_weighted,\n                       f\"all{n_total_slices}slices_weighted_noise{int(noise_pct*10000)/100}pct_early{int(EARLY_PERCENTAGE*100)}pct\",\n                       f\"Score-weighted average of all {n_total_slices} slices with {noise_pct*100:.2f}% noise removal\")\n        \n        # c) Clean slices only (excluding global clean)\n        clean_preds = []\n        for learner_name in test_preds:\n            learner_clean = [test_preds[learner_name][s] for s in test_preds[learner_name] \n                           if \"_clean\" in s and \"global_clean\" not in s]\n            if learner_clean:\n                clean_preds.append(np.mean(learner_clean, axis=0))\n        \n        if clean_preds:\n            final_clean_only = np.mean(clean_preds, axis=0)\n            save_submission(submission_df, final_clean_only,\n                           f\"clean4slices_only_noise{int(noise_pct*10000)/100}pct_early{int(EARLY_PERCENTAGE*100)}pct\",\n                           f\"Clean slices only (4 slices) with {noise_pct*100:.2f}% noise removal\")\n        \n        # d) Global clean only\n        global_clean_name = f\"global_clean{int(noise_pct*10000)/100}pct\"\n        if global_clean_name in test_preds[list(test_preds.keys())[0]]:\n            final_global_clean = np.mean([\n                test_preds[learner_name][global_clean_name]\n                for learner_name in test_preds\n            ], axis=0)\n            \n            save_submission(submission_df, final_global_clean,\n                           f\"globalclean_only_noise{int(noise_pct*10000)/100}pct_early{int(EARLY_PERCENTAGE*100)}pct\",\n                           f\"Global clean only with {noise_pct*100:.2f}% noise removal\")\n        \n        # e) All clean slices including global\n        all_clean_preds = []\n        for learner_name in test_preds:\n            learner_all_clean = [test_preds[learner_name][s] for s in test_preds[learner_name] \n                                if \"_clean\" in s]\n            if learner_all_clean:\n                all_clean_preds.append(np.mean(learner_all_clean, axis=0))\n        \n        if all_clean_preds:\n            final_all_clean = np.mean(all_clean_preds, axis=0)\n            save_submission(submission_df, final_all_clean,\n                           f\"allclean5slices_noise{int(noise_pct*10000)/100}pct_early{int(EARLY_PERCENTAGE*100)}pct\",\n                           f\"All clean slices (5 total) with {noise_pct*100:.2f}% noise removal\")\n        \n        # f) Intelligent ensemble (above threshold)\n        intelligent_ensembles = create_intelligent_ensemble(\n            train_df, oof_preds, test_preds, slice_scores, MIN_SCORE_THRESHOLD\n        )\n        \n        final_intelligent = np.mean([\n            le[\"test\"] for le in intelligent_ensembles.values()\n        ], axis=0)\n        \n        save_submission(submission_df, final_intelligent,\n                       f\"intelligent_noise{int(noise_pct*10000)/100}pct_early{int(EARLY_PERCENTAGE*100)}pct\",\n                       f\"Intelligent ensemble (score>{MIN_SCORE_THRESHOLD}) with {noise_pct*100:.2f}% noise removal\")\n        \n        # g) Top 3 slices only\n        top_3_slices = {}\n        for learner_name in slice_scores:\n            sorted_slices = sorted(slice_scores[learner_name].items(), key=lambda x: x[1], reverse=True)[:3]\n            top_3_slices[learner_name] = dict(sorted_slices)\n        \n        final_top3 = np.zeros(len(test_df))\n        total_weight = 0\n        for learner_name in test_preds:\n            for slice_name, score in top_3_slices[learner_name].items():\n                final_top3 += score * test_preds[learner_name][slice_name]\n                total_weight += score\n        final_top3 /= total_weight\n        \n        save_submission(submission_df, final_top3,\n                       f\"top3slices_noise{int(noise_pct*10000)/100}pct_early{int(EARLY_PERCENTAGE*100)}pct\",\n                       f\"Top 3 performing slices with {noise_pct*100:.2f}% noise removal\")\n    \n    # 3. Save summary report\n    print(\"\\n\" + \"=\"*80)\n    print(\"SUMMARY REPORT\")\n    print(\"=\"*80)\n    \n    summary = {\n        \"early_percentage\": EARLY_PERCENTAGE,\n        \"experiments\": {}\n    }\n    \n    for exp_name, exp_data in all_results.items():\n        if exp_name == \"baseline\":\n            summary[\"experiments\"][exp_name] = {\n                \"final_score\": exp_data[\"final_score\"],\n                \"n_slices\": 4\n            }\n        else:\n            # Calculate average scores for clean vs original\n            avg_clean = np.mean([\n                score for learner_scores in exp_data[\"scores\"].values()\n                for slice_name, score in learner_scores.items()\n                if \"_clean\" in slice_name and \"global_clean\" not in slice_name\n            ])\n            avg_original = np.mean([\n                score for learner_scores in exp_data[\"scores\"].values()\n                for slice_name, score in learner_scores.items()\n                if \"_clean\" not in slice_name\n            ])\n            \n            # Get global clean score if available\n            global_clean_scores = [\n                score for learner_scores in exp_data[\"scores\"].values()\n                for slice_name, score in learner_scores.items()\n                if \"global_clean\" in slice_name\n            ]\n            avg_global_clean = np.mean(global_clean_scores) if global_clean_scores else None\n            \n            summary[\"experiments\"][exp_name] = {\n                \"noise_pct\": exp_data[\"noise_pct\"],\n                \"avg_clean_score\": avg_clean,\n                \"avg_original_score\": avg_original,\n                \"avg_global_clean_score\": avg_global_clean,\n                \"improvement\": avg_clean - avg_original\n            }\n    \n    # Save summary as JSON\n    with open(f\"experiment_summary_early{int(EARLY_PERCENTAGE*100)}pct.json\", \"w\") as f:\n        json.dump(summary, f, indent=2)\n    \n    print(\"\\nExperiment Summary:\")\n    print(f\"{'Experiment':<20} {'Clean Avg':<12} {'Original Avg':<12} {'Global Clean':<12} {'Improvement':<12}\")\n    print(\"-\" * 68)\n    \n    for exp_name, exp_info in summary[\"experiments\"].items():\n        if exp_name == \"baseline\":\n            print(f\"{exp_name:<20} {'N/A':<12} {exp_info['final_score']:.4f}\")\n        else:\n            clean_avg = f\"{exp_info['avg_clean_score']:.4f}\"\n            orig_avg = f\"{exp_info['avg_original_score']:.4f}\"\n            global_avg = f\"{exp_info['avg_global_clean_score']:.4f}\" if exp_info['avg_global_clean_score'] else \"N/A\"\n            improvement = f\"{exp_info['improvement']:.4f}\"\n            print(f\"{exp_name:<20} {clean_avg:<12} {orig_avg:<12} {global_avg:<12} {improvement:<12}\")\n\n# Main\nif __name__ == \"__main__\":\n    print(f\"\\nRunning experiments with EARLY_PERCENTAGE = {EARLY_PERCENTAGE} ({int(EARLY_PERCENTAGE*100)}%)\")\n    print(f\"Testing noise removal percentages: {[p*100 for p in NOISE_REMOVAL_PERCENTAGES]}%\")\n    print(f\"Global outliers: Records appearing as outliers in {MIN_OUTLIER_SLICES}+ slices\")\n    print(f\"\\nThis will create:\")\n    print(f\"  - 1 baseline file (4 original slices)\")\n    print(f\"  - 7 files per noise level × {len(NOISE_REMOVAL_PERCENTAGES)} levels = {7 * len(NOISE_REMOVAL_PERCENTAGES)} files\")\n    print(f\"  - 1 JSON summary\")\n    print(f\"  - Total: {1 + 7 * len(NOISE_REMOVAL_PERCENTAGES) + 1} files\")\n    \n    train_df, test_df, submission_df = load_data()\n    run_all_experiments(train_df, test_df, submission_df)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}