{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Clustering-Based Crypto Price Prediction\n# Advanced Non-Parametric Approach using Cluster Representatives\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.cluster import KMeans, MiniBatchKMeans, AgglomerativeClustering, DBSCAN\nfrom sklearn.preprocessing import StandardScaler, RobustScaler, MinMaxScaler\nfrom sklearn.metrics import silhouette_score, calinski_harabasz_score\nfrom sklearn.neighbors import NearestNeighbors\nfrom sklearn.decomposition import PCA\nfrom sklearn.manifold import TSNE\nfrom scipy.spatial.distance import cdist\nfrom scipy.stats import pearsonr\nimport warnings\nwarnings.filterwarnings('ignore')\n\nprint(\"=== CLUSTERING-BASED CRYPTO PRICE PREDICTION ===\")\nprint(\"Non-Parametric Approach using Cluster Representatives\")\nprint(\"=\"*60)","metadata":{"_uuid":"8681256c-b43b-4242-bb08-23d2d09e11cb","_cell_guid":"bf31ba35-8f2c-417e-9d4a-2be960d587da","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-07-22T13:51:14.812586Z","iopub.execute_input":"2025-07-22T13:51:14.812791Z","iopub.status.idle":"2025-07-22T13:51:20.881117Z","shell.execute_reply.started":"2025-07-22T13:51:14.812774Z","shell.execute_reply":"2025-07-22T13:51:20.880293Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ============================================================================\n# 1. ENHANCED DATA LOADING AND PREPROCESSING\n# ============================================================================","metadata":{"_uuid":"8681256c-b43b-4242-bb08-23d2d09e11cb","_cell_guid":"bf31ba35-8f2c-417e-9d4a-2be960d587da","collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"def load_and_preprocess_for_clustering():\n    \"\"\"Load and preprocess data specifically for clustering approach\"\"\"\n    print(\"\\n1. Loading and Preprocessing Data for Clustering...\")\n    \n    # Load data\n    train_df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\n    test_df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')\n\n    start_date = \"2023-12-01 00:00:00\"\n    end_date = \"2024-02-29 23:59:00\"\n\n    # Filter the DataFrame and update train_df with the subset\n    train_df = train_df.loc[start_date:end_date]\n    \n    print(f\"Training data shape: {train_df.shape}\")\n    print(f\"Test data shape: {test_df.shape}\")\n\n    for col in train_df.columns:\n        if train_df[col].dtype != object:\n            if train_df[col].dtype == 'float64':\n                train_df[col] = train_df[col].astype(np.float32)\n            elif df[col].dtype == 'int64' :\n                train_df[col] = train_df[col].astype(np.int32)\n                \n    \n    # Get feature columns\n    feature_cols = [col for col in train_df.columns if col.startswith('X')]\n    public_cols = ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']\n    all_feature_cols = feature_cols + public_cols\n    \n    print(f\"Total features for clustering: {len(all_feature_cols)}\")\n    \n    # Handle missing values\n    train_df[all_feature_cols] = train_df[all_feature_cols].fillna(method='ffill').fillna(0)\n    test_df[all_feature_cols] = test_df[all_feature_cols].fillna(method='ffill').fillna(0)\n    \n    # Remove infinite values\n    train_df = train_df.replace([np.inf, -np.inf], np.nan).fillna(0)\n    test_df = test_df.replace([np.inf, -np.inf], np.nan).fillna(0)\n    \n    return train_df, test_df, all_feature_cols","metadata":{"_uuid":"8681256c-b43b-4242-bb08-23d2d09e11cb","_cell_guid":"bf31ba35-8f2c-417e-9d4a-2be960d587da","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-07-22T13:51:20.882194Z","iopub.execute_input":"2025-07-22T13:51:20.882971Z","iopub.status.idle":"2025-07-22T13:51:20.891136Z","shell.execute_reply.started":"2025-07-22T13:51:20.882941Z","shell.execute_reply":"2025-07-22T13:51:20.890361Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ============================================================================\n# 2. FEATURE ENGINEERING FOR CLUSTERING\n# ============================================================================","metadata":{"_uuid":"8681256c-b43b-4242-bb08-23d2d09e11cb","_cell_guid":"bf31ba35-8f2c-417e-9d4a-2be960d587da","collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"def create_clustering_features(df):\n    \"\"\"Create features specifically useful for clustering\"\"\"\n    print(\"Creating clustering-specific features...\")\n    \n    df = df.copy()\n    \n    # Market microstructure features\n    df['bid_ask_spread'] = df['ask_qty'] - df['bid_qty']\n    df['bid_ask_ratio'] = df['bid_qty'] / (df['ask_qty'] + 1e-8)\n    df['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-8)\n    df['net_flow'] = df['buy_qty'] - df['sell_qty']\n    df['total_liquidity'] = df['bid_qty'] + df['ask_qty']\n    \n    # Intensity features\n    df['buy_intensity'] = df['buy_qty'] / (df['volume'] + 1e-8)\n    df['sell_intensity'] = df['sell_qty'] / (df['volume'] + 1e-8)\n    \n    # Volatility proxies\n    for col in ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']:\n        df[f'{col}_rolling_mean_5'] = df[col].rolling(window=5).mean()\n        df[f'{col}_rolling_std_5'] = df[col].rolling(window=5).std()\n        df[f'{col}_rolling_mean_15'] = df[col].rolling(window=15).mean()\n        df[f'{col}_z_score'] = (df[col] - df[f'{col}_rolling_mean_15']) / (df[f'{col}_rolling_std_5'] + 1e-8)\n    \n    # Remove columns with NaN values after feature creation\n    df = df.fillna(method='ffill').fillna(0)\n    \n    return df","metadata":{"_uuid":"8681256c-b43b-4242-bb08-23d2d09e11cb","_cell_guid":"bf31ba35-8f2c-417e-9d4a-2be960d587da","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-07-22T13:51:20.892698Z","iopub.execute_input":"2025-07-22T13:51:20.892957Z","iopub.status.idle":"2025-07-22T13:51:20.924733Z","shell.execute_reply.started":"2025-07-22T13:51:20.892935Z","shell.execute_reply":"2025-07-22T13:51:20.923992Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ============================================================================\n# 3. CLUSTERING APPROACH IMPLEMENTATION\n# ============================================================================","metadata":{"_uuid":"8681256c-b43b-4242-bb08-23d2d09e11cb","_cell_guid":"bf31ba35-8f2c-417e-9d4a-2be960d587da","collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"class ClusteringPredictor:\n    \"\"\"\n    Clustering-based predictor that assigns labels based on cluster representatives\n    \"\"\"\n    \n    def __init__(self, n_clusters=1000, clustering_method='kmeans', scaler_type='standard'):\n        self.n_clusters = n_clusters\n        self.clustering_method = clustering_method\n        self.scaler_type = scaler_type\n        self.scaler = None\n        self.clusterer = None\n        self.cluster_representatives = None\n        self.cluster_labels = None\n        self.is_fitted = False\n        \n    def _initialize_scaler(self):\n        \"\"\"Initialize scaler based on type\"\"\"\n        if self.scaler_type == 'standard':\n            self.scaler = StandardScaler()\n        elif self.scaler_type == 'robust':\n            self.scaler = RobustScaler()\n        elif self.scaler_type == 'minmax':\n            self.scaler = MinMaxScaler()\n        else:\n            raise ValueError(\"scaler_type must be 'standard', 'robust', or 'minmax'\")\n    \n    def _initialize_clusterer(self):\n        \"\"\"Initialize clustering algorithm\"\"\"\n        if self.clustering_method == 'kmeans':\n            self.clusterer = KMeans(n_clusters=self.n_clusters, random_state=42, n_init=10)\n        elif self.clustering_method == 'minibatch_kmeans':\n            self.clusterer = MiniBatchKMeans(n_clusters=self.n_clusters, random_state=42, batch_size=1000)\n        else:\n            raise ValueError(\"clustering_method must be 'kmeans' or 'minibatch_kmeans'\")\n    \n    def fit(self, X, y):\n        \"\"\"\n        Fit the clustering model\n        X: feature matrix (n_samples, n_features)\n        y: target values (n_samples,)\n        \"\"\"\n        print(f\"\\nFitting Clustering Predictor:\")\n        print(f\"  Samples: {X.shape[0]}\")\n        print(f\"  Features: {X.shape[1]}\")\n        print(f\"  Clusters: {self.n_clusters}\")\n        print(f\"  Method: {self.clustering_method}\")\n        \n        # Initialize components\n        self._initialize_scaler()\n        self._initialize_clusterer()\n        \n        # Scale features\n        X_scaled = self.scaler.fit_transform(X)\n        \n        # Perform clustering\n        print(\"Performing clustering...\")\n        cluster_assignments = self.clusterer.fit_predict(X_scaled)\n        \n        # Calculate cluster representatives and their labels\n        print(\"Calculating cluster representatives...\")\n        self.cluster_representatives = {}\n        self.cluster_labels = {}\n        \n        for cluster_id in range(self.n_clusters):\n            cluster_mask = cluster_assignments == cluster_id\n            \n            if cluster_mask.sum() > 0:  # Check if cluster has any points\n                # Representative is the centroid (mean of points in cluster)\n                cluster_points = X_scaled[cluster_mask]\n                representative = np.mean(cluster_points, axis=0)\n                \n                # Label is the mean of target values in the cluster\n                cluster_target_values = y[cluster_mask]\n                representative_label = np.mean(cluster_target_values)\n                \n                self.cluster_representatives[cluster_id] = representative\n                self.cluster_labels[cluster_id] = representative_label\n        \n        print(f\"Created {len(self.cluster_representatives)} non-empty clusters\")\n        \n        # Calculate clustering quality metrics\n        if len(self.cluster_representatives) > 1:\n            try:\n                silhouette_avg = silhouette_score(X_scaled, cluster_assignments)\n                calinski_score = calinski_harabasz_score(X_scaled, cluster_assignments)\n                print(f\"Silhouette Score: {silhouette_avg:.4f}\")\n                print(f\"Calinski-Harabasz Score: {calinski_score:.4f}\")\n            except:\n                print(\"Could not calculate clustering quality metrics\")\n        \n        self.is_fitted = True\n        return self\n    \n    def predict(self, X):\n        \"\"\"\n        Predict using cluster representatives\n        \"\"\"\n        if not self.is_fitted:\n            raise ValueError(\"Model must be fitted before prediction\")\n        \n        print(f\"Making predictions for {X.shape[0]} samples...\")\n        \n        # Scale test features\n        X_scaled = self.scaler.transform(X)\n        \n        # Find closest cluster representative for each test point\n        predictions = np.zeros(X.shape[0])\n        \n        # Convert cluster representatives to array for efficient distance calculation\n        if len(self.cluster_representatives) == 0:\n            return predictions\n        \n        cluster_ids = list(self.cluster_representatives.keys())\n        representatives_array = np.array([self.cluster_representatives[cid] for cid in cluster_ids])\n        \n        # Calculate distances and find closest clusters\n        distances = cdist(X_scaled, representatives_array, metric='euclidean')\n        closest_cluster_indices = np.argmin(distances, axis=1)\n        \n        # Assign predictions based on closest cluster representatives\n        for i, closest_idx in enumerate(closest_cluster_indices):\n            closest_cluster_id = cluster_ids[closest_idx]\n            predictions[i] = self.cluster_labels[closest_cluster_id]\n        \n        return predictions\n    \n    def get_cluster_statistics(self):\n        \"\"\"Get statistics about the clusters\"\"\"\n        if not self.is_fitted:\n            return None\n        \n        cluster_stats = {\n            'n_clusters': len(self.cluster_representatives),\n            'label_stats': {\n                'mean': np.mean(list(self.cluster_labels.values())),\n                'std': np.std(list(self.cluster_labels.values())),\n                'min': np.min(list(self.cluster_labels.values())),\n                'max': np.max(list(self.cluster_labels.values()))\n            }\n        }\n        \n        return cluster_stats","metadata":{"_uuid":"8681256c-b43b-4242-bb08-23d2d09e11cb","_cell_guid":"bf31ba35-8f2c-417e-9d4a-2be960d587da","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-07-22T13:51:20.925420Z","iopub.execute_input":"2025-07-22T13:51:20.925665Z","iopub.status.idle":"2025-07-22T13:51:20.952451Z","shell.execute_reply.started":"2025-07-22T13:51:20.925648Z","shell.execute_reply":"2025-07-22T13:51:20.951753Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ============================================================================\n# 4. ADVANCED CLUSTERING STRATEGIES\n# ============================================================================","metadata":{"_uuid":"8681256c-b43b-4242-bb08-23d2d09e11cb","_cell_guid":"bf31ba35-8f2c-417e-9d4a-2be960d587da","collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"class EnsembleClusteringPredictor:\n    \"\"\"\n    Ensemble of multiple clustering approaches\n    \"\"\"\n    \n    def __init__(self):\n        self.predictors = {}\n        self.weights = {}\n        self.is_fitted = False\n    \n    def fit(self, X, y, validation_split=0.2):\n        \"\"\"Fit ensemble of clustering predictors\"\"\"\n        print(\"\\nFitting Ensemble Clustering Predictors...\")\n        \n        # Split data for validation\n        split_idx = int(len(X) * (1 - validation_split))\n        X_train, X_val = X[:split_idx], X[split_idx:]\n        y_train, y_val = y[:split_idx], y[split_idx:]\n        \n        # Different clustering configurations\n        configs = [\n            {'n_clusters': 500, 'method': 'kmeans', 'scaler': 'standard'},\n            {'n_clusters': 1000, 'method': 'kmeans', 'scaler': 'standard'},\n            {'n_clusters': 1500, 'method': 'kmeans', 'scaler': 'standard'},\n            {'n_clusters': 1000, 'method': 'kmeans', 'scaler': 'robust'},\n            {'n_clusters': 1000, 'method': 'minibatch_kmeans', 'scaler': 'standard'},\n            {'n_clusters': 2000, 'method': 'minibatch_kmeans', 'scaler': 'standard'},\n        ]\n        \n        val_scores = {}\n        \n        for i, config in enumerate(configs):\n            config_name = f\"config_{i+1}_{config['n_clusters']}_{config['method']}_{config['scaler']}\"\n            print(f\"\\nTraining {config_name}...\")\n            \n            predictor = ClusteringPredictor(\n                n_clusters=config['n_clusters'],\n                clustering_method=config['method'],\n                scaler_type=config['scaler']\n            )\n            \n            try:\n                predictor.fit(X_train, y_train)\n                val_pred = predictor.predict(X_val)\n                \n                # Calculate validation correlation\n                val_corr = pearsonr(y_val, val_pred)[0]\n                if np.isnan(val_corr):\n                    val_corr = 0\n                \n                val_scores[config_name] = max(0, val_corr)  # Only positive correlations\n                self.predictors[config_name] = predictor\n                \n                print(f\"Validation correlation: {val_corr:.4f}\")\n                \n            except Exception as e:\n                print(f\"Failed to train {config_name}: {e}\")\n                val_scores[config_name] = 0\n        \n        # Calculate ensemble weights\n        total_score = sum(val_scores.values())\n        if total_score > 0:\n            self.weights = {name: score / total_score for name, score in val_scores.items()}\n        else:\n            # Equal weights if no positive correlations\n            self.weights = {name: 1.0 / len(self.predictors) for name in self.predictors.keys()}\n        \n        print(f\"\\nEnsemble weights:\")\n        for name, weight in self.weights.items():\n            print(f\"  {name}: {weight:.4f}\")\n        \n        self.is_fitted = True\n        return self\n    \n    def predict(self, X):\n        \"\"\"Make ensemble predictions\"\"\"\n        if not self.is_fitted:\n            raise ValueError(\"Ensemble must be fitted before prediction\")\n        \n        print(f\"\\nMaking ensemble predictions...\")\n        predictions = np.zeros(X.shape[0])\n        \n        for name, predictor in self.predictors.items():\n            if name in self.weights and self.weights[name] > 0:\n                try:\n                    pred = predictor.predict(X)\n                    predictions += self.weights[name] * pred\n                    print(f\"Added predictions from {name} (weight: {self.weights[name]:.4f})\")\n                except Exception as e:\n                    print(f\"Failed to get predictions from {name}: {e}\")\n        \n        return predictions","metadata":{"_uuid":"8681256c-b43b-4242-bb08-23d2d09e11cb","_cell_guid":"bf31ba35-8f2c-417e-9d4a-2be960d587da","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-07-22T13:51:20.953279Z","iopub.execute_input":"2025-07-22T13:51:20.953543Z","iopub.status.idle":"2025-07-22T13:51:20.976624Z","shell.execute_reply.started":"2025-07-22T13:51:20.953504Z","shell.execute_reply":"2025-07-22T13:51:20.975978Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ============================================================================\n# 5. FEATURE SELECTION FOR CLUSTERING\n# ============================================================================","metadata":{"_uuid":"8681256c-b43b-4242-bb08-23d2d09e11cb","_cell_guid":"bf31ba35-8f2c-417e-9d4a-2be960d587da","collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"def select_features_for_clustering(X, y, n_features=200):\n    \"\"\"Select most informative features for clustering\"\"\"\n    print(f\"\\nSelecting top {n_features} features for clustering...\")\n    \n    from sklearn.feature_selection import SelectKBest, mutual_info_regression, f_regression\n    from sklearn.ensemble import RandomForestRegressor\n    \n    # Method 1: Mutual Information\n    mi_selector = SelectKBest(score_func=mutual_info_regression, k=n_features)\n    mi_selector.fit(X, y)\n    mi_features = mi_selector.get_support()\n    \n    # Method 2: F-regression\n    f_selector = SelectKBest(score_func=f_regression, k=n_features)\n    f_selector.fit(X, y)\n    f_features = f_selector.get_support()\n    \n    # Method 3: Random Forest Feature Importance\n    rf = RandomForestRegressor(n_estimators=100, random_state=42, n_jobs=-1)\n    rf.fit(X, y)\n    rf_importance = rf.feature_importances_\n    rf_top_indices = np.argsort(rf_importance)[-n_features:]\n    rf_features = np.zeros(X.shape[1], dtype=bool)\n    rf_features[rf_top_indices] = True\n    \n    # Combine features (union of all methods)\n    combined_features = mi_features | f_features | rf_features\n    \n    print(f\"Selected {combined_features.sum()} features total\")\n    print(f\"  Mutual Info: {mi_features.sum()}\")\n    print(f\"  F-regression: {f_features.sum()}\")\n    print(f\"  Random Forest: {rf_features.sum()}\")\n    \n    return combined_features","metadata":{"_uuid":"8681256c-b43b-4242-bb08-23d2d09e11cb","_cell_guid":"bf31ba35-8f2c-417e-9d4a-2be960d587da","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-07-22T13:51:20.977449Z","iopub.execute_input":"2025-07-22T13:51:20.977884Z","iopub.status.idle":"2025-07-22T13:51:21.007578Z","shell.execute_reply.started":"2025-07-22T13:51:20.977855Z","shell.execute_reply":"2025-07-22T13:51:21.006821Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ============================================================================\n# 6. MAIN CLUSTERING PIPELINE\n# ============================================================================","metadata":{"_uuid":"8681256c-b43b-4242-bb08-23d2d09e11cb","_cell_guid":"bf31ba35-8f2c-417e-9d4a-2be960d587da","collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"def main_clustering_pipeline():\n    \"\"\"Main clustering-based prediction pipeline\"\"\"\n    print(\"Starting Clustering-Based Prediction Pipeline...\")\n    \n    # Load and preprocess data\n    train_df, test_df, base_feature_cols = load_and_preprocess_for_clustering()\n    \n    # Create clustering-specific features\n    train_df = create_clustering_features(train_df)\n    test_df = create_clustering_features(test_df)\n    \n    # Get all available features\n    all_feature_cols = [col for col in train_df.columns if col not in ['timestamp', 'label']]\n    print(f\"Total features available: {len(all_feature_cols)}\")\n    \n    # Prepare data\n    X = train_df[all_feature_cols].values\n    y = train_df['label'].values\n    X_test = test_df[all_feature_cols].values\n    \n    # Feature selection for clustering\n    selected_features = select_features_for_clustering(X, y, n_features=50)\n    X_selected = X[:, selected_features]\n    X_test_selected = X_test[:, selected_features]\n    \n    print(f\"Using {X_selected.shape[1]} selected features for clustering\")\n    \n    # Use recent data as suggested in competition\n    if 'timestamp' in train_df.columns:\n        recent_cutoff = train_df.index.max() - pd.DateOffset(months=2)\n        recent_mask = train_df.index >= recent_cutoff\n        \n        X_recent = X_selected[recent_mask]\n        y_recent = y[recent_mask]\n        \n        print(f\"Using recent 6 months: {len(X_recent)} samples\")\n    else:\n        # Use all data if no timestamp\n        X_recent = X_selected\n        y_recent = y\n        print(f\"Using all available data: {len(X_recent)} samples\")\n    \n    # Train ensemble clustering predictor\n    ensemble_predictor = EnsembleClusteringPredictor()\n    ensemble_predictor.fit(X_recent, y_recent, validation_split=0.2)\n    \n    # Make predictions\n    predictions = ensemble_predictor.predict(X_test_selected)\n    \n    # Post-process predictions\n    # Remove extreme outliers\n    pred_mean = np.mean(predictions)\n    pred_std = np.std(predictions)\n    predictions = np.clip(predictions, \n                         pred_mean - 3*pred_std, \n                         pred_mean + 3*pred_std)\n    \n    # Create submission\n    submission = pd.DataFrame({\n        'ID': test_df['timestamp'] if 'timestamp' in test_df.columns else range(len(test_df)),\n        'label': predictions\n    })\n    \n    submission.to_csv('/kaggle/working/clustering_submission.csv', index=False)\n    \n    print(f\"\\n\" + \"=\"*50)\n    print(\"CLUSTERING PREDICTION RESULTS\")\n    print(\"=\"*50)\n    print(f\"Submission file: clustering_submission.csv\")\n    print(f\"Predictions statistics:\")\n    print(f\"  Mean: {predictions.mean():.6f}\")\n    print(f\"  Std: {predictions.std():.6f}\")\n    print(f\"  Min: {predictions.min():.6f}\")\n    print(f\"  Max: {predictions.max():.6f}\")\n    print(f\"  Median: {np.median(predictions):.6f}\")\n    \n    # Compare with target statistics\n    print(f\"\\nTarget statistics (for reference):\")\n    print(f\"  Mean: {y.mean():.6f}\")\n    print(f\"  Std: {y.std():.6f}\")\n    print(f\"  Min: {y.min():.6f}\")\n    print(f\"  Max: {y.max():.6f}\")\n    print(f\"  Median: {np.median(y):.6f}\")\n    \n    return ensemble_predictor, predictions, submission","metadata":{"_uuid":"8681256c-b43b-4242-bb08-23d2d09e11cb","_cell_guid":"bf31ba35-8f2c-417e-9d4a-2be960d587da","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-07-22T13:51:21.008323Z","iopub.execute_input":"2025-07-22T13:51:21.009002Z","iopub.status.idle":"2025-07-22T13:51:21.035323Z","shell.execute_reply.started":"2025-07-22T13:51:21.008978Z","shell.execute_reply":"2025-07-22T13:51:21.034653Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ============================================================================\n# 7. VALIDATION AND ANALYSIS\n# ============================================================================","metadata":{"_uuid":"8681256c-b43b-4242-bb08-23d2d09e11cb","_cell_guid":"bf31ba35-8f2c-417e-9d4a-2be960d587da","collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"def validate_clustering_approach(train_df, all_feature_cols, n_folds=3):\n    \"\"\"Validate clustering approach using time series splits\"\"\"\n    print(f\"\\nValidating Clustering Approach with {n_folds} folds...\")\n    \n    X = train_df[all_feature_cols].values\n    y = train_df['label'].values\n    \n    # Feature selection\n    selected_features = select_features_for_clustering(X, y, n_features=50)\n    X_selected = X[:, selected_features]\n    \n    # Time series validation\n    fold_size = len(X_selected) // (n_folds + 1)\n    correlations = []\n    \n    for fold in range(n_folds):\n        print(f\"\\nFold {fold + 1}/{n_folds}\")\n        \n        # Create train/val split\n        train_end = (fold + 1) * fold_size\n        val_start = train_end\n        val_end = min(val_start + fold_size, len(X_selected))\n        \n        X_train_fold = X_selected[:train_end]\n        y_train_fold = y[:train_end]\n        X_val_fold = X_selected[val_start:val_end]\n        y_val_fold = y[val_start:val_end]\n        \n        print(f\"Train: {len(X_train_fold)}, Val: {len(X_val_fold)}\")\n        \n        # Train clustering predictor\n        predictor = ClusteringPredictor(n_clusters=1000, clustering_method='kmeans')\n        predictor.fit(X_train_fold, y_train_fold)\n        \n        # Predict\n        val_pred = predictor.predict(X_val_fold)\n        \n        # Calculate correlation\n        corr = pearsonr(y_val_fold, val_pred)[0]\n        if not np.isnan(corr):\n            correlations.append(corr)\n            print(f\"Fold {fold + 1} correlation: {corr:.4f}\")\n        else:\n            print(f\"Fold {fold + 1} correlation: NaN (skipped)\")\n    \n    if correlations:\n        mean_corr = np.mean(correlations)\n        std_corr = np.std(correlations)\n        print(f\"\\nValidation Results:\")\n        print(f\"Mean correlation: {mean_corr:.4f} ± {std_corr:.4f}\")\n        print(f\"Individual fold correlations: {[f'{c:.4f}' for c in correlations]}\")\n        return mean_corr, std_corr\n    else:\n        print(\"No valid correlations calculated\")\n        return 0, 0","metadata":{"_uuid":"8681256c-b43b-4242-bb08-23d2d09e11cb","_cell_guid":"bf31ba35-8f2c-417e-9d4a-2be960d587da","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-07-22T13:51:21.036164Z","iopub.execute_input":"2025-07-22T13:51:21.036418Z","iopub.status.idle":"2025-07-22T13:51:21.058245Z","shell.execute_reply.started":"2025-07-22T13:51:21.036394Z","shell.execute_reply":"2025-07-22T13:51:21.057571Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ============================================================================\n# 8. EXECUTION\n# ============================================================================","metadata":{"_uuid":"8681256c-b43b-4242-bb08-23d2d09e11cb","_cell_guid":"bf31ba35-8f2c-417e-9d4a-2be960d587da","collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"if __name__ == \"__main__\":\n    # Run main pipeline\n    print(\"Executing Clustering-Based Crypto Price Prediction...\")\n    \n    try:\n        predictor, predictions, submission = main_clustering_pipeline()\n        print(\"\\n✅ Clustering pipeline completed successfully!\")\n        print(f\"📊 Check 'clustering_submission.csv' for results\")\n        \n    except Exception as e:\n        print(f\"\\n❌ Pipeline failed with error: {e}\")\n        import traceback\n        traceback.print_exc()\n    \n    print(\"\\n\" + \"=\"*60)\n    print(\"CLUSTERING APPROACH SUMMARY\")\n    print(\"=\"*60)\n","metadata":{"_uuid":"8681256c-b43b-4242-bb08-23d2d09e11cb","_cell_guid":"bf31ba35-8f2c-417e-9d4a-2be960d587da","trusted":true,"execution":{"iopub.status.busy":"2025-07-22T13:51:21.060328Z","iopub.execute_input":"2025-07-22T13:51:21.060765Z","iopub.status.idle":"2025-07-22T14:47:22.503018Z","shell.execute_reply.started":"2025-07-22T13:51:21.060742Z","shell.execute_reply":"2025-07-22T14:47:22.499559Z"}},"outputs":[],"execution_count":null}]}