{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==================== INSTALLATIONS ====================\n# !pip install -q xgboost lightgbm tensorflow scikit-learn pandas numpy\n\n# ==================== IMPORTS ====================\nimport pandas as pd\nimport numpy as np\nimport lightgbm as lgb\nfrom xgboost import XGBRegressor\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models, callbacks, optimizers\nfrom sklearn.model_selection import TimeSeriesSplit\nfrom sklearn.preprocessing import RobustScaler\nfrom sklearn.metrics import mean_squared_error\nfrom scipy.stats import pearsonr\nimport warnings\nimport gc\nfrom typing import Dict, List, Tuple, Optional\nimport json\n\nwarnings.filterwarnings('ignore')\ntf.random.set_seed(42)\nnp.random.seed(42)\n\n# ==================== CONFIGURATION ====================\nclass CFG:\n    # Data paths\n    train_path = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    test_path = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    sample_sub_path = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n    \n    # Feature selection parameters\n    feature_selection_sample_size = 100000  # Sample size for feature selection\n    top_features_percentage = 0.3  # Select top 30% features\n    \n    # Model parameters\n    use_gpu = True\n    random_state = 42\n    \n    # Neural network parameters\n    nn_epochs = 80\n    nn_batch_size = 2048\n    nn_learning_rate = 0.001\n    nn_patience = 10\n    \n    # Cross-validation\n    n_folds = 5\n\n# ==================== UTILITY FUNCTIONS ====================\ndef reduce_mem_usage(df, verbose=True):\n    \"\"\"Reduce memory usage of dataframe\"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n            else:\n                df[col] = df[col].astype(np.float32)\n    \n    end_mem = df.memory_usage().sum() / 1024**2\n    if verbose:\n        print(f'Memory usage: {start_mem:.2f} MB -> {end_mem:.2f} MB ({100 * (start_mem - end_mem) / start_mem:.1f}% reduction)')\n    \n    return df\n\ndef clean_data(df: pd.DataFrame) -> pd.DataFrame:\n    \"\"\"Clean data by handling inf, -inf, and NaN values\"\"\"\n    # Replace infinite values with NaN\n    df = df.replace([np.inf, -np.inf], np.nan)\n    \n    # Fill NaN values\n    numeric_cols = df.select_dtypes(include=[np.number]).columns\n    for col in numeric_cols:\n        if df[col].isnull().any():\n            if any(keyword in col.lower() for keyword in ['qty', 'volume']):\n                df[col] = df[col].fillna(0)\n            else:\n                df[col] = df[col].fillna(df[col].median())\n    \n    return df\n\ndef create_time_weights(n_samples, decay_factor=0.95):\n    \"\"\"Create time-based weights with exponential decay\"\"\"\n    positions = np.arange(n_samples)\n    weights = decay_factor ** (n_samples - positions - 1)\n    return weights / weights.sum() * len(weights)\n\n# ==================== GRADIENT BOOSTING FEATURE SELECTOR ====================\nclass GradientBoostingFeatureSelector:\n    \"\"\"Feature selector using XGBoost and LightGBM\"\"\"\n    \n    def __init__(self):\n        self.feature_importance = {}\n        self.selected_features = []\n        \n    def calculate_feature_importance(self, X: pd.DataFrame, y: pd.Series) -> Dict[str, float]:\n        \"\"\"Calculate feature importance using gradient boosting models\"\"\"\n        print(\"\\nCalculating feature importance...\")\n        \n        # Split data for validation\n        split_idx = int(len(X) * 0.8)\n        X_train, X_val = X.iloc[:split_idx], X.iloc[split_idx:]\n        y_train, y_val = y.iloc[:split_idx], y.iloc[split_idx:]\n        \n        # XGBoost model\n        print(\"Training XGBoost...\")\n        xgb_params = {\n            'n_estimators': 200,\n            'max_depth': 8,\n            'learning_rate': 0.05,\n            'subsample': 0.8,\n            'colsample_bytree': 0.8,\n            'random_state': CFG.random_state,\n            'tree_method': 'gpu_hist' if CFG.use_gpu else 'hist',\n            'verbosity': 0\n        }\n        \n        xgb_model = XGBRegressor(**xgb_params)\n        xgb_model.fit(\n            X_train, y_train,\n            eval_set=[(X_val, y_val)],\n            early_stopping_rounds=20,\n            verbose=False\n        )\n        xgb_importance = dict(zip(X.columns, xgb_model.feature_importances_))\n        \n        # LightGBM model\n        print(\"Training LightGBM...\")\n        lgb_params = {\n            'n_estimators': 200,\n            'max_depth': 8,\n            'learning_rate': 0.05,\n            'subsample': 0.8,\n            'colsample_bytree': 0.8,\n            'num_leaves': 50,\n            'random_state': CFG.random_state,\n            'device': 'gpu' if CFG.use_gpu else 'cpu',\n            'verbosity': -1\n        }\n        \n        lgb_model = lgb.LGBMRegressor(**lgb_params)\n        lgb_model.fit(\n            X_train, y_train,\n            eval_set=[(X_val, y_val)],\n            callbacks=[lgb.early_stopping(20), lgb.log_evaluation(0)]\n        )\n        lgb_importance = dict(zip(X.columns, lgb_model.feature_importances_))\n        \n        # Combine importances (average of XGBoost and LightGBM)\n        combined_importance = {}\n        for feature in X.columns:\n            combined_importance[feature] = (\n                xgb_importance[feature] + lgb_importance[feature]\n            ) / 2\n        \n        # Normalize\n        max_importance = max(combined_importance.values())\n        for feature in combined_importance:\n            combined_importance[feature] /= max_importance\n        \n        self.feature_importance = combined_importance\n        \n        # Clean up\n        del xgb_model, lgb_model\n        gc.collect()\n        \n        return combined_importance\n    \n    def select_top_features(self, top_percentage: float = 0.3) -> List[str]:\n        \"\"\"Select top features based on importance\"\"\"\n        sorted_features = sorted(\n            self.feature_importance.items(), \n            key=lambda x: x[1], \n            reverse=True\n        )\n        \n        n_features = int(len(sorted_features) * top_percentage)\n        self.selected_features = [feature for feature, _ in sorted_features[:n_features]]\n        \n        print(f\"\\nSelected {len(self.selected_features)} features out of {len(sorted_features)}\")\n        print(f\"Top 10 features: {self.selected_features[:10]}\")\n        \n        return self.selected_features\n\n# ==================== NEURAL NETWORK MODEL ====================\nclass CryptoNeuralNetwork:\n    \"\"\"Deep neural network for crypto prediction\"\"\"\n    \n    def __init__(self, n_features: int):\n        self.n_features = n_features\n        self.model = None\n        self.scaler = RobustScaler()\n        \n    def build_model(self) -> models.Model:\n        \"\"\"Build neural network architecture\"\"\"\n        model = models.Sequential([\n            # Input layer\n            layers.Input(shape=(self.n_features,)),\n            \n            # First block\n            layers.Dense(512, activation='relu'),\n            layers.BatchNormalization(),\n            layers.Dropout(0.3),\n            \n            # Second block\n            layers.Dense(256, activation='relu'),\n            layers.BatchNormalization(),\n            layers.Dropout(0.3),\n            \n            # Third block\n            layers.Dense(128, activation='relu'),\n            layers.BatchNormalization(),\n            layers.Dropout(0.2),\n            \n            # Fourth block\n            layers.Dense(64, activation='relu'),\n            layers.BatchNormalization(),\n            layers.Dropout(0.2),\n            \n            # Fifth block\n            layers.Dense(32, activation='relu'),\n            layers.BatchNormalization(),\n            \n            # Output layer\n            layers.Dense(1, activation='linear')\n        ])\n        \n        model.compile(\n            optimizer=optimizers.Adam(learning_rate=CFG.nn_learning_rate),\n            loss='mse',\n            metrics=['mae']\n        )\n        \n        return model\n    \n    def train(self, X_train: pd.DataFrame, y_train: pd.Series, \n              X_val: pd.DataFrame, y_val: pd.Series,\n              sample_weights: Optional[np.ndarray] = None) -> None:\n        \"\"\"Train the neural network\"\"\"\n        # Scale features\n        X_train_scaled = self.scaler.fit_transform(X_train)\n        X_val_scaled = self.scaler.transform(X_val)\n        \n        # Build model\n        self.model = self.build_model()\n        \n        # Callbacks\n        early_stop = callbacks.EarlyStopping(\n            monitor='val_loss',\n            patience=CFG.nn_patience,\n            restore_best_weights=True,\n            verbose=1\n        )\n        \n        reduce_lr = callbacks.ReduceLROnPlateau(\n            monitor='val_loss',\n            factor=0.5,\n            patience=5,\n            min_lr=1e-6,\n            verbose=1\n        )\n        \n        # Train\n        self.model.fit(\n            X_train_scaled, y_train,\n            validation_data=(X_val_scaled, y_val),\n            epochs=CFG.nn_epochs,\n            batch_size=CFG.nn_batch_size,\n            sample_weight=sample_weights,\n            callbacks=[early_stop, reduce_lr],\n            verbose=1\n        )\n    \n    def predict(self, X: pd.DataFrame) -> np.ndarray:\n        \"\"\"Make predictions\"\"\"\n        X_scaled = self.scaler.transform(X)\n        return self.model.predict(X_scaled, batch_size=CFG.nn_batch_size).flatten()\n\n# ==================== MAIN PIPELINE ====================\ndef main():\n    print(\"=\"*80)\n    print(\"SIMPLIFIED CRYPTO PREDICTION PIPELINE\")\n    print(\"=\"*80)\n    \n    # Load data\n    print(\"\\nLoading data...\")\n    train = pd.read_parquet(CFG.train_path)\n    test = pd.read_parquet(CFG.test_path)\n    sample_submission = pd.read_csv(CFG.sample_sub_path)\n    \n    # Reduce memory usage\n    train = reduce_mem_usage(train)\n    test = reduce_mem_usage(test)\n    \n    # Clean data\n    train = clean_data(train)\n    test = clean_data(test)\n    \n    print(f\"\\nTrain shape: {train.shape}\")\n    print(f\"Test shape: {test.shape}\")\n    \n    # Separate features and target\n    feature_cols = [col for col in train.columns if col not in ['label', 'timestamp']]\n    X_train_full = train[feature_cols]\n    y_train_full = train['label']\n    X_test = test[feature_cols]\n    \n    # Ensure critical features are always included\n    critical_features = ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']\n    \n    # ==================== FEATURE SELECTION ====================\n    print(\"\\n\" + \"=\"*60)\n    print(\"FEATURE SELECTION PHASE\")\n    print(\"=\"*60)\n    \n    # Use recent data for feature selection\n    sample_size = min(CFG.feature_selection_sample_size, len(X_train_full))\n    X_sample = X_train_full.tail(sample_size).reset_index(drop=True)\n    y_sample = y_train_full.tail(sample_size).reset_index(drop=True)\n    \n    # Initialize feature selector\n    selector = GradientBoostingFeatureSelector()\n    \n    # Calculate feature importance\n    feature_importance = selector.calculate_feature_importance(X_sample, y_sample)\n    \n    # Select top features\n    selected_features = selector.select_top_features(CFG.top_features_percentage)\n    \n    # Ensure critical features are included\n    for feature in critical_features:\n        if feature in feature_cols and feature not in selected_features:\n            selected_features.append(feature)\n    \n    print(f\"\\nFinal number of selected features: {len(selected_features)}\")\n    \n    # Apply feature selection\n    X_train_selected = X_train_full[selected_features]\n    X_test_selected = X_test[selected_features]\n    \n    # Save feature importance\n    feature_report = pd.DataFrame({\n        'feature': list(feature_importance.keys()),\n        'importance': list(feature_importance.values()),\n        'selected': [f in selected_features for f in feature_importance.keys()]\n    }).sort_values('importance', ascending=False)\n    feature_report.to_csv('feature_importance.csv', index=False)\n    print(\"Feature importance saved to feature_importance.csv\")\n    \n    # ==================== NEURAL NETWORK TRAINING ====================\n    print(\"\\n\" + \"=\"*60)\n    print(\"NEURAL NETWORK TRAINING PHASE\")\n    print(\"=\"*60)\n    \n    # Create time-based weights\n    sample_weights = create_time_weights(len(X_train_selected))\n    \n    # Time series cross-validation\n    tscv = TimeSeriesSplit(n_splits=CFG.n_folds)\n    oof_predictions = np.zeros(len(X_train_selected))\n    test_predictions = np.zeros(len(X_test_selected))\n    fold_scores = []\n    \n    for fold, (train_idx, val_idx) in enumerate(tscv.split(X_train_selected)):\n        print(f\"\\n--- Fold {fold + 1}/{CFG.n_folds} ---\")\n        \n        X_fold_train = X_train_selected.iloc[train_idx]\n        X_fold_val = X_train_selected.iloc[val_idx]\n        y_fold_train = y_train_full.iloc[train_idx]\n        y_fold_val = y_train_full.iloc[val_idx]\n        fold_weights = sample_weights[train_idx]\n        \n        # Train neural network\n        nn_model = CryptoNeuralNetwork(n_features=len(selected_features))\n        nn_model.train(\n            X_fold_train, y_fold_train,\n            X_fold_val, y_fold_val,\n            sample_weights=fold_weights\n        )\n        \n        # Make predictions\n        val_pred = nn_model.predict(X_fold_val)\n        test_pred = nn_model.predict(X_test_selected)\n        \n        # Store predictions\n        oof_predictions[val_idx] = val_pred\n        test_predictions += test_pred / CFG.n_folds\n        \n        # Calculate fold score\n        fold_score = pearsonr(y_fold_val, val_pred)[0]\n        fold_scores.append(fold_score)\n        print(f\"Fold {fold + 1} Pearson correlation: {fold_score:.4f}\")\n        \n        # Clean up\n        del nn_model\n        gc.collect()\n        tf.keras.backend.clear_session()\n    \n    # ==================== FINAL EVALUATION ====================\n    print(\"\\n\" + \"=\"*60)\n    print(\"FINAL RESULTS\")\n    print(\"=\"*60)\n    \n    # Calculate overall performance\n    overall_score = pearsonr(y_train_full, oof_predictions)[0]\n    print(f\"\\nOverall OOF Pearson correlation: {overall_score:.4f}\")\n    print(f\"Average fold score: {np.mean(fold_scores):.4f} (+/- {np.std(fold_scores):.4f})\")\n    \n    # Create submission\n    submission = sample_submission.copy()\n    submission['prediction'] = test_predictions\n    submission.to_csv('submission.csv', index=False)\n    print(\"\\nSubmission saved to submission.csv\")\n    \n    # Save results summary\n    results = {\n        'n_original_features': len(feature_cols),\n        'n_selected_features': len(selected_features),\n        'overall_oof_score': float(overall_score),\n        'fold_scores': [float(s) for s in fold_scores],\n        'mean_fold_score': float(np.mean(fold_scores)),\n        'std_fold_score': float(np.std(fold_scores))\n    }\n    \n    with open('results_summary.json', 'w') as f:\n        json.dump(results, f, indent=2)\n    \n    print(\"\\nResults summary saved to results_summary.json\")\n    print(\"\\nPipeline completed successfully!\")\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}