{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-25T21:59:52.029606Z","iopub.execute_input":"2025-05-25T21:59:52.029884Z","iopub.status.idle":"2025-05-25T21:59:54.115782Z","shell.execute_reply.started":"2025-05-25T21:59:52.029856Z","shell.execute_reply":"2025-05-25T21:59:54.114942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom prophet import Prophet\nimport warnings\nwarnings.filterwarnings('ignore')\n\ndef simple_prophet_outlier_detection(df, feature_col, timestamp_col='timestamp'):\n    \"\"\"\n    Simple example of using Prophet to detect outliers in a single feature.\n    \n    The idea: Prophet models the expected behavior of the feature over time.\n    Points that fall far outside Prophet's confidence interval are outliers.\n    \"\"\"\n    print(f\"Detecting outliers in {feature_col} using Prophet...\")\n    \n    # 1. Prepare data for Prophet\n    prophet_df = pd.DataFrame({\n        'ds': df[timestamp_col],\n        'y': df[feature_col]\n    })\n    \n    # Remove obvious bad values\n    prophet_df = prophet_df[np.isfinite(prophet_df['y'])]\n    \n    # Resample to hourly for efficiency (adjust based on your needs)\n    prophet_hourly = prophet_df.set_index('ds').resample('1H').mean().reset_index()\n    prophet_hourly = prophet_hourly.dropna()\n    \n    # 2. Fit Prophet model\n    model = Prophet(\n        changepoint_prior_scale=0.05,  # Low value = less sensitive to outliers\n        interval_width=0.95,  # 95% confidence interval\n        yearly_seasonality=False,\n        weekly_seasonality=True,\n        daily_seasonality=True\n    )\n    \n    model.fit(prophet_hourly)\n    \n    # 3. Generate predictions\n    forecast = model.predict(prophet_hourly)\n    \n    # 4. Identify outliers\n    # Method 1: Points outside confidence interval\n    outliers_ci = (\n        (prophet_hourly['y'] < forecast['yhat_lower']) | \n        (prophet_hourly['y'] > forecast['yhat_upper'])\n    )\n    \n    # Method 2: Points with large standardized residuals\n    residuals = prophet_hourly['y'] - forecast['yhat']\n    residual_std = residuals.std()\n    outliers_residual = np.abs(residuals) > 3 * residual_std\n    \n    # Combine both methods\n    outliers = outliers_ci & outliers_residual\n    \n    # 5. Visualize\n    fig, (ax1, ax2) = plt.subplots(2, 1, figsize=(12, 8))\n    \n    # Plot 1: Time series with outliers\n    ax1.plot(prophet_hourly['ds'], prophet_hourly['y'], 'b.', alpha=0.5, label='Actual')\n    ax1.plot(forecast['ds'], forecast['yhat'], 'g-', linewidth=2, label='Prophet Fit')\n    ax1.fill_between(forecast['ds'], forecast['yhat_lower'], forecast['yhat_upper'], \n                     alpha=0.2, color='green', label='95% CI')\n    \n    # Highlight outliers\n    outlier_points = prophet_hourly[outliers]\n    ax1.scatter(outlier_points['ds'], outlier_points['y'], \n               color='red', s=100, edgecolor='darkred', linewidth=2, \n               label=f'Outliers ({outliers.sum()})', zorder=10)\n    \n    ax1.set_title(f'Prophet Outlier Detection: {feature_col}')\n    ax1.set_xlabel('Date')\n    ax1.set_ylabel('Value')\n    ax1.legend()\n    ax1.grid(True, alpha=0.3)\n    \n    # Plot 2: Residual distribution\n    ax2.hist(residuals, bins=50, alpha=0.7, color='blue', edgecolor='black')\n    ax2.axvline(x=-3*residual_std, color='red', linestyle='--', label='±3σ threshold')\n    ax2.axvline(x=3*residual_std, color='red', linestyle='--')\n    ax2.set_title('Residual Distribution')\n    ax2.set_xlabel('Residual (Actual - Predicted)')\n    ax2.set_ylabel('Frequency')\n    ax2.legend()\n    ax2.grid(True, alpha=0.3)\n    \n    plt.tight_layout()\n    plt.show()\n    \n    # 6. Return outlier information\n    outlier_info = {\n        'n_outliers': outliers.sum(),\n        'outlier_pct': outliers.sum() / len(prophet_hourly) * 100,\n        'outlier_timestamps': outlier_points['ds'].tolist(),\n        'outlier_values': outlier_points['y'].tolist(),\n        'residual_std': residual_std\n    }\n    \n    print(f\"Found {outlier_info['n_outliers']} outliers ({outlier_info['outlier_pct']:.2f}%)\")\n    \n    return outlier_info, model, forecast\n\n\n# Quick example for multiple features\ndef detect_outliers_multiple_features(data_path, features_to_check=None):\n    \"\"\"\n    Run outlier detection on multiple features.\n    \"\"\"\n    # Load data\n    if features_to_check is None:\n        # Default to some common features\n        features_to_check = ['volume', 'bid_qty', 'ask_qty', 'buy_qty', 'sell_qty']\n    \n    df = pd.read_parquet(data_path, columns=['timestamp'] + features_to_check)\n    \n    # Ensure timestamp\n    if 'timestamp' not in df.columns:\n        df['timestamp'] = pd.date_range('2023-03-01', periods=len(df), freq='T')\n    \n    # Analyze each feature\n    outlier_summary = {}\n    \n    for feature in features_to_check:\n        if feature in df.columns:\n            print(f\"\\n{'='*60}\")\n            outlier_info, model, forecast = simple_prophet_outlier_detection(df, feature)\n            outlier_summary[feature] = outlier_info\n    \n    # Summary plot\n    fig, ax = plt.subplots(figsize=(10, 6))\n    \n    features = list(outlier_summary.keys())\n    outlier_pcts = [outlier_summary[f]['outlier_pct'] for f in features]\n    \n    bars = ax.bar(features, outlier_pcts, color='coral', edgecolor='darkred', linewidth=2)\n    \n    # Highlight high outlier features\n    for i, pct in enumerate(outlier_pcts):\n        if pct > 1.0:  # More than 1% outliers\n            bars[i].set_color('red')\n    \n    ax.set_title('Outlier Percentage by Feature', fontsize=14, fontweight='bold')\n    ax.set_ylabel('Outlier %')\n    ax.set_xlabel('Feature')\n    ax.grid(True, alpha=0.3, axis='y')\n    \n    # Add value labels\n    for bar, pct in zip(bars, outlier_pcts):\n        height = bar.get_height()\n        ax.text(bar.get_x() + bar.get_width()/2., height,\n                f'{pct:.2f}%', ha='center', va='bottom')\n    \n    plt.tight_layout()\n    plt.show()\n    \n    return outlier_summary\n\n\n# Practical example: Using outlier detection for data cleaning\ndef clean_feature_outliers(df, feature_col, outlier_info, method='cap'):\n    \"\"\"\n    Clean outliers from a feature using different methods.\n    \"\"\"\n    # Get outlier timestamps\n    outlier_timestamps = outlier_info['outlier_timestamps']\n    \n    # Create copy\n    df_clean = df.copy()\n    \n    if method == 'remove':\n        # Remove rows with outliers\n        mask = ~df_clean['timestamp'].isin(outlier_timestamps)\n        df_clean = df_clean[mask]\n        print(f\"Removed {len(outlier_timestamps)} rows with outliers\")\n        \n    elif method == 'cap':\n        # Cap outliers at 99th percentile\n        lower_cap = df_clean[feature_col].quantile(0.01)\n        upper_cap = df_clean[feature_col].quantile(0.99)\n        \n        df_clean[feature_col] = df_clean[feature_col].clip(lower=lower_cap, upper=upper_cap)\n        print(f\"Capped {feature_col} to range [{lower_cap:.2f}, {upper_cap:.2f}]\")\n        \n    elif method == 'interpolate':\n        # Replace outliers with interpolated values\n        outlier_mask = df_clean['timestamp'].isin(outlier_timestamps)\n        df_clean.loc[outlier_mask, feature_col] = np.nan\n        df_clean[feature_col] = df_clean[feature_col].interpolate(method='linear')\n        print(f\"Interpolated {len(outlier_timestamps)} outlier values\")\n    \n    return df_clean\n\n\n# Example usage\nif __name__ == \"__main__\":\n    # Example 1: Single feature analysis\n    data_path = '/kaggle/input/drw-crypto-market-prediction/train.parquet'\n    df = pd.read_parquet(data_path, columns=['timestamp', 'volume'])\n    \n    if 'timestamp' not in df.columns:\n        df['timestamp'] = pd.date_range('2023-03-01', periods=len(df), freq='T')\n    \n    # Detect outliers in volume\n    outlier_info, model, forecast = simple_prophet_outlier_detection(df, 'volume')\n    \n    # Example 2: Multiple features\n    outlier_summary = detect_outliers_multiple_features(\n        data_path,\n        features_to_check=['volume', 'bid_qty', 'ask_qty']\n    )\n    \n    # Example 3: Clean the data\n    df_clean = clean_feature_outliers(df, 'volume', outlier_info, method='cap')\n    \n    print(\"\\nOutlier detection complete!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T21:59:54.117257Z","iopub.execute_input":"2025-05-25T21:59:54.117666Z","iopub.status.idle":"2025-05-25T22:00:17.572131Z","shell.execute_reply.started":"2025-05-25T21:59:54.117643Z","shell.execute_reply":"2025-05-25T22:00:17.571428Z"}},"outputs":[],"execution_count":null}]}