{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport polars as pl\nimport traceback\nfrom sklearn.ensemble import RandomForestRegressor\nfrom typing import List, Tuple, Optional\nimport time\nimport gc\nimport joblib\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nimport pyarrow.parquet as pq\nimport kaggle_evaluation.jane_street_inference_server\n\n\nwarnings.filterwarnings('ignore')\n# Set pandas display options to show all rows and columns\npd.set_option('display.max_rows', None)\npd.set_option('display.max_columns', None)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T15:34:19.944744Z","iopub.execute_input":"2024-11-26T15:34:19.945792Z","iopub.status.idle":"2024-11-26T15:34:23.718412Z","shell.execute_reply.started":"2024-11-26T15:34:19.945724Z","shell.execute_reply":"2024-11-26T15:34:23.717537Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ModelSelector:\n    def __init__(self, threshold: float = 0.8):\n        \"\"\"Initialize model selector with parameters.\"\"\"\n        self.threshold = threshold\n        self.feature_importance = None\n        self.best_model = None\n        self.feature_names = None\n        \n    def train(self, train_data: pd.DataFrame, train_y: np.ndarray, feature_names: List[str]):\n        \"\"\"Train model using either single model or ensemble approach.\"\"\"\n        self.feature_names = feature_names\n        \n        # Use smaller validation split (10%)\n        val_size = int(len(train_data) * 0.1)\n        train_idx = len(train_data) - val_size\n        \n        # Split data\n        X_train = train_data[feature_names].iloc[:train_idx]\n        y_train = train_y[:train_idx]\n        X_val = train_data[feature_names].iloc[train_idx:]\n        y_val = train_y[train_idx:]\n        \n        print(\"\\nTraining base model...\")\n        base_model = RandomForestRegressor(\n            n_estimators=500,\n            max_features=min(int(np.sqrt(len(feature_names))), 10),\n            min_samples_leaf=5,\n            max_leaf_nodes=1000,\n            n_jobs=-1,\n            random_state=42\n        )\n        \n        base_model.fit(X_train, y_train)\n        \n        # Calculate feature importance\n        self.feature_importance = pd.DataFrame({\n            'feature': feature_names,\n            'importance': base_model.feature_importances_\n        }).sort_values('importance', ascending=False)\n        \n        # Calculate validation score\n        val_score = base_model.score(X_val, y_val)\n        print(f\"\\nValidation R² Score: {val_score:.4f}\")\n        \n        # Create ensemble if score is below threshold\n        if val_score < self.threshold:\n            print(f\"R² score ({val_score:.4f}) below threshold ({self.threshold}). Creating ensemble...\")\n            self.best_model = self._create_ensemble(train_data[feature_names], train_y)\n        else:\n            print(f\"R² score ({val_score:.4f}) above threshold. Using single model.\")\n            self.best_model = base_model\n            \n    def _create_ensemble(self, X_train: pd.DataFrame, y_train: np.ndarray) -> List:\n        \"\"\"Create small ensemble of models with different seeds.\"\"\"\n        models = []\n        seeds = [42, 84]\n        \n        for seed in seeds:\n            model = RandomForestRegressor(\n                n_estimators=50,\n                max_features=min(int(np.sqrt(len(self.feature_names))), 10),\n                min_samples_leaf=5,\n                max_leaf_nodes=1000,\n                n_jobs=-1,\n                random_state=seed\n            )\n            model.fit(X_train, y_train)\n            models.append(model)\n            gc.collect()\n            \n        return models\n    \n    def predict(self, test_data: pd.DataFrame) -> np.ndarray:\n        \"\"\"Make predictions using single model or ensemble.\"\"\"\n        if not self.best_model:\n            raise ValueError(\"Model not trained. Please train the model first.\")\n            \n        if isinstance(self.best_model, list):\n            # For ensemble, average predictions\n            predictions = np.mean([\n                model.predict(test_data[self.feature_names])\n                for model in self.best_model\n            ], axis=0)\n        else:\n            # For single model\n            predictions = self.best_model.predict(test_data[self.feature_names])\n            \n        return predictions\n    \n    def print_feature_importance(self, top_n: int = 10):\n        \"\"\"Print top N most important features.\"\"\"\n        if self.feature_importance is not None:\n            print(f\"\\nTop {top_n} Most Important Features:\")\n            print(self.feature_importance.head(top_n))\n        else:\n            print(\"Feature importance not available. Train the model first.\")        ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T15:34:23.720635Z","iopub.execute_input":"2024-11-26T15:34:23.721114Z","iopub.status.idle":"2024-11-26T15:34:23.736071Z","shell.execute_reply.started":"2024-11-26T15:34:23.721082Z","shell.execute_reply":"2024-11-26T15:34:23.734979Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_and_prepare_data(file_path: str) -> Tuple[pd.DataFrame, np.ndarray, List[str]]:\n    train = pq.read_table(file_path).to_pandas()\n    \n    numeric_train = train.copy()\n    for col in train.select_dtypes(include=['category']).columns:\n        numeric_train[col] = pd.to_numeric(train[col].astype(str), errors='coerce')\n    \n    numeric_cols = numeric_train.select_dtypes(include=np.number).columns[4:]\n    \n    features_agg_sum = pd.DataFrame({\n        'feature': numeric_cols,\n        'Value': numeric_train[numeric_cols].sum()\n    })\n    \n    features_agg_sum_large = features_agg_sum[features_agg_sum['Value'] > 2]\n    feature_names = features_agg_sum_large['feature'].tolist()\n    \n    for col in feature_names:\n        numeric_train[col] = numeric_train[col].fillna(numeric_train[col].median())\n        \n    return numeric_train, numeric_train['responder_6'].values, feature_names","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T15:34:23.737319Z","iopub.execute_input":"2024-11-26T15:34:23.737695Z","iopub.status.idle":"2024-11-26T15:34:23.768570Z","shell.execute_reply.started":"2024-11-26T15:34:23.737662Z","shell.execute_reply":"2024-11-26T15:34:23.767438Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Run Training","metadata":{}},{"cell_type":"code","source":"# Second block - Main execution\nif __name__ == \"__main__\":\n    file_path = \"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=9/part-0.parquet\"\n\n    try:\n        print(\"Initializing model selector...\")\n        global model_selector\n        model_selector = ModelSelector(threshold=0.8)\n        \n        # Load and prepare data\n        print(\"\\nLoading and preparing data...\")\n        train_data, train_y, feature_names = load_and_prepare_data(file_path)\n        \n        # Train model\n        print(\"\\nStarting model training...\")\n        model_selector.train(train_data, train_y, feature_names)\n        \n        # Print feature importance\n        print(\"\\nFeature Importance Analysis:\")\n        model_selector.print_feature_importance()\n        \n        # Optional: Save feature importance to CSV\n        if model_selector.feature_importance is not None:\n            model_selector.feature_importance.to_csv('feature_importance.csv', index=False)\n            print(\"\\nFeature importance saved to 'feature_importance.csv'\")\n            \n        # Save model if needed\n        joblib.dump(model_selector.best_model, 'model.joblib')\n        print(\"\\nModel saved to 'model.joblib'\")\n        \n        print(\"\\nTraining and initialization completed successfully!\")\n\n    except Exception as e:\n        print(f\"\\nError during training: {str(e)}\")\n        import traceback\n        print(\"\\nDetailed error info:\")\n        print(traceback.format_exc())\n        print(\"\\nEnsure you have enough memory and all required data is available\")\n        \n    finally:\n        # Clean up memory\n        gc.collect()\n        print(\"\\nMemory cleaned up\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T15:34:23.771345Z","iopub.execute_input":"2024-11-26T15:34:23.772135Z","iopub.status.idle":"2024-11-26T16:11:44.541812Z","shell.execute_reply.started":"2024-11-26T15:34:23.772084Z","shell.execute_reply":"2024-11-26T16:11:44.540514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_feature_importance(model_selector, n_features=20, figsize=(12, 8)):\n    \"\"\"Create a horizontal bar plot of feature importances.\"\"\"\n    if model_selector.feature_importance is None:\n        print(\"No feature importance data available. Please train the model first.\")\n        return\n    \n    # Get top N features\n    top_features = model_selector.feature_importance.head(n_features).copy()\n    \n    # Sort in ascending order for better visualization\n    top_features = top_features.sort_values('importance', ascending=True)\n    \n    # Create figure\n    plt.figure(figsize=figsize)\n    \n    # Create horizontal bar plot\n    bars = plt.barh(y=range(len(top_features)), \n                   width=top_features['importance'],\n                   color=plt.cm.viridis(np.linspace(0, 1, n_features)))\n    \n    # Customize plot\n    plt.yticks(range(len(top_features)), top_features['feature'])\n    plt.xlabel('Relative Importance')\n    plt.title(f'Top {n_features} Most Important Features')\n    \n    # Add value labels on the bars\n    for bar in bars:\n        width = bar.get_width()\n        plt.text(width, \n                bar.get_y() + bar.get_height()/2,\n                f'{width:.3f}',\n                ha='left', \n                va='center',\n                fontweight='bold')\n    \n    plt.tight_layout()\n    plt.show()\n    \n    # Print numerical values\n    print(f\"\\nTop {n_features} Features by Importance:\")\n    print(top_features[['feature', 'importance']].to_string(index=False))\n\ndef plot_feature_importance_advanced(model_selector, n_features=20, figsize=(15, 10)):\n    \"\"\"Create an advanced visualization of feature importances with additional statistics.\"\"\"\n    if model_selector.feature_importance is None:\n        print(\"No feature importance data available. Please train the model first.\")\n        return\n    \n    # Create figure with subplots\n    fig, (ax1, ax2) = plt.subplots(2, 1, figsize=figsize, height_ratios=[3, 1])\n    \n    # Get top N features\n    top_features = model_selector.feature_importance.head(n_features).copy()\n    \n    # Calculate cumulative importance\n    total_importance = model_selector.feature_importance['importance'].sum()\n    model_selector.feature_importance['cumulative_importance'] = model_selector.feature_importance['importance'].cumsum() / total_importance\n    \n    # Plot 1: Horizontal bar chart\n    top_features = top_features.sort_values('importance', ascending=True)\n    bars = ax1.barh(y=range(len(top_features)), \n                   width=top_features['importance'],\n                   color=plt.cm.viridis(np.linspace(0, 1, n_features)))\n    \n    # Customize first plot\n    ax1.set_yticks(range(len(top_features)))\n    ax1.set_yticklabels(top_features['feature'])\n    ax1.set_xlabel('Relative Importance')\n    ax1.set_title(f'Top {n_features} Most Important Features')\n    \n    # Add value labels\n    for bar in bars:\n        width = bar.get_width()\n        ax1.text(width, \n                bar.get_y() + bar.get_height()/2,\n                f'{width:.3f}',\n                ha='left', \n                va='center',\n                fontweight='bold')\n    \n    # Plot 2: Cumulative importance\n    feature_count = range(1, len(model_selector.feature_importance) + 1)\n    ax2.plot(feature_count, \n             model_selector.feature_importance['cumulative_importance'],\n             'b-o',\n             linewidth=2,\n             markersize=6)\n    \n    # Add threshold lines\n    for threshold in [0.5, 0.8, 0.9]:\n        features_needed = sum(model_selector.feature_importance['cumulative_importance'] <= threshold) + 1\n        ax2.axhline(y=threshold, color='r', linestyle='--', alpha=0.5)\n        ax2.axvline(x=features_needed, color='r', linestyle='--', alpha=0.5)\n        ax2.text(features_needed + 0.5, threshold + 0.01, \n                f'{threshold:.0%} importance\\n{features_needed} features',\n                va='bottom')\n    \n    ax2.set_xlabel('Number of Features')\n    ax2.set_ylabel('Cumulative Importance')\n    ax2.set_title('Cumulative Feature Importance')\n    ax2.grid(True)\n    \n    plt.tight_layout()\n    plt.show()\n    \n    # Print summary statistics\n    print(\"\\nFeature Importance Summary:\")\n    print(f\"Total number of features: {len(model_selector.feature_importance)}\")\n    for threshold in [0.5, 0.8, 0.9]:\n        features_needed = sum(model_selector.feature_importance['cumulative_importance'] <= threshold) + 1\n        print(f\"Features needed for {threshold*100:.0f}% importance: {features_needed}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T16:11:44.543475Z","iopub.execute_input":"2024-11-26T16:11:44.543809Z","iopub.status.idle":"2024-11-26T16:11:44.561284Z","shell.execute_reply.started":"2024-11-26T16:11:44.543778Z","shell.execute_reply":"2024-11-26T16:11:44.560113Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Print numerical values first\n# print(\"Feature Importance Rankings:\")\n# model_selector.print_feature_importance(top_n=20)\n\n# Create visualization plots\nplt.figure(figsize=(15, 10))\n\n# Basic feature importance plot\n# print(\"\\nBasic Feature Importance Plot:\")\n# plot_feature_importance(model_selector, n_features=20)\n\n# Advanced plot with cumulative importance\nprint(\"\\nAdvanced Feature Importance Analysis:\")\nplot_feature_importance_advanced(model_selector, n_features=20)\n\n# Optionally save plots\nplt.savefig('feature_importance_plots.png', bbox_inches='tight')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T16:11:44.562566Z","iopub.execute_input":"2024-11-26T16:11:44.562935Z","iopub.status.idle":"2024-11-26T16:11:45.446047Z","shell.execute_reply.started":"2024-11-26T16:11:44.562899Z","shell.execute_reply":"2024-11-26T16:11:45.444892Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    global lags_\n    if lags is not None:\n        lags_ = lags\n    \n    test_data = test.to_pandas()\n    numeric_test = test_data.copy()\n    \n    if 'partition_id' not in numeric_test.columns:\n        numeric_test['partition_id'] = 9\n    if 'responder_1' not in numeric_test.columns:\n        numeric_test['responder_1'] = 0\n    numeric_test['responder_2'] = numeric_test['responder_1']\n    \n    test_features = numeric_test[model_selector.feature_names]\n    for col in test_features.columns:\n        median_val = test_features[col].median()\n        if pd.isna(median_val):\n            test_features[col] = test_features[col].fillna(0)\n        else:\n            test_features[col] = test_features[col].fillna(median_val)\n    \n    predictions = model_selector.predict(test_features)\n    \n     # Create output DataFrame\n    return pd.DataFrame({\n        'row_id': test_data.index if 'row_id' not in test_data.columns else test_data['row_id'],\n        'responder_6': predictions\n    })","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T16:11:45.447272Z","iopub.execute_input":"2024-11-26T16:11:45.447553Z","iopub.status.idle":"2024-11-26T16:11:45.947514Z","shell.execute_reply.started":"2024-11-26T16:11:45.447525Z","shell.execute_reply":"2024-11-26T16:11:45.946478Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    print(\"Running in competition mode\")\n    inference_server.serve()\nelse:\n    print(\"Running in local gateway mode\")\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}