{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div style=\"font-family: 'Poppins'; font-weight: bold; letter-spacing: 0px; color: #FFFFFF; font-size: 300%; text-align: left; padding: 15px; background: #0A0F29; border: 8px solid #00FFFF; border-radius: 15px; box-shadow: 5px 5px 20px rgba(0, 0, 0, 0.5);\">\n    Jane Street Real-Time Market Data Forecasting<br>\n</div>","metadata":{}},{"cell_type":"markdown","source":"# <div style=\"background-color:#0A0F29; font-family:'Poppins', cursive; color:#E0F7FA; font-size:140%; text-align:center; border: 2px solid #00FFFF; border-radius:15px; padding: 15px; box-shadow: 5px 5px 20px rgba(0, 0, 0, 0.5); font-weight: bold; letter-spacing: 1px; text-transform: uppercase;\">Introduction</div>","metadata":{}},{"cell_type":"markdown","source":"This competition challenges participants to build models using real-world financial data derived from Jane Street's production systems.\n\n- **Anonymized and obfuscated features**: The data represents real trading systems but is anonymized to protect proprietary information.\n- **Non-stationary time series**: Participants must deal with evolving data that doesn't follow traditional assumptions, making modeling more difficult.\n- **Fat-tailed distributions**: Extreme events and outliers are common, requiring robust models that can handle unpredictable outcomes.\n- **Submission guidelines**: Models must be trained using **Kaggle Notebooks**, with a time limit of **8 hours** for both CPU and GPU during the training phase.\n- **Forecasting phase**: After the final submission, models will be tested on real market data, with an extended **9-hour runtime** during this phase.","metadata":{}},{"cell_type":"markdown","source":"# <div style=\"background-color:#0A0F29; font-family:'Poppins', cursive; color:#E0F7FA; font-size:140%; text-align:center; border: 2px solid #00FFFF; border-radius:15px; padding: 15px; box-shadow: 5px 5px 20px rgba(0, 0, 0, 0.5); font-weight: bold; letter-spacing: 1px; text-transform: uppercase;\">Automated EDA</div>","metadata":{}},{"cell_type":"code","source":"!pip install -U ipywidgets  > /dev/null 2>&1\n!pip install pydantic > /dev/null 2>&1\n!pip install sweetviz > /dev/null 2>&1\n!pip install langchain-core > /dev/null 2>&1\n!pip install langchain-openai  > /dev/null 2>&1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T01:10:51.564070Z","iopub.execute_input":"2024-11-16T01:10:51.564475Z","iopub.status.idle":"2024-11-16T01:11:47.698386Z","shell.execute_reply.started":"2024-11-16T01:10:51.564434Z","shell.execute_reply":"2024-11-16T01:11:47.696820Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import libraries\n\n# General Purpose Libraries\nimport os\nfrom pathlib import Path\nimport json\nimport logging\nimport numpy as np\nimport polars as pl\nimport pandas as pd\nimport sweetviz as sv\nimport seaborn as sns\nimport pprint\nimport matplotlib.pyplot as plt\nfrom itertools import product\nimport warnings\nfrom IPython.display import Markdown, display\nfrom kaggle_secrets import UserSecretsClient\nfrom scipy.stats import ttest_ind, stats\n\n# LLM Libraries\nfrom langchain_core.output_parsers import StrOutputParser\nfrom langchain_core.prompts import ChatPromptTemplate\nfrom langchain_openai import ChatOpenAI\n\n# Model training\nimport optuna\nfrom sklearn.model_selection import TimeSeriesSplit\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.preprocessing import StandardScaler, RobustScaler\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score\nfrom sklearn.impute import KNNImputer\nfrom sklearn.feature_selection import SelectFromModel\n\n# Suppress specific FutureWarnings\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\n\n# Set matplotlib parameters to handle large datasets\nplt.rcParams['agg.path.chunksize'] = 10000\nplt.rcParams['path.simplify_threshold'] = 0.5","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T01:15:12.725554Z","iopub.execute_input":"2024-11-16T01:15:12.726108Z","iopub.status.idle":"2024-11-16T01:15:12.737158Z","shell.execute_reply.started":"2024-11-16T01:15:12.726065Z","shell.execute_reply":"2024-11-16T01:15:12.736020Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Secrets Management\nuser_secrets = UserSecretsClient()\nOPENAI_API_KEY = user_secrets.get_secret(\"openai_key\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T01:06:57.204905Z","iopub.execute_input":"2024-11-16T01:06:57.206063Z","iopub.status.idle":"2024-11-16T01:06:58.020959Z","shell.execute_reply.started":"2024-11-16T01:06:57.206013Z","shell.execute_reply":"2024-11-16T01:06:58.019896Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the LLM model using LangChain\nmodel = ChatOpenAI(\n    model='gpt-4o-2024-05-13',\n    temperature=0,\n    api_key=OPENAI_API_KEY\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T01:07:04.865624Z","iopub.execute_input":"2024-11-16T01:07:04.866473Z","iopub.status.idle":"2024-11-16T01:07:04.898144Z","shell.execute_reply.started":"2024-11-16T01:07:04.866426Z","shell.execute_reply":"2024-11-16T01:07:04.897239Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Helper functions\n\n# Function to classify columns into continuous and categorical\ndef classify_columns(df):\n    continuous_cols = []\n    categorical_cols = []\n    for column in df.columns:\n        if df[column].dtypes == 'object':\n            categorical_cols.append(column)\n        else:\n            unique_values = df[column].nunique()\n            if unique_values < 15:\n                categorical_cols.append(column)\n            else:\n                continuous_cols.append(column)\n    return continuous_cols, categorical_cols\n\n# Function to perform basic visualizations for continuous and categorical features\ndef eda_visualizations(df, target=None):\n    continuous_cols, categorical_cols = classify_columns(df)\n    \n    # Plotting continuous columns\n    for col in continuous_cols:\n        plt.figure(figsize=(10, 4))\n        sns.histplot(df[col], kde=True)\n        plt.title(f'Distribution of {col}')\n        plt.xlabel(col)\n        plt.ylabel('Frequency')\n        plt.show()\n    \n    # Plotting categorical columns\n    for col in categorical_cols:\n        plt.figure(figsize=(10, 4))\n        sns.countplot(data=df, x=col, hue=target)\n        plt.title(f'Count plot for {col}')\n        plt.xlabel(col)\n        plt.ylabel('Count')\n        plt.xticks(rotation=45)\n        plt.show()\n\n# Function to compare train and test datasets\ndef compare_train_test(train, test):\n    continuous_cols, categorical_cols = classify_columns(train)\n    \n    # Compare continuous columns\n    for col in continuous_cols:\n        plt.figure(figsize=(10, 4))\n        sns.kdeplot(train[col], label='Train', shade=True)\n        sns.kdeplot(test[col], label='Test', shade=True)\n        plt.title(f'Comparison of {col} Distribution in Train vs Test')\n        plt.xlabel(col)\n        plt.ylabel('Density')\n        plt.legend()\n        plt.show()\n    \n    # Compare categorical columns\n    for col in categorical_cols:\n        if col in test.columns:  # Ensure the column exists in the test dataset\n            plt.figure(figsize=(10, 4))\n            train_counts = train[col].value_counts(normalize=True)\n            test_counts = test[col].value_counts(normalize=True)\n            train_counts.plot(kind='bar', alpha=0.5, label='Train', color='blue')\n            test_counts.plot(kind='bar', alpha=0.5, label='Test', color='red')\n            plt.title(f'Comparison of {col} Proportions in Train vs Test')\n            plt.xlabel(col)\n            plt.ylabel('Proportion')\n            plt.legend()\n            plt.xticks(rotation=45)\n            plt.show()\n            \n# Function to create key statistics for a dataset\ndef eda_summary(df):\n    summary = {}\n    \n    # General Info\n    summary['general'] = {\n        'num_rows': df.shape[0],\n        'num_columns': df.shape[1],\n        'num_missing_values': df.isnull().sum().sum(),\n        'percent_missing_values': df.isnull().mean().mean() * 100\n    }\n    \n    # Column Data Types\n    summary['data_types'] = df.dtypes.to_dict()\n    \n    # Missing Value Summary (per column)\n    summary['missing_values'] = (\n        df.isnull()\n        .sum()\n        .to_frame(name='missing_count')\n        .assign(percent_missing=lambda x: (x['missing_count'] / df.shape[0]) * 100)\n        .to_dict(orient='index')\n    )\n    \n    # Numerical Summary (Mean, Median, Std, Min, Max)\n    describe_df = df.describe()\n    numerical_columns = ['mean', '50%', 'std', 'min', 'max']\n    available_columns = [col for col in numerical_columns if col in describe_df.columns]\n    summary['numerical_summary'] = (\n        describe_df[available_columns]\n        .rename(columns={'50%': 'median'})\n        .to_dict(orient='index')\n    )\n    \n    # Unique Counts for Categorical Columns\n    summary['categorical_summary'] = (\n        df.select_dtypes(include=['object', 'category'])\n        .nunique()\n        .to_frame(name='unique_counts')\n        .to_dict(orient='index')\n    )\n    \n    # Skewness and Kurtosis\n    summary['skewness_kurtosis'] = {\n        column: {\n            'skewness': df[column].skew(),\n            'kurtosis': df[column].kurt()\n        } for column in df.select_dtypes(include=[np.number]).columns\n    }\n    \n    # Correlations\n    try:\n        summary['correlations'] = df.corr(numeric_only=True).to_dict()\n    except ValueError:\n        summary['correlations'] = \"Unable to calculate correlations due to data type issues.\"\n    \n    # Outlier Count based on IQR\n    outlier_summary = {}\n    for column in df.select_dtypes(include=[np.number]).columns:\n        Q1 = df[column].quantile(0.25)\n        Q3 = df[column].quantile(0.75)\n        IQR = Q3 - Q1\n        outliers = df[(df[column] < (Q1 - 1.5 * IQR)) | (df[column] > (Q3 + 1.5 * IQR))]\n        outlier_summary[column] = {\n            'outlier_count': outliers.shape[0],\n            'percent_outliers': (outliers.shape[0] / df.shape[0]) * 100\n        }\n    summary['outlier_summary'] = outlier_summary\n\n    return summary\n\n# Function to create a summarized key statistics for a dataset\n# Function to create a summarized key statistics for a dataset\ndef eda_summary_concise(df):\n    summary = {}\n    \n    # General Info\n    summary['general'] = {\n        'num_rows': df.shape[0],\n        'num_columns': df.shape[1],\n        'percent_missing_values': df.isnull().mean().mean() * 100\n    }\n    \n    # Column Data Types Summary\n    data_types = df.dtypes.apply(lambda x: str(x)).value_counts().to_dict()\n    summary['data_types_summary'] = data_types\n    \n    # Columns with the Most Missing Values\n    missing_values = (\n        df.isnull().sum()\n        .sort_values(ascending=False)\n        .head(5)  # Limit to top 5 columns\n        .to_frame(name='missing_count')\n        .assign(percent_missing=lambda x: (x['missing_count'] / df.shape[0]) * 100)\n        .to_dict(orient='index')\n    )\n    # Converting any non-serializable keys to strings\n    missing_values = {str(k): v for k, v in missing_values.items()}\n    summary['most_missing_values'] = missing_values\n\n    # Key Numerical Summary (Mean, Median, Std)\n    describe_df = df.describe()\n    numerical_summary = describe_df.loc[['mean', '50%', 'std']].to_dict(orient='index')\n    numerical_summary = {str(k): v for k, v in numerical_summary.items()}\n    summary['numerical_summary'] = numerical_summary\n    \n    # Correlations Summary (only highest correlations)\n    correlations = df.corr(numeric_only=True)\n    if correlations is not None:\n        highest_corrs = (\n            correlations.unstack()\n            .sort_values(key=lambda x: abs(x), ascending=False)\n            .drop_duplicates()\n            .head(5)\n            .to_dict()\n        )\n        # Converting the keys which are tuples into strings\n        highest_corrs = {f\"{k[0]}_{k[1]}\": v for k, v in highest_corrs.items()}\n        summary['highest_correlations'] = highest_corrs\n\n    return summary","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T01:07:24.486124Z","iopub.execute_input":"2024-11-16T01:07:24.486518Z","iopub.status.idle":"2024-11-16T01:07:24.514411Z","shell.execute_reply.started":"2024-11-16T01:07:24.486481Z","shell.execute_reply":"2024-11-16T01:07:24.513280Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# load datasets\nDATA_DIR = \"/kaggle/input/jane-street-real-time-market-data-forecasting\"\n\nfeatures = pd.read_csv(f\"{DATA_DIR}/features.csv\")\nresponders = pd.read_csv(f\"{DATA_DIR}/responders.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T01:07:33.309280Z","iopub.execute_input":"2024-11-16T01:07:33.309677Z","iopub.status.idle":"2024-11-16T01:07:33.337473Z","shell.execute_reply.started":"2024-11-16T01:07:33.309641Z","shell.execute_reply":"2024-11-16T01:07:33.336551Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# training set\n\ndir_path = Path(DATA_DIR, 'train.parquet/')\n\nfor file in dir_path.glob('*'):\n    print(file)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T01:07:42.224305Z","iopub.execute_input":"2024-11-16T01:07:42.224721Z","iopub.status.idle":"2024-11-16T01:07:42.236412Z","shell.execute_reply.started":"2024-11-16T01:07:42.224683Z","shell.execute_reply":"2024-11-16T01:07:42.235223Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pl.read_parquet(f\"{DATA_DIR}/train.parquet/partition_id=0/part-0.parquet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T01:07:49.764066Z","iopub.execute_input":"2024-11-16T01:07:49.764963Z","iopub.status.idle":"2024-11-16T01:07:51.995788Z","shell.execute_reply.started":"2024-11-16T01:07:49.764921Z","shell.execute_reply":"2024-11-16T01:07:51.994862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_pd = train.to_pandas()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T01:07:55.519899Z","iopub.execute_input":"2024-11-16T01:07:55.521033Z","iopub.status.idle":"2024-11-16T01:07:56.647383Z","shell.execute_reply.started":"2024-11-16T01:07:55.520957Z","shell.execute_reply":"2024-11-16T01:07:56.646240Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# reduce the dataset to a smaller size compatible with Kaggle kernels\nsample_fraction = 0.25\ntrain_pd = train_pd.sample(frac=sample_fraction, random_state=42).reset_index(drop=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T01:08:04.895046Z","iopub.execute_input":"2024-11-16T01:08:04.895447Z","iopub.status.idle":"2024-11-16T01:08:05.454975Z","shell.execute_reply.started":"2024-11-16T01:08:04.895410Z","shell.execute_reply":"2024-11-16T01:08:05.453816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Automated EDA report with sweetviz\nreport = sv.analyze(train_pd, pairwise_analysis=\"off\")\nreport.show_html(filepath='/kaggle/working/train_data_report.html', open_browser=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T01:15:20.464927Z","iopub.execute_input":"2024-11-16T01:15:20.465440Z","iopub.status.idle":"2024-11-16T01:16:57.343385Z","shell.execute_reply.started":"2024-11-16T01:15:20.465396Z","shell.execute_reply":"2024-11-16T01:16:57.341920Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Assuming train_pd is the DataFrame\nsummary_concise = eda_summary_concise(train_pd)\n\n# Serializing to JSON\nsummary_json_concise = json.dumps(summary_concise, indent=4, default=str)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T01:16:57.345537Z","iopub.execute_input":"2024-11-16T01:16:57.345914Z","iopub.status.idle":"2024-11-16T01:17:09.242908Z","shell.execute_reply.started":"2024-11-16T01:16:57.345877Z","shell.execute_reply":"2024-11-16T01:17:09.241786Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"template = \"\"\"Provide an analysis of the following EDA summary:\n{context}\n\nKey insights and observations:\n\"\"\"\n\nprompt = ChatPromptTemplate.from_template(template)\n\n# Create a chain to pass the summary to the model\nchain = prompt | model | StrOutputParser()\n\n# Invoke the chain to analyze the EDA summary\nresult = chain.invoke(summary_json_concise)\n\n# Print the result\ndisplay(Markdown(result))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <div style=\"background-color:#0A0F29; font-family:'Poppins', cursive; color:#E0F7FA; font-size:140%; text-align:center; border: 2px solid #00FFFF; border-radius:15px; padding: 15px; box-shadow: 5px 5px 20px rgba(0, 0, 0, 0.5); font-weight: bold; letter-spacing: 1px; text-transform: uppercase;\">Detailled EDA</div>","metadata":{}},{"cell_type":"code","source":"# General Overview\ndef general_overview(data):\n    print(\"\\n--- General Overview ---\\n\")\n    print(f\"Dataset Shape: {data.shape}\")\n    print(f\"Dataset Info:\\n{data.info()}\")\n    print(f\"Dataset Missing Values:\\n{data.isnull().sum()}\")\n    print(f\"Dataset Missing Values Percentage:\\n{data.isnull().sum() / len(data) * 100}\")\n    print(\"\\n\")\n\n# Data Types Summary\ndef data_types_summary(data):\n    print(\"--- Data Types Summary ---\\n\")\n    print(data.dtypes.value_counts())\n    print(\"\\n\")\n\n# Check Missing Values\ndef missing_values_summary(data):\n    print(\"--- Missing Values ---\\n\")\n    missing_values = data.isnull().sum().sort_values(ascending=False)\n    missing_percent = (missing_values / len(data)) * 100\n    missing_summary = pd.DataFrame({'Missing Values': missing_values, 'Percentage': missing_percent})\n    print(missing_summary[missing_summary['Missing Values'] > 0])\n    print(\"\\n\")\n\n# Descriptive Statistics\ndef descriptive_statistics(data):\n    print(\"--- Descriptive Statistics ---\\n\")\n    desc_stats = data.describe(include='all')\n    print(desc_stats)\n    print(\"\\n\")\n\n# Correlation Analysis\ndef correlation_analysis(data, threshold=0.9):\n    print(\"--- Correlation Analysis ---\\n\")\n    correlations = data.corr()\n    sns.heatmap(correlations, cmap=\"coolwarm\", linewidths=0.5)\n    plt.title(\"Correlation Heatmap\")\n    plt.show()\n\n    # Display high correlation pairs\n    high_corr_pairs = []\n    for i in range(correlations.shape[0]):\n        for j in range(i + 1, correlations.shape[0]):\n            if abs(correlations.iloc[i, j]) > threshold:\n                high_corr_pairs.append((correlations.columns[i], correlations.columns[j], correlations.iloc[i, j]))\n\n    if high_corr_pairs:\n        print(\"High correlation pairs (>0.9):\")\n        for feature1, feature2, corr_value in high_corr_pairs:\n            print(f\"{feature1} and {feature2}: {corr_value:.2f}\")\n    else:\n        print(\"No highly correlated feature pairs found (>0.9)\")\n    print(\"\\n\")\n\n# Missing Values Handling\ndef visualize_missing_data(data):\n    print(\"--- Visualizing Missing Data ---\\n\")\n    plt.figure(figsize=(12, 6))\n    sns.heatmap(data.isnull(), cbar=False, cmap='viridis')\n    plt.title('Heatmap of Missing Data')\n    plt.show()\n\n# Visualizing Selected Feature Relationships\ndef visualize_feature_relationships(data, features):\n    for feature in features:\n        plt.figure(figsize=(10, 5))\n        sns.histplot(data[feature].dropna(), kde=True)\n        plt.title(f'Distribution of {feature}')\n        plt.xlabel(feature)\n        plt.ylabel('Count')\n        plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Running the EDA\ngeneral_overview(train_pd)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_types_summary(train_pd)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_values_summary(train_pd)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"descriptive_statistics(train_pd)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"correlation_analysis(train_pd)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"visualize_missing_data(train_pd)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualize key feature relationships\nselected_features = ['feature_75', 'feature_76', 'feature_77', 'feature_78', 'responder_0']\nvisualize_feature_relationships(train_pd, selected_features)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plotting Feature by Time\ndef plot_feature_by_time(data, feature, time_column='time_id'):\n    print(f\"--- Plotting {feature} by {time_column} ---\\n\")\n    plt.figure(figsize=(14, 7))\n    plt.plot(data[time_column], data[feature], marker='o', linestyle='-', alpha=0.7)\n    plt.xlabel(time_column)\n    plt.ylabel(feature)\n    plt.title(f'{feature} Over {time_column}')\n    plt.grid(True)\n    plt.show()\n\n# Analyzing Symbol Distribution\ndef analyze_symbol_distribution(data):\n    print(\"--- Analyzing Symbol Distribution ---\\n\")\n    symbol_counts = data['symbol_id'].value_counts()\n    print(f\"Number of Unique Symbols: {symbol_counts.count()}\")\n    plt.figure(figsize=(14, 7))\n    sns.histplot(symbol_counts, bins=50, kde=True)\n    plt.title('Distribution of Symbols (symbol_id)')\n    plt.xlabel('Symbol Count')\n    plt.ylabel('Frequency')\n    plt.show()\n\n# Analyzing Weight Distribution\ndef analyze_weight_distribution(data):\n    print(\"--- Analyzing Weight Distribution ---\\n\")\n    plt.figure(figsize=(10, 5))\n    sns.histplot(data['weight'].dropna(), kde=True)\n    plt.title('Distribution of Weight')\n    plt.xlabel('Weight')\n    plt.ylabel('Count')\n    plt.show()\n\n# Analyzing Responder Target (responder_6)\ndef analyze_responder_target(data):\n    print(\"--- Analyzing Responder Target (responder_6) ---\\n\")\n    plt.figure(figsize=(10, 5))\n    sns.histplot(data['responder_6'].dropna(), kde=True)\n    plt.title('Distribution of responder_6 (Target)')\n    plt.xlabel('Responder 6')\n    plt.ylabel('Count')\n    plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plotting feature by time\ntime_column = 'time_id'\nplot_feature_by_time(train_pd, 'feature_75', time_column)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"time_column = 'time_id'\nplot_feature_by_time(train_pd, 'feature_72', time_column)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Analyze Symbol Distribution\nanalyze_symbol_distribution(train_pd)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Analyze Weight Distribution\nanalyze_weight_distribution(train_pd)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Analyze Responder Target (responder_6)\nanalyze_responder_target(train_pd)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plotting Feature for a Given Symbol\ndef plot_feature_for_symbol(data, feature, symbol_id, time_column='time_id'):\n    print(f\"--- Plotting {feature} for Symbol {symbol_id} by {time_column} ---\\n\")\n    symbol_data = data[data['symbol_id'] == symbol_id].sort_values(by=[time_column])\n    if symbol_data.empty:\n        print(f\"No data available for symbol_id {symbol_id}\")\n    else:\n        plt.figure(figsize=(14, 7))\n        plt.plot(symbol_data[time_column], symbol_data[feature], marker='o', linestyle='-', alpha=0.7)\n        plt.xlabel(time_column)\n        plt.ylabel(feature)\n        plt.title(f'{feature} for Symbol {symbol_id} Over {time_column}')\n        plt.grid(True)\n        plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plotting feature_75 for a given symbol (example with symbol_id=5)\nplot_feature_for_symbol(train_pd, 'feature_72', symbol_id=2, time_column='time_id')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <div style=\"background-color:#0A0F29; font-family:'Poppins', cursive; color:#E0F7FA; font-size:140%; text-align:center; border: 2px solid #00FFFF; border-radius:15px; padding: 15px; box-shadow: 5px 5px 20px rgba(0, 0, 0, 0.5); font-weight: bold; letter-spacing: 1px; text-transform: uppercase;\">Random Forest baseline model</div>","metadata":{}},{"cell_type":"code","source":"class MarketDataModel:\n    def __init__(self):\n        self.scaler = RobustScaler()  # More robust to outliers\n        self.imputer = KNNImputer(n_neighbors=5)\n        self.feature_selector = None\n        self.model = None\n        self.feature_names = None\n        \n    def prepare_features(self, df):\n        \"\"\"\n        Prepare features with improved preprocessing\n        \"\"\"\n        # Select features and target\n        feature_cols = [col for col in df.columns if col.startswith('feature_')]\n        \n        # Remove features with >50% missing values\n        missing_pct = df[feature_cols].isnull().mean()\n        valid_features = missing_pct[missing_pct < 0.5].index.tolist()\n        \n        # Add other potentially important columns\n        additional_cols = ['weight', 'symbol_id', 'time_id']\n        valid_features.extend([col for col in additional_cols if col in df.columns])\n        \n        # Create time-based features\n        X = df[valid_features].copy()\n        X['hour'] = df['time_id'] // 60\n        X['minute'] = df['time_id'] % 60\n        \n        # Create symbol-based aggregations\n        symbol_means = df.groupby('symbol_id')[valid_features].transform('mean')\n        symbol_means.columns = [f'{col}_symbol_mean' for col in symbol_means.columns]\n        X = pd.concat([X, symbol_means], axis=1)\n        \n        return X\n        \n    def prepare_target(self, df):\n        \"\"\"\n        Prepare target variable with custom logic\n        \"\"\"\n        # Combine multiple responders for a more robust target\n        y = df[['responder_6', 'responder_7', 'responder_8']].mean(axis=1)\n        return y\n        \n    def optimize_hyperparameters(self, X, y, n_trials=50):\n        \"\"\"\n        Optimize RF hyperparameters using Optuna\n        \"\"\"\n        def objective(trial):\n            params = {\n                'n_estimators': trial.suggest_int('n_estimators', 100, 300),\n                'max_depth': trial.suggest_int('max_depth', 5, 30),\n                'min_samples_split': trial.suggest_int('min_samples_split', 2, 10),\n                'min_samples_leaf': trial.suggest_int('min_samples_leaf', 1, 5),\n                'max_features': trial.suggest_float('max_features', 0.3, 1.0)\n            }\n            \n            rf = RandomForestRegressor(**params, random_state=42, n_jobs=-1)\n            scores = []\n            \n            tscv = TimeSeriesSplit(n_splits=3)\n            for train_idx, val_idx in tscv.split(X):\n                X_train, X_val = X[train_idx], X[val_idx]\n                y_train, y_val = y[train_idx], y[val_idx]\n                \n                rf.fit(X_train, y_train)\n                y_pred = rf.predict(X_val)\n                score = mean_squared_error(y_val, y_pred, squared=False)\n                scores.append(score)\n                \n            return np.mean(scores)\n            \n        study = optuna.create_study(direction='minimize')\n        study.optimize(lambda trial: objective(trial), n_trials=n_trials)\n        \n        return study.best_params\n        \n    def fit(self, X, y):\n        \"\"\"\n        Train the model with preprocessed data\n        \"\"\"\n        # Scale features\n        X_scaled = self.scaler.fit_transform(X)\n        \n        # Impute missing values\n        X_imputed = self.imputer.fit_transform(X_scaled)\n        \n        # Feature selection\n        base_model = RandomForestRegressor(n_estimators=100, random_state=42)\n        self.feature_selector = SelectFromModel(base_model, prefit=False)\n        X_selected = self.feature_selector.fit_transform(X_imputed, y)\n        \n        # Get best hyperparameters\n        best_params = self.optimize_hyperparameters(X_selected, y)\n        \n        # Train final model\n        self.model = RandomForestRegressor(**best_params, random_state=42, n_jobs=-1)\n        self.model.fit(X_selected, y)\n        \n        # Store feature names\n        self.feature_names = X.columns[self.feature_selector.get_support()].tolist()\n        \n    def predict(self, X):\n        \"\"\"\n        Make predictions on new data\n        \"\"\"\n        X_scaled = self.scaler.transform(X)\n        X_imputed = self.imputer.transform(X_scaled)\n        X_selected = self.feature_selector.transform(X_imputed)\n        return self.model.predict(X_selected)\n        \n    def evaluate(self, X, y):\n        \"\"\"\n        Evaluate model performance\n        \"\"\"\n        y_pred = self.predict(X)\n        metrics = {\n            'mae': mean_absolute_error(y, y_pred),\n            'rmse': mean_squared_error(y, y_pred, squared=False),\n            'r2': r2_score(y, y_pred)\n        }\n        return metrics\n        \n    def get_feature_importance(self):\n        \"\"\"\n        Get feature importance scores\n        \"\"\"\n        importance = self.model.feature_importances_\n        return pd.DataFrame({\n            'feature': self.feature_names,\n            'importance': importance\n        }).sort_values('importance', ascending=False)\n\n# Usage example\ndef train_improved_model(train_df):\n    model = MarketDataModel()\n    \n    # Prepare data\n    X = model.prepare_features(train_df)\n    y = model.prepare_target(train_df)\n    \n    # Train-validation split\n    tscv = TimeSeriesSplit(n_splits=5)\n    metrics_list = []\n    \n    for train_idx, val_idx in tscv.split(X):\n        X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n        \n        # Train model\n        model.fit(X_train, y_train)\n        \n        # Evaluate\n        metrics = model.evaluate(X_val, y_val)\n        metrics_list.append(metrics)\n        \n    # Print average metrics\n    avg_metrics = pd.DataFrame(metrics_list).mean()\n    print(\"\\nAverage Cross-validation Metrics:\")\n    print(f\"MAE: {avg_metrics['mae']:.4f}\")\n    print(f\"RMSE: {avg_metrics['rmse']:.4f}\")\n    print(f\"R2: {avg_metrics['r2']:.4f}\")\n    \n    # Plot feature importance\n    importance_df = model.get_feature_importance()\n    plot_feature_importance(importance_df)\n    \n    return model","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}