{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-05T03:23:31.578162Z","iopub.execute_input":"2025-07-05T03:23:31.578458Z","iopub.status.idle":"2025-07-05T03:23:31.982252Z","shell.execute_reply.started":"2025-07-05T03:23:31.578436Z","shell.execute_reply":"2025-07-05T03:23:31.981380Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Path to the Parquet file\ntrain_parquet_path = '/kaggle/input/drw-crypto-market-prediction/train.parquet'\n\n# Load the Parquet file\ntrain_df = pd.read_parquet(train_parquet_path)\n\n# Now train_df contains the same data as if you loaded train.csv,\n# but likely much faster.\nprint(train_df.head())\nprint(train_df.info())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-05T03:23:31.983901Z","iopub.execute_input":"2025-07-05T03:23:31.984258Z","iopub.status.idle":"2025-07-05T03:23:57.101433Z","shell.execute_reply.started":"2025-07-05T03:23:31.984235Z","shell.execute_reply":"2025-07-05T03:23:57.100158Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\nMissing values in 'label':\")\nprint(train_df['label'].isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-05T03:23:57.102186Z","iopub.execute_input":"2025-07-05T03:23:57.102483Z","iopub.status.idle":"2025-07-05T03:23:57.110224Z","shell.execute_reply.started":"2025-07-05T03:23:57.102451Z","shell.execute_reply":"2025-07-05T03:23:57.108976Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\nDescriptive statistics for 'label':\")\nprint(train_df['label'].describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-05T03:23:57.111164Z","iopub.execute_input":"2025-07-05T03:23:57.111452Z","iopub.status.idle":"2025-07-05T03:23:57.153197Z","shell.execute_reply.started":"2025-07-05T03:23:57.111430Z","shell.execute_reply":"2025-07-05T03:23:57.152433Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\nplt.figure(figsize=(10, 6))\nsns.histplot(train_df['label'], bins=100, kde=True)\nplt.title('Distribution of Target Variable (label)')\nplt.xlabel('Label Value')\nplt.ylabel('Frequency')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-05T03:23:57.154006Z","iopub.execute_input":"2025-07-05T03:23:57.154528Z","iopub.status.idle":"2025-07-05T03:24:01.021616Z","shell.execute_reply.started":"2025-07-05T03:23:57.154481Z","shell.execute_reply":"2025-07-05T03:24:01.020199Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"@From the Distribution of Target Variable (label) Histogram:\nHighly Peaked (Leptokurtic): The most striking feature is the extremely sharp peak at label = 0. This tells us that for a vast majority of the time intervals, the market experiences very little or no movement. The frequency count reaches over 160,000 for values very close to zero.\n\n\n@Symmetry around Zero (mostly): The distribution appears largely symmetric around zero, which is common for financial returns, where price movements can be both positive and negative.\n\n\n@\"Fat Tails\": Despite the sharp peak, the distribution extends significantly into the negative (down to -20 and beyond) and positive (up to 20 and beyond) ranges. These \"fat tails\" (or \"leptokurtosis\") are characteristic of financial data, meaning that extreme events (large price changes) occur more frequently than they would in a purely normal (Gaussian) distribution.","metadata":{}},{"cell_type":"markdown","source":"Data Skewness/Outliers: The fat tails mean that outliers (large movements) are inherent to the data, not necessarily errors. You generally shouldn't just remove them, as they contain valuable information about market volatility.","metadata":{}},{"cell_type":"code","source":"print(\"\\nMissing values for core trading features:\")\ncore_features = ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']\nfor col in core_features:\n    print(f\"{col}: {train_df[col].isnull().sum()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-05T03:24:01.023324Z","iopub.execute_input":"2025-07-05T03:24:01.024246Z","iopub.status.idle":"2025-07-05T03:24:01.044185Z","shell.execute_reply.started":"2025-07-05T03:24:01.024207Z","shell.execute_reply":"2025-07-05T03:24:01.043257Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\nDescriptive statistics for core trading features:\")\nprint(train_df[core_features].describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-05T03:24:01.046572Z","iopub.execute_input":"2025-07-05T03:24:01.046826Z","iopub.status.idle":"2025-07-05T03:24:01.200112Z","shell.execute_reply.started":"2025-07-05T03:24:01.046806Z","shell.execute_reply":"2025-07-05T03:24:01.199180Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Example for bid_qty\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nplt.figure(figsize=(10, 6))\nsns.histplot(train_df['bid_qty'], bins=50, kde=True)\nplt.title('Distribution of Bid Quantity')\nplt.xlabel('Bid Quantity')\nplt.ylabel('Frequency')\nplt.show()\n\n# You can repeat this for other core_features.\n# If distributions are heavily skewed (common for volume/quantity),\n# consider using `plt.yscale('log')` to see details for smaller values better.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-05T03:24:01.201004Z","iopub.execute_input":"2025-07-05T03:24:01.201259Z","iopub.status.idle":"2025-07-05T03:24:03.803332Z","shell.execute_reply.started":"2025-07-05T03:24:01.201236Z","shell.execute_reply":"2025-07-05T03:24:03.801874Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"Bid/Ask Quantities (bid_qty, ask_qty):\nThey are very similar to each other, as expected (reflecting opposite sides of the order book).\nMean values (around 10) are much lower than max values (over 1000), indicating a strong positive skew.\nMin values are 0.001, meaning there's almost always a small quantity available, not perfectly zero.","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Overall Skewness: For all these features, the mean is significantly higher than the median (50th percentile), and the max value is orders of magnitude larger than the 75th percentile. This strongly indicates heavy positive skewness and the presence of extreme outliers (spikes).","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"Implications for Feature Engineering and Modeling:\n\n1.Skewness and Transformations: The heavy skewness of these features is a critical observation.\nFor models sensitive to input distributions (e.g., linear models, neural networks), logarithmic transformations (e.g., np.log(1 + feature_value)) would be highly beneficial to normalize these distributions and compress the range of extreme values.\nTree-based models like LightGBM or XGBoost are less sensitive to input distribution, but transformations can still sometimes help by making relationships more linear for the tree splits.\n\n2.Zero/Near-Zero Values:\nThe min of 0.001 for bid_qty and ask_qty is notable. When taking logarithms, you'll need to handle this (e.g., np.log(feature_value) will work, but np.log(1 + feature_value) is safer if actual zeros existed).\nThe min of 0 for buy_qty, sell_qty, and volume means you must use np.log(1 + feature_value) to avoid log(0).\n\n3.Feature Interaction / Ratios: These quantities are intrinsically linked. Consider creating new features:\nOrder Book Imbalance: (bid_qty - ask_qty) / (bid_qty + ask_qty)\n\n\nTotal Order Book Quantity: bid_qty + ask_qty\n\n\nTrade Imbalance: (buy_qty - sell_qty) / (buy_qty + sell_qty)\n\n\nTotal Traded Quantity: buy_qty + sell_qty (should be close to volume)\nRatios: buy_qty / bid_qty, sell_qty / ask_qty (if these make sense in context).\n\n4.Lagged Features: The values of these features from previous minutes will be highly predictive of their current/future values, and thus indirectly, of the label.","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\ncore_features = ['ask_qty', 'buy_qty', 'sell_qty', 'volume']\n\nfor col in core_features:\n    plt.figure(figsize=(10, 6))\n    sns.histplot(train_df[col], bins=50, kde=True)\n    plt.title(f'Distribution of {col.replace(\"_\", \" \").title()}') # Nicer title\n    plt.xlabel(col.replace(\"_\", \" \").title())\n    plt.ylabel('Frequency')\n    # Consider uncommenting the next line if the distribution is very skewed and you want to see low-frequency bins better\n    # plt.yscale('log')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-05T03:24:03.804577Z","iopub.execute_input":"2025-07-05T03:24:03.805044Z","iopub.status.idle":"2025-07-05T03:24:14.028027Z","shell.execute_reply.started":"2025-07-05T03:24:03.805006Z","shell.execute_reply":"2025-07-05T03:24:14.027096Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Summary of Observations for Core Trading Features (bid_qty, ask_qty, buy_qty, sell_qty, volume):\n\n1.Universally Skewed\n2.Order Book vs. Executed Trades:\nbid_qty and ask_qty (quantities available on the order book) are generally lower in magnitude and have shorter tails than buy_qty, sell_qty, and volume (executed quantities). This makes sense: the amount of open orders at the best price is typically less than the total volume traded in a minute.\nbuy_qty and sell_qty show very similar distributions, as do bid_qty and ask_qty, which is expected given they represent opposite sides of the same market activity.\nvolume exhibits the longest tail and highest maximum values, as it represents the total executed quantity.\n","metadata":{}},{"cell_type":"markdown","source":"Step 3: Examine the X Features (X1 to X890):\nSince there are 890 of them, we'll examine a sample for missing values and descriptive statistics.\n\n","metadata":{}},{"cell_type":"code","source":"print(\"\\nMissing values for a sample of X features (e.g., first 10):\")\nx_features_sample = [f'X{i}' for i in range(1, 11) if f'X{i}' in train_df.columns]\nfor col in x_features_sample:\n    print(f\"{col}: {train_df[col].isnull().sum()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-05T03:24:14.029422Z","iopub.execute_input":"2025-07-05T03:24:14.029822Z","iopub.status.idle":"2025-07-05T03:24:14.061307Z","shell.execute_reply.started":"2025-07-05T03:24:14.029771Z","shell.execute_reply":"2025-07-05T03:24:14.060321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\nDescriptive statistics for a sample of X features (e.g., X1-X5):\")\nprint(train_df[['X1', 'X2', 'X3', 'X4', 'X5']].describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-05T03:24:14.062592Z","iopub.execute_input":"2025-07-05T03:24:14.062844Z","iopub.status.idle":"2025-07-05T03:24:14.204260Z","shell.execute_reply.started":"2025-07-05T03:24:14.062825Z","shell.execute_reply":"2025-07-05T03:24:14.203372Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Analysis of the X Features (X1 to X5 as a sample):\n\n\nFrom Descriptive Statistics:\n\n\ncount: 525887.000000: Confirms no missing values across these sampled X columns.\n\n\nmean (close to 0): The mean values for X1 through X5 are very close to zero (e.g., -0.006, -0.0002). This strongly suggests that these features have been standardized or centered around zero.\n\n\nstd (around 0.5 to 0.8): The standard deviations are relatively consistent but not exactly 1 (they range from ~0.46 to ~0.85). This implies they might have been scaled, but not necessarily to a standard normal distribution (where std = 1).\n\n\nmin and max: The range of values generally extends from roughly -6 to +6 (e.g., X2: min -5.86, max 6.15). This is a typical range for standardized features, suggesting there are not excessively extreme outliers compared to the scale of the feature itself.\n\n\n50% (median) (close to 0, slightly negative): The medians are also very close to zero, often slightly negative (e.g., -0.0159 for X1). This, combined with the mean being slightly negative, indicates a very slight negative skew, but for practical purposes, they appear largely symmetrical around zero.\n","metadata":{}},{"cell_type":"markdown","source":"Next Step: Time-Series Specific EDA\nThis is crucial as it's a time-series prediction problem.","metadata":{}},{"cell_type":"code","source":"print(f\"\\nTotal number of entries: {len(train_df)}\")\nprint(f\"Time range: {train_df.index.min()} to {train_df.index.max()}\")\ntime_diffs = train_df.index.to_series().diff().dropna()\nprint(f\"Median time difference between consecutive rows: {time_diffs.median()}\")\nprint(f\"Most frequent time difference: {time_diffs.mode()[0]}\")\nprint(f\"Number of unique time differences: {len(time_diffs.unique())}\")\nprint(f\"Top 5 most frequent time differences:\\n{time_diffs.value_counts().head()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-05T03:24:14.205238Z","iopub.execute_input":"2025-07-05T03:24:14.205538Z","iopub.status.idle":"2025-07-05T03:24:14.253241Z","shell.execute_reply.started":"2025-07-05T03:24:14.205490Z","shell.execute_reply":"2025-07-05T03:24:14.252411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(15, 7))\nplt.plot(train_df.index, train_df['label'], alpha=0.7, label='Original Label')\nplt.title('Target (label) over Time')\nplt.xlabel('Timestamp')\nplt.ylabel('Label Value')\nplt.grid(True)\nplt.legend()\nplt.show()\n\n# To see broader trends, plot a rolling mean\nplt.figure(figsize=(15, 7))\nplt.plot(train_df.index, train_df['label'].rolling(window=60*24).mean(), alpha=0.7, label='24-Hour Rolling Mean') # 24 hours of minutes\nplt.title('Target (label) 24-Hour Rolling Mean over Time')\nplt.xlabel('Timestamp')\nplt.ylabel('Rolling Mean Label Value')\nplt.grid(True)\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-05T03:24:14.254134Z","iopub.execute_input":"2025-07-05T03:24:14.254406Z","iopub.status.idle":"2025-07-05T03:24:15.556616Z","shell.execute_reply.started":"2025-07-05T03:24:14.254378Z","shell.execute_reply":"2025-07-05T03:24:15.555378Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Select a few features to check correlation\n# We already know core_features and some X. Let's make a new list for printing.\nfeatures_for_corr_summary = ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume', 'X1', 'X2', 'X3', 'X4', 'X5']\n\nprint(\"\\nCorrelations with 'label' for selected features:\")\nprint(train_df[features_for_corr_summary + ['label']].corr()['label'].sort_values(ascending=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-05T03:24:15.557816Z","iopub.execute_input":"2025-07-05T03:24:15.558188Z","iopub.status.idle":"2025-07-05T03:24:15.811125Z","shell.execute_reply.started":"2025-07-05T03:24:15.558154Z","shell.execute_reply":"2025-07-05T03:24:15.809754Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Analysis of the Correlation Output:\n\n\nThe single most important takeaway is that all linear correlations between the selected features and the label are extremely weak. The strongest correlation (apart from the label with itself) is ask_qty at a mere -0.015762.\n\n\nHere's what this tells us:\n\n\n1.No \"Silver Bullet\" Feature: There is no single feature in this set that has a strong, direct linear relationship with the target. You cannot simply look at X2 or ask_qty and predict the label with any accuracy.\n\n\n2.The Problem is Non-Linear: \n\nThis is the crucial insight. The relationships are likely complex, conditional, and interactive.\nFor example, a high volume might only be predictive of a positive label if ask_qty is low.\nOr, the change in a feature (e.g., volume this minute vs. volume last minute) might be more important than its absolute value.\n\n\n3.A linear correlation check would completely miss these kinds of sophisticated patterns.\nModel Choice is Critical:\n\nThis result immediately tells you that a simple Linear Regressi\non model will perform very poorly. You need a model that can automatically capture these complex, non-linear interactions. This is precisely why Gradient Boosted Trees (like LightGBM, XGBoost, or CatBoost) are the go-to models for this type of tabular, noisy financial data. They are exceptionally good at finding predictive power from a large number of weak predictors.","metadata":{}},{"cell_type":"markdown","source":"Summary of Entire EDA and Path Forward\nWe have now completed a thorough initial EDA. Here's what we've learned and what to do next:\n\n\nWhat We Know:\n\n\n1.Target (label): A price return, centered around zero, with a very sharp peak and fat tails (many small movements, rare large movements).\n\n\n2.Core Features (bid_qty, etc.): Highly skewed with extreme outliers (spikes). They are all complete with no missing values.\n\n\n3.X Features: Anonymized, pre-processed (centered/standardized), and ready to use. Also complete with no missing values.\n\n\n4.Time: The data is a near-perfect 1-minute interval time series for a single asset/market over one year.\n\n\n5.Relationships: The relationship between any single raw feature and the target is extremely weak linearly, implying the predictive power lies in non-linear interactions and time-dependent patterns.","metadata":{}},{"cell_type":"markdown","source":"\n\n\n\n","metadata":{}},{"cell_type":"markdown","source":"The EDA has given us a clear roadmap. The key to this competition is not in the raw features, but in the ones you create.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# Let's assume train_df is your loaded DataFrame with timestamp as index\n\ndef feature_engineering(df):\n    \"\"\"Creates time-series and interaction features.\"\"\"\n    df_out = df.copy()\n\n    # 1. Log transform skewed features\n    # Using np.log1p which is log(1+x) to handle zeros\n    skewed_features = ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']\n    for col in skewed_features:\n        df_out[f'{col}_log'] = np.log1p(df_out[col])\n\n    # 2. Interaction & Ratio Features (based on EDA insights)\n    df_out['order_book_imbalance'] = (df_out['bid_qty'] - df_out['ask_qty']) / (df_out['bid_qty'] + df_out['ask_qty'])\n    df_out['trade_imbalance'] = (df_out['buy_qty'] - df_out['sell_qty']) / (df_out['buy_qty'] + df_out['sell_qty'])\n    df_out['depth_pressure'] = (df_out['bid_qty'] - df_out['ask_qty']) / (df_out['buy_qty'] + df_out['sell_qty']) # New idea\n\n    # 3. Time-based Features\n    df_out['hour'] = df_out.index.hour\n    df_out['day_of_week'] = df_out.index.dayofweek # Monday=0, Sunday=6\n\n    # 4. Lagged Features (CRITICAL)\n    # Using a few lags for a simple start\n    lags = [1, 2, 5, 10] # 1, 2, 5, 10 minutes ago\n    features_to_lag = ['volume_log', 'order_book_imbalance', 'trade_imbalance', 'X1', 'X2']\n    for lag in lags:\n        for feat in features_to_lag:\n            df_out[f'{feat}_lag_{lag}'] = df_out[feat].shift(lag)\n\n    # 5. Rolling Window Features (CRITICAL for volatility)\n    windows = [5, 10, 30] # 5, 10, 30 minute windows\n    features_to_roll = ['volume_log', 'label', 'X1', 'X2'] # Rolling on 'label' is ok for PAST values\n    for window in windows:\n        for feat in features_to_roll:\n            # Shift by 1 to prevent using current value to predict itself, especially for label\n            df_out[f'{feat}_roll_std_{window}'] = df_out[feat].shift(1).rolling(window=window).std()\n            df_out[f'{feat}_roll_mean_{window}'] = df_out[feat].shift(1).rolling(window=window).mean()\n\n    # Clean up NaNs created by lagging/rolling\n    df_out = df_out.replace([np.inf, -np.inf], np.nan) # Replace infs created by division by zero\n    df_out = df_out.fillna(0) # Simple strategy: fill NaNs with 0. Forward fill is another option.\n\n    return df_out\n\n# Apply the function\ntrain_featured_df = feature_engineering(train_df)\n\nprint(\"Shape of original df:\", train_df.shape)\nprint(\"Shape of featured df:\", train_featured_df.shape)\nprint(\"\\nSome new features:\")\nprint(train_featured_df[['order_book_imbalance', 'volume_log', 'X1_lag_5', 'label_roll_std_10']].head(15))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-05T03:24:15.812188Z","iopub.execute_input":"2025-07-05T03:24:15.812432Z","iopub.status.idle":"2025-07-05T03:24:36.165217Z","shell.execute_reply.started":"2025-07-05T03:24:15.812410Z","shell.execute_reply":"2025-07-05T03:24:36.164240Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Step 2: Set Up and Train a Baseline Model (LightGBM)\nNow for the exciting part. We'll set up a strong baseline model using LightGBM, a fast and efficient gradient boosting framework that is excellent for this kind of tabular data.\n","metadata":{}},{"cell_type":"markdown","source":"1. Define Features and Target:\nFirst, we need to create our list of features (X) and our target variable (y).","metadata":{}},{"cell_type":"code","source":"# The target variable is the 'label' column\ny_v2 = train_featured_df['label']\n\n# The features are all columns EXCEPT the original label and any other columns we want to exclude\n# We should exclude the raw skewed features now that we have the log-transformed versions\nexcluded_features = ['label', 'bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']\nfeatures = [col for col in train_featured_df.columns if col not in excluded_features]\nX_v2 = train_featured_df[features]\n\nprint(f\"Number of features: {len(X_v2.columns)}\")\nprint(f\"Shape of X: {X_v2.shape}\")\nprint(f\"Shape of y: {y_v2.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-05T03:24:36.166126Z","iopub.execute_input":"2025-07-05T03:24:36.166372Z","iopub.status.idle":"2025-07-05T03:24:37.443084Z","shell.execute_reply.started":"2025-07-05T03:24:36.166352Z","shell.execute_reply":"2025-07-05T03:24:37.440839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import optuna\nimport lightgbm as lgb\nimport numpy as np","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-05T03:24:37.444318Z","iopub.execute_input":"2025-07-05T03:24:37.444629Z","iopub.status.idle":"2025-07-05T03:24:43.886122Z","shell.execute_reply.started":"2025-07-05T03:24:37.444584Z","shell.execute_reply":"2025-07-05T03:24:43.884799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def objective(trial):\n#     dtrain = lgb.Dataset(\n#         X_v2.values.astype('float32'),\n#         label=y_v2.values.astype('float32'),\n#         free_raw_data=False,\n#     )\n\n#     params = {\n#         \"objective\": \"regression\",\n#         \"metric\": \"rmse\",\n#         \"boosting_type\": \"gbdt\",\n#         \"learning_rate\": 0.05,\n#         \"verbosity\": -1,\n#         \"num_leaves\": trial.suggest_int(\"num_leaves\", 20, 80),\n#         \"lambda_l1\": trial.suggest_float(\"lambda_l1\", 1e-3, 10.0, log=True),\n#         \"lambda_l2\": trial.suggest_float(\"lambda_l2\", 1e-3, 10.0, log=True),\n#         \"feature_fraction\": trial.suggest_float(\"feature_fraction\", 0.6, 1.0),\n#         \"bagging_fraction\": trial.suggest_float(\"bagging_fraction\", 0.6, 1.0),\n#         \"bagging_freq\": trial.suggest_int(\"bagging_freq\", 1, 7),\n#         \"min_child_samples\": trial.suggest_int(\"min_child_samples\", 5, 100),\n#         \"num_threads\": 1,\n#     }\n\n#     cv_results = lgb.cv(\n#         params,\n#         dtrain,\n#         nfold=3,                # or 5 if your memory allows\n#         num_boost_round=300,    # or your chosen budget\n#         shuffle=False,\n#         stratified=False,\n#         seed=None,\n#         callbacks=[\n#             lgb.early_stopping(stopping_rounds=50),\n#             lgb.log_evaluation(period=0),\n#         ],\n#     )\n\n#     # pick up the right '-mean' key no matter what prefix\n#     mean_key = next(key for key in cv_results if key.endswith(\"-mean\"))\n#     result = min(cv_results[mean_key])\n\n#     # clean up\n#     del dtrain, cv_results\n#     import gc; gc.collect()\n\n#     return result\n\n\n\n# # 3️⃣ Create Optuna study\n# study = optuna.create_study(direction=\"minimize\", pruner=optuna.pruners.MedianPruner())\n\n# print(\"--- Starting Hyperparameter Tuning ---\")\n# study.optimize(objective, n_trials=30)\n# print(\"--- Tuning Complete ---\\n\")\n\n# # 4️⃣ Print results (just like before)\n# print(\"--- Tuning Results ---\")\n# print(f\"Number of finished trials: {len(study.trials)}\")\n# print(f\"Best trial's Average CV RMSE: {study.best_value:.5f}\\n\")\n\n# print(\"Best trial's parameters:\")\n# for key, value in study.best_params.items():\n#     print(f\"    {key}: {value}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-05T03:24:43.887436Z","iopub.execute_input":"2025-07-05T03:24:43.888301Z","iopub.status.idle":"2025-07-05T03:24:43.894497Z","shell.execute_reply.started":"2025-07-05T03:24:43.888267Z","shell.execute_reply":"2025-07-05T03:24:43.893585Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" study = optuna.create_study(direction=\"minimize\", pruner=optuna.pruners.MedianPruner())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-05T03:24:43.895798Z","iopub.execute_input":"2025-07-05T03:24:43.896154Z","iopub.status.idle":"2025-07-05T03:24:43.928635Z","shell.execute_reply.started":"2025-07-05T03:24:43.896125Z","shell.execute_reply":"2025-07-05T03:24:43.926983Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_params_from_cv = {\n    'num_leaves': 45,\n    'lambda_l1': 8.787667104782061,\n    'lambda_l2': 0.005771739600459773,\n    'feature_fraction': 0.8918692418029897,\n    'bagging_fraction': 0.8283591790068191,\n    'bagging_freq': 1,\n    'min_child_samples': 5,\n}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-05T03:24:43.930584Z","iopub.execute_input":"2025-07-05T03:24:43.930923Z","iopub.status.idle":"2025-07-05T03:24:43.951380Z","shell.execute_reply.started":"2025-07-05T03:24:43.930894Z","shell.execute_reply":"2025-07-05T03:24:43.949933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import lightgbm as lgb\nimport numpy as np\n\n\nprint(\"--- Training Final Model with OPTUNA‑TUNED Hyperparameters ---\")\n\n# 2️⃣ Base params that you always want\nfinal_params = {\n    'objective': 'rmse',\n    'metric': 'rmse',\n    'boosting_type': 'gbdt',\n    'learning_rate': 0.05,\n    'verbose': -1,\n    'n_jobs': -1,         # uses all CPU cores\n}\n\n# 3️⃣ Merge in your tuned hyperparameters\nfinal_params.update(best_params_from_cv)\n\n# 4️⃣ Decide on how many trees to grow\n#    - You tuned up to 2000 in CV, but you can train fewer now.\n#    - If you want to leverage early‑stopping, you could:\n#         n_estimators=2000, callbacks=[lgb.early_stopping(…)]\n#      But here we’ll go with a fixed budget:\nfinal_params['n_estimators'] = 500\n\n# 5️⃣ Instantiate and fit on ALL of your data\nfinal_model = lgb.LGBMRegressor(**final_params)\nfinal_model.fit(X_v2, y_v2)\n\nprint(\"\\n--- Final Model is Trained and Ready for Submission ---\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-05T03:24:43.953011Z","iopub.execute_input":"2025-07-05T03:24:43.953376Z","iopub.status.idle":"2025-07-05T03:31:38.969831Z","shell.execute_reply.started":"2025-07-05T03:24:43.953346Z","shell.execute_reply":"2025-07-05T03:31:38.967423Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_model.booster_.save_model('lgbm_final_model.txt')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-05T03:31:38.976076Z","iopub.execute_input":"2025-07-05T03:31:38.976574Z","iopub.status.idle":"2025-07-05T03:31:39.013883Z","shell.execute_reply.started":"2025-07-05T03:31:38.976521Z","shell.execute_reply":"2025-07-05T03:31:39.013048Z"}},"outputs":[],"execution_count":null}]}