{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":247603950,"sourceType":"kernelVersion"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport lightgbm as lgb\nimport gc\n\n# --- 1. Load the Pre-trained Model ---\n# IMPORTANT: Before running this notebook, you must add the output of your training notebook as a data source.\n# 1. Go to your notebook viewer for THIS submission notebook.\n# 2. Click \"+ Add data\" in the top-right.\n# 3. Select the \"Notebook Output Files\" tab.\n# 4. Find your training notebook and click \"Add\".\n# The model will then be available at a path like the one below.\n# You MUST change 'your-training-notebook-name' to the actual URL name of your training notebook.\nmodel_path = '/kaggle/input/notebook5f77c8150b/lgbm_final_model.txt'  # <-- CHANGE THIS PATH\nmodel = lgb.Booster(model_file=model_path)\n\n# --- 2. Re-create the EXACT SAME Feature Engineering Function ---\n# This must be identical to the one used for training.\ndef enhanced_feature_engineering_mem_safe(df):\n    \"\"\"V2.1 of feature engineering, slightly trimmed for memory.\"\"\"\n    df_out = df.copy()\n    # Log transform\n    skewed_features = ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']\n    for col in skewed_features:\n        df_out[f'{col}_log'] = np.log1p(df_out[col])\n    # Ratios\n    df_out['order_book_imbalance'] = (df_out['bid_qty'] - df_out['ask_qty']) / (df_out['bid_qty'] + df_out['ask_qty'])\n    df_out['trade_imbalance'] = (df_out['buy_qty'] - df_out['sell_qty']) / (df_out['buy_qty'] + df_out['sell_qty'])\n    # Expanded Rolling Windows\n    windows = [5, 10, 30, 60]\n    features_to_roll = ['label', 'X719']\n    for window in windows:\n        for feat in features_to_roll:\n            shifted_feat = df_out[feat].shift(1)\n            df_out[f'{feat}_roll_std_{window}'] = shifted_feat.rolling(window=window).std()\n            df_out[f'{feat}_roll_mean_{window}'] = shifted_feat.rolling(window=window).mean()\n    # Lag the most important raw X features\n    important_x_features = ['X719', 'X235', 'X718']\n    for lag in [1, 2, 5]:\n        for feat in important_x_features:\n            df_out[f'{feat}_lag_{lag}'] = df_out[feat].shift(lag)\n    # Momentum Features\n    momentum_features = ['label', 'X719', 'trade_imbalance']\n    for lag in [1, 5]:\n        for feat in momentum_features:\n            df_out[f'{feat}_mom_{lag}'] = df_out[feat] - df_out[feat].shift(lag)\n    # The label in the test set is always 0, so rolling features on it will be 0, which is fine.\n    df_out = df_out.replace([np.inf, -np.inf], np.nan)\n    return df_out\n\n# --- 3. Load and Process Test Data ---\n# Load the test data from the Parquet file\ntest_df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet', engine='pyarrow')\n\n# Apply feature engineering\ndf_proc = enhanced_feature_engineering_mem_safe(test_df)\n\n# Get the feature names from the trained model\nfeatures = model.feature_name()\n\n# Select only the features the model was trained on\nX_test = df_proc[features]\n\n# Fill any NaNs that might still exist (e.g., from lags or rolling features)\nX_test = X_test.fillna(0)\n\n# --- 4. Make Predictions ---\npredictions = model.predict(X_test)\n\n# --- 5. Create Submission File ---\n# Load the sample submission file\nsample_submission = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\n\n# Assign predictions to the submission file\n# Assuming the sample submission expects a 'prediction' column (adjust if the column name is 'label')\nsample_submission['prediction'] = predictions\n\n# Save the submission file\nsample_submission.to_csv('submission.csv', index=False)\n\n# --- 6. Clean Up Memory ---\ndel df_proc, X_test, predictions\ngc.collect()\n\nprint(\"Submission complete.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T11:40:25.117865Z","iopub.execute_input":"2025-06-30T11:40:25.118178Z","iopub.status.idle":"2025-06-30T11:41:41.506194Z","shell.execute_reply.started":"2025-06-30T11:40:25.118155Z","shell.execute_reply":"2025-06-30T11:41:41.505203Z"}},"outputs":[],"execution_count":null}]}