{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# Garbage Collector\nimport gc \n\nimport pandas as pd\nimport numpy as np\nimport os\n\n# Time Modules\nimport calendar\nimport time\nimport datetime\nfrom datetime import datetime, timedelta\n\npd.set_option('display.max_rows', None)\npd.set_option('display.max_columns', None)\n\n\n# Plots\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom matplotlib import cm\nimport plotly.graph_objects as go\nimport plotly.express as px\nimport plotly.subplots as sp\nsns.set_style(\"whitegrid\")\nsns.set(rc={'figure.figsize':(18, 12)})\n%matplotlib inline\n\n# Statistics \nfrom scipy.stats import norm\nfrom scipy.stats import zscore\nfrom scipy import stats\n\nimport warnings\nwarnings.filterwarnings('ignore')\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-24T14:33:52.481792Z","iopub.execute_input":"2025-07-24T14:33:52.482085Z","iopub.status.idle":"2025-07-24T14:33:52.496644Z","shell.execute_reply.started":"2025-07-24T14:33:52.482044Z","shell.execute_reply":"2025-07-24T14:33:52.495868Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **Model.**","metadata":{}},{"cell_type":"code","source":"import lightgbm as lgb\nfrom xgboost import XGBRegressor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T14:33:52.497929Z","iopub.execute_input":"2025-07-24T14:33:52.498154Z","iopub.status.idle":"2025-07-24T14:33:52.509379Z","shell.execute_reply.started":"2025-07-24T14:33:52.498138Z","shell.execute_reply":"2025-07-24T14:33:52.508729Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import optuna\noptuna.logging.set_verbosity(optuna.logging.CRITICAL)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T14:33:52.510116Z","iopub.execute_input":"2025-07-24T14:33:52.510353Z","iopub.status.idle":"2025-07-24T14:33:52.522547Z","shell.execute_reply.started":"2025-07-24T14:33:52.510337Z","shell.execute_reply":"2025-07-24T14:33:52.521781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"##################################################################\n# Installing GPU driver for LightGBM:-\n!mkdir -p /etc/OpenCL/vendors && echo \"libnvidia-opencl.so.1\" > /etc/OpenCL/vendors/nvidia.icd\n!sudo apt install nvidia-driver-460 nvidia-cuda-toolkit clinfo\n!apt-get update --fix-missing\n!pip install -q  lightgbm==4.1.0 \\\n  --config-settings=cmake.define.USE_GPU=ON \\\n  --config-settings=cmake.define.OpenCL_INCLUDE_DIR=\"/usr/local/cuda/include/\" \\\n  --config-settings=cmake.define.OpenCL_LIBRARY=\"/usr/local/cuda/lib64/libOpenCL.so\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T14:33:52.523248Z","iopub.execute_input":"2025-07-24T14:33:52.523480Z","iopub.status.idle":"2025-07-24T14:34:02.216783Z","shell.execute_reply.started":"2025-07-24T14:33:52.523459Z","shell.execute_reply":"2025-07-24T14:34:02.215773Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Metric.**","metadata":{}},{"cell_type":"code","source":"from scipy.stats import pearsonr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T14:34:02.219002Z","iopub.execute_input":"2025-07-24T14:34:02.219264Z","iopub.status.idle":"2025-07-24T14:34:02.223317Z","shell.execute_reply.started":"2025-07-24T14:34:02.219240Z","shell.execute_reply":"2025-07-24T14:34:02.222590Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **Interpretation.**\n* r=1: Perfect positive linear correlation\n\n* r=−1: Perfect negative linear correlation\n\n* r=0: No linear correlation\n\n* Closer to ±1: Stronger linear relationship\n\n### **When to Use Pearson Correlation**.\n\n* Both variables are continuous.\n\n* The relationship is approximately linear.\n\n* Data is normally distributed (ideally).\n\n### **How to Use as a Metric.**\n\n* Measure similarity between predicted and actual values in regression tasks.\n\n* Check relationships between features in data exploration.\n\n* Feature selection: Remove highly correlated features (to avoid multicollinearity).\n\n* Validate models: High correlation between predictions and actual targets indicates good performance (when paired with other metrics).","metadata":{}},{"cell_type":"code","source":"main_columns = ['bid_qty','ask_qty','buy_qty','sell_qty','volume','label']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T14:34:02.224190Z","iopub.execute_input":"2025-07-24T14:34:02.224398Z","iopub.status.idle":"2025-07-24T14:34:02.240595Z","shell.execute_reply.started":"2025-07-24T14:34:02.224375Z","shell.execute_reply":"2025-07-24T14:34:02.239891Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"main_features = [\n#  'X420','X726','X724','X124','X338','X137','X337','X334','X175',\n# 'X184','X721','X86','X39','X580','X371','X415','X325','X125',\n# 'X4','X531','X149','X120','X662','X426','X219','X728','X281','X279',\n# 'X245','X230','X271','X674','X256','X244','X247','X232','X210','X218',\n# 'X729','X119','X123','X136','X156','X161','X162','X167',\n# 'X169','X174','X188','X195','X207','X286','X213','X215','X730','X284',\n# 'X651','X574','X398','X404','X565','X561','X560','X414','X421', 'X430',\n# 'X443','X445','X530','X457','X524','X464','X465','X473','X509','X505',\n 'X492','X396','X391','X649','X386','X301','X303','X637','X619','X112',\n'bid_qty','ask_qty','buy_qty','sell_qty','volume','label'\n]\nlen(sorted(main_features))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T14:34:02.254520Z","iopub.execute_input":"2025-07-24T14:34:02.254717Z","iopub.status.idle":"2025-07-24T14:34:02.268672Z","shell.execute_reply.started":"2025-07-24T14:34:02.254694Z","shell.execute_reply":"2025-07-24T14:34:02.267961Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Target.**","metadata":{}},{"cell_type":"code","source":"y = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet').label.values\ny[0:10]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T14:34:02.269525Z","iopub.execute_input":"2025-07-24T14:34:02.269737Z","iopub.status.idle":"2025-07-24T14:34:07.408904Z","shell.execute_reply.started":"2025-07-24T14:34:02.269721Z","shell.execute_reply":"2025-07-24T14:34:07.408267Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nclass Processor:\n    def __init__(self, train_path, test_path, main_features):\n        self.train = pd.read_parquet(train_path, columns=main_features)\n        self.test = pd.read_parquet(test_path, columns=main_features)\n        self.indexes = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv').index\n        self.target = np.array(self.train.label)\n\n    def reduce_mem_usage(self, dataframe, dataset_name=\"\"):\n        \"\"\"\n        Function taken from: https://www.kaggle.com/code/ravaghi/drw-crypto-market-prediction-ensemble\n        Reduces memory usage of a DataFrame by downcasting numeric types.\n        \"\"\"\n        print('Reducing memory usage for:', dataset_name)\n        initial_mem_usage = dataframe.memory_usage().sum() / 1024**2\n\n        for col in dataframe.columns:\n            col_type = dataframe[col].dtype\n\n            c_min = dataframe[col].min()\n            c_max = dataframe[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    dataframe[col] = dataframe[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    dataframe[col] = dataframe[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    dataframe[col] = dataframe[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    dataframe[col] = dataframe[col].astype(np.int64)\n            elif str(col_type)[:5] == 'float':\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    dataframe[col] = dataframe[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    dataframe[col] = dataframe[col].astype(np.float32)\n                else:\n                    dataframe[col] = dataframe[col].astype(np.float64)\n\n        final_mem_usage = dataframe.memory_usage().sum() / 1024**2\n        print('--- Memory usage before: {:.2f} MB'.format(initial_mem_usage))\n        print('--- Memory usage after: {:.2f} MB'.format(final_mem_usage))\n        print('--- Decreased memory usage by {:.1f}%\\n'.format(100 * (initial_mem_usage - final_mem_usage) / initial_mem_usage))\n\n\n            \n        return dataframe","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T14:34:07.409740Z","iopub.execute_input":"2025-07-24T14:34:07.409999Z","iopub.status.idle":"2025-07-24T14:34:07.419591Z","shell.execute_reply.started":"2025-07-24T14:34:07.409969Z","shell.execute_reply":"2025-07-24T14:34:07.418957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"processor = Processor(\n    train_path='/kaggle/input/drw-crypto-market-prediction/train.parquet',\n    test_path='/kaggle/input/drw-crypto-market-prediction/test.parquet',\n    main_features = main_features\n)\n\n\nprocessor.train = processor.reduce_mem_usage(processor.train, \"Train Dataset\")\nprocessor.test = processor.reduce_mem_usage(processor.test, \"Test Dataset\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T14:34:07.420459Z","iopub.execute_input":"2025-07-24T14:34:07.420709Z","iopub.status.idle":"2025-07-24T14:34:08.133260Z","shell.execute_reply.started":"2025-07-24T14:34:07.420685Z","shell.execute_reply":"2025-07-24T14:34:08.132498Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.decomposition import IncrementalPCA\nfrom sklearn.preprocessing import StandardScaler\n\nclass FeatureEngineering:\n    def __init__(self, main_columns=None, n_components=50, batch_size=100):\n        self.main_columns = main_columns\n        self.n_components = n_components\n        self.batch_size = batch_size\n        self.ipca = IncrementalPCA(n_components=n_components, batch_size=batch_size)\n        self.scaler = StandardScaler()\n        self.fitted = False\n        self.scaler_fitted = False\n\n    def transform_columns(self, df):\n        df = df.copy()\n\n        for col in ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']:\n            df[col] = pd.to_numeric(df[col], errors='coerce')\n\n        df.replace(0, np.nan, inplace=True)\n\n        # Binary indicators\n        df['bid_ask_binary'] = np.where(df['bid_qty'] > df['ask_qty'], 1, 0)\n        df['buy_sell_binary'] = np.where(df['buy_qty'] > df['sell_qty'], 1, 0)\n\n        df['bid_ask_interaction'] = df['bid_qty'] * df['ask_qty']\n        df['bid_buy_interaction'] = df['bid_qty'] * df['buy_qty']\n        df['bid_sell_interaction'] = df['bid_qty'] * df['sell_qty']\n        df['ask_buy_interaction'] = df['ask_qty'] * df['buy_qty']\n        df['ask_sell_interaction'] = df['ask_qty'] * df['sell_qty']\n        df['buy_sell_interaction'] = df['buy_qty'] * df['sell_qty']\n\n        df['spread_indicator'] = (df['ask_qty'] - df['bid_qty']) / (df['ask_qty'] + df['bid_qty'] + 1e-8)\n\n        df['volume_weighted_buy'] = df['buy_qty'] * df['volume']\n        df['volume_weighted_sell'] = df['sell_qty'] * df['volume']\n        df['volume_weighted_bid'] = df['bid_qty'] * df['volume']\n        df['volume_weighted_ask'] = df['ask_qty'] * df['volume']\n\n        df['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-8)\n        df['bid_ask_ratio'] = df['bid_qty'] / (df['ask_qty'] + 1e-8)\n\n        df['order_flow_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-8)\n\n        df['buying_pressure'] = df['buy_qty'] / (df['volume'] + 1e-8)\n        df['selling_pressure'] = df['sell_qty'] / (df['volume'] + 1e-8)\n\n        df['total_liquidity'] = df['bid_qty'] + df['ask_qty']\n        df['liquidity_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['total_liquidity'] + 1e-8)\n        df['relative_spread'] = (df['ask_qty'] - df['bid_qty']) / (df['volume'] + 1e-8)\n\n        df['trade_intensity'] = (df['buy_qty'] + df['sell_qty']) / (df['volume'] + 1e-8)\n        df['avg_trade_size'] = df['volume'] / (df['buy_qty'] + df['sell_qty'] + 1e-8)\n        df['net_trade_flow'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-8)\n\n        df['depth_ratio'] = df['total_liquidity'] / (df['volume'] + 1e-8)\n        df['volume_participation'] = (df['buy_qty'] + df['sell_qty']) / (df['total_liquidity'] + 1e-8)\n        df['market_activity'] = df['volume'] * df['total_liquidity']\n\n        df['effective_spread_proxy'] = np.abs(df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-8)\n        df['realized_volatility_proxy'] = np.abs(df['order_flow_imbalance']) * df['volume']\n\n        df['normalized_buy_volume'] = df['buy_qty'] / (df['bid_qty'] + 1e-8)\n        df['normalized_sell_volume'] = df['sell_qty'] / (df['ask_qty'] + 1e-8)\n\n        df['liquidity_adjusted_imbalance'] = df['order_flow_imbalance'] * df['depth_ratio']\n        df['pressure_spread_interaction'] = df['buying_pressure'] * df['spread_indicator']\n\n        df.replace([np.inf, -np.inf], 0, inplace=True)\n        df.fillna(0, inplace=True)\n\n        return df\n\n    def drop_columns(self, df):\n        df = df.copy()\n        if self.main_columns is not None:\n            missing_cols = [col for col in self.main_columns if col not in df.columns]\n            if missing_cols:\n                print(f\"Warning: columns {missing_cols} not found in DataFrame.\")\n            df = df.drop(columns=[col for col in self.main_columns if col in df.columns])\n        return df\n\n    def generate_lag_features(self, df, cols, lags=[1,3,5,7,10,20,60,120,180,240,60*24,60*24*2,60*24*3,60*24*4,60*24*5], windows=[60*7, 60*14], dropna=True):\n        df = df.copy()\n\n        for col in cols:\n            for lag in lags:\n                df[f'{col}_lag_{lag}'] = df[col].shift(lag)\n\n            for window in windows:\n                df[f'{col}_roll_mean_{window}'] = df[col].shift(window).ewm(halflife=window).mean()\n                df[f'{col}_roll_std_{window}'] = df[col].shift(window).ewm(halflife=window).std()\n\n        if dropna:\n            df.dropna(inplace=True)\n        else:\n            df.replace([np.inf, -np.inf], 0, inplace=True)\n            df.fillna(0, inplace=True)\n\n        return df\n\n    def fit_scaler(self, df):\n        df = df.copy()\n        self.scaler.fit(df)\n        self.scaler_fitted = True\n\n    def transform_scaler(self, df):\n        if not self.scaler_fitted:\n            raise RuntimeError(\"Scaler is not fitted. Call fit_scaler() on training data first.\")\n        df_scaled = self.scaler.transform(df)\n        return pd.DataFrame(df_scaled, columns=df.columns, index=df.index)\n\n    def fit_ipca(self, df):\n        df = df.copy()\n        self.ipca.fit(df)\n        self.fitted = True\n\n    def transform_ipca(self, df):\n        if not self.fitted:\n            raise RuntimeError(\"IPCA model is not fitted. Call fit_ipca() on training data first.\")\n        df_ipca = self.ipca.transform(df)\n        df_ipca = pd.DataFrame(df_ipca, columns=[f'ipca_{i}' for i in range(self.n_components)], index=df.index)\n        return df_ipca\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T14:34:08.134539Z","iopub.execute_input":"2025-07-24T14:34:08.134835Z","iopub.status.idle":"2025-07-24T14:34:08.152473Z","shell.execute_reply.started":"2025-07-24T14:34:08.134809Z","shell.execute_reply":"2025-07-24T14:34:08.151907Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define main columns to remove later (raw input features)\nmain_columns = ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume', 'label']\n\nfe = FeatureEngineering()\n\ntrain = fe.transform_columns(processor.train)\ntest = fe.transform_columns(processor.test)\n\n# Add lag features to e.g. 'volume'\ntrain = fe.generate_lag_features(train, cols=main_columns, lags=[60*7, 60*14, 60*28], windows=[60*7, 60*14], dropna=False)\ntest = fe.generate_lag_features(test, cols=main_columns, lags=[60*7, 60*14, 60*28], windows=[60*7, 60*14], dropna=False)\n\n\nfe.fit_scaler(train)\nfe.fit_scaler(test)\n\ntrain = fe.transform_scaler(train)\ntest = fe.transform_scaler(test)\n\nfe.fit_ipca(train)\nfe.fit_ipca(train)\n\ntrain = fe.transform_ipca(train)\ntest = fe.transform_ipca(test)\n\n\ntrain = fe.drop_columns(train)\ntest = fe.drop_columns(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T14:34:08.153201Z","iopub.execute_input":"2025-07-24T14:34:08.153441Z","iopub.status.idle":"2025-07-24T14:34:59.510949Z","shell.execute_reply.started":"2025-07-24T14:34:08.153415Z","shell.execute_reply":"2025-07-24T14:34:59.510193Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T14:34:59.513652Z","iopub.execute_input":"2025-07-24T14:34:59.514426Z","iopub.status.idle":"2025-07-24T14:34:59.518652Z","shell.execute_reply.started":"2025-07-24T14:34:59.514405Z","shell.execute_reply":"2025-07-24T14:34:59.518121Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T14:34:59.519596Z","iopub.execute_input":"2025-07-24T14:34:59.519818Z","iopub.status.idle":"2025-07-24T14:34:59.561246Z","shell.execute_reply.started":"2025-07-24T14:34:59.519793Z","shell.execute_reply":"2025-07-24T14:34:59.560335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T14:34:59.561992Z","iopub.execute_input":"2025-07-24T14:34:59.562304Z","iopub.status.idle":"2025-07-24T14:34:59.566777Z","shell.execute_reply.started":"2025-07-24T14:34:59.562286Z","shell.execute_reply":"2025-07-24T14:34:59.566243Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Submission.**","metadata":{}},{"cell_type":"code","source":"xgb_params = {\n    \"tree_method\": \"hist\",\n    \"device\": \"gpu\",\n    'metric': ['l1', 'l2'],\n    \"colsample_bylevel\": 0.4778,\n    \"colsample_bynode\": 0.3628,\n    \"colsample_bytree\": 0.7107,\n    \"gamma\": 1.7095,\n    \"learning_rate\": 0.02213,\n    \"max_depth\": 20,\n    \"max_leaves\": 12,\n    \"min_child_weight\": 16,\n    \"n_estimators\": 1667,\n    \"subsample\": 0.06567,\n    \"reg_alpha\": 39.3524,\n    \"reg_lambda\": 75.4484,\n    \"random_state\": 700,\n    \"verbose\": False,\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T14:34:59.567474Z","iopub.execute_input":"2025-07-24T14:34:59.567667Z","iopub.status.idle":"2025-07-24T14:34:59.581492Z","shell.execute_reply.started":"2025-07-24T14:34:59.567645Z","shell.execute_reply":"2025-07-24T14:34:59.580884Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"params_gdbt = {\n    'boosting_type': 'gbdt',\n    'objective': 'regression',\n    'metric': ['l1', 'l2'],\n    'seed':42,\n    'device': 'gpu',\n    'learning_rate': 0.06459880558047476,\n    'n_estimators': 1196,\n    'max_depth': 8, \n    'num_leaves': 483,\n    'min_child_samples': 172, \n    'subsample': 0.10298004227879802, \n    'colsample_bytree': 0.9034687230448682, \n    'reg_alpha': 0.7684295974829274, \n    'reg_lambda': 0.49761953142451365,\n    'verbose':-1,\n    'verbosity':0,\n    'predict_disable_shape_check':True,\n\n\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T14:34:59.582226Z","iopub.execute_input":"2025-07-24T14:34:59.582472Z","iopub.status.idle":"2025-07-24T14:34:59.598113Z","shell.execute_reply.started":"2025-07-24T14:34:59.582452Z","shell.execute_reply":"2025-07-24T14:34:59.597475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"params_goss = {\n    'boosting_type': 'goss',\n    'objective': 'regression',\n    'metric': ['l1', 'l2'],\n    'seed':42,\n    'device': 'gpu',\n    'learning_rate': 0.06459880558047476,\n    'n_estimators': 1196,\n    'max_depth': 8, \n    'num_leaves': 483,\n    'min_child_samples': 172, \n    'subsample': 0.10298004227879802, \n    'colsample_bytree': 0.9034687230448682, \n    'reg_alpha': 0.7684295974829274, \n    'reg_lambda': 0.49761953142451365,\n    'verbose':-1,\n    'verbosity':0,\n    'predict_disable_shape_check':True,\n\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T14:34:59.598861Z","iopub.execute_input":"2025-07-24T14:34:59.599188Z","iopub.status.idle":"2025-07-24T14:34:59.614219Z","shell.execute_reply.started":"2025-07-24T14:34:59.599169Z","shell.execute_reply":"2025-07-24T14:34:59.613405Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **GDBT/Goss/XGB.**","metadata":{}},{"cell_type":"code","source":"goss = lgb.LGBMRegressor(**params_goss)\ngbdt = lgb.LGBMRegressor(**params_gdbt)\nmodel_xgb = XGBRegressor(**xgb_params)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T14:34:59.615014Z","iopub.execute_input":"2025-07-24T14:34:59.615289Z","iopub.status.idle":"2025-07-24T14:34:59.635146Z","shell.execute_reply.started":"2025-07-24T14:34:59.615266Z","shell.execute_reply":"2025-07-24T14:34:59.634500Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport time\nfrom sklearn.model_selection import KFold, TimeSeriesSplit, GroupKFold\nimport lightgbm as lgb\nfrom xgboost import XGBClassifier\n\n# Parameters\nn_splits = 10\ngkf = GroupKFold(n_splits=n_splits)\ngroups = np.array(train.ipca_5)\n\n# Initialize predictions\noof_gbdt = np.zeros(len(train))\noof_goss = np.zeros(len(train))\noof_xgb  = np.zeros(len(train))\n\ntest_preds_gbdt = np.zeros(len(test))\ntest_preds_goss = np.zeros(len(test))\ntest_preds_xgb  = np.zeros(len(test))\n\n# CV loop\nfor fold, (train_idx, val_idx) in enumerate(gkf.split(train, y, groups)):\n    print(f\"\\n🌀 Fold {fold + 1}/{n_splits}\")\n\n    X_train_fold, X_val_fold = train.iloc[train_idx], train.iloc[val_idx]\n    y_train_fold, y_val_fold = y[train_idx], y[val_idx]\n\n    ### LightGBM GBDT\n    print(\"Training LightGBM (GBDT)...\")\n    gbdt.fit(\n        X_train_fold, y_train_fold,\n        eval_set=[(X_val_fold, y_val_fold)],\n        callbacks=[\n            lgb.early_stopping(stopping_rounds=100),\n            lgb.log_evaluation(period=10)\n        ]\n    )\n    oof_gbdt[val_idx] = gbdt.predict(X_val_fold)\n    test_preds_gbdt += gbdt.predict(test) / n_splits\n\n    ### LightGBM GOSS\n    print(\"Training LightGBM (GOSS)...\")\n    goss.fit(\n        X_train_fold, y_train_fold,\n        eval_set=[(X_val_fold, y_val_fold)],\n        callbacks=[\n            lgb.early_stopping(stopping_rounds=100),\n            lgb.log_evaluation(period=10)\n        ]\n    )\n    oof_goss[val_idx] = goss.predict(X_val_fold)\n    test_preds_goss += goss.predict(test) / n_splits\n\n    ### XGBoost\n    print(\"Training XGBoost...\")\n    model_xgb.fit(\n        X_train_fold, y_train_fold,\n        eval_set=[(X_val_fold, y_val_fold)],\n        early_stopping_rounds=100,\n        verbose=False\n    )\n    oof_xgb[val_idx] = model_xgb.predict(X_val_fold)\n    test_preds_xgb += model_xgb.predict(test) / n_splits\n\n# Create submission files\nfor name, preds in zip(\n    [\"gbdt\", \"goss\", \"xgb\"],\n    [test_preds_gbdt, test_preds_goss, test_preds_xgb]\n):\n    submission = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\n    submission[\"prediction\"] = preds\n    submission.to_csv(f\"submission_{name}.csv\", index=False)\n    print(f\"✅ Saved: submission_{name}.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T14:35:40.684699Z","iopub.execute_input":"2025-07-24T14:35:40.685216Z","execution_failed":"2025-07-24T14:37:17.774Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# total_weight = 1 / avg_pc_gbdt + 1 / avg_pc_goss + 1 / avg_pc_xgb\n# weight_gbdt = (1 / avg_pc_gbdt) / total_weight\n# weight_goss = (1 / avg_pc_goss) / total_weight\n# weight_xgb = (1 / avg_pc_xgb) / total_weight","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T14:34:59.685690Z","iopub.status.idle":"2025-07-24T14:34:59.685901Z","shell.execute_reply.started":"2025-07-24T14:34:59.685799Z","shell.execute_reply":"2025-07-24T14:34:59.685809Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# final_test_predictions = (\n#     weight_gbdt * test_predictions_gbdt.mean(axis=1) +\n#     weight_goss * test_predictions_goss.mean(axis=1) + \n#     weight_xgb * test_predictions_xgb.mean(axis=1)\n# )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T14:34:59.686668Z","iopub.status.idle":"2025-07-24T14:34:59.686876Z","shell.execute_reply.started":"2025-07-24T14:34:59.686778Z","shell.execute_reply":"2025-07-24T14:34:59.686787Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **XGB & LGB & Cat Models.**","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import TimeSeriesSplit, cross_val_predict\nfrom sklearn.linear_model import RidgeCV\nfrom sklearn.metrics import mean_squared_error\nfrom scipy.stats import pearsonr\nimport lightgbm as lgb\nimport xgboost as xgb\nimport catboost as cb\n\n# Parameters\nn_splits = 10\n\n# Placeholders for OOF and test predictions\noof_preds = {\n    \"gbdt\": np.zeros(len(train)),\n    \"goss\": np.zeros(len(train)),\n    \"xgb\": np.zeros(len(train)),\n    \"cat\": np.zeros(len(train)),\n}\ntest_preds = {\n    \"gbdt\": np.zeros((len(test), n_splits)),\n    \"goss\": np.zeros((len(test), n_splits)),\n    \"xgb\": np.zeros((len(test), n_splits)),\n    \"cat\": np.zeros((len(test), n_splits)),\n}\nfold_pc = {k: [] for k in oof_preds}\n\n# Model definitions\nmodels = {\n    \"gbdt\": lgb.LGBMRegressor(**params_gdbt),\n    \"goss\": lgb.LGBMRegressor(**params_goss),\n    \"xgb\": xgb.XGBRegressor(**xgb_params),\n    \"cat\": cb.CatBoostRegressor(iterations=1000, learning_rate=0.05, verbose=0)\n}\n\n# Time series CV training\nfor fold, (train_idx, val_idx) in enumerate(gkf.split(train, y, groups)):\n    print(f\"\\nFold {fold + 1}/{n_splits}\")\n    \n    X_train_fold = train.iloc[train_idx]\n    X_val_fold = train.iloc[val_idx]\n    y_train_fold = y.iloc[train_idx] if isinstance(y, pd.Series) else y[train_idx]\n    y_val_fold = y.iloc[val_idx] if isinstance(y, pd.Series) else y[val_idx]\n\n    for name, model in models.items():\n        print(f\" Training {name.upper()}...\")\n\n        model.fit(\n            X_train_fold, y_train_fold,\n            eval_set=[(X_val_fold, y_val_fold)],\n            callbacks=[\n                lgb.early_stopping(stopping_rounds=100),\n                lgb.log_evaluation(period=10)\n            ] if 'lgb' in str(type(model)) else None\n        )\n\n        # OOF and test predictions\n        oof_preds[name][val_idx] = model.predict(X_val_fold)\n        test_preds[name][:, fold] = model.predict(test)\n        \n        # Pearson correlation\n        pc = pearsonr(y_val_fold, oof_preds[name][val_idx])[0]\n        fold_pc[name].append(pc)\n        print(f\"  → Fold Pearson: {pc:.4f}\")\n\n# Show average PC\nfor name in oof_preds:\n    avg_pc = np.mean(fold_pc[name])\n    print(f\"\\n{name.upper()} Average Pearson Correlation: {avg_pc:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T14:34:59.688452Z","iopub.status.idle":"2025-07-24T14:34:59.688725Z","shell.execute_reply.started":"2025-07-24T14:34:59.688575Z","shell.execute_reply":"2025-07-24T14:34:59.688585Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Combine OOF predictions into stacking features\nX_stack_train = np.column_stack([oof_preds[k] for k in oof_preds])\nX_stack_test = np.column_stack([test_preds[k].mean(axis=1) for k in test_preds])\n\n# Meta-model: RidgeCV with CV internally\nprint(\"\\nTraining Meta-Model (RidgeCV)...\")\nmeta_model = RidgeCV(alphas=np.logspace(-3, 3, 50), cv=5)\nmeta_model.fit(X_stack_train, y)\n\n# Final stacked predictions\nfinal_oof = meta_model.predict(X_stack_train)\nfinal_test = meta_model.predict(X_stack_test)\n\n# Evaluation\nfinal_pc = pearsonr(y, final_oof)[0]\n\nprint(f\"\\n=== FINAL STACKED MODEL ===\")\nprint(f\"Stacked Model Pearson Correlation: {final_pc:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T14:34:59.689911Z","iopub.status.idle":"2025-07-24T14:34:59.690157Z","shell.execute_reply.started":"2025-07-24T14:34:59.690017Z","shell.execute_reply":"2025-07-24T14:34:59.690025Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\nsubmission[\"prediction\"] = final_test\nsubmission.to_csv(\"submission_meta_model.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T14:34:59.691181Z","iopub.status.idle":"2025-07-24T14:34:59.691414Z","shell.execute_reply.started":"2025-07-24T14:34:59.691297Z","shell.execute_reply":"2025-07-24T14:34:59.691308Z"}},"outputs":[],"execution_count":null}]}