{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"},{"sourceId":249869065,"sourceType":"kernelVersion"},{"sourceId":250989304,"sourceType":"kernelVersion"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.linear_model import Ridge\nimport os\nimport gc\n\ndef optimize_memory(df, verbose=True):\n    \"\"\"\n    Optimize memory usage by downcasting numeric types where possible.\n    \"\"\"\n\n    if verbose:\n        start_mem = df.memory_usage().sum() / 1024**2\n        print(f'Memory usage before optimization: {start_mem:.2f} MB')\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        \n        if col_type != 'object':\n            c_min = df[col].min()\n            c_max = df[col].max()\n            \n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                # For float types, we'll use float32 instead of float64\n                # This is usually sufficient for ML models\n                if c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n    \n    if verbose:\n        end_mem = df.memory_usage().sum() / 1024**2\n        print(f'Memory usage after optimization: {end_mem:.2f} MB')\n        print(f'Decreased by {100 * (start_mem - end_mem) / start_mem:.1f}%')\n    \n    return df\n\ndef feature_engineering(df):\n    df['exp_302P289M125'] = np.exp(df['X302'] + df['X289'] - df['X125'])\n    df['289xexp_289M125'] = df['X289'] * np.exp(df['X289'] - df['X125'])\n    df['385xexp_289M125'] = df['X385'] * np.exp(df['X289'] - df['X125'])\n    \n    df['bid_ask_interaction'] = df['bid_qty'] * df['ask_qty']\n    df['bid_buy_interaction'] = df['bid_qty'] * df['buy_qty']\n    df['bid_sell_interaction'] = df['bid_qty'] * df['sell_qty']\n    df['ask_buy_interaction'] = df['ask_qty'] * df['buy_qty']\n    df['ask_sell_interaction'] = df['ask_qty'] * df['sell_qty']\n    df['volume_weighted_sell'] = df['sell_qty'] * df['volume']\n    df['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-10)\n    df['selling_pressure'] = df['sell_qty'] / (df['volume'] + 1e-10)\n    df['log_volume'] = np.log1p(df['volume'])\n\n    df['effective_spread_proxy'] = np.abs(df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-10)\n    df['bid_ask_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['order_flow_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['liquidity_ratio'] = (df['bid_qty'] + df['ask_qty']) / (df['volume'] + 1e-10)\n    \n    df['ask_buy_interaction_x_X293']=df['X293']*df['ask_buy_interaction']\n    \n     # Price Pressure Indicators\n    df['net_order_flow'] = df['buy_qty'] - df['sell_qty']\n    df['normalized_net_flow'] = df['net_order_flow'] / (df['volume'] + 1e-10)\n    df['buying_pressure'] = df['buy_qty'] / (df['volume'] + 1e-10)\n    df['volume_weighted_buy'] = df['buy_qty'] * df['volume']\n    \n    # Liquidity Depth Measures\n    df['total_depth'] = df['bid_qty'] + df['ask_qty']\n    df['depth_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['total_depth'] + 1e-10)\n    df['relative_spread'] = np.abs(df['bid_qty'] - df['ask_qty']) / (df['total_depth'] + 1e-10)\n    df['log_depth'] = np.log1p(df['total_depth'])\n    \n    # Order Flow Toxicity Proxies\n    df['kyle_lambda'] = np.abs(df['net_order_flow']) / (df['volume'] + 1e-10)\n    df['flow_toxicity'] = np.abs(df['order_flow_imbalance']) * df['volume']\n    df['aggressive_flow_ratio'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    \n    # Market Activity Indicators\n    df['volume_depth_ratio'] = df['volume'] / (df['total_depth'] + 1e-10)\n    df['activity_intensity'] = (df['buy_qty'] + df['sell_qty']) / (df['volume'] + 1e-10)\n    df['log_buy_qty'] = np.log1p(df['buy_qty'])\n    df['log_sell_qty'] = np.log1p(df['sell_qty'])\n    df['log_bid_qty'] = np.log1p(df['bid_qty'])\n    df['log_ask_qty'] = np.log1p(df['ask_qty'])\n    \n    # Microstructure Volatility Proxies\n    df['realized_spread_proxy'] = 2 * np.abs(df['net_order_flow']) / (df['volume'] + 1e-10)\n    df['price_impact_proxy'] = df['net_order_flow'] / (df['total_depth'] + 1e-10)\n    df['quote_volatility_proxy'] = np.abs(df['depth_imbalance'])\n    \n    # Complex Interaction Terms\n    df['flow_depth_interaction'] = df['net_order_flow'] * df['total_depth']\n    df['imbalance_volume_interaction'] = df['order_flow_imbalance'] * df['volume']\n    df['depth_volume_interaction'] = df['total_depth'] * df['volume']\n    df['buy_sell_spread'] = np.abs(df['buy_qty'] - df['sell_qty'])\n    df['bid_ask_spread'] = np.abs(df['bid_qty'] - df['ask_qty'])\n    \n    # Information Asymmetry Measures\n    df['trade_informativeness'] = df['net_order_flow'] / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['execution_shortfall_proxy'] = df['buy_sell_spread'] / (df['volume'] + 1e-10)\n    df['adverse_selection_proxy'] = df['net_order_flow'] / (df['total_depth'] + 1e-10) * df['volume']\n    \n    # Market Efficiency Indicators\n    df['fill_probability'] = df['volume'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['execution_rate'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    df['market_efficiency'] = df['volume'] / (df['bid_ask_spread'] + 1e-10)\n    \n    # Non-linear Transformations\n    df['sqrt_volume'] = np.sqrt(df['volume'])\n    df['sqrt_depth'] = np.sqrt(df['total_depth'])\n    df['volume_squared'] = df['volume'] ** 2\n    df['imbalance_squared'] = df['order_flow_imbalance'] ** 2\n    \n    # Relative Measures\n    df['bid_ratio'] = df['bid_qty'] / (df['total_depth'] + 1e-10)\n    df['ask_ratio'] = df['ask_qty'] / (df['total_depth'] + 1e-10)\n    df['buy_ratio'] = df['buy_qty'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['sell_ratio'] = df['sell_qty'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    \n    # Market Stress Indicators\n    df['liquidity_consumption'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    df['market_stress'] = df['volume'] / (df['total_depth'] + 1e-10) * np.abs(df['order_flow_imbalance'])\n    df['depth_depletion'] = df['volume'] / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    \n    # Directional Indicators\n    df['net_buying_ratio'] = df['net_order_flow'] / (df['volume'] + 1e-10)\n    df['directional_volume'] = df['net_order_flow'] * np.log1p(df['volume'])\n    df['signed_volume'] = np.sign(df['net_order_flow']) * df['volume']\n\n    #etc\n    # df['sqrt_volume_div_log_volume'] = df['sqrt_volume'] / (df['log_volume'] + 1e-6)\n    # df['sqrt_volume_div_activity_intensity'] = df['sqrt_volume'] / (df['activity_intensity'] + 1e-6)\n    # df['sqrt_volume_mul_fill_probability'] = df['sqrt_volume'] * df['fill_probability']\n    # df['volume_div_sqrt_volume'] = df['volume'] / (df['sqrt_volume'] + 1e-6)\n    # df['sqrt_volume_div_fill_probability'] = df['sqrt_volume'] / (df['fill_probability'] + 1e-6)\n    # df['sqrt_volume_mul_activity_intensity'] = df['sqrt_volume'] * df['activity_intensity']\n    # df['sqrt_volume_div_log_sell_qty'] = df['sqrt_volume'] / (df['log_sell_qty'] + 1e-6)\n    # df['log_buy_qty_mul_sqrt_volume'] = df['log_buy_qty'] * df['sqrt_volume']\n    # df['sqrt_volume_mul_log_buy_qty'] = df['sqrt_volume'] * df['log_buy_qty']\n    # df['log_volume_mul_sqrt_volume'] = df['log_volume'] * df['sqrt_volume']\n    \n    # df['log_sell_qty_mul_X598'] = df['log_sell_qty'] * df['X598']\n    # df['log_buy_qty_mul_X598'] = df['log_buy_qty'] * df['X598']\n    # df['log_volume_mul_X598'] = df['log_volume'] * df['X598']\n    \n    # # df['sqrt_volume_mul_X856'] = df['sqrt_volume'] * df['X856']\n    \n    # df['log_sell_qty_mul_X302'] = df['log_sell_qty'] * df['X302']\n    # df['log_volume_mul_X302'] = df['log_volume'] * df['X302']\n    # df['log_buy_qty_mul_X302'] = df['log_buy_qty'] * df['X302']\n    \n    # df['log_sell_qty_mul_X292'] = df['log_sell_qty'] * df['X292']\n\n    df = df.replace([np.inf, -np.inf], np.nan)\n    df = df.fillna(0)\n    return df\n\ndef preprocess_data_chunked(raw_df, chunk_size=10):\n    \"\"\"\n    Preprocess data with memory-efficient chunked lag creation.\n    \"\"\"\n    assert len(raw_df.shape) == 2\n\n    y = raw_df['label'].to_numpy().astype(np.float32)  # Use float32 for labels\n    assert y.shape == (raw_df.shape[0],)\n\n    # Original features\n    cols = [\n        'X363', 'X405', 'X321',\n        'X175', 'X179', 'X137', 'X197', 'X22', 'X40', 'X181',\n        'X28', 'X169', 'X198', 'X173',\n        'X338', 'X288', 'X385', 'X344', 'X427', 'X587', 'X450',\n        'X97', 'X52', 'X444',\n        'X598', 'X379', 'X696', 'X297', 'X138',\n        'X572', 'X343', 'X586', 'X466', 'X438', 'X452', 'X459',\n        'X435', 'X386', 'X55', 'X341', 'X683', 'X428', 'X605',\n        'X445', 'X272', 'X180', 'X593', 'X680',\n        'X686', 'X692', 'X695',\n        \"X603\", \"X674\", \"X421\", \"X333\",\n        \"X415\", \"X345\", \"X174\", \"X302\", \"X178\", \"X168\", \"X612\",\n        'X298', 'X45', 'X46', 'X39', 'X752', 'X759', 'X41', 'X42',\n        \"buy_qty\", \"sell_qty\", \"volume\",\n        \"bid_qty\", \"ask_qty\",\n        'X465','X153','X289','X125','X21',\"X293\",'X540','X493',\n        'X425',\"X292\", \"X532\",\n    ]\n    \n    # Add new top important features\n    new_features = [\n        'X758',  # Importance: 0.0260, Consistency: 83.3%\n        'X296',  # Importance: 0.0170, Consistency: 66.7%\n        'X611',  # Importance: 0.0133, Consistency: 66.7%\n        'X780',  # Importance: 0.0084, Consistency: 100.0%\n        'X451',  # Importance: 0.0449, Consistency: 16.7%\n        'X25',   # Importance: 0.0148, Consistency: 50.0%\n        'X591',  # Importance: 0.0138, Consistency: 50.0%\n        'X363', 'X321', 'X405', 'X730', 'X523', 'X756', 'X589', 'X462', 'X779', 'log_liquidity',\n        'X25', 'X532', 'X520', 'X329', 'X383', 'X751', 'X535', 'X639', 'X596', 'X761',\n        'X145', 'X709', 'X173', 'X245', 'X168', 'X171', 'X241', 'X31', 'X105', 'X63',\n        'X263', 'X426', 'X286', 'X357', 'X399', 'X315', 'X468', 'X131', 'X647', 'log_spread',\n        'X752', 'X254', 'X592', 'X733', 'X636', 'X394', 'X527', 'X180', 'X367', 'X38',\n        'X634', 'X718', 'X387', 'X429', 'X345', 'X344', 'X253', 'X469', 'X446', 'X125',\n        'X760', 'X186', 'X711', 'X150', 'X661', 'X215', 'X403', 'X141', 'X771', 'X453',\n        'X401', 'X629', 'X616', 'X281', 'X432', 'X283', 'X244', 'X440', 'X430', 'X382',\n        'X175', 'X95', 'X444', 'X189', 'X55', 'X605', 'X663', 'X194', 'X439', 'X670',\n        'X483', 'X163', 'X376', 'X71', 'X650', 'X203', 'X8', 'X624', 'X160', 'X100',\n        'X14', 'X511', 'X59', 'X302', 'X81', 'X325', 'X514', 'X649', 'X447', 'X538',\n        'X443', 'X39', 'X343', 'X12', 'X678', 'X775', 'X498', 'X249', 'X42', 'X384',\n        'kyle_lambda', 'X349', 'X356', 'X2', 'X250', 'X397', 'X685', 'X568', 'X136', 'X496',\n        'X53', 'X66', 'X374', 'X590', 'X668', 'X585', 'X677', 'X667', 'X530', 'X28',\n        'X64', 'X407', 'X494', 'X770', 'X710', 'X526', 'X644', 'X167', 'X190', 'X723',\n        'X33', 'X579', 'X206',\n    ]\n\n    # ADDITIONAL features requested by user\n    additional_features = [\n        'X525', 'X267', 'X166', 'X719', 'X489', 'X758', 'X652', 'X433', 'X778', 'X428',\n        'X617', 'X259', 'X633', 'X565', 'X364', 'depth_ratio', 'X550', 'X687', 'X610', 'X599',\n        'X717', 'X587', 'X143', 'X506', 'X546', 'X505', 'X159', 'X574', 'X278', 'X458',\n        'X1', 'X749', 'X155', 'X651', 'X470', 'X580', 'X445', 'X373', 'X82', 'X607',\n        'X298', 'X221', 'X388', 'X120', 'X391', 'X23', 'X679', 'X377', 'X767', 'X755',\n        'X566', 'X424', 'X438', 'X198', 'X300', 'X268', 'X434', 'X290', 'X368', 'X464',\n        'X119', 'X197', 'X597', 'X157', 'X485', 'X127', 'X101', 'X533', 'X235', 'X712',\n        'X154', 'X239', 'X10', 'X420', 'X449', 'X740', 'X227', 'X36', 'X358', 'X551',\n        'X528', 'X285', 'X335', 'X152', 'X110', 'X68', 'X713', 'X402', 'X370', 'X735',\n        'X200', 'X331', 'X473', 'X162', 'X213', 'X322', 'X289', 'X477', 'X113', 'X560',\n        'X672', 'X621', 'X682', 'X5', 'X72', 'X44', 'X419', 'buy_pressure', 'X242', 'volume',\n        'X472', 'X332', 'X441', 'buy_sell_ratio', 'pressure_ratio', 'X508', 'X594', 'X191',\n        'X261', 'X603', 'net_pressure', 'order_flow_imbalance', 'sell_pressure', 'X240', 'X673',\n        'X608', 'X509', 'X165', 'X720', 'X314', 'X522', 'X531', 'X625', 'bid_depth_ratio',\n        'X435', 'X293', 'X486', 'price_efficiency', 'X716', 'X627', 'X626', 'X169', 'X613',\n        'X680', 'X544', 'X115', 'X307', 'X665', 'X465', 'X347', 'X728', 'X70', 'log_volume',\n        'X340', 'X459', 'X56', 'X395', 'X354', 'X51', 'X732', 'X247', 'X324', 'X316',\n        'X76', 'X341', 'X739', 'X601', 'X386', 'X683', 'X149', 'X193', 'X628', 'X309',\n        'X351', 'X393'\n    ]\n    extended_features = [\n        # 'X727', 'X427', 'X288', 'X721', 'X312', 'X421', 'X471', 'X573', 'X780', 'X255',\n        # 'X144', 'X299', 'X301', 'X563', 'X737', 'X702', 'ask_qty', 'X507', 'X306', 'X501',\n        # 'X303', 'amihud_illiquidity', 'X586', 'X43', 'X517', 'X248', 'X137', 'X757', 'X196',\n        # 'X777', 'X280', 'X266', 'X689', 'X294', 'X492', 'X555', 'X731', 'X262', 'X576',\n        # 'X13', 'X518', 'X502', 'X558', 'pin_proxy', 'X6', 'X602', 'X695', 'X703', 'X413',\n        # 'X660', 'X37', 'X15', 'X310', 'X512', 'X362', 'X631', 'X214', 'X562', 'X488',\n        # 'X510', 'X256', 'X35', 'X128', 'X86', 'X170', 'X30', 'X265', 'X323', 'X559',\n        # 'X348', 'X130', 'X529', 'X20', 'X4', 'X90', 'X192', 'X91', 'X582', 'X99',\n        # 'X24', 'X317', 'X707', 'X653', 'X519', 'X557', 'X371', 'X415', 'X84', 'X83',\n        # 'order_toxicity', 'X360', 'X111', 'X699', 'X187', 'X591', 'X637', 'X567', 'X577',\n        # 'X313', 'X60', 'X671', 'X698', 'X701', 'X725', 'X292', 'X638', 'X741', 'X379',\n        # 'X700', 'X614', 'X676', 'X516', 'X697', 'X611', 'X311', 'X615', 'X706', 'X466',\n        # 'X571', 'X451', 'X17', 'X584', 'X436', 'X305', 'liquidity_consumption', 'X34', 'X282',\n        # 'X681', 'X7', 'X208', 'X41', 'X536', 'X548', 'X296', 'X776', 'X87', 'X40',\n        # 'X570', 'X539', 'X474', 'X753', 'X425', 'X217', 'X199', 'X18', 'X609', 'X21',\n        # 'X277', 'X279', 'X326', 'X540', 'X688', 'X553', 'X452', 'X738', 'X183', 'X759',\n        # 'bid_ask_ratio', 'X495', 'volume_participation', 'X715', 'X385', 'X291', 'X409', 'X112',\n        # 'X693', 'X102', 'X318', 'X705', 'X556', 'X547'\n    ]\n\n    # Combine all features and remove duplicates while preserving order\n    cols = list(dict.fromkeys(cols + new_features+additional_features))\n    \n    # Check which features actually exist in the dataframe\n    available_cols = [col for col in cols if col in raw_df.columns]\n    missing_cols = [col for col in cols if col not in raw_df.columns]\n    \n    if missing_cols:\n        print(f\"Warning: The following features are not in the dataset: {missing_cols}\")\n    \n    print(f\"Using {len(available_cols)} features out of {len(cols)} requested\")\n\n    # Select and optimize base features\n    # print(\"feature engineering...\")\n    # df = feature_engineering(df)\n    df = raw_df[available_cols].copy()\n    print(df.shape)\n    # df = optimize_memory(df, verbose=True)\n    \n    assert df.isna().sum().sum() == 0\n\n    # Extended lag features\n    lag_periods = [\n        1, 3, 5, 6, 7, 8, 9,  # Very short-term (1-10)\n        12, 15, 18, 20, 25, 30,          # Short-term (12-30)\n        # 40, 50, 60, 75, 90,              # Medium-term (40-90)\n        # 120,150,\n        60, 120, 180,\n        # 180, 222,              # Long-term (2-4 hours if minute data)\n        365,800, 1600\n        # 480, 600,              # Longer-term (6-12 hours)\n        # 960, 1440,                 # Very long-term (16-24 hours)\n        # 2880                     # Multi-day (2-3 days)\n    ]\n    \n    # Process lags in chunks to manage memory\n    print(\"Creating lagged features in chunks...\")\n    \n    # Start with base features\n    result_df = df.copy()\n    \n    # Process lags in chunks\n    for i in range(0, len(lag_periods), chunk_size):\n        chunk_lags = lag_periods[i:i+chunk_size]\n        print(f\"  Processing lags: {chunk_lags}\")\n        \n        # Create lagged features for this chunk\n        chunk_dfs = []\n        for lag in chunk_lags:\n            lagged = df.shift(-lag).add_suffix(f'_lead_{lag}')\n            lagged = lagged.fillna(0.0).astype(np.float32)  # Fill NaN and convert to float32\n            chunk_dfs.append(lagged)\n        \n        # Concatenate chunk\n        if chunk_dfs:\n            chunk_combined = pd.concat(chunk_dfs, axis=1)\n            result_df = pd.concat([result_df, chunk_combined], axis=1)\n            \n            # Clean up\n            del chunk_dfs, chunk_combined\n            gc.collect()\n    \n    # Final optimization\n    # result_df = optimize_memory(result_df, verbose=True)\n    # print(\"feature engineering...\")\n    # result_df = feature_engineering(result_df)\n    \n    assert 'label' not in result_df.columns\n    assert raw_df.shape[0] == result_df.shape[0] and (raw_df.index == result_df.index).all()\n    assert result_df.isna().sum().sum() == 0\n    assert result_df.shape[0] == y.shape[0]\n    \n    print(f\"Final feature count: {result_df.shape[1]}\")\n    \n    return result_df, y\n","metadata":{"execution":{"iopub.status.busy":"2025-07-17T11:19:58.179438Z","iopub.execute_input":"2025-07-17T11:19:58.179744Z","iopub.status.idle":"2025-07-17T11:19:58.224701Z","shell.execute_reply.started":"2025-07-17T11:19:58.179719Z","shell.execute_reply":"2025-07-17T11:19:58.223922Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Set memory-efficient options for pandas\npd.options.mode.chained_assignment = None  # Disable SettingWithCopyWarning\npd.options.display.max_columns = None\n\n# Load and preprocess training data\nprint(\"Loading training data...\")\ntrain_df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\n\n# Display available columns to verify feature existence\nprint(f\"\\nTotal columns in training data: {len(train_df.columns)}\")\nprint(f\"Sample columns: {list(train_df.columns[:20])}\")\n\n# Optimize memory for the raw training data\n# print(\"\\nOptimizing memory for raw training data...\")\n# train_df = optimize_memory(train_df, verbose=True)\n\nX_train, y_train = preprocess_data_chunked(train_df, chunk_size=10)\n\n# Clean up training dataframe\ndel train_df\ngc.collect()\n\nprint(f\"\\nTraining data shape: X={X_train.shape}, y={y_train.shape}\")\n","metadata":{"execution":{"iopub.status.busy":"2025-07-17T11:20:04.115108Z","iopub.execute_input":"2025-07-17T11:20:04.115428Z","execution_failed":"2025-07-17T11:21:14.084Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# XGB Model - COMMENTED OUT\n# from sklearn.model_selection import KFold\n# from xgboost import XGBRegressor\n\n# LightGBM Model\nfrom sklearn.model_selection import KFold\nimport lightgbm as lgb\nimport importlib\nimport pandas as pd\nimportlib.reload(pd)\n\n\n# XGB_PARAMS = {\n#     \"tree_method\": \"hist\",\n#     \"device\": \"gpu\",\n#     \"colsample_bylevel\": 0.4778,\n#     \"colsample_bynode\": 0.3628,\n#     \"colsample_bytree\": 0.7107,\n#     \"gamma\": 1.7095,\n#     \"learning_rate\": 0.02213,\n#     \"max_depth\": 20,\n#     \"max_leaves\": 12,\n#     \"min_child_weight\": 16,\n#     \"n_estimators\": 1500,\n#     \"subsample\": 0.06567,\n#     \"reg_alpha\": 39.3524,\n#     \"reg_lambda\": 75.4484,\n#     \"verbosity\": 0,\n#     \"random_state\": 42,\n#     \"n_jobs\": -1,\n#     \"early_stopping_rounds\":100\n# }\n\nLGBM_PARAMS = {\n    \"objective\": \"regression\",\n    \"metric\": \"rmse\",\n    \"boosting_type\": \"gbdt\",\n    \"num_leaves\": 31,\n    \"learning_rate\": 0.05,\n    \"feature_fraction\": 0.9,\n    \"bagging_fraction\": 0.8,\n    \"bagging_freq\": 5,\n    \"verbose\": 0,\n    \"random_state\": 42,\n    \"n_jobs\": -1,\n    \"device\": \"gpu\"\n}\nmodels = []\n\nkf = KFold(n_splits=5, shuffle=False)\nfor fold, (train_idx, valid_idx) in enumerate(kf.split(X_train), start=1):\n    print(f'*** ----------- FOLD {fold} ----------- ***')\n    X_val = X_train.iloc[valid_idx]\n    X_trn = X_train.iloc[train_idx]\n    y_val = y_train[valid_idx]\n    y_trn = y_train[train_idx]\n    \n    train_data = lgb.Dataset(X_trn, label=y_trn)\n    valid_data = lgb.Dataset(X_val, label=y_val, reference=train_data)\n    print('start train')\n    model = lgb.train(\n        LGBM_PARAMS,\n        train_data,\n        valid_sets=[valid_data],\n        num_boost_round=1000,\n        callbacks=[lgb.early_stopping(100), lgb.log_evaluation(50)]\n    )\n    models.append(model)\n\n# # Clean up training data\n# del X_train, y_train\n# gc.collect()\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-07-17T11:21:14.084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tqdm import tqdm\n# Load test data\nprint(\"\\nLoading test data...\")\ntest_df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')\n\n# Optimize memory for test data\n# print(\"\\nOptimizing memory for raw test data...\")\n# test_df = optimize_memory(test_df, verbose=True)\n\n# Try to load precomputed timestamp reconstruction data\ntimestamp_recon_path = '/kaggle/input/the-order-of-the-test-rows-2/closest_rows.csv'\nuse_timestamp_reconstruction = os.path.exists(timestamp_recon_path)\n\nif use_timestamp_reconstruction:\n    print(\"Found timestamp reconstruction file, loading...\")\n    \n    # Load precomputed timestamp reconstruction data\n    t = pd.Series(pd.read_csv(timestamp_recon_path)['0'].to_numpy())\n    assert t.shape == (test_df.shape[0],)\n    print('Reconstructed timestamps share:', len(t[t >= 0]) / len(t))\n\n    # Visualize the reconstructed timestamps\n    plt.figure(figsize=(16, 4))\n    plt.plot(t.sort_values().to_numpy())\n    plt.title('Sorted Reconstructed Timestamps')\n    plt.show()\n\n    plt.figure(figsize=(16, 4))\n    plt.plot(t[t >= 0].sort_values().iloc[:1000].to_numpy())\n    plt.axhline(10080, color='r', linestyle='--')\n    plt.title('First 1000 Valid Reconstructed Timestamps')\n    plt.show()\n\n    # Process timestamp reconstruction\n    t -= 10080\n    t[t < 0] = 538149\n\n    t = t.sort_values()\n    t[t <= len(t)] = np.arange(t[t <= len(t)].shape[0])\n    t = t.sort_index()\n\n    t = pd.Series(np.arange(538150), index=t.to_numpy()).sort_index()\n\n    # Visualize test data before sorting\n    if 'X656' in test_df.columns:\n        plt.figure(figsize=(16, 4))\n        plt.plot(test_df['X656'].to_numpy())\n        plt.title('Test Data Feature X656 - Before Sorting')\n        plt.show()\n\n    # Sort test dataset by reconstructed time order\n    test_df = test_df.iloc[t.to_numpy()]\n\n    # Visualize test data after sorting\n    if 'X656' in test_df.columns:\n        plt.figure(figsize=(16, 4))\n        plt.plot(test_df['X656'].to_numpy())\n        plt.title('Test Data Feature X656 - After Sorting')\n        plt.show()\nelse:\n    print(\"WARNING: Timestamp reconstruction file not found!\")\n    print(f\"Expected path: {timestamp_recon_path}\")\n    print(\"Proceeding without timestamp reconstruction...\")\n    print(\"This may significantly impact model performance since lagged features assume temporal order.\")\n    \n    t = pd.Series(np.arange(len(test_df)))\n\n# Preprocess test data\nprint(\"\\nPreprocessing test data...\")\n\nX_test, _ = preprocess_data_chunked(test_df, chunk_size=10)\n\n# Clean up test dataframe\ndel test_df\ngc.collect()\n\nprint(f\"Test data shape: {X_test.shape}\")\n\n# LightGBM Make predictions in batches to save memory\nprint(\"\\nMaking predictions...\")\nbatch_size = 100000\nn_samples = X_test.shape[0]\ny_pred = np.zeros(n_samples, dtype=np.float32)\n\nfor i in range(0, n_samples, batch_size):\n    end_idx = min(i + batch_size, n_samples)\n    print(f\"  Predicting batch {i//batch_size + 1}/{(n_samples + batch_size - 1)//batch_size}\")\n    preds = []\n    for j in tqdm([0,1,2,3,4]):\n        preds.append(models[j].predict(X_test.iloc[i:end_idx], num_iteration=models[j].best_iteration).astype(np.float32))\n    y_pred[i:end_idx] = np.mean(preds,0)\n    # y_pred[i:end_idx] = model.predict(X_test.iloc[i:end_idx]).astype(np.float32)\n\n# Clean up test features\ndel X_test\ngc.collect()\n\n# Display prediction statistics\nprint(\"\\nPrediction statistics:\")\nprint(pd.Series(y_pred).describe())\n\n# Plot cumulative predictions\nplt.figure(figsize=(16, 4))\nplt.plot(np.cumsum(y_pred))\nplt.title('Cumulative Predictions')\nplt.xlabel('Sample Index')\nplt.ylabel('Cumulative Sum')\nplt.grid(True, alpha=0.3)\nplt.show()\n\n# Plot prediction distribution\nplt.figure(figsize=(12, 6))\nplt.subplot(1, 2, 1)\nplt.hist(y_pred, bins=100, alpha=0.7, edgecolor='black')\nplt.title('Prediction Distribution')\nplt.xlabel('Predicted Value')\nplt.ylabel('Frequency')\n\nplt.subplot(1, 2, 2)\nplt.plot(y_pred[:1000])\nplt.title('First 1000 Predictions')\nplt.xlabel('Sample Index')\nplt.ylabel('Predicted Value')\nplt.tight_layout()\nplt.show()\n\n# Prepare submission\nprint(\"\\nPreparing submission...\")\nsubmission = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\n\nif use_timestamp_reconstruction:\n    # Reorder submission to match original test order\n    submission = submission.iloc[t.to_numpy()]\n    submission['prediction'] = y_pred\n    submission = submission.sort_index()\nelse:\n    # If no timestamp reconstruction, just use predictions in order\n    submission['prediction'] = y_pred\n\n# Save submission\nsubmission.to_csv('submission_lgbm.csv', index=False)\nprint(\"Submission saved to 'submission_lgbm.csv'\")\n\n# Display submission\nprint(\"\\nSubmission preview:\")\nprint(submission.head())\nprint(f\"\\nSubmission shape: {submission.shape}\")\nprint(f\"Prediction range: [{submission['prediction'].min():.6f}, {submission['prediction'].max():.6f}]\")\n\n# Final memory cleanup\ngc.collect()\nprint(\"\\nDone!\")","metadata":{"execution":{"iopub.execute_input":"2025-07-13T15:16:58.823525Z","iopub.status.busy":"2025-07-13T15:16:58.823242Z","iopub.status.idle":"2025-07-13T15:18:55.602041Z","shell.execute_reply":"2025-07-13T15:18:55.601204Z","shell.execute_reply.started":"2025-07-13T15:16:58.823504Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n# sub1 = pd.read_csv(\"submission_xgb.csv\",index_col=None)  # Changed from XGB to LGBM\n# sub2 = pd.read_csv(\"submission_sgd.csv\",index_col=None)\n# sub3 = pd.read_csv(\"submission_turkish.csv\",index_col=None)\n# sub4 = pd.read_csv(\"submission_iblend.csv\",index_col=None)\n# sub5 = pd.read_csv(\"submission_DL.csv\",index_col=None)\n# sub6 = pd.read_csv(\"submission_ensemble.csv\",index_col=None)\nsub7 = pd.read_csv(\"submission_lgbm.csv\",index_col=None)\nsub8 = pd.read_csv(\"/kaggle/input/drw-blend-h-v-remix-higher-changepoint/submission_prophet_enhanced.csv\",index_col=None)\nsub9 = pd.read_csv(\"\")","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub1['prediction'] = sub7['prediction'] * 0.6 + sub8['prediction'] * 0.4\nsub1.to_csv(\"submission.csv\",index=False)","metadata":{},"outputs":[],"execution_count":null}]}