{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"## libraries are loaded\nimport matplotlib.pyplot as plt\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nimport shap\nfrom sklearn.model_selection import KFold\nfrom scipy.stats import pearsonr\n\n\n\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T01:56:09.136332Z","iopub.execute_input":"2025-06-12T01:56:09.136792Z","iopub.status.idle":"2025-06-12T01:56:09.144160Z","shell.execute_reply.started":"2025-06-12T01:56:09.136754Z","shell.execute_reply":"2025-06-12T01:56:09.142853Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n\nclass CFG:\n    train_path = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    test_path = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    sample_sub_path = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n\ndef add_features(df):\n    df['bid_ask_interaction'] = df['bid_qty'] * df['ask_qty']\n    df['bid_buy_interaction'] = df['bid_qty'] * df['buy_qty']\n    df['bid_sell_interaction'] = df['bid_qty'] * df['sell_qty']\n    df['ask_buy_interaction'] = df['ask_qty'] * df['buy_qty']\n    df['ask_sell_interaction'] = df['ask_qty'] * df['sell_qty']\n    df['buy_sell_interaction'] = df['buy_qty'] * df['sell_qty']\n\n    df['spread_indicator'] = (df['ask_qty'] - df['bid_qty']) / (df['ask_qty'] + df['bid_qty'] + 1e-8)\n\n    df['volume_weighted_buy'] = df['buy_qty'] * df['volume']\n    df['volume_weighted_sell'] = df['sell_qty'] * df['volume']\n    df['volume_weighted_bid'] = df['bid_qty'] * df['volume']\n    df['volume_weighted_ask'] = df['ask_qty'] * df['volume']\n\n    df['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-8)\n    df['bid_ask_ratio'] = df['bid_qty'] / (df['ask_qty'] + 1e-8)\n\n    df['order_flow_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-8)\n\n    df['buying_pressure'] = df['buy_qty'] / (df['volume'] + 1e-8)\n    df['selling_pressure'] = df['sell_qty'] / (df['volume'] + 1e-8)\n\n    df['total_liquidity'] = df['bid_qty'] + df['ask_qty']\n    df['liquidity_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['total_liquidity'] + 1e-8)\n    df['relative_spread'] = (df['ask_qty'] - df['bid_qty']) / (df['volume'] + 1e-8)\n\n    df['trade_intensity'] = (df['buy_qty'] + df['sell_qty']) / (df['volume'] + 1e-8)\n    df['avg_trade_size'] = df['volume'] / (df['buy_qty'] + df['sell_qty'] + 1e-8)\n    df['net_trade_flow'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-8)\n\n    df['depth_ratio'] = df['total_liquidity'] / (df['volume'] + 1e-8)\n    df['volume_participation'] = (df['buy_qty'] + df['sell_qty']) / (df['total_liquidity'] + 1e-8)\n    df['market_activity'] = df['volume'] * df['total_liquidity']\n\n    df['effective_spread_proxy'] = np.abs(df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-8)\n    df['realized_volatility_proxy'] = np.abs(df['order_flow_imbalance']) * df['volume']\n\n    df['normalized_buy_volume'] = df['buy_qty'] / (df['bid_qty'] + 1e-8)\n    df['normalized_sell_volume'] = df['sell_qty'] / (df['ask_qty'] + 1e-8)\n\n    df['liquidity_adjusted_imbalance'] = df['order_flow_imbalance'] * df['depth_ratio']\n    df['pressure_spread_interaction'] = df['buying_pressure'] * df['spread_indicator']\n\n\n    df.replace([np.inf, -np.inf], np.nan, inplace=True)\n    df.fillna(0, inplace=True)\n\n\n    \n    return df\n\ndef chunks(lst, n):\n    for i in range(0, len(lst), n):\n        yield lst[i:i + n]\n\n\ndef reduce_mem_usage(dataframe, dataset):    \n    print('Reducing memory usage for:', dataset)\n    initial_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    \n    for col in dataframe.columns:\n        col_type = dataframe[col].dtype\n\n        c_min = dataframe[col].min()\n        c_max = dataframe[col].max()\n        if str(col_type)[:3] == 'int':\n            if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                dataframe[col] = dataframe[col].astype(np.int8)\n            elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                dataframe[col] = dataframe[col].astype(np.int16)\n            elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                dataframe[col] = dataframe[col].astype(np.int32)\n            elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                dataframe[col] = dataframe[col].astype(np.int64)\n        else:\n            if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                dataframe[col] = dataframe[col].astype(np.float16)\n            elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                dataframe[col] = dataframe[col].astype(np.float32)\n            else:\n                dataframe[col] = dataframe[col].astype(np.float64)\n\n    final_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    print('--- Memory usage before: {:.2f} MB'.format(initial_mem_usage))\n    print('--- Memory usage after: {:.2f} MB'.format(final_mem_usage))\n    print('--- Decreased memory usage by {:.1f}%\\n'.format(100 * (initial_mem_usage - final_mem_usage) / initial_mem_usage))\n\n    return dataframe\n\n# Create time-based sample weights\ndef create_time_weights(n_samples, decay_factor=0.98):\n    \"\"\"\n    Create exponentially decaying weights based on sample position.\n    More recent samples (higher indices) get higher weights.\n    decay_factor controls the rate of decay (0.98 = 2% decay per time unit)\n    \"\"\"\n    positions = np.arange(n_samples)\n    # Normalize positions to [0, 1] range\n    normalized_positions = positions / (n_samples - 1)\n    # Apply exponential weighting\n    weights = decay_factor ** (1 - normalized_positions)\n    # Normalize weights to sum to n_samples (maintains scale)\n    weights = weights * n_samples / weights.sum()\n    return weights\n\ndef log_scale_columns(df, cols):\n    for col in cols:\n        df[col] = np.log(df[col])\n    return df\n\ndef safe_log_transform(df, cols):\n    for col in cols:\n        safe_vals = df[col].copy()\n        # 무한대값을 NaN으로 변환\n        safe_vals.replace([np.inf, -np.inf], np.nan, inplace=True)\n        \n        # NaN 아닌 값들만 비교해서 0 이하 처리\n        mask = safe_vals.notna() & (safe_vals <= 0)\n        safe_vals[mask] = np.nan\n        \n        df[col] = np.log(safe_vals)\n    return df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T02:55:10.805140Z","iopub.execute_input":"2025-06-12T02:55:10.805478Z","iopub.status.idle":"2025-06-12T02:55:10.838924Z","shell.execute_reply.started":"2025-06-12T02:55:10.805453Z","shell.execute_reply":"2025-06-12T02:55:10.837768Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load data\ntrain = pd.read_parquet(CFG.train_path).reset_index(drop=True)\ntest = pd.read_parquet(CFG.test_path).reset_index(drop=True)\ntrain=add_features(train)\ntest=add_features(test)\nsample = pd.read_csv(CFG.sample_sub_path)\n\n\n\n# Select features\nselected_features = [\n\n    \n    \"X863\", \"X856\", \"X344\", \"X598\", \"X862\", \"X385\", \"X852\", \"X603\", \"X860\", \"X674\",\n    \"X415\", \"X345\", \"X137\", \"X855\", \"X174\", \"X302\", \"X178\", \"X532\", \"X168\", \"X612\",\n    \"X888\", \"X421\", \"X333\",\n    'bid_ask_interaction',\n    'bid_buy_interaction', 'bid_sell_interaction', 'ask_buy_interaction', 'ask_sell_interaction', 'buy_sell_interaction',\n    'spread_indicator',\n    'volume_weighted_buy', 'volume_weighted_sell', 'volume_weighted_bid', 'volume_weighted_ask',\n    'buy_sell_ratio', 'bid_ask_ratio',\n    'order_flow_imbalance',\n    'buying_pressure', 'selling_pressure',\n    'total_liquidity', 'liquidity_imbalance', 'relative_spread',\n    'net_trade_flow',\n    'depth_ratio', 'volume_participation', 'market_activity',\n    'effective_spread_proxy', 'realized_volatility_proxy',\n    'normalized_buy_volume', 'normalized_sell_volume',\n    'liquidity_adjusted_imbalance', 'pressure_spread_interaction'\n]\n\n\n\ntrain = train[selected_features + [\"label\"]]\ntest = test[selected_features]\n\n\nbin_features = [\n    \"X863\", \"X856\", \"X344\", \"X598\", \"X862\", \"X385\", \"X852\", \"X603\", \"X860\", \"X674\",\n    \"X415\", \"X345\", \"X137\", \"X855\", \"X174\", \"X302\", \"X178\", \"X532\", \"X168\", \"X612\"\n]\n\nfor col in bin_features:\n    train[f\"{col}_exp\"] = np.exp(train[col])\n    test[f\"{col}_exp\"] = np.exp(test[col])\n\n# 파생 피처 이름 리스트\ntransformed_features = [f\"{col}_exp\" for col in bin_features] \n\n# 최종 사용 피처\nselected_features +=  transformed_features\n\n\ntrain = train[selected_features + [\"label\"]]\ntest = test[selected_features]\n\n\ntrain = reduce_mem_usage(train, \"train\")\ntest = reduce_mem_usage(test, \"test\")\n\nprint(\"Train=\", train.shape)\nprint(\"Test=\", test.shape)\nprint(\"Sample=\", sample.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T01:56:09.199607Z","iopub.execute_input":"2025-06-12T01:56:09.200004Z","iopub.status.idle":"2025-06-12T01:58:26.735164Z","shell.execute_reply.started":"2025-06-12T01:56:09.199978Z","shell.execute_reply":"2025-06-12T01:58:26.733931Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# selected_features에 해당하는 열만 선택\ndata = train[selected_features]\n\n# 상관관계 행렬 계산\ncorr_matrix = data.corr()\n\n# 상관관계 히트맵 시각화\nplt.figure(figsize=(20, 16))\nsns.heatmap(corr_matrix, cmap='coolwarm', center=0, annot=False)\nplt.title(\"Correlation Matrix of Selected Features\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T01:58:26.736227Z","iopub.execute_input":"2025-06-12T01:58:26.736617Z","iopub.status.idle":"2025-06-12T01:58:36.085226Z","shell.execute_reply.started":"2025-06-12T01:58:26.736589Z","shell.execute_reply":"2025-06-12T01:58:36.084181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"time_index = train.index\n# 그래프를 그릴 피처 리스트 (label 제외)\n\ncheck_features = [\n    \"X863\", \"X856\", \"X344\", \"X598\", \"X862\", \"X385\", \"X852\", \"X603\", \"X860\", \"X674\",\n    \"X415\", \"X345\", \"X137\", \"X855\", \"X174\", \"X302\", \"X178\", \"X532\", \"X168\", \"X612\",\n    \"X888\", \"X421\", \"X333\"\n]\ncheck_features += transformed_features\n\n# 4개씩 나누기\nfeature_groups_4 = [check_features[i:i+4] for i in range(0, len(check_features), 4)]\n\n# 전체 시각화\nfor i, group in enumerate(feature_groups_4, 1):\n    plt.figure(figsize=(15, 6))\n    for feat in group:\n        plt.plot(time_index, train[feat], label=feat, alpha=0.7, linewidth=1)\n    plt.plot(time_index, train['label'], color='red', alpha=0.3, linewidth=1.5, label='label')\n    plt.title(f'Features Group {i} (4 variables) with Label over Time - Full Data')\n    plt.ylim(-20, 20)\n    plt.xlabel('Time')\n    plt.ylabel('Value')\n    plt.legend(loc='upper right', fontsize=9)\n    plt.grid(True)\n    plt.show()\n\n# 최신 10,000개만 선택\nlatest_index = time_index[-10000:]\ntrain_latest = train.loc[latest_index]\n\n# 최신 10,000개 시각화\nfor i, group in enumerate(feature_groups_4, 1):\n    plt.figure(figsize=(15, 6))\n    for feat in group:\n        plt.plot(latest_index, train_latest[feat], label=feat, alpha=0.7, linewidth=1)\n    plt.plot(latest_index, train_latest['label'], color='red', alpha=0.3, linewidth=1.5, label='label')\n    plt.title(f'Features Group {i} (4 variables) with Label over Time - Last 10,000')\n    plt.ylim(-20, 20)\n    plt.xlabel('Time')\n    plt.ylabel('Value')\n    plt.legend(loc='upper right', fontsize=9)\n    plt.grid(True)\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T02:35:23.186058Z","iopub.execute_input":"2025-06-12T02:35:23.186395Z","iopub.status.idle":"2025-06-12T02:35:34.720279Z","shell.execute_reply.started":"2025-06-12T02:35:23.186372Z","shell.execute_reply":"2025-06-12T02:35:34.718613Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"derived_features = [\n    'spread_indicator', 'buy_sell_ratio', 'bid_ask_ratio', 'order_flow_imbalance',\n    'buying_pressure', 'selling_pressure', 'liquidity_imbalance', 'relative_spread',\n    'net_trade_flow', 'depth_ratio',\n    'volume_participation', 'market_activity', 'effective_spread_proxy',\n    'realized_volatility_proxy', 'normalized_buy_volume', 'normalized_sell_volume'\n]\n\n# 2개씩 나누기\nderived_feature_groups = [derived_features[i:i+2] for i in range(0, len(derived_features), 2)]\n\n# 시계열 인덱스\ntime_index = train.index\n\n# 최근 10,000개만 선택\nlatest_index = time_index[-10000:]\ntrain_latest = train.loc[latest_index]\n\n# 시각화\nfor i, group in enumerate(derived_feature_groups, 1):\n    plt.figure(figsize=(15, 6))\n    for feat in group:\n        if feat in train.columns:\n            plt.plot(latest_index, train_latest[feat], label=feat, alpha=0.7, linewidth=1)\n    plt.plot(latest_index, train_latest['label'], color='red', alpha=0.4, linewidth=1.5, label='label x10')\n    plt.title(f'Derived Feature Group {i} with Label - Last 10,000')\n    plt.xlabel('Time')\n    plt.ylabel('Value')\n    plt.legend(loc='upper right', fontsize=9)\n    plt.grid(True)\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T02:16:33.238471Z","iopub.execute_input":"2025-06-12T02:16:33.238935Z","iopub.status.idle":"2025-06-12T02:16:35.831473Z","shell.execute_reply.started":"2025-06-12T02:16:33.238906Z","shell.execute_reply":"2025-06-12T02:16:35.830206Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cols_to_check = [\n    'buy_sell_ratio', 'bid_ask_ratio', \n    'total_liquidity', 'relative_spread', \n    'depth_ratio', 'volume_participation', 'market_activity',\n    'realized_volatility_proxy', 'normalized_buy_volume', 'normalized_sell_volume'\n]\n\nfor col in cols_to_check:\n    plt.figure(figsize=(12, 4))\n    data = train[col].astype('float32').replace([np.inf, -np.inf], np.nan).dropna()\n\n    # 인덱스 기준 시간 흐름으로 그리기\n    plt.plot(data.index, data.values, alpha=0.6, label=col)\n    plt.title(f'Time Series of {col}')\n    plt.xlabel('Index (Time Order)')\n    plt.ylabel(col)\n    plt.grid(True)\n    plt.tight_layout()\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T01:59:51.711705Z","iopub.execute_input":"2025-06-12T01:59:51.712956Z","iopub.status.idle":"2025-06-12T01:59:54.995837Z","shell.execute_reply.started":"2025-06-12T01:59:51.712919Z","shell.execute_reply":"2025-06-12T01:59:54.994571Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cols_to_log = [\n    'relative_spread',\n    'depth_ratio', 'volume_participation', 'market_activity',\n    'normalized_buy_volume', 'normalized_sell_volume'\n]\n\ntrain = safe_log_transform(train, cols_to_log)\ntest = safe_log_transform(test, cols_to_log)\n\ncols_log = [col  for col in cols_to_log]\n\n# 타입 변환\nfor col in cols_log:\n    train[col] = train[col].astype('float32')\n\n# 시각화\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfor col in cols_to_log:\n    plt.figure(figsize=(10, 4))\n    plt.plot(train[col].values, linewidth=0.5)\n    plt.title(f'{col} over time in train set (log-transformed)')\n    plt.xlabel('Index (time proxy)')\n    plt.ylabel(col)\n    plt.tight_layout()\n    plt.show()\n    \nfor col in cols_log:\n    train[col] = train[col].astype('float16')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T02:01:38.560798Z","iopub.execute_input":"2025-06-12T02:01:38.561188Z","iopub.status.idle":"2025-06-12T02:01:41.748572Z","shell.execute_reply.started":"2025-06-12T02:01:38.561163Z","shell.execute_reply":"2025-06-12T02:01:41.747184Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Feature 1: X344 * X385 / e**(X888/10)\ntrain['feature_1'] = train['X344'] * train['X385'] / np.exp(train['X888']/10)\ntest['feature_1'] = test['X344'] * test['X385'] / np.exp(test['X888']/10)\n\n# Feature 2: X863 * X856\ntrain['feature_2'] = train['X863'] * train['X856']\ntest['feature_2'] = test['X863'] * test['X856']\n\n# Feature 3: e**(X612/2) + X333\ntrain['feature_3'] = np.exp(train['X612']/2) + train['X333']\ntest['feature_3'] = np.exp(test['X612']/2) + test['X333']\n\n# Feature 4: X345 * X863\ntrain['feature_4'] = train['X345'] * train['X863']\ntest['feature_4'] = test['X345'] * test['X863']\n\n#Feature 5:\ntrain['feature_5'] = np.exp(-train['relative_spread'] / 5) * train['X603']\ntest['feature_5'] = np.exp(-test['relative_spread'] / 5) * test['X603']\n\n\nnew_features = ['feature_1', 'feature_2', 'feature_3','feature_4','feature_5']\nselected_features += new_features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T01:17:26.086311Z","iopub.execute_input":"2025-06-12T01:17:26.086661Z","iopub.status.idle":"2025-06-12T01:17:26.322789Z","shell.execute_reply.started":"2025-06-12T01:17:26.086631Z","shell.execute_reply":"2025-06-12T01:17:26.321831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from itertools import combinations\n\nfeature1 = [\"X862_exp\", \"X178_exp\", \"X598_exp\", \"X612_exp\"]\nfeature2 = [\"X863\", \"X344\", \"X598\", \"X603\", \"X415\", \"X860\", \"X137\"]\n\n# 파생변수를 담을 딕셔너리\nnew_train_cols = {}\nnew_test_cols = {}\n\n# f2 * f1 조합\nfor v in feature2:\n    for l in feature1:\n        new_col = f\"{v}_x_{l}\"\n        new_train_cols[new_col] = train[v] * train[l]\n        new_test_cols[new_col] = test[v] * test[l]\n\n# f2 * f2 조합\nfor v1, v2 in combinations(feature2, 2):\n    new_col = f\"{v1}_x_{v2}\"\n    new_train_cols[new_col] = train[v1] * train[v2]\n    new_test_cols[new_col] = test[v1] * test[v2]\n\n# DataFrame으로 만들어 한 번에 concat\nnew_train_df = pd.DataFrame(new_train_cols)\nnew_test_df = pd.DataFrame(new_test_cols)\n\ntrain = pd.concat([train, new_train_df], axis=1)\ntest = pd.concat([test, new_test_df], axis=1)\n\n# 단편화 해제 (필요 시)\ntrain = train.copy()\ntest = test.copy()\n\n# selected_features에 추가\nselected_features += list(new_train_cols.keys())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T02:50:31.973835Z","iopub.execute_input":"2025-06-12T02:50:31.974950Z","iopub.status.idle":"2025-06-12T02:50:33.787917Z","shell.execute_reply.started":"2025-06-12T02:50:31.974910Z","shell.execute_reply":"2025-06-12T02:50:33.786713Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"time_index = train.index  # 시계열 인덱스\n\nfor feat in new_features:\n    plt.figure(figsize=(15, 5))\n    plt.plot(time_index, train[feat], label=feat, color='blue', linewidth=1.5, alpha=0.8)\n    plt.plot(time_index, train['label'], label='label', color='red', linewidth=1, alpha=0.3)\n    plt.title(f\"{feat} vs label over time\")\n    plt.xlabel('Time')\n    plt.ylabel('Value')\n    plt.legend()\n    plt.grid(True)\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T01:17:26.326558Z","iopub.execute_input":"2025-06-12T01:17:26.327132Z","iopub.status.idle":"2025-06-12T01:17:31.150343Z","shell.execute_reply.started":"2025-06-12T01:17:26.327100Z","shell.execute_reply":"2025-06-12T01:17:31.149006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 파생된 feature 리스트\nderived_feature_names = list(new_train_cols.keys())\n\nderived_feature_groups = list(chunks(derived_feature_names, 4))\n\n# 최근 10,000개 데이터 선택\nlatest_index = train.index[-10000:]\ntrain_latest = train.loc[latest_index]\n\nfor i, group in enumerate(derived_feature_groups, 1):\n    plt.figure(figsize=(15, 6))\n    for feat in group:\n        if feat in train.columns:\n            plt.plot(latest_index, train_latest[feat], label=feat, alpha=0.7, linewidth=1)\n    \n    # label 함께 시각화\n    if 'label' in train.columns:\n        plt.plot(latest_index, train_latest['label'] , color='red', alpha=0.4, linewidth=1.5, label='label x10')\n    \n    plt.title(f'Derived Feature Group {i} - Last 10,000')\n    plt.xlabel('Time')\n    plt.ylabel('Value')\n    plt.legend(loc='upper right', fontsize=9)\n    plt.grid(True)\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T02:56:38.421434Z","iopub.execute_input":"2025-06-12T02:56:38.421965Z","iopub.status.idle":"2025-06-12T02:56:43.822175Z","shell.execute_reply.started":"2025-06-12T02:56:38.421933Z","shell.execute_reply":"2025-06-12T02:56:43.821084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nRMV = [\"label\"]\nFEATURES = [c for c in train.columns if c not in RMV]\nprint(f\"There are {len(FEATURES)} FEATURES: {FEATURES}\")\n\n# Define cross-validation\nFOLDS = 5\nkf = KFold(n_splits=FOLDS, shuffle=True, random_state=42)\n\n# XGBoost parameters (same for all models)\nxgb_params = {\n    \"tree_method\": \"gpu_hist\",\n    \"colsample_bylevel\": 0.4778015829774066,\n    \"colsample_bynode\": 0.362764358742407,\n    \"colsample_bytree\": 0.7107423488010493,\n    \"gamma\": 1.7094857725240398,\n    \"learning_rate\": 0.02213323588455387,\n    \"max_depth\": 20,\n    \"max_leaves\": 12,\n    \"min_child_weight\": 16,\n    \"n_estimators\": 2000,\n    \"n_jobs\": -1,\n    \"random_state\": 42,\n    \"reg_alpha\": 39.352415706891264,\n    \"reg_lambda\": 75.44843704068275,\n    \"subsample\": 0.05,\n    \"verbosity\": 0\n}\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T01:17:31.151266Z","iopub.execute_input":"2025-06-12T01:17:31.151618Z","iopub.status.idle":"2025-06-12T01:17:31.161204Z","shell.execute_reply.started":"2025-06-12T01:17:31.151592Z","shell.execute_reply":"2025-06-12T01:17:31.160036Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_params = {\n    \"boosting_type\": \"gbdt\",\n    \"colsample_bytree\": 0.5625888953382505,\n    \"learning_rate\": 0.03,\n    \"min_child_samples\": 63,\n    \"min_child_weight\": 0.11456572852335424,\n    \"n_estimators\": 300,\n    \"n_jobs\": -1,\n    \"num_leaves\": 37,\n    \"random_state\": 42,\n    \"reg_alpha\": 85.2476527854083,\n    \"reg_lambda\": 99.38305361388907,\n    \"subsample\": 0.45,\n    \"verbose\": -1\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T01:17:31.162396Z","iopub.execute_input":"2025-06-12T01:17:31.162908Z","iopub.status.idle":"2025-06-12T01:17:31.185315Z","shell.execute_reply.started":"2025-06-12T01:17:31.162879Z","shell.execute_reply":"2025-06-12T01:17:31.183834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize predictions for all three models\noof_preds_model1 = np.zeros(len(train))\ntest_preds_model1 = np.zeros(len(test))\noof_preds_model2 = np.zeros(len(train))\ntest_preds_model2 = np.zeros(len(test))\noof_preds_model3 = np.zeros(len(train))  \ntest_preds_model3 = np.zeros(len(test))   \noof_preds_model3_lb = np.zeros(len(train))  # NEW: Model 3 predictions\ntest_preds_model3_lb = np.zeros(len(test))   # NEW: Model 3 predictions\n\n\n# Generate sample weights for Model 1 (full data)\nsample_weights_full = create_time_weights(len(train), decay_factor=0.98)\nprint(f\"\\nModel 1 - Full data sample weights range: [{sample_weights_full.min():.4f}, {sample_weights_full.max():.4f}]\")\nprint(f\"Model 1 - Full data sample weights mean: {sample_weights_full.mean():.4f}\")\n\n# Calculate the cutoff for 75% most recent data\ncutoff_idx_75 = int(len(train) * 0.25)\nprint(f\"\\nModel 2 - Using most recent {len(train) - cutoff_idx_75} samples (75% of data)\")\n\n# Calculate the cutoff for 50% most recent data\ncutoff_idx_50 = int(len(train) * 0.50)  # NEW: 50% cutoff\nprint(f\"\\nModel 3 - Using most recent {len(train) - cutoff_idx_50} samples (50% of data)\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T01:17:31.186671Z","iopub.execute_input":"2025-06-12T01:17:31.187077Z","iopub.status.idle":"2025-06-12T01:17:31.242512Z","shell.execute_reply.started":"2025-06-12T01:17:31.187051Z","shell.execute_reply":"2025-06-12T01:17:31.241136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Cross-validation loop\nfor i, (train_idx, valid_idx) in enumerate(kf.split(train)):\n    print(\"\\n\" + \"#\" * 50)\n    print(f\"### Fold {i + 1}\")\n    print(\"#\" * 50)\n    \n    # ========== MODEL 1: FULL DATA WITH TIME WEIGHTS ==========\n    print(\"\\n--- Model 1: Full Data with Time Weights ---\")\n    \n    X_train_m1 = train.iloc[train_idx][FEATURES]\n    y_train_m1 = train.iloc[train_idx][\"label\"]\n    X_valid = train.iloc[valid_idx][FEATURES]\n    y_valid = train.iloc[valid_idx][\"label\"]\n    X_test = test[FEATURES]\n    \n    # Extract sample weights for this fold's training data\n    train_weights_m1 = sample_weights_full[train_idx]\n    \n    model1 = XGBRegressor(**xgb_params)\n    model1.fit(\n        X_train_m1, y_train_m1,\n        sample_weight=train_weights_m1,\n        eval_set=[(X_valid, y_valid)],\n        early_stopping_rounds=25,\n        verbose=200\n    )\n    \n    oof_preds_model1[valid_idx] = model1.predict(X_valid)\n    test_preds_model1 += model1.predict(X_test)\n    \n    # ========== MODEL 2: 75% MOST RECENT DATA ==========\n    print(\"\\n--- Model 2: 75% Most Recent Data ---\")\n    \n    # Filter train indices to only include those from the recent 75% of data\n    train_idx_recent_75 = train_idx[train_idx >= cutoff_idx_75]\n    \n    # Adjust indices to start from 0 for the recent subset\n    train_idx_recent_adjusted_75 = train_idx_recent_75 - cutoff_idx_75\n    \n    # Get the recent subset of training data\n    train_recent_75 = train.iloc[cutoff_idx_75:].reset_index(drop=True)\n    \n    X_train_m2 = train_recent_75.iloc[train_idx_recent_adjusted_75][FEATURES]\n    y_train_m2 = train_recent_75.iloc[train_idx_recent_adjusted_75][\"label\"]\n    \n    # Create time weights for the recent data subset\n    sample_weights_recent_75 = create_time_weights(len(train_recent_75), decay_factor=0.98)\n    train_weights_m2 = sample_weights_recent_75[train_idx_recent_adjusted_75]\n    \n    model2 = XGBRegressor(**xgb_params)\n    model2.fit(\n        X_train_m2, y_train_m2,\n        sample_weight=train_weights_m2,\n        eval_set=[(X_valid, y_valid)],\n        early_stopping_rounds=25,\n        verbose=200\n    )\n    \n    # For validation predictions, we need to handle cases where validation indices\n    # might be from the older 25% of data\n    valid_idx_in_range_75 = valid_idx[valid_idx >= cutoff_idx_75]\n    if len(valid_idx_in_range_75) > 0:\n        X_valid_m2 = train.iloc[valid_idx_in_range_75][FEATURES]\n        oof_preds_model2[valid_idx_in_range_75] = model2.predict(X_valid_m2)\n    \n    # For indices before cutoff, use Model 1 predictions\n    valid_idx_out_range_75 = valid_idx[valid_idx < cutoff_idx_75]\n    if len(valid_idx_out_range_75) > 0:\n        oof_preds_model2[valid_idx_out_range_75] = oof_preds_model1[valid_idx_out_range_75]\n    \n    test_preds_model2 += model2.predict(X_test)\n    \n    # ========== MODEL 3: 50% MOST RECENT DATA ========== NEW\n    print(\"\\n--- Model 3: 50% Most Recent Data ---\")\n    \n    # Filter train indices to only include those from the recent 50% of data\n    train_idx_recent_50 = train_idx[train_idx >= cutoff_idx_50]\n    \n    # Adjust indices to start from 0 for the recent subset\n    train_idx_recent_adjusted_50 = train_idx_recent_50 - cutoff_idx_50\n    \n    # Get the recent subset of training data\n    train_recent_50 = train.iloc[cutoff_idx_50:].reset_index(drop=True)\n    \n    X_train_m3 = train_recent_50.iloc[train_idx_recent_adjusted_50][FEATURES]\n    y_train_m3 = train_recent_50.iloc[train_idx_recent_adjusted_50][\"label\"]\n    \n    # Create time weights for the recent data subset\n    sample_weights_recent_50 = create_time_weights(len(train_recent_50), decay_factor=0.98)\n    train_weights_m3 = sample_weights_recent_50[train_idx_recent_adjusted_50]\n    \n    model3 = XGBRegressor(**xgb_params)\n    model3.fit(\n        X_train_m3, y_train_m3,\n        sample_weight=train_weights_m3,\n        eval_set=[(X_valid, y_valid)],\n        early_stopping_rounds=25,\n        verbose=200\n    )\n    \n    # For validation predictions, we need to handle cases where validation indices\n    # might be from the older 50% of data\n    valid_idx_in_range_50 = valid_idx[valid_idx >= cutoff_idx_50]\n    if len(valid_idx_in_range_50) > 0:\n        X_valid_m3 = train.iloc[valid_idx_in_range_50][FEATURES]\n        oof_preds_model3[valid_idx_in_range_50] = model3.predict(X_valid_m3)\n    \n    # For indices before cutoff, use Model 1 predictions\n    valid_idx_out_range_50 = valid_idx[valid_idx < cutoff_idx_50]\n    if len(valid_idx_out_range_50) > 0:\n        oof_preds_model3[valid_idx_out_range_50] = oof_preds_model1[valid_idx_out_range_50]\n    \n    test_preds_model3 += model3.predict(X_test)\n\n    # ========== MODEL 3: 50% MOST RECENT DATA  LGBM========== NEW\n    print(\"\\n--- Model 3: 50% Most Recent Data with LGBM ---\")\n    \n    model3_lb = LGBMRegressor(**lgbm_params)\n    model3_lb.fit(\n        X_train_m3, y_train_m3,\n        sample_weight=train_weights_m3,\n        eval_set=[(X_valid, y_valid)]\n    )\n    \n    # For validation predictions, we need to handle cases where validation indices\n    # might be from the older 50% of data\n    valid_idx_in_range_50 = valid_idx[valid_idx >= cutoff_idx_50]\n    if len(valid_idx_in_range_50) > 0:\n        X_valid_m3 = train.iloc[valid_idx_in_range_50][FEATURES]\n        oof_preds_model3_lb[valid_idx_in_range_50] = model3_lb.predict(X_valid_m3)\n    \n    # For indices before cutoff, use Model 1 predictions\n    valid_idx_out_range_50 = valid_idx[valid_idx < cutoff_idx_50]\n    if len(valid_idx_out_range_50) > 0:\n        oof_preds_model3_lb[valid_idx_out_range_50] = oof_preds_model1[valid_idx_out_range_50]\n    \n    test_preds_model3_lb += model3_lb.predict(X_test)\n\n# Average test predictions across folds\ntest_preds_model1 /= FOLDS\ntest_preds_model2 /= FOLDS\ntest_preds_model3 /= FOLDS  # NEW\ntest_preds_model3_lb /= FOLDS  # NEW\n\n# Calculate individual model scores\npearson_score_model1 = pearsonr(train[\"label\"], oof_preds_model1)[0]\npearson_score_model2 = pearsonr(train[\"label\"], oof_preds_model2)[0]\npearson_score_model3 = pearsonr(train[\"label\"], oof_preds_model3)[0]  \npearson_score_model3_lb = pearsonr(train[\"label\"], oof_preds_model3_lb)[0]  # NEW\n\nprint(\"\\n\" + \"=\" * 50)\nprint(\"INDIVIDUAL MODEL PERFORMANCE\")\nprint(\"=\" * 50)\nprint(f\"Model 1 (Full Data) Pearson Correlation: {pearson_score_model1:.4f}\")\nprint(f\"Model 2 (75% Recent) Pearson Correlation: {pearson_score_model2:.4f}\")\nprint(f\"Model 3 XB (50% Recent) Pearson Correlation: {pearson_score_model3:.4f}\")  # NEW\nprint(f\"Model 3 LB (50% Recent) Pearson Correlation: {pearson_score_model3_lb:.4f}\")  # NEW\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T01:17:31.243691Z","iopub.execute_input":"2025-06-12T01:17:31.244079Z","iopub.status.idle":"2025-06-12T01:17:36.920508Z","shell.execute_reply.started":"2025-06-12T01:17:31.244052Z","shell.execute_reply":"2025-06-12T01:17:36.918578Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\n# inf, -inf 존재 여부 확인\nprint(np.isinf(X_test).sum())  # inf 값 개수 출력\n\n# inf를 nan으로 바꾸기\nX_test = X_test.replace([np.inf, -np.inf], np.nan)\n\n# nan을 0 또는 다른 값으로 채우기 (필요에 따라 다름)\nX_test = X_test.fillna(0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T01:17:36.921575Z","iopub.status.idle":"2025-06-12T01:17:36.921974Z","shell.execute_reply.started":"2025-06-12T01:17:36.921802Z","shell.execute_reply":"2025-06-12T01:17:36.921819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create ensemble predictions\n# Simple average ensemble (now with 3 models)\nensemble_oof_preds = (oof_preds_model1 + oof_preds_model2 + oof_preds_model3 + oof_preds_model3_lb) / 4  # UPDATED\nensemble_test_preds = (test_preds_model1 + test_preds_model2 + test_preds_model3 + test_preds_model3_lb) / 4  # UPDATED\n\n# Calculate ensemble score\nensemble_pearson_score = pearsonr(train[\"label\"], ensemble_oof_preds)[0]\n\nprint(\"\\n\" + \"=\" * 50)\nprint(\"ENSEMBLE PERFORMANCE\")\nprint(\"=\" * 50)\nprint(f\"Ensemble (Equal Weight) Pearson Correlation: {ensemble_pearson_score:.4f}\")\n\n# Performance-weighted ensemble (now with 3 models)\ntotal_score = pearson_score_model1 + pearson_score_model2 + pearson_score_model3+ pearson_score_model3_lb  # UPDATED\nweight_model1 = pearson_score_model1 / total_score  # UPDATED\nweight_model2 = pearson_score_model2 / total_score  # UPDATED\nweight_model3 = pearson_score_model3 / total_score  # NEW\nweight_model3_lb = pearson_score_model3_lb / total_score  # NEW\n\nweighted_ensemble_oof = (weight_model1 * oof_preds_model1 + \n                        weight_model2 * oof_preds_model2 + \n                        weight_model3 * oof_preds_model3 + \n                        weight_model3_lb * oof_preds_model3_lb)  # UPDATED\nweighted_ensemble_test = (weight_model1 * test_preds_model1 + \n                         weight_model2 * test_preds_model2 + \n                         weight_model3 * test_preds_model3  + \n                         weight_model3_lb * test_preds_model3_lb)  # UPDATED\n\nweighted_ensemble_score = pearsonr(train[\"label\"], weighted_ensemble_oof)[0]\n\nprint(f\"\\nWeighted Ensemble Performance:\")\nprint(f\"  Model 1 weight: {weight_model1:.3f}\")\nprint(f\"  Model 2 weight: {weight_model2:.3f}\")\nprint(f\"  Model 3 weight: {weight_model3:.3f}\")  # NEW\nprint(f\"  Model 3 weight: {weight_model3_lb:.3f}\")  # NEW\nprint(f\"  Weighted Ensemble Pearson Correlation: {weighted_ensemble_score:.4f}\")\n\n# Use the better ensemble for final predictions\nif weighted_ensemble_score > ensemble_pearson_score:\n    final_test_preds = weighted_ensemble_test\n    print(\"\\nUsing weighted ensemble for final predictions\")\nelse:\n    final_test_preds = ensemble_test_preds\n    print(\"\\nUsing simple average ensemble for final predictions\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T01:17:36.923764Z","iopub.status.idle":"2025-06-12T01:17:36.924572Z","shell.execute_reply.started":"2025-06-12T01:17:36.924311Z","shell.execute_reply":"2025-06-12T01:17:36.924336Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# SHAP analysis (using Model 1 as representative)\nprint(\"\\nGenerating SHAP analysis for model 1...\")\nexplainer = shap.TreeExplainer(model1, feature_perturbation=\"tree_path_dependent\", model_output=\"raw\")\nshap_values = explainer.shap_values(X_test)\nshap.summary_plot(shap_values, X_test, max_display=100)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T01:17:36.926434Z","iopub.status.idle":"2025-06-12T01:17:36.926877Z","shell.execute_reply.started":"2025-06-12T01:17:36.926668Z","shell.execute_reply":"2025-06-12T01:17:36.926690Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# SHAP analysis (using Model 1 as representative)\nprint(\"\\nGenerating SHAP analysis for model 2...\")\nexplainer = shap.TreeExplainer(model2, feature_perturbation=\"tree_path_dependent\", model_output=\"raw\")\nshap_values = explainer.shap_values(X_test)\nshap.summary_plot(shap_values, X_test, max_display=100)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T01:17:36.927936Z","iopub.status.idle":"2025-06-12T01:17:36.928240Z","shell.execute_reply.started":"2025-06-12T01:17:36.928109Z","shell.execute_reply":"2025-06-12T01:17:36.928121Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save predictions\nsample[\"prediction\"] = final_test_preds\nsample.to_csv(\"submission.csv\", index=False)\nprint(\"\\nPredictions saved to submission.csv\")\nprint(sample.head())\n\n# Save detailed results (now with 3 models)\nensemble_results = pd.DataFrame({\n    'model': ['Model 1 (Full Data)', 'Model 2 (75% Recent)', 'Model 3 (50% Recent)', \n              'Simple Ensemble', 'Weighted Ensemble'],  # UPDATED\n    'pearson_correlation': [pearson_score_model1, pearson_score_model2, pearson_score_model3, \n                           ensemble_pearson_score, weighted_ensemble_score],  # UPDATED\n    'weight_in_final': [weight_model1 if weighted_ensemble_score > ensemble_pearson_score else 1/3,\n                        weight_model2 if weighted_ensemble_score > ensemble_pearson_score else 1/3,\n                        weight_model3 if weighted_ensemble_score > ensemble_pearson_score else 1/3,\n                        np.nan, np.nan]  # UPDATED\n})\nensemble_results.to_csv(\"ensemble_results.csv\", index=False)\nprint(\"\\nEnsemble results saved to ensemble_results.csv\")\nprint(ensemble_results)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-12T01:17:36.929660Z","iopub.status.idle":"2025-06-12T01:17:36.930004Z","shell.execute_reply.started":"2025-06-12T01:17:36.929866Z","shell.execute_reply":"2025-06-12T01:17:36.929879Z"}},"outputs":[],"execution_count":null}]}