{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.17","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport lightgbm as lgb\nfrom sklearn.model_selection import TimeSeriesSplit\nfrom sklearn.decomposition import PCA\nimport gc\n\n# --- Configuration ---\nTRAIN_FILE = '/kaggle/input/drw-crypto-market-prediction/train.parquet'\nN_SPLITS = 5\n\n# --- LightGBM Parameters (from your code) ---\nlgb_params_baseline = {\n    'objective': 'regression_l1', 'metric': 'mae', 'n_estimators': 1500,\n    'learning_rate': 0.015, 'feature_fraction': 0.7, 'bagging_fraction': 0.7,\n    'bagging_freq': 1, 'lambda_l1': 0.1, 'lambda_l2': 0.1, 'num_leaves': 31,\n    'verbose': -1, 'n_jobs': -1, 'seed': 42, 'boosting_type': 'gbdt',\n}\n\n# --- Utility Functions ---\ndef pearson_correlation(y_true, y_pred):\n    y_true = np.nan_to_num(y_true);\n    y_pred = np.nan_to_num(y_pred)\n    if np.std(y_true) < 1e-9 or np.std(y_pred) < 1e-9 or len(y_true) < 2: return 0.0\n    try: corr = np.corrcoef(y_true, y_pred)[0, 1]; return np.nan_to_num(corr)\n    except Exception: return 0.0\n\n# --- 1. Load Data (DRW Only) ---\nprint(\"Loading DRW data ONLY for baseline...\")\ntrain_df_orig = pd.read_parquet(TRAIN_FILE) # Keep original for y\n\n# Prepare train_df for feature engineering\ntrain_df = train_df_orig.copy()\nif isinstance(train_df.index, pd.DatetimeIndex) or train_df.index.name == 'timestamp':\n    train_df = train_df.reset_index(drop=True)\nelif 'timestamp' in train_df.columns:\n    train_df = train_df.drop(columns=['timestamp'])\n\ntms = TimeSeriesSplit(n_splits=N_SPLITS)\nprint(f\"Train shape for FE: {train_df.shape}\")\n\n# --- 2. Feature Engineering (Your ORIGINAL feature_engineer for 0.053 score) ---\nprint(\"Starting feature engineering (original DRW-only function)...\")\n# proprietary_features defined here, based on current train_df (after timestamp drop)\nproprietary_features = [col for col in train_df.columns if (col.startswith('X') or col.startswith('x')) and col != 'label']\nif not proprietary_features:\n    print(\"Warning: No proprietary features found in train_df columns.\")\nelse:\n    print(f\"Found {len(proprietary_features)} proprietary features in train_df.\")\ntrain_medians_for_X = {col: train_df[col].median() for col in proprietary_features if col in train_df.columns}\n\n# Using your provided feature_engineer function structure\ndef feature_engineer_original_drw(df_input, medians_dict, list_of_prop_features_arg): # Renamed arg\n    df_fe = df_input.copy() # Work on a copy\n    \n    # Imbalance features\n    # Ensure base columns exist before creating derived ones\n    if all(c in df_fe.columns for c in ['bid_qty', 'ask_qty']):\n        df_fe['bid_ask_qty_imbalance'] = (df_fe['bid_qty'] - df_fe['ask_qty']) / (df_fe['bid_qty'] + df_fe['ask_qty'] + 1e-6)\n        df_fe['bid_ask_ratio'] = df_fe['bid_qty'] / (df_fe['ask_qty'] + 1e-6)\n    else: # Create dummy if base cols missing, so subsequent FE doesn't fail\n        df_fe['bid_ask_qty_imbalance'] = 0\n        df_fe['bid_ask_ratio'] = 0\n        \n    if all(c in df_fe.columns for c in ['buy_qty', 'sell_qty']):\n        df_fe['buy_sell_qty_imbalance'] = (df_fe['buy_qty'] - df_fe['sell_qty']) / (df_fe['buy_qty'] + df_fe['sell_qty'] + 1e-6)\n    else:\n        df_fe['buy_sell_qty_imbalance'] = 0\n\n    # Lags\n    lags_orig = [1, 2, 3, 5, 10]\n    features_to_lag_orig = ['volume', 'bid_qty', 'ask_qty', 'bid_ask_qty_imbalance', 'buy_sell_qty_imbalance'] # Matching your list\n    \n    for feature_name_to_lag in features_to_lag_orig:\n        if feature_name_to_lag in df_fe.columns: # Check if base feature for lag exists\n            for lag_val in lags_orig:\n                new_col_name = f'{feature_name_to_lag}_lag_{lag_val}'\n                df_fe[new_col_name] = df_fe[feature_name_to_lag].shift(lag_val)\n                df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n\n    # Rolling features for public features\n    windows_orig = [5, 10, 30]\n    # features_for_rolling_orig from your code: ['volume', 'bid_ask_qty_imbalance']\n    features_for_rolling_orig = ['volume', 'bid_ask_qty_imbalance'] \n    # If you also rolled 'buy_sell_qty_imbalance', add it here:\n    # if 'buy_sell_qty_imbalance' in df_fe.columns: features_for_rolling_orig.append('buy_sell_qty_imbalance')\n\n    for feature_name_to_roll in features_for_rolling_orig:\n        if feature_name_to_roll in df_fe.columns: # Check if base feature for rolling exists\n            for w_val in windows_orig:\n                df_fe[f'{feature_name_to_roll}_roll_mean_{w_val}'] = df_fe[feature_name_to_roll].rolling(window=w_val, min_periods=1).mean()\n                df_fe[f'{feature_name_to_roll}_roll_std_{w_val}'] = df_fe[feature_name_to_roll].rolling(window=w_val, min_periods=1).std()\n\n    # Rolling features for proprietary features (top 15 from the provided list)\n    prop_to_roll_orig = [f for f in list_of_prop_features_arg if f in df_fe.columns][:15]\n    for feature_name_prop_roll in prop_to_roll_orig:\n        for w_val in [5, 10]: # Your windows for prop features\n            df_fe[f'{feature_name_prop_roll}_roll_mean_{w_val}'] = df_fe[feature_name_prop_roll].rolling(window=w_val, min_periods=1).mean()\n            df_fe[f'{feature_name_prop_roll}_roll_std_{w_val}'] = df_fe[feature_name_prop_roll].rolling(window=w_val, min_periods=1).std()\n    \n    # Interaction features (top 8 from the provided list)\n    prop_for_interaction_orig = [f for f in list_of_prop_features_arg if f in df_fe.columns][:8]\n    for i, col_i in enumerate(prop_for_interaction_orig):\n        for col_j in prop_for_interaction_orig[i + 1:]: # Ensure col_j is also in df_fe.columns\n             if col_i in df_fe.columns and col_j in df_fe.columns:\n                df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n\n    # Fill NaNs for ALL proprietary features (not just top 15 or 8)\n    for col_prop_fill in list_of_prop_features_arg: # Iterate over the full list passed\n        if col_prop_fill in df_fe.columns and df_fe[col_prop_fill].isnull().any():\n            df_fe[col_prop_fill].fillna(medians_dict.get(col_prop_fill, 0), inplace=True)\n            \n    df_fe.fillna(0, inplace=True) # Global fill for any other NaNs (e.g. from initial shifts)\n    df_fe.replace([np.inf, -np.inf], 0, inplace=True)\n    return df_fe\n\ntrain_df_fe = feature_engineer_original_drw(train_df.copy(), train_medians_for_X, proprietary_features)\n# `proprietary_features` here is the list from train_df (before FE).\n# It's passed to the function and used for specific X_ feature processing.\ngc.collect()\n\n# --- Define Features X and Target y ---\n# 'label' should not be in features list.\n# 'timestamp' was already dropped from train_df.\nfeatures = [col for col in train_df_fe.columns if col != 'label']\nX = train_df_fe[features].copy()\ny = train_df_orig['label'].copy() # Use label from original DF to ensure it's clean\n\n# Align X and y. train_df_orig and train_df_fe (and thus X) should have same index if FE doesn't change rows.\n# If train_df had timestamp dropped, its index is RangeIndex. train_df_orig might have DatetimeIndex.\n# Let's ensure y is aligned to X's index (which is likely RangeIndex now)\nif not X.index.equals(y.index):\n    print(\"Aligning y to X's index.\")\n    # If X's index is a subset of y's original index (e.g. if y came from train_df_orig with DatetimeIndex)\n    # this alignment might be tricky if FE changed row count/order significantly.\n    # Assuming FE preserves row order and count from the input `train_df` (which had timestamp dropped).\n    # `train_df` and `train_df_orig` will have same number of rows if no NaN drop on label yet.\n    y = y.iloc[X.index] # This assumes X.index are valid iloc positions for y if y is from train_df_orig.\n                       # Safer: y = train_df_orig['label'].loc[train_df_orig.index[X.index]].copy() if X.index maps to original index\n                       # Simplest if FE doesn't change row order/count:\n    if len(y) == len(X):\n        y.index = X.index\n    else: # If lengths differ, need more careful alignment or check FE\n        print(f\"WARNING: Length mismatch after FE. X:{len(X)}, y_orig:{len(train_df_orig['label'])}. Trying common index.\")\n        common_idx = X.index.intersection(train_df_orig.index)\n        X = X.loc[common_idx]\n        y = train_df_orig['label'].loc[common_idx]\n\n\nprint(f\"Shape X after FE: {X.shape}, y: {y.shape}\")\nif X.shape[1] == 0: print(\"ERROR: X is empty after FE! Check feature_engineer function.\"); exit()\nif len(X) != len(y): print(f\"CRITICAL ERROR: X and y length mismatch! X:{len(X)}, y:{len(y)}\"); exit()\n\n\n# --- PCA and Feature Selection (Your ORIGINAL logic from 0.053 script) ---\n# For PCA, use proprietary features that are present in the current X\npca_prop_features_in_X = [f for f in proprietary_features if f in X.columns]\nprint(f\"Applying PCA on {len(pca_prop_features_in_X)} original DRW proprietary features found in X.\")\n\nif pca_prop_features_in_X and len(pca_prop_features_in_X) >=30:\n    temp_model_pca_baseline = lgb.LGBMRegressor(**lgb_params_baseline)\n    temp_model_pca_baseline.fit(X, y)\n    importances_pca_baseline = pd.Series(temp_model_pca_baseline.feature_importances_, index=X.columns)\n    \n    # Select top_prop_features from pca_prop_features_in_X based on importance\n    top_prop_features_for_pca_logic = importances_pca_baseline[importances_pca_baseline.index.isin(pca_prop_features_in_X)].nlargest(20).index.tolist()\n    # Ensure these selected features actually exist (they should)\n    top_prop_features_for_pca_logic = [f for f in top_prop_features_for_pca_logic if f in X.columns]\n\n\n    pca = PCA(n_components=30, random_state=42)\n    X_pca_train = pca.fit_transform(X[pca_prop_features_in_X]) # Fit PCA on the identified prop features in X\n    X_pca_train_df = pd.DataFrame(X_pca_train, columns=[f'PCA_{i}' for i in range(30)], index=X.index)\n    \n    # Features to drop: those in pca_prop_features_in_X that are NOT in top_prop_features_for_pca_logic\n    cols_to_drop_for_pca_baseline = [f for f in pca_prop_features_in_X if f not in top_prop_features_for_pca_logic]\n    \n    X_after_pca_drops = X.drop(columns=cols_to_drop_for_pca_baseline, errors='ignore')\n    # Concatenate remaining X features with PCA components.\n    # The top_prop_features_for_pca_logic are ALREADY in X_after_pca_drops if not dropped.\n    X = pd.concat([X_after_pca_drops, X_pca_train_df], axis=1)\n    print(\"PCA applied and combined.\")\nelse:\n    print(f\"Skipping PCA: Found {len(pca_prop_features_in_X)} prop features for PCA in X (need >=30).\")\ngc.collect()\n\nprint(\"Starting feature selection (Quantile based)...\")\nfs_model_baseline = lgb.LGBMRegressor(**lgb_params_baseline)\nfs_model_baseline.fit(X, y)\nimportances_fs_baseline = pd.Series(fs_model_baseline.feature_importances_, index=X.columns)\n# Your original quantile selection\nselected_features_final = importances_fs_baseline[importances_fs_baseline > importances_fs_baseline.quantile(0.15)].index.tolist()\nif not selected_features_final : selected_features_final = X.columns.tolist() # Fallback\nX = X[selected_features_final]\nprint(f\"Number of features after selection: {len(X.columns)}\")\ngc.collect()\n\nif X.shape[1] == 0: print(\"ERROR: X is empty after FS!\"); exit()\n\n# --- 3. Model Training (LGBM Only) ---\nprint(\"\\nStarting LGBM model training (DRW-Only Baseline from 0.053 code)...\")\noof_predictions_lgbm = np.zeros(len(X)) # Sized to current X\noof_val_indices_collector = [] # To collect actual indices of validation folds\n\nfor fold, (train_idx_iloc, val_idx_iloc) in enumerate(tms.split(X)): # tms splits X by row order\n    # Collect actual DataFrame indices for OOF evaluation\n    current_val_indices = X.iloc[val_idx_iloc].index \n    oof_val_indices_collector.extend(current_val_indices.tolist())\n\n    X_train, X_val = X.iloc[train_idx_iloc], X.iloc[val_idx_iloc]\n    y_train, y_val = y.iloc[train_idx_iloc], y.iloc[val_idx_iloc] # y should be aligned with X here\n    \n    model = lgb.LGBMRegressor(**lgb_params_baseline) # Using your original params\n    model.fit(X_train, y_train, eval_set=[(X_val, y_val)],\n              eval_metric=lambda yt, yp: [('pearson_corr', pearson_correlation(yt, yp), True)],\n              callbacks=[lgb.early_stopping(100, verbose=False)]) # verbose=False as per your original\n    val_preds = model.predict(X_val)\n    oof_predictions_lgbm[val_idx_iloc] = val_preds # Store using iloc indices\n    fold_pearson = pearson_correlation(y_val, val_preds)\n    print(f\"LGBM Fold {fold+1} Pearson Correlation: {fold_pearson:.4f}\")\ngc.collect()\n\n# More robust OOF score calculation\nunique_oof_indices = sorted(list(set(oof_val_indices_collector)))\nif unique_oof_indices:\n    y_true_oof = y.loc[unique_oof_indices] # Select true y values for these OOF indices\n    oof_preds_iloc_indices = X.index.get_indexer(unique_oof_indices)\n    valid_iloc_indices_for_oof_array = oof_preds_iloc_indices[oof_preds_iloc_indices != -1]\n    \n    if len(valid_iloc_indices_for_oof_array) == len(y_true_oof) and len(valid_iloc_indices_for_oof_array) > 0:\n        oof_lgbm_actual_values = oof_predictions_lgbm[valid_iloc_indices_for_oof_array]\n        overall_oof_pearson_lgbm = pearson_correlation(y_true_oof, oof_lgbm_actual_values)\n        print(f\"\\nOverall LGBM OOF Pearson (DRW-Only Baseline - Original FE): {overall_oof_pearson_lgbm:.4f}\")\n    else:\n        print(f\"LGBM OOF (DRW-Only Baseline) alignment issue: y_true_len={len(y_true_oof)}, selected_oof_len={len(valid_iloc_indices_for_oof_array)}\")\nelse:\n    print(\"No OOF predictions were generated for LGBM baseline.\")\n\nprint(\"\\nDRW-Only Baseline (Original FE) script finished.\")\n# No CatBoost, no ensembling, no submission for this specific baseline validation.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-14T10:09:23.882954Z","iopub.execute_input":"2025-06-14T10:09:23.883232Z","iopub.status.idle":"2025-06-14T10:14:47.755933Z","shell.execute_reply.started":"2025-06-14T10:09:23.883208Z","shell.execute_reply":"2025-06-14T10:14:47.750032Z"}},"outputs":[{"name":"stdout","text":"Loading DRW data ONLY for baseline...\nTrain shape for FE: (525887, 896)\nStarting feature engineering (original DRW-only function)...\nFound 890 proprietary features in train_df.\n","output_type":"stream"},{"name":"stderr","text":"/tmp/ipykernel_367/2467504299.py:78: FutureWarning: A value is trying to be set on a copy of a DataFrame or Series through chained assignment using an inplace method.\nThe behavior will change in pandas 3.0. This inplace method will never work because the intermediate object on which we are setting values always behaves as a copy.\n\nFor example, when doing 'df[col].method(value, inplace=True)', try using 'df.method({col: value}, inplace=True)' or df[col] = df[col].method(value) instead, to perform the operation inplace on the original object.\n\n\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: Series.fillna with 'method' is deprecated and will raise in a future version. Use obj.ffill() or obj.bfill() instead.\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: A value is trying to be set on a copy of a DataFrame or Series through chained assignment using an inplace method.\nThe behavior will change in pandas 3.0. This inplace method will never work because the intermediate object on which we are setting values always behaves as a copy.\n\nFor example, when doing 'df[col].method(value, inplace=True)', try using 'df.method({col: value}, inplace=True)' or df[col] = df[col].method(value) instead, to perform the operation inplace on the original object.\n\n\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: Series.fillna with 'method' is deprecated and will raise in a future version. Use obj.ffill() or obj.bfill() instead.\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: A value is trying to be set on a copy of a DataFrame or Series through chained assignment using an inplace method.\nThe behavior will change in pandas 3.0. This inplace method will never work because the intermediate object on which we are setting values always behaves as a copy.\n\nFor example, when doing 'df[col].method(value, inplace=True)', try using 'df.method({col: value}, inplace=True)' or df[col] = df[col].method(value) instead, to perform the operation inplace on the original object.\n\n\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: Series.fillna with 'method' is deprecated and will raise in a future version. Use obj.ffill() or obj.bfill() instead.\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: A value is trying to be set on a copy of a DataFrame or Series through chained assignment using an inplace method.\nThe behavior will change in pandas 3.0. This inplace method will never work because the intermediate object on which we are setting values always behaves as a copy.\n\nFor example, when doing 'df[col].method(value, inplace=True)', try using 'df.method({col: value}, inplace=True)' or df[col] = df[col].method(value) instead, to perform the operation inplace on the original object.\n\n\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: Series.fillna with 'method' is deprecated and will raise in a future version. Use obj.ffill() or obj.bfill() instead.\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: A value is trying to be set on a copy of a DataFrame or Series through chained assignment using an inplace method.\nThe behavior will change in pandas 3.0. This inplace method will never work because the intermediate object on which we are setting values always behaves as a copy.\n\nFor example, when doing 'df[col].method(value, inplace=True)', try using 'df.method({col: value}, inplace=True)' or df[col] = df[col].method(value) instead, to perform the operation inplace on the original object.\n\n\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: Series.fillna with 'method' is deprecated and will raise in a future version. Use obj.ffill() or obj.bfill() instead.\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: A value is trying to be set on a copy of a DataFrame or Series through chained assignment using an inplace method.\nThe behavior will change in pandas 3.0. This inplace method will never work because the intermediate object on which we are setting values always behaves as a copy.\n\nFor example, when doing 'df[col].method(value, inplace=True)', try using 'df.method({col: value}, inplace=True)' or df[col] = df[col].method(value) instead, to perform the operation inplace on the original object.\n\n\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: Series.fillna with 'method' is deprecated and will raise in a future version. Use obj.ffill() or obj.bfill() instead.\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: A value is trying to be set on a copy of a DataFrame or Series through chained assignment using an inplace method.\nThe behavior will change in pandas 3.0. This inplace method will never work because the intermediate object on which we are setting values always behaves as a copy.\n\nFor example, when doing 'df[col].method(value, inplace=True)', try using 'df.method({col: value}, inplace=True)' or df[col] = df[col].method(value) instead, to perform the operation inplace on the original object.\n\n\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: Series.fillna with 'method' is deprecated and will raise in a future version. Use obj.ffill() or obj.bfill() instead.\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: A value is trying to be set on a copy of a DataFrame or Series through chained assignment using an inplace method.\nThe behavior will change in pandas 3.0. This inplace method will never work because the intermediate object on which we are setting values always behaves as a copy.\n\nFor example, when doing 'df[col].method(value, inplace=True)', try using 'df.method({col: value}, inplace=True)' or df[col] = df[col].method(value) instead, to perform the operation inplace on the original object.\n\n\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: Series.fillna with 'method' is deprecated and will raise in a future version. Use obj.ffill() or obj.bfill() instead.\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: A value is trying to be set on a copy of a DataFrame or Series through chained assignment using an inplace method.\nThe behavior will change in pandas 3.0. This inplace method will never work because the intermediate object on which we are setting values always behaves as a copy.\n\nFor example, when doing 'df[col].method(value, inplace=True)', try using 'df.method({col: value}, inplace=True)' or df[col] = df[col].method(value) instead, to perform the operation inplace on the original object.\n\n\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: Series.fillna with 'method' is deprecated and will raise in a future version. Use obj.ffill() or obj.bfill() instead.\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: A value is trying to be set on a copy of a DataFrame or Series through chained assignment using an inplace method.\nThe behavior will change in pandas 3.0. This inplace method will never work because the intermediate object on which we are setting values always behaves as a copy.\n\nFor example, when doing 'df[col].method(value, inplace=True)', try using 'df.method({col: value}, inplace=True)' or df[col] = df[col].method(value) instead, to perform the operation inplace on the original object.\n\n\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: Series.fillna with 'method' is deprecated and will raise in a future version. Use obj.ffill() or obj.bfill() instead.\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: A value is trying to be set on a copy of a DataFrame or Series through chained assignment using an inplace method.\nThe behavior will change in pandas 3.0. This inplace method will never work because the intermediate object on which we are setting values always behaves as a copy.\n\nFor example, when doing 'df[col].method(value, inplace=True)', try using 'df.method({col: value}, inplace=True)' or df[col] = df[col].method(value) instead, to perform the operation inplace on the original object.\n\n\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: Series.fillna with 'method' is deprecated and will raise in a future version. Use obj.ffill() or obj.bfill() instead.\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: A value is trying to be set on a copy of a DataFrame or Series through chained assignment using an inplace method.\nThe behavior will change in pandas 3.0. This inplace method will never work because the intermediate object on which we are setting values always behaves as a copy.\n\nFor example, when doing 'df[col].method(value, inplace=True)', try using 'df.method({col: value}, inplace=True)' or df[col] = df[col].method(value) instead, to perform the operation inplace on the original object.\n\n\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: Series.fillna with 'method' is deprecated and will raise in a future version. Use obj.ffill() or obj.bfill() instead.\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: A value is trying to be set on a copy of a DataFrame or Series through chained assignment using an inplace method.\nThe behavior will change in pandas 3.0. This inplace method will never work because the intermediate object on which we are setting values always behaves as a copy.\n\nFor example, when doing 'df[col].method(value, inplace=True)', try using 'df.method({col: value}, inplace=True)' or df[col] = df[col].method(value) instead, to perform the operation inplace on the original object.\n\n\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: Series.fillna with 'method' is deprecated and will raise in a future version. Use obj.ffill() or obj.bfill() instead.\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: A value is trying to be set on a copy of a DataFrame or Series through chained assignment using an inplace method.\nThe behavior will change in pandas 3.0. This inplace method will never work because the intermediate object on which we are setting values always behaves as a copy.\n\nFor example, when doing 'df[col].method(value, inplace=True)', try using 'df.method({col: value}, inplace=True)' or df[col] = df[col].method(value) instead, to perform the operation inplace on the original object.\n\n\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: Series.fillna with 'method' is deprecated and will raise in a future version. Use obj.ffill() or obj.bfill() instead.\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: A value is trying to be set on a copy of a DataFrame or Series through chained assignment using an inplace method.\nThe behavior will change in pandas 3.0. This inplace method will never work because the intermediate object on which we are setting values always behaves as a copy.\n\nFor example, when doing 'df[col].method(value, inplace=True)', try using 'df.method({col: value}, inplace=True)' or df[col] = df[col].method(value) instead, to perform the operation inplace on the original object.\n\n\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: Series.fillna with 'method' is deprecated and will raise in a future version. Use obj.ffill() or obj.bfill() instead.\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: A value is trying to be set on a copy of a DataFrame or Series through chained assignment using an inplace method.\nThe behavior will change in pandas 3.0. This inplace method will never work because the intermediate object on which we are setting values always behaves as a copy.\n\nFor example, when doing 'df[col].method(value, inplace=True)', try using 'df.method({col: value}, inplace=True)' or df[col] = df[col].method(value) instead, to perform the operation inplace on the original object.\n\n\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: Series.fillna with 'method' is deprecated and will raise in a future version. Use obj.ffill() or obj.bfill() instead.\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: A value is trying to be set on a copy of a DataFrame or Series through chained assignment using an inplace method.\nThe behavior will change in pandas 3.0. This inplace method will never work because the intermediate object on which we are setting values always behaves as a copy.\n\nFor example, when doing 'df[col].method(value, inplace=True)', try using 'df.method({col: value}, inplace=True)' or df[col] = df[col].method(value) instead, to perform the operation inplace on the original object.\n\n\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: Series.fillna with 'method' is deprecated and will raise in a future version. Use obj.ffill() or obj.bfill() instead.\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: A value is trying to be set on a copy of a DataFrame or Series through chained assignment using an inplace method.\nThe behavior will change in pandas 3.0. This inplace method will never work because the intermediate object on which we are setting values always behaves as a copy.\n\nFor example, when doing 'df[col].method(value, inplace=True)', try using 'df.method({col: value}, inplace=True)' or df[col] = df[col].method(value) instead, to perform the operation inplace on the original object.\n\n\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: Series.fillna with 'method' is deprecated and will raise in a future version. Use obj.ffill() or obj.bfill() instead.\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: A value is trying to be set on a copy of a DataFrame or Series through chained assignment using an inplace method.\nThe behavior will change in pandas 3.0. This inplace method will never work because the intermediate object on which we are setting values always behaves as a copy.\n\nFor example, when doing 'df[col].method(value, inplace=True)', try using 'df.method({col: value}, inplace=True)' or df[col] = df[col].method(value) instead, to perform the operation inplace on the original object.\n\n\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: Series.fillna with 'method' is deprecated and will raise in a future version. Use obj.ffill() or obj.bfill() instead.\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: A value is trying to be set on a copy of a DataFrame or Series through chained assignment using an inplace method.\nThe behavior will change in pandas 3.0. This inplace method will never work because the intermediate object on which we are setting values always behaves as a copy.\n\nFor example, when doing 'df[col].method(value, inplace=True)', try using 'df.method({col: value}, inplace=True)' or df[col] = df[col].method(value) instead, to perform the operation inplace on the original object.\n\n\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: Series.fillna with 'method' is deprecated and will raise in a future version. Use obj.ffill() or obj.bfill() instead.\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: A value is trying to be set on a copy of a DataFrame or Series through chained assignment using an inplace method.\nThe behavior will change in pandas 3.0. This inplace method will never work because the intermediate object on which we are setting values always behaves as a copy.\n\nFor example, when doing 'df[col].method(value, inplace=True)', try using 'df.method({col: value}, inplace=True)' or df[col] = df[col].method(value) instead, to perform the operation inplace on the original object.\n\n\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: Series.fillna with 'method' is deprecated and will raise in a future version. Use obj.ffill() or obj.bfill() instead.\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: A value is trying to be set on a copy of a DataFrame or Series through chained assignment using an inplace method.\nThe behavior will change in pandas 3.0. This inplace method will never work because the intermediate object on which we are setting values always behaves as a copy.\n\nFor example, when doing 'df[col].method(value, inplace=True)', try using 'df.method({col: value}, inplace=True)' or df[col] = df[col].method(value) instead, to perform the operation inplace on the original object.\n\n\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: Series.fillna with 'method' is deprecated and will raise in a future version. Use obj.ffill() or obj.bfill() instead.\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: A value is trying to be set on a copy of a DataFrame or Series through chained assignment using an inplace method.\nThe behavior will change in pandas 3.0. This inplace method will never work because the intermediate object on which we are setting values always behaves as a copy.\n\nFor example, when doing 'df[col].method(value, inplace=True)', try using 'df.method({col: value}, inplace=True)' or df[col] = df[col].method(value) instead, to perform the operation inplace on the original object.\n\n\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: Series.fillna with 'method' is deprecated and will raise in a future version. Use obj.ffill() or obj.bfill() instead.\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: A value is trying to be set on a copy of a DataFrame or Series through chained assignment using an inplace method.\nThe behavior will change in pandas 3.0. This inplace method will never work because the intermediate object on which we are setting values always behaves as a copy.\n\nFor example, when doing 'df[col].method(value, inplace=True)', try using 'df.method({col: value}, inplace=True)' or df[col] = df[col].method(value) instead, to perform the operation inplace on the original object.\n\n\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: Series.fillna with 'method' is deprecated and will raise in a future version. Use obj.ffill() or obj.bfill() instead.\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: A value is trying to be set on a copy of a DataFrame or Series through chained assignment using an inplace method.\nThe behavior will change in pandas 3.0. This inplace method will never work because the intermediate object on which we are setting values always behaves as a copy.\n\nFor example, when doing 'df[col].method(value, inplace=True)', try using 'df.method({col: value}, inplace=True)' or df[col] = df[col].method(value) instead, to perform the operation inplace on the original object.\n\n\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:78: FutureWarning: Series.fillna with 'method' is deprecated and will raise in a future version. Use obj.ffill() or obj.bfill() instead.\n  df_fe[new_col_name].fillna(method='ffill', inplace=True) # Your ffill\n/tmp/ipykernel_367/2467504299.py:98: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{feature_name_prop_roll}_roll_std_{w_val}'] = df_fe[feature_name_prop_roll].rolling(window=w_val, min_periods=1).std()\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n/tmp/ipykernel_367/2467504299.py:105: PerformanceWarning: DataFrame is highly fragmented.  This is usually the result of calling `frame.insert` many times, which has poor performance.  Consider joining all columns at once using pd.concat(axis=1) instead. To get a de-fragmented frame, use `newframe = frame.copy()`\n  df_fe[f'{col_i}_{col_j}_product'] = df_fe[col_i] * df_fe[col_j]\n","output_type":"stream"},{"name":"stdout","text":"Aligning y to X's index.\nShape X after FE: (525887, 1023), y: (525887,)\nApplying PCA on 890 original DRW proprietary features found in X.\nPCA applied and combined.\nStarting feature selection (Quantile based)...\nNumber of features after selection: 140\n\nStarting LGBM model training (DRW-Only Baseline from 0.053 code)...\nLGBM Fold 1 Pearson Correlation: 0.0127\nLGBM Fold 2 Pearson Correlation: 0.0429\nLGBM Fold 3 Pearson Correlation: 0.0923\nLGBM Fold 4 Pearson Correlation: 0.0896\nLGBM Fold 5 Pearson Correlation: 0.0175\n\nOverall LGBM OOF Pearson (DRW-Only Baseline - Original FE): 0.0495\n\nDRW-Only Baseline (Original FE) script finished.\n","output_type":"stream"}],"execution_count":1},{"cell_type":"code","source":"pip install lightgbm","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}