{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":105399,"databundleVersionId":12733338,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ! pip install -U xgboost lightgbm tensorflow scikit-learn","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport lightgbm as lgb\nfrom sklearn.model_selection import GroupKFold\nfrom sklearn.preprocessing import LabelEncoder\nimport gc\nimport warnings\nimport pandas as pd\nimport numpy as np\nimport gc\nwarnings.filterwarnings('ignore')\npd.set_option('display.max_columns', None)\nimport numpy as np\nimport pandas as pd\nimport xgboost as xgb\nimport lightgbm as lgb\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder, OneHotEncoder\nfrom sklearn.metrics import log_loss\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -U xgboost lightgbm tensorflow scikit-learn","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reduce_mem_usage(df, verbose=True):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2\n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n    end_mem = df.memory_usage().sum() / 1024**2\n    if verbose: print(f'Mem. usage decreased to {end_mem:5.2f} Mb ({100 * (start_mem - end_mem) / start_mem:.1f}% reduction)')\n    return df\n\ndef calculate_hit_rate_at_3(df_preds_with_true_and_rank):\n    \"\"\"\n    Calculates HitRate@3.\n    df_preds_with_true_and_rank must have:\n        - 'ranker_id'\n        - 'selected' (true binary target, 1 for chosen)\n        - 'predicted_rank' (rank assigned by the model, 1 is best)\n    \"\"\"\n    hits = 0\n    valid_queries_count = 0\n    \n    for ranker_id, group in df_preds_with_true_and_rank.groupby('ranker_id'):\n        if len(group) <= 10:\n            continue  # Skip groups with 10 or fewer options as per competition rules\n        \n        valid_queries_count += 1\n        \n        true_selected_item = group[group['selected'] == 1]\n        \n        if not true_selected_item.empty:\n            # Get the rank of the true selected item\n            rank_of_true_item = true_selected_item.iloc[0]['predicted_rank']\n            if rank_of_true_item <= 3:\n                hits += 1\n        # else:\n            # This shouldn't happen in validation if data is prepared correctly from train\n            # print(f\"Warning: No selected item found for ranker_id {ranker_id} in HitRate calculation.\")\n            \n    if valid_queries_count == 0:\n        return 0.0\n    return hits / valid_queries_count","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"initial_core_columns = [\n    'Id', 'ranker_id', 'selected', 'profileId', 'companyID',\n    'requestDate', 'totalPrice', 'taxes',\n    'legs0_departureAt', 'legs0_arrivalAt', 'legs0_duration',\n    'legs1_departureAt', 'legs1_arrivalAt', 'legs1_duration',\n    'legs0_segments0_departureFrom_airport_iata', 'legs0_segments0_arrivalTo_airport_iata',\n    'legs0_segments0_marketingCarrier_code', 'legs0_segments0_cabinClass',\n    'legs0_segments0_baggageAllowance_quantity',\n    'searchRoute',\n    'pricingInfo_isAccessTP', 'pricingInfo_passengerCount',\n    'sex', 'nationality', 'isVip',\n    'miniRules0_monetaryAmount', 'miniRules0_percentage', \n    'miniRules1_monetaryAmount', 'miniRules1_percentage'\n]\ninitial_core_columns_test = [col for col in initial_core_columns if col != 'selected']\n\nprint(\"Loading a subset of columns for train_df...\")\ntrain_df = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/train.parquet', columns=initial_core_columns)\nprint(\"Loading a subset of columns for test_df...\")\ntest_df = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/test.parquet', columns=initial_core_columns_test)\nsample_submission_df = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/sample_submission.parquet')\n\nprint(\"\\nTrain DataFrame (after loading subset - BEFORE reduce_mem_usage and any FE):\")\ntrain_df.info(memory_usage='deep')\nprint(f\"\\nShape: {train_df.shape}\")\nprint(\"\\nTest DataFrame (after loading subset - BEFORE reduce_mem_usage and any FE):\")\ntest_df.info(memory_usage='deep')\nprint(f\"\\nShape: {test_df.shape}\")\n\nif 'Id' in test_df.columns and 'ranker_id' in test_df.columns:\n    test_ids_df = test_df[['Id', 'ranker_id']].copy()\nelse:\n    print(\"Warning: 'Id' or 'ranker_id' not found in loaded test_df columns. Submission might fail.\")\n    try:\n        temp_ids = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/test.parquet', columns=['Id', 'ranker_id'])\n        test_ids_df = temp_ids.copy()\n        del temp_ids\n    except Exception as e:\n        print(f\"Fallback to load test Ids failed: {e}\")\n        test_ids_df = pd.DataFrame()\ngc.collect()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df['ranker_id'].unique()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_initial_datetime_features(df):\n    loaded_cols = df.columns\n    potential_dt_cols = ['requestDate', 'legs0_departureAt', 'legs0_arrivalAt', 'legs1_departureAt', 'legs1_arrivalAt']\n    for col in potential_dt_cols:\n        if col in loaded_cols:\n            if not pd.api.types.is_datetime64_any_dtype(df[col]):\n                current_dtype = df[col].dtype\n                print(f\"Converting column {col} (current dtype: {current_dtype}) to datetime.\")\n                df[col] = pd.to_datetime(df[col].astype(str), errors='coerce')\n    return df\n\ndef create_remaining_features(df, is_train=True):\n    # --- Date/Time Component Extraction ---\n    potential_dt_cols_for_components = ['legs0_departureAt', 'legs0_arrivalAt', 'legs1_departureAt', 'legs1_arrivalAt']\n    for col in potential_dt_cols_for_components:\n        if col in df.columns and pd.api.types.is_datetime64_any_dtype(df[col]):\n             df[col + '_hour'] = df[col].dt.hour.astype(np.int8, errors='ignore')\n             df[col + '_dow'] = df[col].dt.dayofweek.astype(np.int8, errors='ignore')\n\n    # --- Booking Lead Time ---\n    if 'legs0_departureAt' in df.columns and 'requestDate' in df.columns and \\\n       pd.api.types.is_datetime64_any_dtype(df['legs0_departureAt']) and \\\n       pd.api.types.is_datetime64_any_dtype(df['requestDate']):\n        df['booking_lead_days'] = (df['legs0_departureAt'] - df['requestDate']).dt.total_seconds() / (24 * 60 * 60)\n        df['booking_lead_days'] = df['booking_lead_days'].fillna(-1).astype(np.float32)\n    else:\n        missing_cols = [c for c in ['legs0_departureAt', 'requestDate'] if c not in df.columns]\n        if missing_cols: print(f\"Warning: Columns {missing_cols} not found for booking_lead_days.\")\n        else: print(f\"Warning: Dtype issue for booking_lead_days. legs0_dep: {df.get('legs0_departureAt', pd.Series(dtype='object')).dtype}, reqDate: {df.get('requestDate', pd.Series(dtype='object')).dtype}\")\n        df['booking_lead_days'] = -1.0\n\n    # --- Route Features ---\n    if 'searchRoute' in df.columns: df['is_round_trip'] = df['searchRoute'].astype(str).str.contains('/').astype(np.int8)\n    else: df['is_round_trip'] = -1 \n    \n    if 'legs1_departureAt' in df.columns and pd.api.types.is_datetime64_any_dtype(df['legs1_departureAt']):\n        df['num_legs'] = 1 + df['legs1_departureAt'].notna().astype(np.int8)\n    elif 'legs1_departureAt' in df.columns :\n         df['num_legs'] = 1 + pd.to_datetime(df['legs1_departureAt'].astype(str),errors='coerce').notna().astype(np.int8)\n    else: df['num_legs'] = 1\n\n    # --- Segment Count ---\n    df['num_segments_leg0'] = 0; df['num_segments_leg1'] = 0\n    if 'legs0_segments0_departureFrom_airport_iata' in df.columns: df['num_segments_leg0'] += df['legs0_segments0_departureFrom_airport_iata'].notna().astype(np.int8)\n    if 'legs1_segments0_departureFrom_airport_iata' in df.columns: df['num_segments_leg1'] += df['legs1_segments0_departureFrom_airport_iata'].notna().astype(np.int8)\n   \n    df['total_segments'] = (df['num_segments_leg0'] + df['num_segments_leg1']).astype(np.int8)\n    \n    # --- Flight Duration ---\n    for dur_col in ['legs0_duration', 'legs1_duration']:\n        if dur_col in df.columns:\n            if not pd.api.types.is_numeric_dtype(df[dur_col]):\n                df[dur_col] = pd.to_numeric(df[dur_col].astype(str), errors='coerce').fillna(0)\n            else: df[dur_col] = df[dur_col].fillna(0)\n        else: df[dur_col] = 0 \n    df['total_flight_duration'] = (df['legs0_duration'] + df['legs1_duration']).astype(np.float32)\n\n    # --- Price Features ---\n    if 'totalPrice' in df.columns and 'taxes' in df.columns:\n        df['price_per_duration'] = (df['totalPrice'] / (df['total_flight_duration'] + 1e-6)).fillna(0).astype(np.float32)\n        df['tax_percentage'] = (df['taxes'] / (df['totalPrice'] + 1e-6)).fillna(0) * 100\n        df['tax_percentage'] = df['tax_percentage'].astype(np.float32)\n    else: df['price_per_duration'] = 0.0; df['tax_percentage'] = 0.0\n\n    # --- Policy/Convenience ---\n    if 'pricingInfo_isAccessTP' in df.columns: df['is_compliant'] = df['pricingInfo_isAccessTP'].fillna(0).astype(np.int8)\n    else: df['is_compliant'] = -1\n    \n    if 'legs0_segments0_baggageAllowance_quantity' in df.columns: df['baggage_leg0_included'] = (df['legs0_segments0_baggageAllowance_quantity'].fillna(0) > 0).astype(np.int8)\n    else: df['baggage_leg0_included'] = -1\n        \n    if 'legs1_segments0_baggageAllowance_quantity' in df.columns: # Giả sử chỉ có segment 0 được tải cho leg 1\n        df['baggage_leg1_included'] = (df['legs1_segments0_baggageAllowance_quantity'].fillna(0) > 0).astype(np.int8)\n        if 'baggage_leg0_included' in df.columns and df['baggage_leg0_included'].iloc[0] != -1:\n             df['baggage_both_legs_included'] = (df['baggage_leg0_included'] & df['baggage_leg1_included']).astype(np.int8)\n        else: df['baggage_both_legs_included'] = -1\n    else: \n        df['baggage_leg1_included'] = 0 \n        if 'baggage_leg0_included' in df.columns and df['baggage_leg0_included'].iloc[0] != -1:\n            df['baggage_both_legs_included'] = df['baggage_leg0_included'].astype(np.int8)\n        else: df['baggage_both_legs_included'] = -1\n    \n    # --- Cancellation/Exchange ---\n    df['free_cancel'] = -1; df['free_exchange'] = -1\n    if 'miniRules0_monetaryAmount' in df.columns and 'miniRules0_percentage' in df.columns:\n        df['free_cancel'] = ((pd.to_numeric(df['miniRules0_monetaryAmount'], errors='coerce').fillna(1) == 0) & \\\n                             (pd.to_numeric(df['miniRules0_percentage'], errors='coerce').fillna(1) == 0)).astype(np.int8)\n    if 'miniRules1_monetaryAmount' in df.columns and 'miniRules1_percentage' in df.columns:\n        df['free_exchange'] = ((pd.to_numeric(df['miniRules1_monetaryAmount'], errors='coerce').fillna(1) == 0) & \\\n                              (pd.to_numeric(df['miniRules1_percentage'], errors='coerce').fillna(1) == 0)).astype(np.int8)\n\n    # --- Group-wise Features ---\n    group_key = 'ranker_id'\n    if group_key not in df.columns: return df\n\n    cols_for_group_features = []\n    if 'totalPrice' in df.columns and pd.api.types.is_numeric_dtype(df['totalPrice']):\n        cols_for_group_features.append('totalPrice')\n        \n    print(f\"Processing group-wise features for {'train' if is_train else 'test'} on columns: {cols_for_group_features}\")\n    for col in cols_for_group_features:\n        if col in df.columns and pd.api.types.is_numeric_dtype(df[col]):\n            print(f\"  Calculating rank for {col}...\") # Chỉ giữ lại rank\n            df[f'{col}_rank_in_group'] = df.groupby(group_key)[col].rank(method='dense', ascending=True).astype(np.float32)\n            gc.collect() \n        elif col in df.columns:\n             print(f\"Warning: Column '{col}' for group feature is not numeric (dtype: {df[col].dtype}). Skipping.\")\n\n    # --- User/Company Categorical ---\n    user_company_cats_loaded = [c for c in ['sex', 'nationality', 'isVip'] if c in df.columns]\n    for col in user_company_cats_loaded:\n        if df[col].dtype == 'bool': df[col] = df[col].astype(str)\n        df[col] = df[col].fillna('MISSING').astype('category')\n    \n    binary_cols_loaded = [c for c in ['bySelf', 'isAccess3D'] if c in df.columns] # Thêm các cột này vào initial_core_columns nếu muốn sử dụng\n    for col in binary_cols_loaded: df[col] = df[col].fillna(0).astype(np.int8)\n    return df\n\n# --- Execution part of Cell 4 ---\nprint(\"--- Processing TRAIN_DF ---\")\nprint(\"Initial datetime conversion for train_df...\")\ntrain_df_processed = create_initial_datetime_features(train_df.copy())\ndel train_df; gc.collect()\n\nprint(\"Applying reduce_mem_usage to train_df_processed...\")\ntrain_df_processed = reduce_mem_usage(train_df_processed) # reduce_mem_usage from Cell 2\ngc.collect()\n\nprint(\"Creating remaining features for train_df_processed...\")\ntrain_df_processed = create_remaining_features(train_df_processed, is_train=True)\ngc.collect()\n\ntrain_labels = train_df_processed['selected']\ntrain_ids = train_df_processed['Id']\ntrain_ranker_ids = train_df_processed['ranker_id']\n\nraw_datetime_col_names = ['requestDate', 'legs0_departureAt', 'legs0_arrivalAt', 'legs1_departureAt', 'legs1_arrivalAt']\nid_cols_and_target = ['Id', 'ranker_id', 'selected', 'profileId', 'companyID', 'searchRoute']\nexcluded_for_X_train = id_cols_and_target + raw_datetime_col_names\ntrain_feature_cols = [col for col in train_df_processed.columns if col not in excluded_for_X_train]\n\nX = train_df_processed[train_feature_cols].copy()\ny = train_labels.copy()\nprint(f\"Shape of X_train: {X.shape}\")\nprint(f\"X_train memory usage: {X.memory_usage(deep=True).sum() / 1024**2:.2f} MB\")\ndel train_df_processed; gc.collect()\n\nprint(\"\\n--- Processing TEST_DF ---\")\nprint(\"Initial datetime conversion for test_df...\")\ntest_df_processed = create_initial_datetime_features(test_df.copy())\ndel test_df; gc.collect()\n\nprint(\"Applying reduce_mem_usage to test_df_processed...\")\ntest_df_processed = reduce_mem_usage(test_df_processed)\ngc.collect()\n\nprint(\"Creating remaining features for test_df_processed...\")\ntest_df_processed = create_remaining_features(test_df_processed, is_train=False)\ngc.collect()\n\n# Create X_test\nX_test = pd.DataFrame(columns=train_feature_cols, index=test_df_processed.index)\nfor col in train_feature_cols:\n    if col in test_df_processed.columns:\n        X_test[col] = test_df_processed[col]\n    else:\n        print(f\"Warning: Feature '{col}' from train not found in processed test_df. Filling with 0.\")\n        X_test[col] = 0 \n\n# ✅ Save group IDs before cleaning up\ngroups_test = test_df_processed['ranker_id'].values\n\ndel test_df_processed; gc.collect()\n\n\nprint(f\"Shape of X_test: {X_test.shape}\")\nprint(f\"X_test memory usage: {X_test.memory_usage(deep=True).sum() / 1024**2:.2f} MB\")\nprint(f\"\\nFinal shapes before LabelEncoding: X_train: {X.shape}, X_test: {X_test.shape}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_features_for_encoding = []\nprint(\"\\nIdentifying categorical features for Label Encoding from X.columns...\")\nfor col in X.columns:\n    if X[col].dtype.name == 'object' or X[col].dtype.name == 'category':\n        print(f\"Column '{col}' (dtype: {X[col].dtype}) identified as categorical for encoding.\")\n        categorical_features_for_encoding.append(col)\n        \n        le = LabelEncoder()\n        if col in X_test.columns:\n            combined_col_data = pd.concat([X[col].astype(str), X_test[col].astype(str)], axis=0).unique()\n            le.fit(combined_col_data)\n            X[col] = le.transform(X[col].astype(str))\n            X_test[col] = le.transform(X_test[col].astype(str))\n        else:\n            X[col] = le.fit_transform(X[col].astype(str))\n\nprint(f\"\\nCategorical features processed with LabelEncoder: {categorical_features_for_encoding}\")\n\nprint(\"\\nChecking for non-numeric columns after LabelEncoding...\")\nfor col in X.columns:\n    if not pd.api.types.is_numeric_dtype(X[col]):\n        print(f\"Warning: Non-numeric column post-LE: {col}, dtype: {X[col].dtype}. Forcing numeric.\")\n        X[col] = pd.to_numeric(X[col], errors='coerce').fillna(-1)\n        if col in X_test.columns: X_test[col] = pd.to_numeric(X_test[col], errors='coerce').fillna(-1)\n\nfinal_features_list = list(X.columns)\nprint(f\"\\nFinal features for model ({len(final_features_list)}): {final_features_list}\")\nprint(\"\\nX dtypes after all processing:\")\nprint(X.dtypes.value_counts())\ngc.collect()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"RANDOM_STATE = 42\n\n# Utility functions\ndef sigmoid(x):\n    return 1 / (1 + np.exp(-np.clip(x / 10, -500, 500)))\n\ndef calculate_hitrate_at_k(df, k=3):\n    hits = []\n    for ranker_id, group in df.groupby('ranker_id'):\n        if len(group) > 10:\n            top_k = group.nlargest(k, 'pred')\n            hit = (top_k['selected'] == 1).any()\n            hits.append(hit)\n    return np.mean(hits) if hits else 0.0\n\ndef evaluate_model(y_true, y_pred, groups, model_name=\"Model\"):\n    df = pd.DataFrame({\n        'ranker_id': groups,\n        'pred': y_pred,\n        'selected': y_true\n    })\n    top_preds = df.loc[df.groupby('ranker_id')['pred'].idxmax()]\n    top_preds['prob'] = sigmoid(top_preds['pred'])\n    logloss = log_loss(top_preds['selected'], top_preds['prob'])\n    hitrate_at_3 = calculate_hitrate_at_k(df, k=3)\n    accuracy = (top_preds['selected'] == 1).mean()\n    print(f\"{model_name} Validation Metrics:\")\n    print(f\"HitRate@3 (groups >10): {hitrate_at_3:.4f}\")\n    print(f\"LogLoss:                {logloss:.4f}\")\n    print(f\"Top-1 Accuracy:         {accuracy:.4f}\")\n    print(\"-\" * 40)\n    return df, hitrate_at_3, logloss, accuracy\n\ndef create_proportional_splits(X, y, train_frac=0.4, val_frac=0.2, test_frac=0.4, random_state=42):\n    assert abs(train_frac + val_frac + test_frac - 1.0) < 1e-5, \"Fractions must sum to 1.\"\n    X_train_val, X_test, y_train_val, y_test = train_test_split(\n        X, y, test_size=test_frac, random_state=random_state, stratify=y)\n    val_frac_adj = val_frac / (train_frac + val_frac)\n    X_train, X_val, y_train, y_val = train_test_split(\n        X_train_val, y_train_val, test_size=val_frac_adj, random_state=random_state, stratify=y_train_val)\n    return X_train, X_val, X_test, y_train, y_val, y_test\n\ndef prepare_tree_data(X_tr, X_val, X_test, cat_features):\n    if len(cat_features) == 0:\n        return X_tr.values, X_val.values, X_test.values\n    ohe = OneHotEncoder(handle_unknown='ignore', sparse=False)\n    ohe.fit(X_tr[cat_features])\n    X_tr_cat = ohe.transform(X_tr[cat_features])\n    X_val_cat = ohe.transform(X_val[cat_features])\n    X_test_cat = ohe.transform(X_test[cat_features])\n    X_tr_final = np.hstack([X_tr.drop(columns=cat_features).values, X_tr_cat])\n    X_val_final = np.hstack([X_val.drop(columns=cat_features).values, X_val_cat])\n    X_test_final = np.hstack([X_test.drop(columns=cat_features).values, X_test_cat])\n    return X_tr_final, X_val_final, X_test_final\n\ndef prepare_nn_data(X_tr, X_val, X_test, cat_features):\n    num_features = [col for col in X_tr.columns if col not in cat_features]\n    scaler = StandardScaler()\n    X_tr_num = scaler.fit_transform(X_tr[num_features])\n    X_val_num = scaler.transform(X_val[num_features])\n    X_test_num = scaler.transform(X_test[num_features])\n    label_encoders = {}\n    X_tr_cat = np.zeros((len(X_tr), len(cat_features)))\n    X_val_cat = np.zeros((len(X_val), len(cat_features)))\n    X_test_cat = np.zeros((len(X_test), len(cat_features)))\n    for i, col in enumerate(cat_features):\n        le = LabelEncoder()\n        all_vals = pd.concat([X_tr[col], X_val[col], X_test[col]]).astype(str)\n        le.fit(all_vals)\n        label_encoders[col] = le\n        X_tr_cat[:, i] = le.transform(X_tr[col].astype(str))\n        X_val_cat[:, i] = le.transform(X_val[col].astype(str))\n        X_test_cat[:, i] = le.transform(X_test[col].astype(str))\n    X_tr_nn = np.concatenate([X_tr_num, X_tr_cat], axis=1)\n    X_val_nn = np.concatenate([X_val_num, X_val_cat], axis=1)\n    X_test_nn = np.concatenate([X_test_num, X_test_cat], axis=1)\n    return X_tr_nn, X_val_nn, X_test_nn, scaler, label_encoders\n\ndef create_nn_model(input_dim):\n    model = keras.Sequential([\n        layers.Dense(512, activation='relu', input_shape=(input_dim,)),\n        layers.BatchNormalization(),\n        layers.Dropout(0.3),\n        layers.Dense(256, activation='relu'),\n        layers.BatchNormalization(),\n        layers.Dropout(0.3),\n        layers.Dense(128, activation='relu'),\n        layers.BatchNormalization(),\n        layers.Dropout(0.2),\n        layers.Dense(64, activation='relu'),\n        layers.Dropout(0.2),\n        layers.Dense(32, activation='relu'),\n        layers.Dense(1, activation='sigmoid')\n    ])\n    model.compile(optimizer=keras.optimizers.Adam(learning_rate=0.001),\n                  loss='binary_crossentropy', metrics=['accuracy'])\n    return model\n\n\nfrom sklearn.utils import resample\n\ndef stratified_downsample(X, y, fraction=0.6, random_state=42):\n    df = X.copy()\n    df['target'] = y\n    downsampled_df = []\n\n    for label in df['target'].unique():\n        class_subset = df[df['target'] == label]\n        n_samples = int(len(class_subset) * fraction)\n        class_downsampled = resample(\n            class_subset, replace=False, n_samples=n_samples, random_state=random_state\n        )\n        downsampled_df.append(class_downsampled)\n\n    final_df = pd.concat(downsampled_df)\n    X_down = final_df.drop(columns='target')\n    y_down = final_df['target']\n    return X_down.reset_index(drop=True), y_down.reset_index(drop=True)\n\n\n\n# Reduce dataset size with class balance\nX_reduced, y_reduced = stratified_downsample(X, y, fraction=0.4, random_state=RANDOM_STATE)\n\n# Now split the reduced set\nX_tr, X_val, X_test, y_tr, y_val, y_test = create_proportional_splits(X_reduced, y_reduced)\n\ncat_features_final = X_tr.select_dtypes(include=['object', 'category']).columns.tolist()\ngroups_val = np.arange(len(X_val))\n\n# Train XGBoost\nprint(\"Training XGBoost\")\nX_tr_xgb, X_val_xgb, X_test_xgb = prepare_tree_data(X_tr, X_val, X_test, cat_features_final)\ndtrain = xgb.DMatrix(X_tr_xgb, label=y_tr)\ndval = xgb.DMatrix(X_val_xgb, label=y_val)\nxgb_model = xgb.train({\n    'objective': 'binary:logistic', 'eval_metric': 'logloss', 'max_depth': 10, 'min_child_weight': 5,\n    'subsample': 0.8, 'colsample_bytree': 0.8, 'lambda': 15.0, 'alpha': 5.0, 'learning_rate': 0.05,\n    'seed': RANDOM_STATE, 'n_jobs': -1, 'tree_method': 'hist'\n}, dtrain, num_boost_round=300, evals=[(dtrain, 'train'), (dval, 'val')],\n   early_stopping_rounds=150, verbose_eval=100)\nxgb_val_preds = np.log(xgb_model.predict(dval) / (1 - xgb_model.predict(dval) + 1e-8))\nxgb_val_df, xgb_hr3, xgb_logloss, xgb_acc = evaluate_model(y_val, xgb_val_preds, groups_val, \"XGBoost\")\n\n# Train LightGBM\nprint(\"Training LightGBM\")\nX_tr_lgb, X_val_lgb, X_test_lgb = prepare_tree_data(X_tr, X_val, X_test, cat_features_final)\nlgb_train = lgb.Dataset(X_tr_lgb, label=y_tr)\nlgb_val = lgb.Dataset(X_val_lgb, label=y_val, reference=lgb_train)\nlgb_model = lgb.train({\n    'objective': 'binary', 'metric': 'binary_logloss', 'num_leaves': 255, 'max_depth': 12,\n    'min_data_in_leaf': 20, 'feature_fraction': 0.8, 'bagging_fraction': 0.8, 'bagging_freq': 5,\n    'lambda_l1': 5.0, 'lambda_l2': 15.0, 'learning_rate': 0.05, 'seed': RANDOM_STATE, 'n_jobs': -1, 'verbose': -1\n}, lgb_train, valid_sets=[lgb_train, lgb_val], valid_names=['train', 'val'], num_boost_round=300,\n   callbacks=[lgb.early_stopping(150), lgb.log_evaluation(100)])\nlgb_val_preds = np.log(lgb_model.predict(X_val_lgb) / (1 - lgb_model.predict(X_val_lgb) + 1e-8))\nlgb_val_df, lgb_hr3, lgb_logloss, lgb_acc = evaluate_model(y_val, lgb_val_preds, groups_val, \"LightGBM\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dtest_xgb = xgb.DMatrix(X_test_xgb)\nxgb_test_preds = xgb_model.predict(dtest_xgb)\nlgb_test_preds = lgb_model.predict(X_test_lgb)\n\n# Ensemble (average) or pick one\nfinal_test_preds = (xgb_test_preds + lgb_test_preds) / 2","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"groups_test = test_df_processed.loc[X_test.index, 'ranker_id'].values","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predictions_to_ranks(predictions, group_ids):\n    df = pd.DataFrame({\n        'pred': predictions,\n        'ranker_id': group_ids,\n        'idx': range(len(predictions))\n    })\n    df['rank'] = df.groupby('ranker_id')['pred'].rank(method='first', ascending=False)\n    df = df.sort_values('idx')\n    return df['rank'].values.astype(int)\n\ntest_ranks = predictions_to_ranks(final_test_preds, groups_test)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.DataFrame({\n    'Id': range(len(final_test_preds)),  # Replace with real ID if available\n    'ranker_id': groups_test,\n    'selected': test_ranks\n})\n\nsubmission.to_csv('xgb_ranking_submission.csv', index=False)\nprint(\"Submission saved as 'xgb_ranking_submission.csv'\")\n\n# Show sample and statistics\nprint(\"\\nSample predictions:\")\nprint(submission.head(10))\n\nprint(\"\\nSubmission stats:\")\nprint(f\"Total: {len(submission)}\")\nprint(f\"Unique ranker_ids: {submission['ranker_id'].nunique()}\")\nprint(f\"Avg group size: {len(submission) / submission['ranker_id'].nunique():.2f}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}