{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.17","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":105399,"databundleVersionId":12733338,"sourceType":"competition"}],"dockerImageVersionId":31042,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n!pip install lightgbm\nimport lightgbm as lgb\nfrom sklearn.model_selection import GroupKFold\nfrom sklearn.preprocessing import LabelEncoder\nimport gc # Garbage Collector\nimport warnings\n\nwarnings.filterwarnings('ignore')\npd.set_option('display.max_columns', None)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T05:19:42.400560Z","iopub.execute_input":"2025-06-24T05:19:42.401324Z","iopub.status.idle":"2025-06-24T05:19:50.078675Z","shell.execute_reply.started":"2025-06-24T05:19:42.401292Z","shell.execute_reply":"2025-06-24T05:19:50.072456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"##set---->>>\ndef reduce_mem_usage(df, verbose=True):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2\n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n    end_mem = df.memory_usage().sum() / 1024**2\n    if verbose: print(f'Mem. usage decreased to {end_mem:5.2f} Mb ({100 * (start_mem - end_mem) / start_mem:.1f}% reduction)')\n    return df\n\ndef calculate_hit_rate_at_3(df_preds_with_true_and_rank):\n    \"\"\"\n    Calculates HitRate@3.\n    df_preds_with_true_and_rank must have:\n        - 'ranker_id'\n        - 'selected' (true binary target, 1 for chosen)\n        - 'predicted_rank' (rank assigned by the model, 1 is best)\n    \"\"\"\n    hits = 0\n    valid_queries_count = 0\n    \n    for ranker_id, group in df_preds_with_true_and_rank.groupby('ranker_id'):\n        if len(group) <= 10:\n            continue  # Skip groups with 10 or fewer options as per competition rules\n        \n        valid_queries_count += 1\n        \n        true_selected_item = group[group['selected'] == 1]\n        \n        if not true_selected_item.empty:\n            # Get the rank of the true selected item\n            rank_of_true_item = true_selected_item.iloc[0]['predicted_rank']\n            if rank_of_true_item <= 3:\n                hits += 1\n        # else:\n            # This shouldn't happen in validation if data is prepared correctly from train\n            # print(f\"Warning: No selected item found for ranker_id {ranker_id} in HitRate calculation.\")\n            \n    if valid_queries_count == 0:\n        return 0.0\n    return hits / valid_queries_count","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T05:19:50.080945Z","iopub.execute_input":"2025-06-24T05:19:50.081282Z","iopub.status.idle":"2025-06-24T05:19:50.098423Z","shell.execute_reply.started":"2025-06-24T05:19:50.081258Z","shell.execute_reply":"2025-06-24T05:19:50.093438Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# block 3: Load the Data\nimport pandas as pd\nimport numpy as np\nimport gc\n\ninitial_core_columns = [\n    'Id', 'ranker_id', 'selected', 'profileId', 'companyID',\n    'requestDate', 'totalPrice', 'taxes',\n    'legs0_departureAt', 'legs0_arrivalAt', 'legs0_duration',\n    'legs1_departureAt', 'legs1_arrivalAt', 'legs1_duration',\n    'legs0_segments0_departureFrom_airport_iata', 'legs0_segments0_arrivalTo_airport_iata',\n    'legs0_segments0_marketingCarrier_code', 'legs0_segments0_cabinClass',\n    'legs0_segments0_baggageAllowance_quantity',\n    'searchRoute',\n    'pricingInfo_isAccessTP', 'pricingInfo_passengerCount',\n    'sex', 'nationality', 'isVip',\n    'miniRules0_monetaryAmount', 'miniRules0_percentage', \n    'miniRules1_monetaryAmount', 'miniRules1_percentage'\n]\ninitial_core_columns_test = [col for col in initial_core_columns if col != 'selected']\n\nprint(\"Loading a subset of columns for train_df...\")\ntrain_df = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/train.parquet', columns=initial_core_columns)\nprint(\"Loading a subset of columns for test_df...\")\ntest_df = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/test.parquet', columns=initial_core_columns_test)\nsample_submission_df = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/sample_submission.parquet')\n\nprint(\"\\nTrain DataFrame (after loading subset - BEFORE reduce_mem_usage and any FE):\")\ntrain_df.info(memory_usage='deep')\nprint(f\"\\nShape: {train_df.shape}\")\nprint(\"\\nTest DataFrame (after loading subset - BEFORE reduce_mem_usage and any FE):\")\ntest_df.info(memory_usage='deep')\nprint(f\"\\nShape: {test_df.shape}\")\n\nif 'Id' in test_df.columns and 'ranker_id' in test_df.columns:\n    test_ids_df = test_df[['Id', 'ranker_id']].copy()\nelse:\n    print(\"Warning: 'Id' or 'ranker_id' not found in loaded test_df columns. Submission might fail.\")\n    try:\n        temp_ids = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/test.parquet', columns=['Id', 'ranker_id'])\n        test_ids_df = temp_ids.copy()\n        del temp_ids\n    except Exception as e:\n        print(f\"Fallback to load test Ids failed: {e}\")\n        test_ids_df = pd.DataFrame()\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T05:19:50.100733Z","iopub.execute_input":"2025-06-24T05:19:50.100974Z","iopub.status.idle":"2025-06-24T05:20:52.824475Z","shell.execute_reply.started":"2025-06-24T05:19:50.100951Z","shell.execute_reply":"2025-06-24T05:20:52.819940Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Block 4: Feature Engineering there in it \n\ndef create_initial_datetime_features(df):\n    loaded_cols = df.columns\n    potential_dt_cols = ['requestDate', 'legs0_departureAt', 'legs0_arrivalAt', 'legs1_departureAt', 'legs1_arrivalAt']\n    for col in potential_dt_cols:\n        if col in loaded_cols:\n            if not pd.api.types.is_datetime64_any_dtype(df[col]):\n                current_dtype = df[col].dtype\n                print(f\"Converting column {col} (current dtype: {current_dtype}) to datetime.\")\n                df[col] = pd.to_datetime(df[col].astype(str), errors='coerce')\n    return df\n\ndef create_remaining_features(df, is_train=True):\n    # --- Date/Time Component Extraction ---\n    potential_dt_cols_for_components = ['legs0_departureAt', 'legs0_arrivalAt', 'legs1_departureAt', 'legs1_arrivalAt']\n    for col in potential_dt_cols_for_components:\n        if col in df.columns and pd.api.types.is_datetime64_any_dtype(df[col]):\n             df[col + '_hour'] = df[col].dt.hour.astype(np.int8, errors='ignore')\n             df[col + '_dow'] = df[col].dt.dayofweek.astype(np.int8, errors='ignore')\n\n    # --- Booking Lead Time ---\n    if 'legs0_departureAt' in df.columns and 'requestDate' in df.columns and \\\n       pd.api.types.is_datetime64_any_dtype(df['legs0_departureAt']) and \\\n       pd.api.types.is_datetime64_any_dtype(df['requestDate']):\n        df['booking_lead_days'] = (df['legs0_departureAt'] - df['requestDate']).dt.total_seconds() / (24 * 60 * 60)\n        df['booking_lead_days'] = df['booking_lead_days'].fillna(-1).astype(np.float32)\n    else:\n        missing_cols = [c for c in ['legs0_departureAt', 'requestDate'] if c not in df.columns]\n        if missing_cols: print(f\"Warning: Columns {missing_cols} not found for booking_lead_days.\")\n        else: print(f\"Warning: Dtype issue for booking_lead_days. legs0_dep: {df.get('legs0_departureAt', pd.Series(dtype='object')).dtype}, reqDate: {df.get('requestDate', pd.Series(dtype='object')).dtype}\")\n        df['booking_lead_days'] = -1.0\n\n    # --- Route Features ---\n    if 'searchRoute' in df.columns: df['is_round_trip'] = df['searchRoute'].astype(str).str.contains('/').astype(np.int8)\n    else: df['is_round_trip'] = -1 \n    \n    if 'legs1_departureAt' in df.columns and pd.api.types.is_datetime64_any_dtype(df['legs1_departureAt']):\n        df['num_legs'] = 1 + df['legs1_departureAt'].notna().astype(np.int8)\n    elif 'legs1_departureAt' in df.columns :\n         df['num_legs'] = 1 + pd.to_datetime(df['legs1_departureAt'].astype(str),errors='coerce').notna().astype(np.int8)\n    else: df['num_legs'] = 1\n\n    # --- Segment Count ---\n    df['num_segments_leg0'] = 0; df['num_segments_leg1'] = 0\n    if 'legs0_segments0_departureFrom_airport_iata' in df.columns: df['num_segments_leg0'] += df['legs0_segments0_departureFrom_airport_iata'].notna().astype(np.int8)\n    if 'legs1_segments0_departureFrom_airport_iata' in df.columns: df['num_segments_leg1'] += df['legs1_segments0_departureFrom_airport_iata'].notna().astype(np.int8)\n   \n    df['total_segments'] = (df['num_segments_leg0'] + df['num_segments_leg1']).astype(np.int8)\n    \n    # --- Flight Duration ---\n    for dur_col in ['legs0_duration', 'legs1_duration']:\n        if dur_col in df.columns:\n            if not pd.api.types.is_numeric_dtype(df[dur_col]):\n                df[dur_col] = pd.to_numeric(df[dur_col].astype(str), errors='coerce').fillna(0)\n            else: df[dur_col] = df[dur_col].fillna(0)\n        else: df[dur_col] = 0 \n    df['total_flight_duration'] = (df['legs0_duration'] + df['legs1_duration']).astype(np.float32)\n\n    # --- Price Features ---\n    if 'totalPrice' in df.columns and 'taxes' in df.columns:\n        df['price_per_duration'] = (df['totalPrice'] / (df['total_flight_duration'] + 1e-6)).fillna(0).astype(np.float32)\n        df['tax_percentage'] = (df['taxes'] / (df['totalPrice'] + 1e-6)).fillna(0) * 100\n        df['tax_percentage'] = df['tax_percentage'].astype(np.float32)\n    else: df['price_per_duration'] = 0.0; df['tax_percentage'] = 0.0\n\n    # --- Policy/Convenience ---\n    if 'pricingInfo_isAccessTP' in df.columns: df['is_compliant'] = df['pricingInfo_isAccessTP'].fillna(0).astype(np.int8)\n    else: df['is_compliant'] = -1\n    \n    if 'legs0_segments0_baggageAllowance_quantity' in df.columns: df['baggage_leg0_included'] = (df['legs0_segments0_baggageAllowance_quantity'].fillna(0) > 0).astype(np.int8)\n    else: df['baggage_leg0_included'] = -1\n        \n    if 'legs1_segments0_baggageAllowance_quantity' in df.columns: # Giả sử chỉ có segment 0 được tải cho leg 1\n        df['baggage_leg1_included'] = (df['legs1_segments0_baggageAllowance_quantity'].fillna(0) > 0).astype(np.int8)\n        if 'baggage_leg0_included' in df.columns and df['baggage_leg0_included'].iloc[0] != -1:\n             df['baggage_both_legs_included'] = (df['baggage_leg0_included'] & df['baggage_leg1_included']).astype(np.int8)\n        else: df['baggage_both_legs_included'] = -1\n    else: \n        df['baggage_leg1_included'] = 0 \n        if 'baggage_leg0_included' in df.columns and df['baggage_leg0_included'].iloc[0] != -1:\n            df['baggage_both_legs_included'] = df['baggage_leg0_included'].astype(np.int8)\n        else: df['baggage_both_legs_included'] = -1\n    \n    # --- Cancellation/Exchange ---\n    df['free_cancel'] = -1; df['free_exchange'] = -1\n    if 'miniRules0_monetaryAmount' in df.columns and 'miniRules0_percentage' in df.columns:\n        df['free_cancel'] = ((pd.to_numeric(df['miniRules0_monetaryAmount'], errors='coerce').fillna(1) == 0) & \\\n                             (pd.to_numeric(df['miniRules0_percentage'], errors='coerce').fillna(1) == 0)).astype(np.int8)\n    if 'miniRules1_monetaryAmount' in df.columns and 'miniRules1_percentage' in df.columns:\n        df['free_exchange'] = ((pd.to_numeric(df['miniRules1_monetaryAmount'], errors='coerce').fillna(1) == 0) & \\\n                              (pd.to_numeric(df['miniRules1_percentage'], errors='coerce').fillna(1) == 0)).astype(np.int8)\n\n    # --- Group-wise Features ---\n    group_key = 'ranker_id'\n    if group_key not in df.columns: return df\n\n    cols_for_group_features = []\n    if 'totalPrice' in df.columns and pd.api.types.is_numeric_dtype(df['totalPrice']):\n        cols_for_group_features.append('totalPrice')\n        \n    print(f\"Processing group-wise features for {'train' if is_train else 'test'} on columns: {cols_for_group_features}\")\n    for col in cols_for_group_features:\n        if col in df.columns and pd.api.types.is_numeric_dtype(df[col]):\n            print(f\"  Calculating rank for {col}...\") # Chỉ giữ lại rank\n            df[f'{col}_rank_in_group'] = df.groupby(group_key)[col].rank(method='dense', ascending=True).astype(np.float32)\n            gc.collect() \n        elif col in df.columns:\n             print(f\"Warning: Column '{col}' for group feature is not numeric (dtype: {df[col].dtype}). Skipping.\")\n\n    # --- User/Company Categorical ---\n    user_company_cats_loaded = [c for c in ['sex', 'nationality', 'isVip'] if c in df.columns]\n    for col in user_company_cats_loaded:\n        if df[col].dtype == 'bool': df[col] = df[col].astype(str)\n        df[col] = df[col].fillna('MISSING').astype('category')\n    \n    binary_cols_loaded = [c for c in ['bySelf', 'isAccess3D'] if c in df.columns] # Thêm các cột này vào initial_core_columns nếu muốn sử dụng\n    for col in binary_cols_loaded: df[col] = df[col].fillna(0).astype(np.int8)\n    return df\n\n# --- Execution part of Cell 4 ---\nprint(\"--- Processing TRAIN_DF ---\")\nprint(\"Initial datetime conversion for train_df...\")\ntrain_df_processed = create_initial_datetime_features(train_df.copy())\ndel train_df; gc.collect()\n\nprint(\"Applying reduce_mem_usage to train_df_processed...\")\ntrain_df_processed = reduce_mem_usage(train_df_processed) # reduce_mem_usage from Cell 2\ngc.collect()\n\nprint(\"Creating remaining features for train_df_processed...\")\ntrain_df_processed = create_remaining_features(train_df_processed, is_train=True)\ngc.collect()\n\ntrain_labels = train_df_processed['selected']\ntrain_ids = train_df_processed['Id']\ntrain_ranker_ids = train_df_processed['ranker_id']\n\nraw_datetime_col_names = ['requestDate', 'legs0_departureAt', 'legs0_arrivalAt', 'legs1_departureAt', 'legs1_arrivalAt']\nid_cols_and_target = ['Id', 'ranker_id', 'selected', 'profileId', 'companyID', 'searchRoute']\nexcluded_for_X_train = id_cols_and_target + raw_datetime_col_names\ntrain_feature_cols = [col for col in train_df_processed.columns if col not in excluded_for_X_train]\n\nX = train_df_processed[train_feature_cols].copy()\ny = train_labels.copy()\nprint(f\"Shape of X_train: {X.shape}\")\nprint(f\"X_train memory usage: {X.memory_usage(deep=True).sum() / 1024**2:.2f} MB\")\ndel train_df_processed; gc.collect()\n\nprint(\"\\n--- Processing TEST_DF ---\")\nprint(\"Initial datetime conversion for test_df...\")\ntest_df_processed = create_initial_datetime_features(test_df.copy())\ndel test_df; gc.collect()\n\nprint(\"Applying reduce_mem_usage to test_df_processed...\")\ntest_df_processed = reduce_mem_usage(test_df_processed)\ngc.collect()\n\nprint(\"Creating remaining features for test_df_processed...\")\ntest_df_processed = create_remaining_features(test_df_processed, is_train=False)\ngc.collect()\n\nX_test = pd.DataFrame(columns=train_feature_cols, index=test_df_processed.index)\nfor col in train_feature_cols:\n    if col in test_df_processed.columns:\n        X_test[col] = test_df_processed[col]\n    else:\n        print(f\"Warning: Feature '{col}' from train not found in processed test_df. Filling with 0.\")\n        X_test[col] = 0 \ndel test_df_processed; gc.collect()\n\nprint(f\"Shape of X_test: {X_test.shape}\")\nprint(f\"X_test memory usage: {X_test.memory_usage(deep=True).sum() / 1024**2:.2f} MB\")\nprint(f\"\\nFinal shapes before LabelEncoding: X_train: {X.shape}, X_test: {X_test.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T05:20:52.827061Z","iopub.execute_input":"2025-06-24T05:20:52.827302Z","iopub.status.idle":"2025-06-24T05:23:04.116395Z","shell.execute_reply.started":"2025-06-24T05:20:52.827279Z","shell.execute_reply":"2025-06-24T05:23:04.110346Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Block 5: Label Encoding there\ncategorical_features_for_encoding = []\nprint(\"\\nIdentifying categorical features for Label Encoding from X.columns...\")\nfor col in X.columns:\n    if X[col].dtype.name == 'object' or X[col].dtype.name == 'category':\n        print(f\"Column '{col}' (dtype: {X[col].dtype}) identified as categorical for encoding.\")\n        categorical_features_for_encoding.append(col)\n        \n        le = LabelEncoder()\n        if col in X_test.columns:\n            combined_col_data = pd.concat([X[col].astype(str), X_test[col].astype(str)], axis=0).unique()\n            le.fit(combined_col_data)\n            X[col] = le.transform(X[col].astype(str))\n            X_test[col] = le.transform(X_test[col].astype(str))\n        else:\n            X[col] = le.fit_transform(X[col].astype(str))\n\nprint(f\"\\nCategorical features processed with LabelEncoder: {categorical_features_for_encoding}\")\n\nprint(\"\\nChecking for non-numeric columns after LabelEncoding...\")\nfor col in X.columns:\n    if not pd.api.types.is_numeric_dtype(X[col]):\n        print(f\"Warning: Non-numeric column post-LE: {col}, dtype: {X[col].dtype}. Forcing numeric.\")\n        X[col] = pd.to_numeric(X[col], errors='coerce').fillna(-1)\n        if col in X_test.columns: X_test[col] = pd.to_numeric(X_test[col], errors='coerce').fillna(-1)\n\nfinal_features_list = list(X.columns)\nprint(f\"\\nFinal features for model ({len(final_features_list)}): {final_features_list}\")\nprint(\"\\nX dtypes after all processing:\")\nprint(X.dtypes.value_counts())\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T05:23:04.117233Z","iopub.execute_input":"2025-06-24T05:23:04.117483Z","iopub.status.idle":"2025-06-24T05:24:11.961081Z","shell.execute_reply.started":"2025-06-24T05:23:04.117459Z","shell.execute_reply":"2025-06-24T05:24:11.957888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Block 6: Model Training\n\nparams = {\n    'objective': 'lambdarank',\n    'metric': 'ndcg', \n    'eval_at': [2],   \n    'boosting_type': 'gbdt',\n    'n_estimators': 500,\n    'learning_rate': 0.15,\n    'num_leaves': 7,\n    'max_depth': 4,\n    'min_child_samples': 200,\n    'subsample': 0.5,\n    'colsample_bytree': 0.5,\n    'max_bin': 63,\n    'random_state': 42,\n    'n_jobs': -1,\n    'importance_type': 'gain',\n    'verbose': -1,\n    'seed': 42    \n}\n\nNFOLDS = 5 \ngroup_kfold = GroupKFold(n_splits=NFOLDS)\n\noof_preds_scores = np.zeros(len(X))\ntest_preds_scores = np.zeros(len(X_test))\nmodels = []\nfold_hit_rates = []\n\n# categorical_features_for_encoding \ncat_features_for_lgbm_indices_final = [X.columns.get_loc(col_name) for col_name in categorical_features_for_encoding if col_name in X.columns]\nif cat_features_for_lgbm_indices_final:\n    print(f\"Using categorical feature indices for LightGBM: {cat_features_for_lgbm_indices_final}\")\n    print(f\"Corresponding feature names: {[X.columns[i] for i in cat_features_for_lgbm_indices_final]}\")\nelse:\n    print(\"No categorical features identified for LightGBM native handling.\")\n\n\nfor fold_, (train_idx, val_idx) in enumerate(group_kfold.split(X, y, groups=train_ranker_ids)): # train_ranker_ids từ Cell 4\n    print(f\"====== Fold {fold_+1}/{NFOLDS} ======\")\n    \n    if fold_ > 0: gc.collect() \n\n    X_train_fold, y_train_fold = X.iloc[train_idx], y.iloc[train_idx]\n    X_val_fold, y_val_fold = X.iloc[val_idx], y.iloc[val_idx]\n    print(f\"  Train fold shape: {X_train_fold.shape}, Val fold shape: {X_val_fold.shape}\")\n\n    current_train_fold_ranker_ids = train_ranker_ids.iloc[train_idx]\n    current_val_fold_ranker_ids = train_ranker_ids.iloc[val_idx]\n\n    train_fold_groups = pd.DataFrame({'ranker_id': current_train_fold_ranker_ids}).groupby('ranker_id', sort=False).size().to_list()\n    val_fold_groups = pd.DataFrame({'ranker_id': current_val_fold_ranker_ids}).groupby('ranker_id', sort=False).size().to_list()\n\n    ranker = lgb.LGBMRanker(**params)\n    try:\n        print(f\"  Starting LightGBM fit for fold {fold_+1}...\")\n        ranker.fit(\n            X_train_fold, y_train_fold,\n            group=train_fold_groups,\n            eval_set=[(X_val_fold, y_val_fold)],\n            eval_group=[val_fold_groups],\n            eval_metric='ndcg',\n            callbacks=[lgb.early_stopping(10, verbose=False)],\n            categorical_feature=cat_features_for_lgbm_indices_final if cat_features_for_lgbm_indices_final else 'auto'\n        )\n        print(f\"  LightGBM fit completed for fold {fold_+1}.\")\n    except Exception as e:\n        print(f\"Error during LightGBM fit in fold {fold_+1}: {e}\")\n        print(f\"  X_train_fold mem: {X_train_fold.memory_usage(deep=True).sum() / 1024**2:.2f} MB\")\n        print(f\"  X_val_fold mem: {X_val_fold.memory_usage(deep=True).sum() / 1024**2:.2f} MB\")\n        break \n\n    models.append(ranker)\n    val_fold_scores = ranker.predict(X_val_fold)\n    oof_preds_scores[val_idx] = val_fold_scores\n    \n    print(f\"  Predicting on X_test (shape: {X_test.shape})...\")\n    current_test_preds = ranker.predict(X_test)\n    test_preds_scores += current_test_preds / NFOLDS\n    del current_test_preds; gc.collect()\n\n    val_df_for_metric = pd.DataFrame({\n        'ranker_id': current_val_fold_ranker_ids,\n        'selected': y_val_fold,\n        'score': val_fold_scores\n    })\n    val_df_for_metric['predicted_rank'] = val_df_for_metric.groupby('ranker_id')['score'].rank(method='first', ascending=False).astype(int)\n    fold_hr3 = calculate_hit_rate_at_3(val_df_for_metric) # calculate_hit_rate_at_3 \n    fold_hit_rates.append(fold_hr3)\n    print(f\"Fold {fold_+1} HitRate@3: {fold_hr3:.4f}\")\n    \n    del X_train_fold, y_train_fold, X_val_fold, y_val_fold\n    del current_train_fold_ranker_ids, current_val_fold_ranker_ids\n    del train_fold_groups, val_fold_groups, ranker, val_fold_scores, val_df_for_metric\n    gc.collect()\n\nif models and len(models) == NFOLDS:\n    oof_df_for_metric = pd.DataFrame({\n        'ranker_id': train_ranker_ids,\n        'selected': y,\n        'score': oof_preds_scores\n    })\n    oof_df_for_metric['predicted_rank'] = oof_df_for_metric.groupby('ranker_id')['score'].rank(method='first', ascending=False).astype(int)\n    overall_oof_hr3 = calculate_hit_rate_at_3(oof_df_for_metric)\n    print(f\"\\nOverall OOF HitRate@3: {overall_oof_hr3:.4f}\")\n    if fold_hit_rates: print(f\"Mean Fold HitRate@3: {np.mean(fold_hit_rates):.4f}\")\n\n    print(\"\\nFeature Importances (from last model):\")\n    try:\n        lgb.plot_importance(models[-1], figsize=(10, max(15, len(X.columns)//2)), max_num_features=len(X.columns), importance_type='gain')\n    except Exception as e:\n        print(f\"Could not plot feature importance: {e}\")\nelif models:\n     print(f\"\\nTraining completed for {len(models)} out of {NFOLDS} folds. Cannot reliably calculate overall OOF score.\")\nelse:\n    print(\"No models were trained successfully.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T05:24:11.964859Z","iopub.execute_input":"2025-06-24T05:24:11.966268Z","iopub.status.idle":"2025-06-24T05:31:02.171222Z","shell.execute_reply.started":"2025-06-24T05:24:11.966237Z","shell.execute_reply":"2025-06-24T05:31:02.165132Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Laat Block\n#Use the test_ids_df we saved earlier which has original Id and ranker_id\nsubmission_df = test_ids_df.copy()\nsubmission_df['score'] = test_preds_scores \n\nsubmission_df['selected'] = submission_df.groupby('ranker_id')['score'].rank(method='first', ascending=False).astype(int)\n\n# Select only required columns and ensure correct order\nsubmission_df = submission_df[['Id', 'ranker_id', 'selected']]\n\n# Check submission format against sample\nprint(\"\\nSample Submission:\")\nprint(sample_submission_df.head())\nprint(\"\\nOur Submission:\")\nprint(submission_df.head())\n\n# Save submission\nsubmission_df.to_parquet('submission.parquet', index=False)\nsubmission_df.to_csv('submission.csv', index=False)\nprint(\"\\nSubmission file 'submission.parquet' created successfully.\")\nprint(f\"Submission shape: {submission_df.shape}\")\n\n# Basic validation of submission\n# 1. All Ids from test set are present\nassert len(submission_df) == len(test_ids_df), \"Number of rows doesn't match test set\"\nassert submission_df['Id'].nunique() == len(test_ids_df['Id'].unique()), \"Mismatch in unique Ids\"\n\n# 2. Ranks are integers and start from 1\nassert submission_df['selected'].min() >= 1, \"Ranks should be >= 1\"\nassert submission_df['selected'].dtype == 'int', \"Ranks should be integers\"\n\n# 3. Ranks are a valid permutation within each group\ndef check_rank_permutation(group):\n    N = len(group)\n    sorted_ranks = sorted(list(group['selected']))\n    expected_ranks = list(range(1, N + 1))\n    if sorted_ranks != expected_ranks:\n        print(f\"Invalid rank permutation for ranker_id: {group['ranker_id'].iloc[0]}\")\n        print(f\"Expected: {expected_ranks}, Got: {sorted_ranks}\")\n        return False\n    return True\n\nprint(\"Basic submission validation checks passed (row count, Id uniqueness, rank min value, rank dtype).\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T05:31:02.172925Z","iopub.execute_input":"2025-06-24T05:31:02.173348Z","iopub.status.idle":"2025-06-24T05:31:19.475926Z","shell.execute_reply.started":"2025-06-24T05:31:02.173324Z","shell.execute_reply":"2025-06-24T05:31:19.470225Z"}},"outputs":[],"execution_count":null}]}