{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.17","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":105399,"databundleVersionId":12733338,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\ndf = pd.read_parquet(\"/kaggle/input/aeroclub-recsys-2025/train.parquet\")\ndf.sample(10)  # Random 10 rows\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T09:56:51.492914Z","iopub.execute_input":"2025-07-30T09:56:51.493230Z","iopub.status.idle":"2025-07-30T09:57:16.399418Z","shell.execute_reply.started":"2025-07-30T09:56:51.493206Z","shell.execute_reply":"2025-07-30T09:57:16.392846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_parquet(\"/kaggle/input/aeroclub-recsys-2025/test.parquet\")\ndf.sample(10)  # Random 10 rows\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T09:57:16.400210Z","iopub.execute_input":"2025-07-30T09:57:16.400515Z","iopub.status.idle":"2025-07-30T09:57:29.496653Z","shell.execute_reply.started":"2025-07-30T09:57:16.400491Z","shell.execute_reply":"2025-07-30T09:57:29.493466Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_parquet(\"/kaggle/input/aeroclub-recsys-2025/sample_submission.parquet\")\ndf.sample(10)  # Random 10 rows\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T09:57:29.499602Z","iopub.execute_input":"2025-07-30T09:57:29.499816Z","iopub.status.idle":"2025-07-30T09:57:32.217683Z","shell.execute_reply.started":"2025-07-30T09:57:29.499795Z","shell.execute_reply":"2025-07-30T09:57:32.213531Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install lightgbm\n!pip install polars","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T09:57:32.219929Z","iopub.execute_input":"2025-07-30T09:57:32.220157Z","iopub.status.idle":"2025-07-30T09:57:44.616863Z","shell.execute_reply.started":"2025-07-30T09:57:32.220136Z","shell.execute_reply":"2025-07-30T09:57:44.612168Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ Optimized Polars + LightGBM Pipeline for AeroClub RecSys 2025\nimport polars as pl\nimport lightgbm as lgb\nfrom sklearn.model_selection import train_test_split\nimport numpy as np\nimport time\n\n# --- Configuration ---\n# Grouping settings here makes the script easier to manage and tweak.\nclass CFG:\n    # File paths\n    TRAIN_PATH = \"/kaggle/input/aeroclub-recsys-2025/train.parquet\"\n    TEST_PATH = \"/kaggle/input/aeroclub-recsys-2025/test.parquet\"\n    SUBMISSION_PATH = \"submission.csv\"\n    TOP3_FLAT_PATH = \"top3_choices_flat.csv\"\n    TOP3_PIVOTED_PATH = \"top3_choices_pivoted.csv\"\n\n    # Feature lists\n    # Defining features centrally to avoid repetition.\n    BASE_FEATURES = [\n        \"totalPrice\", \"taxes\",\n        \"legs0_duration\", \"legs1_duration\",\n        \"miniRules0_monetaryAmount\", \"miniRules0_percentage\",\n        \"miniRules1_monetaryAmount\", \"miniRules1_percentage\",\n        \"pricingInfo_passengerCount\"\n    ]\n    BOOLEAN_FEATURES = [\"sex\", \"isVip\", \"isAccess3D\", \"bySelf\", \"pricingInfo_isAccessTP\"]\n    CATEGORICAL_FEATURES = [\"frequentFlyer\"]\n    RANK_COLS = [\"totalPrice\", \"taxes\", \"legs0_duration\", \"legs1_duration\"]\n\n    # Model parameters\n    # LightGBM is often significantly faster than CatBoost for similar performance.\n    LGBM_PARAMS = {\n        'objective': 'binary',\n        'metric': 'logloss',\n        'boosting_type': 'gbdt',\n        'n_estimators': 1500, # Increased estimators, balanced by early stopping\n        'learning_rate': 0.03,\n        'num_leaves': 31,\n        'max_depth': -1,\n        'seed': 42,\n        'n_jobs': -1,\n        'verbose': -1,\n        'colsample_bytree': 0.7,\n        'subsample': 0.7,\n        'reg_alpha': 0.1,\n        'reg_lambda': 0.1,\n        'class_weight': 'balanced'\n    }\n    LGBM_FIT_PARAMS = {\n        \"callbacks\": [lgb.early_stopping(100, verbose=True)] # Increased patience\n    }\n\n    # Other settings\n    TEST_SIZE = 0.2\n    RANDOM_STATE = 42\n\ndef feature_engineer(df: pl.DataFrame) -> pl.DataFrame:\n    \"\"\"\n    Applies all feature engineering steps to the dataframe.\n    This function encapsulates the logic to avoid code duplication.\n    \"\"\"\n    # 1. Duration Conversion: Fast, idiomatic Polars expression.\n    duration_cols = [\"legs0_duration\", \"legs1_duration\"]\n    duration_expressions = []\n    for col in duration_cols:\n        duration_expr = pl.col(col).str.split(\":\")\n        conversion_expr = (\n            pl.when(duration_expr.list.len() == 3)\n            .then(\n                duration_expr.list.get(0).cast(pl.Int64, strict=False) * 60\n                + duration_expr.list.get(1).cast(pl.Int64, strict=False)\n                + duration_expr.list.get(2).cast(pl.Float64, strict=False) / 60\n            )\n            .otherwise(None) # Use None for failed conversions\n            .cast(pl.Float32) # Use Float32 to save memory\n            .fill_null(strategy=\"mean\") # Impute with mean for robustness\n            .alias(col)\n        )\n        duration_expressions.append(conversion_expr)\n\n    df = df.with_columns(duration_expressions)\n\n    # 2. Boolean to Integer Conversion\n    # Explicitly cast boolean columns to integers using `cast(pl.Int8)`\n    df = df.with_columns(\n        [pl.col(c).cast(pl.Boolean, strict=False).cast(pl.Int8) for c in CFG.BOOLEAN_FEATURES]\n    )\n\n    # 3. Categorical Null Filling and Encoding\n    # LightGBM works best with integer-encoded categoricals.\n    df = df.with_columns(\n        [pl.col(c).fill_null(\"None\").cast(pl.Categorical) for c in CFG.CATEGORICAL_FEATURES]\n    )\n\n    # 4. Rank Features\n    # This creates ranking within each group, a powerful feature.\n    rank_expressions = []\n    for col in CFG.RANK_COLS:\n        rank_expressions.append(\n            pl.col(col).rank(method='ordinal').over(\"ranker_id\").alias(f\"rank_{col}\")\n        )\n    df = df.with_columns(rank_expressions)\n\n    return df\n\ndef main():\n    \"\"\"Main function to run the entire pipeline.\"\"\"\n    start_time = time.time()\n\n    # ========== Step 1: Load Train & Test Data ==========\n    print(\"Step 1: Loading data...\")\n    train_df = pl.read_parquet(CFG.TRAIN_PATH)\n    test_df = pl.read_parquet(CFG.TEST_PATH).with_row_index(name=\"row_order\")\n    print(f\"Data loaded successfully. Train shape: {train_df.shape}, Test shape: {test_df.shape}\")\n\n    # ========== Step 2: Feature Engineering ==========\n    print(\"Step 2: Performing feature engineering...\")\n    train_df = feature_engineer(train_df)\n    test_df = feature_engineer(test_df)\n\n    # Combine feature lists, including boolean features\n    all_features = CFG.BASE_FEATURES + CFG.BOOLEAN_FEATURES + [f\"rank_{col}\" for col in CFG.RANK_COLS]\n\n    # Ensure selected features exist in the dataframe\n    missing_train_features = [f for f in all_features if f not in train_df.columns]\n    missing_test_features = [f for f in all_features if f not in test_df.columns]\n\n    if missing_train_features:\n        print(f\"Warning: Missing features in train_df: {missing_train_features}\")\n    if missing_test_features:\n         print(f\"Warning: Missing features in test_df: {missing_test_features}\")\n\n    all_features = [f for f in all_features if f in train_df.columns and f in test_df.columns]\n\n\n    # Find categorical feature indices for LightGBM\n    categorical_feature_indices = [all_features.index(col) for col in CFG.CATEGORICAL_FEATURES if col in all_features]\n\n    print(\"Feature engineering complete.\")\n\n    # ========== Step 3: Prepare Data for Model ==========\n    print(\"Step 3: Preparing data for LightGBM...\")\n    # Create labels\n    train_df = train_df.with_columns(pl.col(\"selected\").cast(pl.Int8).alias(\"label\"))\n\n    # Split data into train and validation sets as Polars DataFrames\n    train_data, val_data = train_test_split(\n        train_df.select(all_features + [\"label\"]), test_size=CFG.TEST_SIZE, random_state=CFG.RANDOM_STATE, stratify=train_df[\"label\"]\n    )\n\n    # Convert Polars DataFrames to NumPy arrays for LightGBM\n    X_train = train_data.select(all_features).to_numpy()\n    y_train = train_data.select(\"label\").to_numpy().flatten()\n    X_val = val_data.select(all_features).to_numpy()\n    y_val = val_data.select(\"label\").to_numpy().flatten()\n\n    print(f\"Data prepared. Train shape: {X_train.shape}, Validation shape: {X_val.shape}\")\n\n    # ========== Step 4: Train LightGBM ==========\n    print(\"Step 4: Training LightGBM model...\")\n    model = lgb.LGBMClassifier(**CFG.LGBM_PARAMS)\n\n    model.fit(\n        X_train, y_train,\n        eval_set=[(X_val, y_val)],\n        eval_metric='logloss',\n        callbacks=CFG.LGBM_FIT_PARAMS['callbacks'],\n        # categorical_feature=[all_features.index(col) for col in CFG.CATEGORICAL_FEATURES if col in all_features], # Pass indices, not names\n        feature_name=all_features # Pass feature names as a list\n    )\n    print(\"Model training complete.\")\n\n    # ========== Step 5: Predict & Rank ==========\n    print(\"Step 5: Predicting on test set and ranking...\")\n    # Predict probabilities for the positive class (class 1) using the Polars DataFrame directly\n    predicted_probs = model.predict_proba(test_df.select(all_features))[:, 1]\n\n    test_df = test_df.with_columns(pl.Series(name=\"predicted_prob\", values=predicted_probs))\n\n    # Rank results based on prediction probability\n    ranked_df = test_df.sort([\"ranker_id\", \"predicted_prob\"], descending=[False, True])\n    ranked_df = ranked_df.with_columns(\n        (pl.int_range(0, pl.len()).over(\"ranker_id\") + 1).alias(\"selected\")\n    )\n    print(\"Prediction and ranking complete.\")\n\n    # ========== Step 6: Output Files ==========\n    print(\"Step 6: Generating output files...\")\n    # Create main submission file\n    submission = ranked_df.sort(\"row_order\").select([\"Id\", \"ranker_id\", \"selected\"])\n    submission.write_csv(CFG.SUBMISSION_PATH)\n\n    # Create file with top 3 choices\n    top_k = ranked_df.filter(pl.col(\"selected\").is_in([1, 2, 3]))\n\n    choice_rank_expr = (\n        pl.when(pl.col(\"selected\") == 1).then(pl.lit(\"best\"))\n          .when(pl.col(\"selected\") == 2).then(pl.lit(\"second_best\"))\n          .when(pl.col(\"selected\") == 3).then(pl.lit(\"third_best\"))\n          .otherwise(pl.lit(\"other\"))\n          .alias(\"choice_rank\")\n    )\n\n    top_k = top_k.with_columns(choice_rank_expr.cast(pl.Categorical))\n    top_k.select([\"Id\", \"ranker_id\", \"choice_rank\", \"predicted_prob\"]).write_csv(CFG.TOP3_FLAT_PATH)\n\n    # Create pivoted top 3 choices file\n    pivoted = top_k.pivot(values=\"Id\", index=\"ranker_id\", on=\"choice_rank\") # Changed columns to on\n    pivoted.write_csv(CFG.TOP3_PIVOTED_PATH)\n\n    total_time = time.time() - start_time\n    print(f\"All done! Submission files are ready. Total runtime: {total_time:.2f} seconds.\")\n\nif __name__ == '__main__':\n    main()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T10:06:35.658943Z","iopub.execute_input":"2025-07-30T10:06:35.659432Z","iopub.status.idle":"2025-07-30T10:12:55.348753Z","shell.execute_reply.started":"2025-07-30T10:06:35.659388Z","shell.execute_reply":"2025-07-30T10:12:55.344160Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/working/submission.csv\")\ndf.sample(10)  # Random 10 rows\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T10:17:30.032894Z","iopub.execute_input":"2025-07-30T10:17:30.033470Z","iopub.status.idle":"2025-07-30T10:17:33.320167Z","shell.execute_reply.started":"2025-07-30T10:17:30.033423Z","shell.execute_reply":"2025-07-30T10:17:33.314756Z"}},"outputs":[],"execution_count":null}]}