{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":105399,"databundleVersionId":12733338,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-25T22:52:32.690126Z","iopub.execute_input":"2025-06-25T22:52:32.691021Z","iopub.status.idle":"2025-06-25T22:52:32.697836Z","shell.execute_reply.started":"2025-06-25T22:52:32.690991Z","shell.execute_reply":"2025-06-25T22:52:32.696856Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install lightgbm pyarrow","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T22:52:32.699417Z","iopub.execute_input":"2025-06-25T22:52:32.700034Z","iopub.status.idle":"2025-06-25T22:52:38.914883Z","shell.execute_reply.started":"2025-06-25T22:52:32.700003Z","shell.execute_reply":"2025-06-25T22:52:38.913770Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ntrain_path = '/kaggle/input/aeroclub-recsys-2025/train.parquet'\ntest_path = '/kaggle/input/aeroclub-recsys-2025/test.parquet'\n\nimportant_columns = [\n    'Id', 'ranker_id', 'selected', 'totalPrice', 'taxes',\n    'frequentFlyer', 'isVip', 'bySelf', 'isAccess3D',\n    'legs0_duration', 'legs1_duration', \n    'pricingInfo_isAccessTP', 'pricingInfo_passengerCount'\n]\n\ntrain_df = pd.read_parquet(train_path, columns=important_columns)\ntest_df = pd.read_parquet(train_path, columns=[col for col in important_columns if col != 'selected'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T22:52:38.916199Z","iopub.execute_input":"2025-06-25T22:52:38.916547Z","execution_failed":"2025-06-25T22:53:28.359Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\nexclude_cols = [\n    'Id', 'ranker_id', 'profileId', 'requestDate', 'searchRoute',\n    'legs0_arrivalAt', 'legs0_departureAt', 'legs1_arrivalAt', 'legs1_departureAt',\n    'selected'  # only in train\n]\n\nfeature_cols = [col for col in train_df.columns if col not in exclude_cols]\n\ntrain_df[feature_cols] = train_df[feature_cols].fillna(-1)\ntest_df[feature_cols] = test_df[feature_cols].fillna(-1)\n\nobject_cols = train_df.select_dtypes(include=['object', 'bool']).columns\n\nfrom sklearn.preprocessing import LabelEncoder\nimport hashlib\n\ncat_cols = train_df.select_dtypes(include=['object', 'bool']).columns\n\nlow_card_cols = [col for col in cat_cols if train_df[col].nunique() <= 200]\nhigh_card_cols = [col for col in cat_cols if col not in low_card_cols]\n\nfor col in low_card_cols:\n    le = LabelEncoder()\n    le.fit(pd.concat([train_df[col], test_df[col]]).astype(str))\n    train_df[col] = le.transform(train_df[col].astype(str))\n    test_df[col] = le.transform(test_df[col].astype(str))\n\ndef hash_encode(series, n_buckets=1000):\n    return series.astype(str).apply(lambda x: int(hashlib.md5(x.encode()).hexdigest(), 16) % n_buckets)\n\nfor col in high_card_cols:\n    train_df[col] = hash_encode(train_df[col])\n    test_df[col] = hash_encode(test_df[col])","metadata":{"trusted":true,"execution":{"execution_failed":"2025-06-25T22:53:28.359Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from lightgbm import LGBMRanker\n\ngroup_sizes = train_df.groupby('ranker_id').size()\nlarge_groups = group_sizes[group_sizes > 10000].index\nprint(f\"Removing {len(large_groups)} oversized groups\")\n\ntrain_df = train_df[~train_df['ranker_id'].isin(large_groups)]\n\nX_train = train_df[feature_cols]\ny_train = train_df['selected']\ngroup_train = train_df.groupby('ranker_id').size().values\n\nranker = LGBMRanker(\n    n_estimators=100,\n    random_state=42,\n    objective='lambdarank'  \n)\n\nranker.fit(X_train, y_train, group=group_train)\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-06-25T22:53:28.360Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = test_df[feature_cols]\ntest_df['score'] = ranker.predict(X_test)\n\ntest_df['rank'] = test_df.groupby('ranker_id')['score'].rank(ascending=False, method='first').astype(int)\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-06-25T22:53:28.360Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df = test_df[['Id', 'ranker_id', 'rank']].copy()\nsubmission_df.rename(columns={'rank': 'selected'}, inplace=True)\n\nsubmission_df.to_csv(\"/kaggle/working/submission.csv\", index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def hit_rate_at_3(df, true_col='selected', rank_col='rank'):\n    hit_count = 0\n    total = 0\n\n    for _, group in df.groupby('ranker_id'):\n        # Get all rows where selected == 1\n        true_rows = group[group[true_col] == 1]\n        top3 = group.nsmallest(3, rank_col)\n\n        if not true_rows.empty:\n            total += 1\n            # Check if any of the true rows are in top-3\n            if true_rows.index[0] in top3.index:\n                hit_count += 1\n\n    if total == 0:\n        print(\"⚠️ No valid ranker_id groups with selected=1 found. HitRate@3 undefined.\")\n        return 0.0\n\n    return hit_count / total\n\n# --- Evaluate (safe check) ---\nif 'selected' in test_df.columns:\n    hr3 = hit_rate_at_3(test_df)\n    print(f\"✅ HitRate@3: {hr3:.4f}\")\nelse:\n    print(\"⚠️ 'selected' column not found in test_df. Cannot compute HitRate@3.\")","metadata":{"trusted":true,"execution":{"execution_failed":"2025-06-25T22:53:28.360Z"}},"outputs":[],"execution_count":null}]}