{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":105399,"databundleVersionId":12733338,"sourceType":"competition"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div style=\"background-color: #d3d3d3; padding: 10px; border: 5px solid #FF845E; border-radius: 10px; text-align: center;\">\n    <span style=\"color: blue; font-size: 24px; font-weight: bold; font-family: Arial, sans-serif;\">✔️ Welcome</span>\n</div>","metadata":{}},{"cell_type":"markdown","source":"## More information about metric: [Metric](https://www.kaggle.com/competitions/aeroclub-recsys-2025/discussion/585621)\n\n## More information, if you have pd.read_parquet problem: [Problem](https://www.kaggle.com/competitions/aeroclub-recsys-2025/discussion/585622)","metadata":{}},{"cell_type":"code","source":"!pip install xgboost > /dev/null","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T04:44:15.687171Z","iopub.execute_input":"2025-07-02T04:44:15.687995Z","iopub.status.idle":"2025-07-02T04:44:19.527545Z","shell.execute_reply.started":"2025-07-02T04:44:15.687956Z","shell.execute_reply":"2025-07-02T04:44:19.526678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport warnings\nimport numpy as np\nimport pandas as pd\nimport polars as pl # read train -> pd\nimport xgboost as xgb\n\nfrom sklearn.metrics import ndcg_score\nfrom ydata_profiling import ProfileReport # Exploratory Data Analysis(EDA)\nfrom sklearn.model_selection import train_test_split\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-02T04:44:19.529001Z","iopub.execute_input":"2025-07-02T04:44:19.529252Z","iopub.status.idle":"2025-07-02T04:44:24.295952Z","shell.execute_reply.started":"2025-07-02T04:44:19.529229Z","shell.execute_reply":"2025-07-02T04:44:24.295063Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Submission example","metadata":{}},{"cell_type":"code","source":"pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/sample_submission.parquet').head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T04:44:24.296742Z","iopub.execute_input":"2025-07-02T04:44:24.297119Z","iopub.status.idle":"2025-07-02T04:44:25.604608Z","shell.execute_reply.started":"2025-07-02T04:44:24.297100Z","shell.execute_reply":"2025-07-02T04:44:25.603906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pl.read_parquet('/kaggle/input/aeroclub-recsys-2025/train.parquet')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T04:44:25.606031Z","iopub.execute_input":"2025-07-02T04:44:25.606279Z","iopub.status.idle":"2025-07-02T04:44:37.884567Z","shell.execute_reply.started":"2025-07-02T04:44:25.606261Z","shell.execute_reply":"2025-07-02T04:44:37.883891Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T04:44:37.885359Z","iopub.execute_input":"2025-07-02T04:44:37.885595Z","iopub.status.idle":"2025-07-02T04:44:37.906363Z","shell.execute_reply.started":"2025-07-02T04:44:37.885577Z","shell.execute_reply":"2025-07-02T04:44:37.905616Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.select(['ranker_id', 'taxes', 'totalPrice', 'selected'])\ntrain = train.to_pandas()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T04:44:37.907241Z","iopub.execute_input":"2025-07-02T04:44:37.907783Z","iopub.status.idle":"2025-07-02T04:44:42.667090Z","shell.execute_reply.started":"2025-07-02T04:44:37.907762Z","shell.execute_reply":"2025-07-02T04:44:42.666209Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ProfileReport(train, title=\"EDA\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T04:44:42.668004Z","iopub.execute_input":"2025-07-02T04:44:42.668212Z","iopub.status.idle":"2025-07-02T04:46:29.323123Z","shell.execute_reply.started":"2025-07-02T04:44:42.668198Z","shell.execute_reply":"2025-07-02T04:46:29.322029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/test.parquet')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T04:46:29.324205Z","iopub.execute_input":"2025-07-02T04:46:29.324444Z","iopub.status.idle":"2025-07-02T04:46:42.381650Z","shell.execute_reply.started":"2025-07-02T04:46:29.324424Z","shell.execute_reply":"2025-07-02T04:46:42.380966Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T04:46:42.382715Z","iopub.execute_input":"2025-07-02T04:46:42.382927Z","iopub.status.idle":"2025-07-02T04:46:42.405901Z","shell.execute_reply.started":"2025-07-02T04:46:42.382910Z","shell.execute_reply":"2025-07-02T04:46:42.405332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['selected'].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T04:46:42.408742Z","iopub.execute_input":"2025-07-02T04:46:42.408960Z","iopub.status.idle":"2025-07-02T04:46:42.515667Z","shell.execute_reply.started":"2025-07-02T04:46:42.408945Z","shell.execute_reply":"2025-07-02T04:46:42.515068Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# set(train['Id']) & set(test['Id']) # return set() =(","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T04:46:42.516160Z","iopub.execute_input":"2025-07-02T04:46:42.516346Z","iopub.status.idle":"2025-07-02T04:46:42.519619Z","shell.execute_reply.started":"2025-07-02T04:46:42.516333Z","shell.execute_reply":"2025-07-02T04:46:42.518914Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Preparing for coding","metadata":{}},{"cell_type":"code","source":"train['ranker_id'] = train['ranker_id'].astype('category')\n\nX = train[['ranker_id', 'taxes', 'totalPrice']]\ny = train['selected']\n\ntrain_categories = X['ranker_id'].cat.categories","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T04:46:42.520334Z","iopub.execute_input":"2025-07-02T04:46:42.520951Z","iopub.status.idle":"2025-07-02T04:46:44.339358Z","shell.execute_reply.started":"2025-07-02T04:46:42.520926Z","shell.execute_reply":"2025-07-02T04:46:44.338699Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Creating an array of group sizes for a training dataset","metadata":{}},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(\n    X.copy(), y.copy(), test_size=0.2, random_state=42\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T04:46:44.340074Z","iopub.execute_input":"2025-07-02T04:46:44.340308Z","iopub.status.idle":"2025-07-02T04:46:46.393207Z","shell.execute_reply.started":"2025-07-02T04:46:44.340270Z","shell.execute_reply":"2025-07-02T04:46:46.392598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Обязательно сохраняем строковый ranker_id для группировки\nX_train['ranker_id_str'] = X_train['ranker_id'].astype(str)\nX_test['ranker_id_str'] = X_test['ranker_id'].astype(str)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T04:46:46.394374Z","iopub.execute_input":"2025-07-02T04:46:46.394597Z","iopub.status.idle":"2025-07-02T04:46:55.811561Z","shell.execute_reply.started":"2025-07-02T04:46:46.394579Z","shell.execute_reply":"2025-07-02T04:46:55.810672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#3. Кодируем ranker_id для подачи в модель:\n# Получаем категории\ntrain_categories = X_train['ranker_id'].astype('category').cat.categories\n\n# Применяем к train\nX_train['ranker_id'] = X_train['ranker_id'].astype('category')\nX_train['ranker_id'] = X_train['ranker_id'].cat.set_categories(train_categories)\nX_train['ranker_id_code'] = X_train['ranker_id'].cat.codes\n\n# Применяем те же категории к test\nX_test['ranker_id'] = X_test['ranker_id'].astype('category')\nX_test['ranker_id'] = X_test['ranker_id'].cat.set_categories(train_categories)\nX_test['ranker_id_code'] = X_test['ranker_id'].cat.codes\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T04:46:55.812394Z","iopub.execute_input":"2025-07-02T04:46:55.812618Z","iopub.status.idle":"2025-07-02T04:46:56.024363Z","shell.execute_reply.started":"2025-07-02T04:46:55.812600Z","shell.execute_reply":"2025-07-02T04:46:56.023710Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 4. Вычисляем group для модели:\ntrain_group_sizes = X_train.groupby('ranker_id_str').size().tolist()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T04:46:56.025114Z","iopub.execute_input":"2025-07-02T04:46:56.025356Z","iopub.status.idle":"2025-07-02T04:46:58.650959Z","shell.execute_reply.started":"2025-07-02T04:46:56.025338Z","shell.execute_reply":"2025-07-02T04:46:58.650112Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Пример базовых params для задачи ранжирования (Learning to Rank):\nparams = {\n    \"objective\": \"lambdarank\",     # или \"rank_xendcg\"\n    \"metric\": \"ndcg\",              # или \"map\", \"ndcg@10\"\n    \"ndcg_eval_at\": [1, 3, 5, 10], # на каких позициях оценивать\n    \"learning_rate\": 0.1,\n    \"num_leaves\": 31,\n    \"min_data_in_leaf\": 20,\n    \"verbose\": -1\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T04:46:58.651751Z","iopub.execute_input":"2025-07-02T04:46:58.651989Z","iopub.status.idle":"2025-07-02T04:46:58.656141Z","shell.execute_reply.started":"2025-07-02T04:46:58.651972Z","shell.execute_reply":"2025-07-02T04:46:58.655334Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#📦 Теперь для обучения модели:\nimport lightgbm as lgb\n\ntrain_data = lgb.Dataset(\n    X_train[['ranker_id_code', 'taxes', 'totalPrice']],\n    label=y_train,\n    group=train_group_sizes\n)\n\n# Обучение\nmodel = lgb.train(params, train_data)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T04:46:58.656965Z","iopub.execute_input":"2025-07-02T04:46:58.657632Z","iopub.status.idle":"2025-07-02T04:48:50.294809Z","shell.execute_reply.started":"2025-07-02T04:46:58.657610Z","shell.execute_reply":"2025-07-02T04:48:50.294084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test['score'] = model.predict(X_test[['ranker_id_code', 'taxes', 'totalPrice']])\n\nX_test['rank'] = (\n    X_test.sort_values(['ranker_id_str', 'score'], ascending=[True, False])\n          .groupby('ranker_id_str')\n          .cumcount() + 1\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T04:48:50.295414Z","iopub.execute_input":"2025-07-02T04:48:50.296260Z","iopub.status.idle":"2025-07-02T04:49:01.407401Z","shell.execute_reply.started":"2025-07-02T04:48:50.296235Z","shell.execute_reply":"2025-07-02T04:49:01.406741Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import ndcg_score\n\n# Предсказанные скоры\nX_test['score'] = model.predict(X_test[['ranker_id_code', 'taxes', 'totalPrice']])\nX_test['true_label'] = y_test.values\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T04:49:01.408255Z","iopub.execute_input":"2025-07-02T04:49:01.408555Z","iopub.status.idle":"2025-07-02T04:49:07.906099Z","shell.execute_reply.started":"2025-07-02T04:49:01.408531Z","shell.execute_reply":"2025-07-02T04:49:07.905511Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(y_test.shape)  # Должно быть (1, N)\nprint(type(y_test))  # Должен быть numpy.ndarray","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T04:49:07.906722Z","iopub.execute_input":"2025-07-02T04:49:07.906985Z","iopub.status.idle":"2025-07-02T04:49:07.911231Z","shell.execute_reply.started":"2025-07-02T04:49:07.906961Z","shell.execute_reply":"2025-07-02T04:49:07.910338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import ndcg_score\nimport numpy as np\n\nndcgs = []\n\nfor _, group in X_test.groupby('ranker_id_str'):\n    if len(group) < 2:\n        continue  # ⚠️ Пропускаем слишком маленькие группы\n\n    true_labels = group['true_label'].values\n    scores = group['score'].values\n\n    if true_labels.sum() == 0:\n        continue  # ⚠️ Пропускаем группы без выбранного рейса\n\n    y_true = np.array([true_labels])\n    y_score = np.array([scores])\n\n    ndcg = ndcg_score(y_true, y_score)\n    ndcgs.append(ndcg)\n\nprint(f\"Mean NDCG: {np.mean(ndcgs):.4f}\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T04:51:51.690507Z","iopub.execute_input":"2025-07-02T04:51:51.690807Z","iopub.status.idle":"2025-07-02T04:52:07.908162Z","shell.execute_reply.started":"2025-07-02T04:51:51.690787Z","shell.execute_reply":"2025-07-02T04:52:07.907334Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### XGBoost Training","metadata":{}},{"cell_type":"code","source":"# 1. Категории из train\ntest['ranker_id_cat'] = test['ranker_id'].astype('category')\ntest['ranker_id_cat'] = test['ranker_id_cat'].cat.set_categories(train_categories)\ntest['ranker_id_code'] = test['ranker_id_cat'].cat.codes\ntest['ranker_id_str'] = test['ranker_id'].astype(str)\n\n# 2. Предсказание\ntest['score'] = model.predict(test[['ranker_id_code', 'taxes', 'totalPrice']])\n\n# 3. Ранжирование внутри каждого ranker_id\ntest['selected'] = (\n    test.sort_values(['ranker_id_str', 'score'], ascending=[True, False])\n        .groupby('ranker_id_str')\n        .cumcount() + 1\n)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T04:54:53.914217Z","iopub.execute_input":"2025-07-02T04:54:53.915229Z","iopub.status.idle":"2025-07-02T04:55:17.571005Z","shell.execute_reply.started":"2025-07-02T04:54:53.915191Z","shell.execute_reply":"2025-07-02T04:55:17.570160Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 4. Сохраняем submission\nsubmission = test[['Id', 'ranker_id', 'selected']]\nsubmission.to_csv('OneLove56.csv', index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T04:55:23.074673Z","iopub.execute_input":"2025-07-02T04:55:23.074958Z","iopub.status.idle":"2025-07-02T04:55:35.294003Z","shell.execute_reply.started":"2025-07-02T04:55:23.074936Z","shell.execute_reply":"2025-07-02T04:55:35.293176Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(submission.head(10))\nprint(submission['selected'].min(), submission['selected'].max())\nprint(submission.groupby('ranker_id').size().head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T05:00:46.133395Z","iopub.execute_input":"2025-07-02T05:00:46.133744Z","iopub.status.idle":"2025-07-02T05:00:46.800948Z","shell.execute_reply.started":"2025-07-02T05:00:46.133723Z","shell.execute_reply":"2025-07-02T05:00:46.800273Z"}},"outputs":[],"execution_count":null}]}