{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":105399,"databundleVersionId":12733338}],"dockerImageVersionId":31259,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport polars as pl\nimport polars.selectors as cs\nfrom datetime import datetime\nfrom collections import Counter\nfrom sklearn.model_selection import train_test_split\nimport xgboost as xgb\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder\nimport matplotlib.pyplot as plt\nimport gc\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-01T07:10:16.228336Z","iopub.execute_input":"2026-05-01T07:10:16.228951Z","iopub.status.idle":"2026-05-01T07:10:17.958833Z","shell.execute_reply.started":"2026-05-01T07:10:16.228922Z","shell.execute_reply":"2026-05-01T07:10:17.95818Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pl.read_parquet('/kaggle/input/aeroclub-recsys-2025/train.parquet', n_rows=10_000_000).drop('__index_level_0__')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T07:10:19.498765Z","iopub.execute_input":"2026-05-01T07:10:19.499422Z","iopub.status.idle":"2026-05-01T07:10:25.383872Z","shell.execute_reply.started":"2026-05-01T07:10:19.499395Z","shell.execute_reply":"2026-05-01T07:10:25.383009Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T01:11:51.529513Z","iopub.execute_input":"2026-04-26T01:11:51.529849Z","iopub.status.idle":"2026-04-26T01:11:51.5596Z","shell.execute_reply.started":"2026-04-26T01:11:51.52982Z","shell.execute_reply":"2026-04-26T01:11:51.558413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.null_count()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-07T05:51:43.029477Z","iopub.execute_input":"2026-04-07T05:51:43.029818Z","iopub.status.idle":"2026-04-07T05:51:43.039118Z","shell.execute_reply.started":"2026-04-07T05:51:43.029792Z","shell.execute_reply":"2026-04-07T05:51:43.037975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.schema","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-07T06:03:18.986828Z","iopub.execute_input":"2026-04-07T06:03:18.987253Z","iopub.status.idle":"2026-04-07T06:03:18.99802Z","shell.execute_reply.started":"2026-04-07T06:03:18.987215Z","shell.execute_reply":"2026-04-07T06:03:18.996688Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Counter(train.dtypes)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T06:28:05.595576Z","iopub.execute_input":"2026-04-26T06:28:05.596361Z","iopub.status.idle":"2026-04-26T06:28:05.603808Z","shell.execute_reply.started":"2026-04-26T06:28:05.59633Z","shell.execute_reply":"2026-04-26T06:28:05.60274Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['selected'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-07T05:54:38.133565Z","iopub.execute_input":"2026-04-07T05:54:38.133848Z","iopub.status.idle":"2026-04-07T05:54:38.346348Z","shell.execute_reply.started":"2026-04-07T05:54:38.133828Z","shell.execute_reply":"2026-04-07T05:54:38.345445Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"null_cols_to_drop = []\nconstant_cols = []\ncorr_to_drop = set()\n\ndef feature_preparation(df, is_train=True):\n\n    target = []\n    groups_train = []\n    test_ids = []\n    test_ranker_ids = []\n    \n    df = df.cast({col: pl.Int32 for col in df.select(pl.col(pl.Int64)).columns})\n    df = df.cast({col: pl.Float32 for col in df.select(pl.col(pl.Float64)).columns})\n\n    if is_train:\n        target = df['selected'].to_numpy().ravel()\n        groups_train = df.select(\"ranker_id\")\n    else:\n        test_ids = df['Id'].to_numpy()\n        test_ranker_ids = df['ranker_id'].to_numpy()\n    \n\n    id_cols = ['Id', 'companyID', 'profileId', 'ranker_id']\n    feature_cols = [col for col in df.columns if col not in id_cols and col != \"selected\"]\n    df = df[feature_cols]\n\n    return df, target, groups_train, test_ids, test_ranker_ids\n\ndef feature_processing(df, is_train=True):\n\n    global null_cols_to_drop\n    global constant_cols\n    global corr_to_drop\n\n    threshold = 0.9\n\n    cols_to_keep = [\n        col for col in df.columns \n        if df[col].null_count() / len(df) <= threshold\n    ]\n\n    df = df.select(cols_to_keep)\n    \n    bool_cols = df.select(cs.boolean()).columns\n    for col in bool_cols:\n        df = df.with_columns(pl.col(col).cast(pl.Int32))\n\n    string_to_date = ['legs0_arrivalAt', 'legs0_departureAt', 'legs1_arrivalAt', 'legs1_departureAt']\n    for col in string_to_date:\n        df = df.with_columns(pl.col(col).str.to_datetime(time_unit='ns'))\n\n    date_cols = df.select(cs.datetime()).columns\n    for col in date_cols:\n        df = df.with_columns(\n            pl.col(col).dt.year().alias(col + \"_year\"),\n            pl.col(col).dt.month().alias(col + \"_month\"),\n            pl.col(col).dt.day().alias(col + \"_day\"),\n            pl.col(col).dt.hour().alias(col + \"_hour\"),\n            pl.col(col).dt.minute().alias(col + \"_minute\"),\n            pl.col(col).dt.second().alias(col + \"_second\")\n        )\n    \n    df = df.drop(date_cols)\n\n    string_with_duration_to_int = df.select(cs.contains(\"duration\")).columns\n    for col in string_with_duration_to_int:\n        df = df.with_columns([\n    \n        pl.col(col)\n        .str.extract(r\"^(\\d+)\\.\", group_index=1)\n        .cast(pl.Int32)\n        .fill_null(0)\n        .alias(col + \"_days\"),\n    \n        pl.col(col)\n        .str.extract(r\"(?:^(\\d+)\\.)?(\\d+):\", group_index=2)\n        .cast(pl.Int32)\n        .fill_null(0)\n        .alias(col + \"_hours\"),\n    \n        pl.col(col)\n        .str.extract(r\":(\\d+):\", group_index=1)\n        .cast(pl.Int32)\n        .fill_null(0)\n        .alias(col + \"_minutes\"),\n    \n        pl.col(col)\n        .str.extract(r\":(\\d+)$\", group_index=1)\n        .cast(pl.Int32)\n        .fill_null(0)\n        .alias(col + \"_seconds\")\n    ])\n        \n        df = df.with_columns(\n        (pl.col(col + \"_days\") * 24 + pl.col(col + \"_hours\")).alias(col + \"_total_hours\")\n        )\n        \n\n    df = df.drop(string_with_duration_to_int)\n\n    string_with_flightnumber_to_int = df.select(cs.contains(\"flightNumber\")).columns\n    for col in string_with_flightnumber_to_int:\n        df = df.with_columns(pl.col(col).cast(pl.Int32))\n\n    categorical_cols = df.select(cs.string()).columns\n    for col in categorical_cols:\n        df = df.with_columns(pl.col(col).cast(pl.Categorical).to_physical())\n\n    df = df.with_columns([\n        (pl.col(\"totalPrice\") / (pl.col(\"taxes\") + 1)).alias(\"price_per_tax\"),\n        (pl.col(\"taxes\") / (pl.col(\"totalPrice\") + 1)).alias(\"tax_rate\"),\n        pl.col(\"totalPrice\").log1p().alias(\"log_price\")])\n    \n    if is_train:\n        \n        null_counts = df.null_count()\n        total_rows = len(df)\n\n        null_cols_to_drop = [\n            col for col in df.columns \n            if null_counts.get_column(col)[0] / total_rows >= 0.8\n        ]\n\n        df = df.drop(null_cols_to_drop)\n        df = df.fill_null(0)\n\n        constant_cols = [\n            col for col in df.columns \n            if df[col].n_unique() == 1\n        ]\n\n        df = df.drop(constant_cols)\n\n        corr_matrix = df.select(cs.numeric()).corr()\n        columns = corr_matrix.columns\n\n        matrix_values = corr_matrix.to_numpy()\n\n        threshold = 0.8\n\n        for i in range(len(columns)):\n            for j in range(i + 1, len(columns)):\n                if abs(matrix_values[i, j]) > threshold:\n                    col_name = columns[j]\n                    corr_to_drop.add(col_name)\n\n        df = df.drop(corr_to_drop)\n\n    else:\n\n        df = df.drop(null_cols_to_drop)\n        df = df.fill_null(0)\n        df = df.drop(constant_cols)\n        df = df.drop(corr_to_drop)\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T07:10:25.385368Z","iopub.execute_input":"2026-05-01T07:10:25.385716Z","iopub.status.idle":"2026-05-01T07:10:25.471603Z","shell.execute_reply.started":"2026-05-01T07:10:25.385689Z","shell.execute_reply":"2026-05-01T07:10:25.470882Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data, target, groups_train, test_ids, test_ranker_ids = feature_preparation(train)\n\ndel train\ngc.collect()\n\ntrain_data = feature_processing(train_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T07:10:36.194854Z","iopub.execute_input":"2026-05-01T07:10:36.195409Z","iopub.status.idle":"2026-05-01T07:11:34.263533Z","shell.execute_reply.started":"2026-05-01T07:10:36.195387Z","shell.execute_reply":"2026-05-01T07:11:34.26257Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T01:13:20.582741Z","iopub.execute_input":"2026-05-01T01:13:20.583094Z","iopub.status.idle":"2026-05-01T01:13:20.599933Z","shell.execute_reply.started":"2026-05-01T01:13:20.583072Z","shell.execute_reply":"2026-05-01T01:13:20.599358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = train_data\ny = target","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T07:11:34.264919Z","iopub.execute_input":"2026-05-01T07:11:34.265238Z","iopub.status.idle":"2026-05-01T07:11:34.268538Z","shell.execute_reply.started":"2026-05-01T07:11:34.265217Z","shell.execute_reply":"2026-05-01T07:11:34.267961Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"group_sizes_train = (groups_train\n                    .group_by('ranker_id', maintain_order=True)\n                    .agg(pl.len())['len']\n                    .to_numpy())\n\ndtrain = xgb.DMatrix(\n    X, \n    label=y, \n    group=group_sizes_train\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T07:11:34.269357Z","iopub.execute_input":"2026-05-01T07:11:34.269642Z","iopub.status.idle":"2026-05-01T07:11:45.555066Z","shell.execute_reply.started":"2026-05-01T07:11:34.269624Z","shell.execute_reply":"2026-05-01T07:11:45.554411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgb_params = {\n    'objective': 'rank:pairwise',  \n    'eval_metric': 'ndcg@3',       \n    'max_depth': 6,                \n    'min_child_weight': 10,       \n    'subsample': 0.8,             \n    'colsample_bytree': 0.8,       \n    'reg_lambda': 10.0,            \n    'learning_rate': 0.05,         \n    'random_state': 42,\n    'device': 'cuda',\n    'n_jobs': -1                  \n}\n\nxgb_model = xgb.train(\n    xgb_params,\n    dtrain,\n    num_boost_round=1000,\n    verbose_eval=100\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T07:11:45.556512Z","iopub.execute_input":"2026-05-01T07:11:45.556919Z","iopub.status.idle":"2026-05-01T07:16:30.166766Z","shell.execute_reply.started":"2026-05-01T07:11:45.556887Z","shell.execute_reply":"2026-05-01T07:16:30.166154Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def hitrate_at_3(y_true, y_pred, groups):\n    df = pl.DataFrame({\n        'group': groups,\n        'pred': y_pred,\n        'true': y_true\n    })\n    \n    return (\n        df.filter(pl.col(\"group\").count().over(\"group\") > 10)\n        .sort([\"group\", \"pred\"], descending=[False, True])\n        .group_by(\"group\", maintain_order=True)\n        .head(3)\n        .group_by(\"group\")\n        .agg(pl.col(\"true\").max())\n        .select(pl.col(\"true\").mean())\n        .item()\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T07:16:30.167472Z","iopub.execute_input":"2026-05-01T07:16:30.167739Z","iopub.status.idle":"2026-05-01T07:16:30.173032Z","shell.execute_reply.started":"2026-05-01T07:16:30.167721Z","shell.execute_reply":"2026-05-01T07:16:30.172318Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Predictions on train...\")\ntrain_predictions_xgb = xgb_model.predict(dtrain)\n\nhitrate_xgb = hitrate_at_3(y, train_predictions_xgb, groups_train.to_numpy().ravel())\nprint(f\"XGBoost HitRate@3 on train: {hitrate_xgb:.5f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T07:16:30.173915Z","iopub.execute_input":"2026-05-01T07:16:30.174238Z","iopub.status.idle":"2026-05-01T07:16:39.750561Z","shell.execute_reply.started":"2026-05-01T07:16:30.174219Z","shell.execute_reply":"2026-05-01T07:16:39.749898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = pl.read_parquet('/kaggle/input/aeroclub-recsys-2025/test.parquet').drop('__index_level_0__')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T05:38:58.69439Z","iopub.execute_input":"2026-05-01T05:38:58.69486Z","iopub.status.idle":"2026-05-01T05:39:02.523947Z","shell.execute_reply.started":"2026-05-01T05:38:58.694828Z","shell.execute_reply":"2026-05-01T05:39:02.521931Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data, target, groups_train, test_ids, test_ranker_ids = feature_preparation(test, is_train=False)\n\ndel test\ngc.collect()\n\ntest_data = feature_processing(test_data, is_train=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T05:39:02.52517Z","iopub.execute_input":"2026-05-01T05:39:02.525419Z","iopub.status.idle":"2026-05-01T05:39:30.516895Z","shell.execute_reply.started":"2026-05-01T05:39:02.525398Z","shell.execute_reply":"2026-05-01T05:39:30.516075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-30T03:34:05.247938Z","iopub.execute_input":"2026-04-30T03:34:05.248216Z","iopub.status.idle":"2026-04-30T03:34:05.272761Z","shell.execute_reply.started":"2026-04-30T03:34:05.248189Z","shell.execute_reply":"2026-04-30T03:34:05.271666Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dtest = xgb.DMatrix(test_data)\ntest_predictions_xgb = xgb_model.predict(dtest)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T05:39:30.518212Z","iopub.execute_input":"2026-05-01T05:39:30.518636Z","iopub.status.idle":"2026-05-01T05:39:52.523081Z","shell.execute_reply.started":"2026-05-01T05:39:30.518614Z","shell.execute_reply":"2026-05-01T05:39:52.522403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result = pl.DataFrame({\n    'Id': test_ids,\n    'ranker_id': test_ranker_ids,\n    'pred_score': test_predictions_xgb\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T05:39:52.52409Z","iopub.execute_input":"2026-05-01T05:39:52.524539Z","iopub.status.idle":"2026-05-01T05:39:53.209812Z","shell.execute_reply.started":"2026-05-01T05:39:52.524518Z","shell.execute_reply":"2026-05-01T05:39:53.208783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result = result.with_columns(\n    pl.col('pred_score').rank('ordinal', descending=True)\n    .over('ranker_id')\n    .alias('selected')\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T05:39:53.211596Z","iopub.execute_input":"2026-05-01T05:39:53.211985Z","iopub.status.idle":"2026-05-01T05:39:53.756668Z","shell.execute_reply.started":"2026-05-01T05:39:53.211965Z","shell.execute_reply":"2026-05-01T05:39:53.756042Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = result.select(['Id', 'ranker_id', 'selected'])\nsubmission.write_csv(f'submission_14.csv')\nprint(f\"Submission saved\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T05:39:53.75758Z","iopub.execute_input":"2026-05-01T05:39:53.757854Z","iopub.status.idle":"2026-05-01T05:39:54.916129Z","shell.execute_reply.started":"2026-05-01T05:39:53.757827Z","shell.execute_reply":"2026-05-01T05:39:54.915225Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"importance_dict = xgb_model.get_score(importance_type='gain')\nimportance_df = pl.DataFrame([\n    {'feature': k, 'importance': v} \n    for k, v in importance_dict.items()\n]).sort('importance', descending=True)\n\nprint(\"Топ-20 importance features:\")\nprint(importance_df.head(20))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-29T10:07:04.30535Z","iopub.execute_input":"2026-04-29T10:07:04.30568Z","iopub.status.idle":"2026-04-29T10:07:04.319748Z","shell.execute_reply.started":"2026-04-29T10:07:04.305656Z","shell.execute_reply":"2026-04-29T10:07:04.318927Z"}},"outputs":[],"execution_count":null}]}