{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":105399,"databundleVersionId":12733338,"sourceType":"competition"},{"sourceId":12602982,"sourceType":"datasetVersion","datasetId":7960530},{"sourceId":252145884,"sourceType":"kernelVersion"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-28T22:20:57.100530Z","iopub.execute_input":"2025-07-28T22:20:57.100831Z","iopub.status.idle":"2025-07-28T22:20:57.117725Z","shell.execute_reply.started":"2025-07-28T22:20:57.100808Z","shell.execute_reply":"2025-07-28T22:20:57.116655Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"🛫 Ranking Model for Flight Search — Baseline Notebook","metadata":{}},{"cell_type":"code","source":"# 📦 Step 1: Import required libraries\nimport pandas as pd\nimport numpy as np\nimport lightgbm as lgb\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:14:34.787471Z","iopub.execute_input":"2025-07-30T16:14:34.787772Z","iopub.status.idle":"2025-07-30T16:14:34.792558Z","shell.execute_reply.started":"2025-07-30T16:14:34.787747Z","shell.execute_reply":"2025-07-30T16:14:34.791671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 📄 Step 2: Load and preview the dataset\ndf = pd.read_csv(\"/kaggle/input/sample/ranking_sample.csv\")\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:14:37.866071Z","iopub.execute_input":"2025-07-30T16:14:37.866425Z","iopub.status.idle":"2025-07-30T16:14:37.904176Z","shell.execute_reply.started":"2025-07-30T16:14:37.866390Z","shell.execute_reply":"2025-07-30T16:14:37.903259Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 🧹 Step 3: Basic data cleaning\ndf.drop_duplicates(inplace=True)\n\n# Convert to datetime\ndf['requestDate'] = pd.to_datetime(df['requestDate'])\ndf['legs0_departureAt'] = pd.to_datetime(df['legs0_departureAt'])\ndf['legs0_arrivalAt'] = pd.to_datetime(df['legs0_arrivalAt'])\n\nif 'legs1_departureAt' in df.columns:\n    df['legs1_departureAt'] = pd.to_datetime(df['legs1_departureAt'])\n    df['legs1_arrivalAt'] = pd.to_datetime(df['legs1_arrivalAt'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:14:40.223146Z","iopub.execute_input":"2025-07-30T16:14:40.223436Z","iopub.status.idle":"2025-07-30T16:14:40.255407Z","shell.execute_reply.started":"2025-07-30T16:14:40.223414Z","shell.execute_reply":"2025-07-30T16:14:40.254481Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 🧠 Step 4: Feature engineering\ndf['total_duration'] = df['legs0_duration']\nif 'legs1_duration' in df.columns:\n    df['total_duration'] += df['legs1_duration']\n\ndf['days_before_departure'] = (df['legs0_departureAt'] - df['requestDate']).dt.days\ndf['tax_ratio'] = df['taxes'] / df['totalPrice']\n\n# Boolean to int\ndf['isVip'] = df['isVip'].astype(int)\ndf['hasFrequentFlyer'] = df['frequentFlyer'].apply(lambda x: int(pd.notnull(x) and str(x).strip() != ''))\ndf['bySelf'] = df['bySelf'].astype(int)\n\n# Fill missing values\ndf.fillna(0, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:14:42.784102Z","iopub.execute_input":"2025-07-30T16:14:42.784421Z","iopub.status.idle":"2025-07-30T16:14:42.805183Z","shell.execute_reply.started":"2025-07-30T16:14:42.784392Z","shell.execute_reply":"2025-07-30T16:14:42.804256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 🏷️ Step 5: Define features and target\nfeatures = [\n    'totalPrice', 'taxes', 'tax_ratio', 'total_duration',\n    'isVip', 'frequentFlyer', 'bySelf', 'days_before_departure',\n    'pricingInfo_passengerCount'\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:14:46.081564Z","iopub.execute_input":"2025-07-30T16:14:46.081835Z","iopub.status.idle":"2025-07-30T16:14:46.086780Z","shell.execute_reply.started":"2025-07-30T16:14:46.081815Z","shell.execute_reply":"2025-07-30T16:14:46.085816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[features].dtypes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:14:48.974727Z","iopub.execute_input":"2025-07-30T16:14:48.975062Z","iopub.status.idle":"2025-07-30T16:14:48.983776Z","shell.execute_reply.started":"2025-07-30T16:14:48.975036Z","shell.execute_reply":"2025-07-30T16:14:48.982832Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['total_duration'] = pd.to_numeric(df['total_duration'], errors='coerce')  # convert or set NaN\ndf['total_duration'].fillna(0, inplace=True)  # заменим NaN на 0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:14:51.789972Z","iopub.execute_input":"2025-07-30T16:14:51.790274Z","iopub.status.idle":"2025-07-30T16:14:51.796607Z","shell.execute_reply.started":"2025-07-30T16:14:51.790250Z","shell.execute_reply":"2025-07-30T16:14:51.795745Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['frequentFlyer'] = df['frequentFlyer'].apply(lambda x: 0 if pd.isna(x) or x in ['0', '', 'None'] else 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:14:54.671791Z","iopub.execute_input":"2025-07-30T16:14:54.672139Z","iopub.status.idle":"2025-07-30T16:14:54.678215Z","shell.execute_reply.started":"2025-07-30T16:14:54.672106Z","shell.execute_reply":"2025-07-30T16:14:54.677395Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[features].dtypes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:14:57.546070Z","iopub.execute_input":"2025-07-30T16:14:57.546384Z","iopub.status.idle":"2025-07-30T16:14:57.555703Z","shell.execute_reply.started":"2025-07-30T16:14:57.546359Z","shell.execute_reply":"2025-07-30T16:14:57.554944Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Проверим есть ли строки среди значений\nprint(df['total_duration'].apply(type).value_counts())\nprint(df['frequentFlyer'].apply(type).value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:15:00.510578Z","iopub.execute_input":"2025-07-30T16:15:00.510879Z","iopub.status.idle":"2025-07-30T16:15:00.518772Z","shell.execute_reply.started":"2025-07-30T16:15:00.510856Z","shell.execute_reply":"2025-07-30T16:15:00.517851Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = df[features]\ny = df['selected']\ngroup = df.groupby('ranker_id').size().to_list()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:15:03.776541Z","iopub.execute_input":"2025-07-30T16:15:03.776844Z","iopub.status.idle":"2025-07-30T16:15:03.784655Z","shell.execute_reply.started":"2025-07-30T16:15:03.776822Z","shell.execute_reply":"2025-07-30T16:15:03.783626Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✂️ Step 6: Split the dataset\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\ngroup_train = df.loc[X_train.index].groupby('ranker_id').size().to_list()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:15:06.402547Z","iopub.execute_input":"2025-07-30T16:15:06.402843Z","iopub.status.idle":"2025-07-30T16:15:06.416718Z","shell.execute_reply.started":"2025-07-30T16:15:06.402818Z","shell.execute_reply":"2025-07-30T16:15:06.415524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 🚂 Step 7: Train LightGBM Ranker\nranker = lgb.LGBMRanker(\n    objective='lambdarank',\n    metric='ndcg',\n    boosting_type='gbdt',\n    num_leaves=31,\n    learning_rate=0.05,\n    n_estimators=100\n)\n\nranker.fit(X_train, y_train, group=group_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:15:08.701727Z","iopub.execute_input":"2025-07-30T16:15:08.702041Z","iopub.status.idle":"2025-07-30T16:15:08.872138Z","shell.execute_reply.started":"2025-07-30T16:15:08.702015Z","shell.execute_reply":"2025-07-30T16:15:08.871296Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 📊 Step 8: Prediction and ranking\ndf_test = df.loc[X_test.index].copy()\ndf_test['pred'] = ranker.predict(X_test)\ndf_test['rank'] = df_test.groupby('ranker_id')['pred'].rank(ascending=False, method='first')\ndf_test[['ranker_id', 'pred', 'rank']].head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:15:57.318149Z","iopub.execute_input":"2025-07-30T16:15:57.318485Z","iopub.status.idle":"2025-07-30T16:15:57.345144Z","shell.execute_reply.started":"2025-07-30T16:15:57.318462Z","shell.execute_reply":"2025-07-30T16:15:57.344269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 🔍 Step 9: Feature importance\nlgb.plot_importance(ranker, max_num_features=10, importance_type='gain')\nplt.title(\"Top 10 Feature Importances\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:16:20.934470Z","iopub.execute_input":"2025-07-30T16:16:20.934776Z","iopub.status.idle":"2025-07-30T16:16:21.209542Z","shell.execute_reply.started":"2025-07-30T16:16:20.934752Z","shell.execute_reply":"2025-07-30T16:16:21.208448Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"используем это решение с файлом test.parquet","metadata":{}},{"cell_type":"code","source":"# Загрузка test-данных\ntest_df = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/test.parquet')\n\nprint(test_df.shape)\ntest_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:21:51.398165Z","iopub.execute_input":"2025-07-30T16:21:51.398496Z","iopub.status.idle":"2025-07-30T16:22:08.742970Z","shell.execute_reply.started":"2025-07-30T16:21:51.398475Z","shell.execute_reply":"2025-07-30T16:22:08.742182Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"2. Повтор фичеинженеринга (тот же, что и в train)","metadata":{}},{"cell_type":"code","source":"# 🧹 Step 3: Basic data cleaning\ntest_df.drop_duplicates(inplace=True)\n\n# Convert to datetime\ntest_df['requestDate'] = pd.to_datetime(test_df['requestDate'])\ntest_df['legs0_departureAt'] = pd.to_datetime(test_df['legs0_departureAt'])\ntest_df['legs0_arrivalAt'] = pd.to_datetime(test_df['legs0_arrivalAt'])\n\nif 'legs1_departureAt' in test_df.columns:\n    test_df['legs1_departureAt'] = pd.to_datetime(test_df['legs1_departureAt'])\n    test_df['legs1_arrivalAt'] = pd.to_datetime(test_df['legs1_arrivalAt'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:28:42.431146Z","iopub.execute_input":"2025-07-30T16:28:42.431429Z","iopub.status.idle":"2025-07-30T16:29:57.246236Z","shell.execute_reply.started":"2025-07-30T16:28:42.431408Z","shell.execute_reply":"2025-07-30T16:29:57.245178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 🧠 Step 4: Feature engineering\ntest_df['total_duration'] = test_df['legs0_duration']\nif 'legs1_duration' in test_df.columns:\n    test_df['total_duration'] += test_df['legs1_duration']\n\ntest_df['days_before_departure'] = (test_df['legs0_departureAt'] - test_df['requestDate']).dt.days\ntest_df['tax_ratio'] = test_df['taxes'] / test_df['totalPrice']\n\n# Boolean to int\ntest_df['isVip'] = test_df['isVip'].astype(int)\ntest_df['hasFrequentFlyer'] = test_df['frequentFlyer'].apply(lambda x: int(pd.notnull(x) and str(x).strip() != ''))\ntest_df['bySelf'] = test_df['bySelf'].astype(int)\n\n# Fill missing values\ntest_df.fillna(0, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:30:41.608503Z","iopub.execute_input":"2025-07-30T16:30:41.608882Z","iopub.status.idle":"2025-07-30T16:32:26.536487Z","shell.execute_reply.started":"2025-07-30T16:30:41.608850Z","shell.execute_reply":"2025-07-30T16:32:26.535501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df['total_duration'] = pd.to_numeric(test_df['total_duration'], errors='coerce')  # convert or set NaN\ntest_df['total_duration'].fillna(0, inplace=True)  # заменим NaN на 0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:32:31.631678Z","iopub.execute_input":"2025-07-30T16:32:31.631929Z","iopub.status.idle":"2025-07-30T16:32:31.678638Z","shell.execute_reply.started":"2025-07-30T16:32:31.631891Z","shell.execute_reply":"2025-07-30T16:32:31.677750Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df['frequentFlyer'] = test_df['frequentFlyer'].apply(lambda x: 0 if pd.isna(x) or x in ['0', '', 'None'] else 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:33:10.960406Z","iopub.execute_input":"2025-07-30T16:33:10.960729Z","iopub.status.idle":"2025-07-30T16:33:15.481742Z","shell.execute_reply.started":"2025-07-30T16:33:10.960702Z","shell.execute_reply":"2025-07-30T16:33:15.480791Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(test_df['total_duration'].apply(type).value_counts())\nprint(test_df['frequentFlyer'].apply(type).value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:34:12.182985Z","iopub.execute_input":"2025-07-30T16:34:12.183324Z","iopub.status.idle":"2025-07-30T16:34:14.738263Z","shell.execute_reply.started":"2025-07-30T16:34:12.183300Z","shell.execute_reply":"2025-07-30T16:34:14.737288Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Готовим признаки\nfeatures = [\n    'totalPrice', 'taxes', 'tax_ratio', 'total_duration',\n    'isVip', 'frequentFlyer', 'bySelf', 'days_before_departure',\n    'pricingInfo_passengerCount'\n]\n\nX_test = test_df[features].copy()\n\n# Убедимся, что всё числовое\nfor col in X_test.columns:\n    X_test[col] = pd.to_numeric(X_test[col], errors='coerce').fillna(0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:35:17.743639Z","iopub.execute_input":"2025-07-30T16:35:17.743971Z","iopub.status.idle":"2025-07-30T16:35:19.075775Z","shell.execute_reply.started":"2025-07-30T16:35:17.743942Z","shell.execute_reply":"2025-07-30T16:35:19.074939Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Предсказание (модель должна быть обучена ранее)\ntest_df['pred'] = ranker.predict(X_test)\n\n# Построение ранга внутри каждой группы\ntest_df['rank'] = test_df.groupby('ranker_id')['pred'].rank(ascending=False, method='first')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:36:37.352500Z","iopub.execute_input":"2025-07-30T16:36:37.352836Z","iopub.status.idle":"2025-07-30T16:37:06.322078Z","shell.execute_reply.started":"2025-07-30T16:36:37.352808Z","shell.execute_reply":"2025-07-30T16:37:06.321195Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Шаг 3: Фильтрация top-3\ntop3_df = test_df[test_df['rank'] <= 3].copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:49:16.507041Z","iopub.execute_input":"2025-07-30T16:49:16.507370Z","iopub.status.idle":"2025-07-30T16:49:18.076584Z","shell.execute_reply.started":"2025-07-30T16:49:16.507345Z","shell.execute_reply":"2025-07-30T16:49:18.075528Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"top3_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:49:37.358883Z","iopub.execute_input":"2025-07-30T16:49:37.359264Z","iopub.status.idle":"2025-07-30T16:49:37.382875Z","shell.execute_reply.started":"2025-07-30T16:49:37.359228Z","shell.execute_reply":"2025-07-30T16:49:37.381884Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"top3_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:50:34.747291Z","iopub.execute_input":"2025-07-30T16:50:34.747671Z","iopub.status.idle":"2025-07-30T16:50:34.753876Z","shell.execute_reply.started":"2025-07-30T16:50:34.747643Z","shell.execute_reply":"2025-07-30T16:50:34.752987Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(len(test_df.groupby('ranker_id')))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:52:04.606597Z","iopub.execute_input":"2025-07-30T16:52:04.606920Z","iopub.status.idle":"2025-07-30T16:52:05.906340Z","shell.execute_reply.started":"2025-07-30T16:52:04.606877Z","shell.execute_reply":"2025-07-30T16:52:05.905409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(test_df.groupby('ranker_id').size())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:53:23.362671Z","iopub.execute_input":"2025-07-30T16:53:23.362983Z","iopub.status.idle":"2025-07-30T16:53:24.064645Z","shell.execute_reply.started":"2025-07-30T16:53:23.362959Z","shell.execute_reply":"2025-07-30T16:53:24.063659Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Заполнить пропуски, например, самым большим ранком\nsubmission['rank'] = submission['rank'].fillna(9999).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:54:41.479774Z","iopub.execute_input":"2025-07-30T16:54:41.480183Z","iopub.status.idle":"2025-07-30T16:54:41.528965Z","shell.execute_reply.started":"2025-07-30T16:54:41.480151Z","shell.execute_reply":"2025-07-30T16:54:41.528029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = test_df[['Id', 'rank']].copy()\nsubmission['rank'] = submission['rank'].astype(int)  # Преобразуем rank к целому типу\nsubmission.to_csv('with_traning_set1.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:55:32.914496Z","iopub.execute_input":"2025-07-30T16:55:32.914781Z","iopub.status.idle":"2025-07-30T16:55:39.213818Z","shell.execute_reply.started":"2025-07-30T16:55:32.914760Z","shell.execute_reply":"2025-07-30T16:55:39.213086Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(submission.head())\nprint(submission.dtypes)\nprint(submission.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T16:55:19.107716Z","iopub.execute_input":"2025-07-30T16:55:19.108134Z","iopub.status.idle":"2025-07-30T16:55:19.115457Z","shell.execute_reply.started":"2025-07-30T16:55:19.108104Z","shell.execute_reply":"2025-07-30T16:55:19.114696Z"}},"outputs":[],"execution_count":null}]}