{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":8586,"databundleVersionId":868729,"isSourceIdPinned":false}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# В обучающей выборке 1000 объектов и 40 признаков. В Тестовой выборке 9000 объектов и 40 признаков\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-04-13T06:25:43.583988Z","iopub.execute_input":"2026-04-13T06:25:43.584771Z","iopub.status.idle":"2026-04-13T06:25:43.592071Z","shell.execute_reply.started":"2026-04-13T06:25:43.584733Z","shell.execute_reply":"2026-04-13T06:25:43.591061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport lightgbm as lgb\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_squared_error","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T06:25:45.315425Z","iopub.execute_input":"2026-04-13T06:25:45.316219Z","iopub.status.idle":"2026-04-13T06:25:45.320936Z","shell.execute_reply.started":"2026-04-13T06:25:45.316181Z","shell.execute_reply":"2026-04-13T06:25:45.319796Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA = '/kaggle/input/competitions/avito-demand-prediction/'\ntrain = pd.read_csv(DATA + 'train.csv', nrows=400000, parse_dates=['activation_date']) # берём меньше строк, чтобы обучение быстрее\ntest  = pd.read_csv(DATA + 'test.csv', parse_dates=['activation_date'])\n\nprint(f\"Train shape: {train.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T06:26:57.241818Z","iopub.execute_input":"2026-04-13T06:26:57.242629Z","iopub.status.idle":"2026-04-13T06:27:11.540247Z","shell.execute_reply.started":"2026-04-13T06:26:57.242592Z","shell.execute_reply":"2026-04-13T06:27:11.539223Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_features(df):\n    df = df.copy()\n    \n    # Текстовые признаки\n    df['title_len'] = df['title'].fillna('').str.len()\n    df['desc_len']  = df['description'].fillna('').str.len()\n    \n    # Цена\n    df['price'] = df['price'].fillna(-1)\n    df['log_price'] = np.log1p(df['price'].clip(0))\n    \n    # Временные признаки\n    df['dayofweek'] = df['activation_date'].dt.dayofweek\n    \n    return df\n\ntrain = create_features(train)\ntest  = create_features(test)\n\n# Категориальные признаки\ncat_cols = ['region', 'city', 'category_name', 'user_type', 'param_1']\nfor col in cat_cols:\n    if col in train.columns:\n        train[col] = train[col].astype('category')\n        test[col]  = test[col].astype('category')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T06:27:13.458465Z","iopub.execute_input":"2026-04-13T06:27:13.459428Z","iopub.status.idle":"2026-04-13T06:27:15.079259Z","shell.execute_reply.started":"2026-04-13T06:27:13.45939Z","shell.execute_reply":"2026-04-13T06:27:15.07822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Признаки для модели\nfeatures = ['title_len', 'desc_len', 'price', 'log_price', 'dayofweek',\n            'region', 'city', 'category_name', 'user_type']\n\nX = train[features]\ny = train['deal_probability']\nX_test = test[features]\n\nprint(f\"Используем {len(features)} признаков\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T06:27:17.056162Z","iopub.execute_input":"2026-04-13T06:27:17.056909Z","iopub.status.idle":"2026-04-13T06:27:17.082763Z","shell.execute_reply.started":"2026-04-13T06:27:17.056874Z","shell.execute_reply":"2026-04-13T06:27:17.081796Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# LightGBM\nparams = {\n    'objective': 'regression',\n    'metric': 'rmse',\n    'learning_rate': 0.05,\n    'num_leaves': 31,\n    'n_estimators': 500,\n    'random_state': 42,\n}\n\nkf = KFold(n_splits=5, shuffle=True, random_state=42)\ntest_preds = np.zeros(len(X_test))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T06:27:19.995832Z","iopub.execute_input":"2026-04-13T06:27:19.996252Z","iopub.status.idle":"2026-04-13T06:27:20.002582Z","shell.execute_reply.started":"2026-04-13T06:27:19.99622Z","shell.execute_reply":"2026-04-13T06:27:20.001525Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for fold, (train_idx, val_idx) in enumerate(kf.split(X)):\n    X_tr, X_val = X.iloc[train_idx], X.iloc[val_idx]\n    y_tr, y_val = y.iloc[train_idx], y.iloc[val_idx]\n    \n    model = lgb.LGBMRegressor(**params)\n    model.fit(X_tr, y_tr,\n              eval_set=[(X_val, y_val)],\n              callbacks=[lgb.early_stopping(50, verbose=False)])\n    \n    test_preds += model.predict(X_test) / 5\n    print(f\"Fold {fold+1} завершён\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T06:27:21.997696Z","iopub.execute_input":"2026-04-13T06:27:21.998711Z","iopub.status.idle":"2026-04-13T06:28:32.009355Z","shell.execute_reply.started":"2026-04-13T06:27:21.998663Z","shell.execute_reply":"2026-04-13T06:28:32.00827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Ограничиваем вероятность диапазоном [0, 1]\ntest_preds = np.clip(test_preds, 0, 1)\n\nsubmission = pd.DataFrame({\n    'item_id': test['item_id'],\n    'deal_probability': test_preds\n})\nsubmission.to_csv('submission.csv', index=False)\nprint(\"submission.csv сохранён\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T06:28:34.549688Z","iopub.execute_input":"2026-04-13T06:28:34.550034Z","iopub.status.idle":"2026-04-13T06:28:35.812549Z","shell.execute_reply.started":"2026-04-13T06:28:34.549918Z","shell.execute_reply":"2026-04-13T06:28:35.811657Z"}},"outputs":[],"execution_count":null}]}