{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":13211974,"sourceType":"datasetVersion","datasetId":8373971}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ====================================================\n# 📦 Імпорти\n# ====================================================\nimport gc\nimport pandas as pd\nimport numpy as np\nimport xgboost as xgb\nfrom sklearn.model_selection import train_test_split\n\n# ====================================================\n# 📂 Завантаження даних\n# ====================================================\ntrain = pd.read_parquet('/kaggle/input/pre-aggregated-amex-dataset-v1/train_final.parquet')\n\n# ====================================================\n# 🧹 Підготовка\n# ====================================================\nX = train.drop(['customer_ID', 'target'], axis=1)\ny = train['target']\n\nX = X.select_dtypes(include=[np.number])\n\ndel train\ngc.collect()\n\n# Розділення train/valid\nX_train, X_valid, y_train, y_valid = train_test_split(\n    X, y, test_size=0.2, random_state=42, stratify=y\n)\n\ndel X, y\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T21:34:52.997329Z","iopub.execute_input":"2025-10-06T21:34:52.997823Z","iopub.status.idle":"2025-10-06T21:35:00.023606Z","shell.execute_reply.started":"2025-10-06T21:34:52.997802Z","shell.execute_reply":"2025-10-06T21:35:00.023023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ====================================================\n# 🧮 AMEX Metric (official)\n# ====================================================\ndef amex_metric(y_true, y_pred):\n    labels = np.transpose(np.array([y_true, y_pred]))\n    labels = labels[labels[:, 1].argsort()[::-1]]\n    weights = np.where(labels[:,0]==0, 20, 1)\n    cut_vals = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n    gini = [0,0]\n    for i in [1,0]:\n        labels = np.transpose(np.array([y_true, y_pred]))\n        labels = labels[labels[:, i].argsort()[::-1]]\n        weight = np.where(labels[:,0]==0, 20, 1)\n        weight_random = np.cumsum(weight / np.sum(weight))\n        total_pos = np.sum(labels[:, 0] *  weight)\n        cum_pos_found = np.cumsum(labels[:, 0] * weight)\n        lorentz = cum_pos_found / total_pos\n        gini[i] = np.sum((lorentz - weight_random) * weight)\n    return 0.5 * (gini[1]/gini[0] + top_four)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T21:39:42.115375Z","iopub.execute_input":"2025-10-06T21:39:42.115632Z","iopub.status.idle":"2025-10-06T21:39:42.121870Z","shell.execute_reply.started":"2025-10-06T21:39:42.115616Z","shell.execute_reply":"2025-10-06T21:39:42.121076Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ====================================================\n# ⚙️ GPU XGBoost параметри\n# ====================================================\nparams = {\n    \"objective\": \"binary:logistic\",\n    \"eval_metric\": \"auc\",\n    \"tree_method\": \"hist\",   # <-- більше не gpu_hist\n    \"device\": \"cuda\",        # <-- ось цей параметр активує GPU\n    \"learning_rate\": 0.05,\n    \"max_depth\": 6,\n    \"subsample\": 0.8,\n    \"colsample_bytree\": 0.8,\n    \"seed\": 42,\n    \"nthread\": -1,\n}\n\n\n# ====================================================\n# 🧠 DMatrix + тренування\n# ====================================================\ndtrain = xgb.DMatrix(X_train, label=y_train)\ndvalid = xgb.DMatrix(X_valid, label=y_valid)\n\nmodel = xgb.train(\n    params=params,\n    dtrain=dtrain,\n    num_boost_round=2000,\n    evals=[(dtrain, \"train\"), (dvalid, \"valid\")],\n    early_stopping_rounds=100,\n    verbose_eval=100\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T21:36:01.731828Z","iopub.execute_input":"2025-10-06T21:36:01.732064Z","iopub.status.idle":"2025-10-06T21:38:10.366992Z","shell.execute_reply.started":"2025-10-06T21:36:01.732049Z","shell.execute_reply":"2025-10-06T21:38:10.366097Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ====================================================\n# 💾 Збереження моделі\n# ====================================================\nmodel.save_model(\"xgb_amex_gpu.json\")\nprint(\"💾 Модель збережено у файл xgb_amex_gpu.json\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T21:41:57.053110Z","iopub.execute_input":"2025-10-06T21:41:57.053823Z","iopub.status.idle":"2025-10-06T21:41:57.126375Z","shell.execute_reply.started":"2025-10-06T21:41:57.053800Z","shell.execute_reply":"2025-10-06T21:41:57.125702Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport gc\nimport pandas as pd\nimport numpy as np\nimport pyarrow.parquet as pq\nimport xgboost as xgb\n\n# ====================================================\n# 💾 Завантаження моделі\n# ====================================================\nmodel = xgb.Booster()\nmodel.load_model(\"xgb_amex_gpu.json\")\nprint(\"✅ Модель успішно завантажена\")\n\n# ====================================================\n# ⚙️ Шлях до тестових parquet-файлів\n# ====================================================\nTEST_DIR = '/kaggle/input/pre-aggregated-amex-dataset-v1/test_agg.parquet'\nOUTPUT_FILE = 'submission.csv'\n\n# Отримуємо список усіх файлів у цій папці\ntest_files = sorted([os.path.join(TEST_DIR, f) for f in os.listdir(TEST_DIR) if f.endswith('.parquet')])\nprint(f\"📦 Знайдено {len(test_files)} parquet-файлів\")\n\n# Якщо файл уже існує — видаляємо\nif os.path.exists(OUTPUT_FILE):\n    os.remove(OUTPUT_FILE)\n\n# ====================================================\n# 🚀 Обробка по частинах\n# ====================================================\nfor i, file_path in enumerate(test_files):\n    print(f\"🧩 Обробляємо файл {i+1}/{len(test_files)}: {os.path.basename(file_path)}\")\n\n    # Зчитуємо parquet\n    df = pq.read_table(file_path).to_pandas()\n    \n    # customer_ID_\n    customer_ids = df['customer_ID_'].values\n\n    # Тільки числові фічі\n    X_chunk = df.select_dtypes(include=[np.number])\n\n    # GPU-предикт\n    dchunk = xgb.DMatrix(X_chunk)\n    preds = model.predict(dchunk)\n\n    # Результат\n    sub_chunk = pd.DataFrame({\n        'customer_ID': customer_ids,\n        'prediction': preds\n    })\n\n    # Запис у CSV (append)\n    sub_chunk.to_csv(OUTPUT_FILE, mode='a', header=not os.path.exists(OUTPUT_FILE), index=False)\n\n    del df, X_chunk, dchunk, sub_chunk\n    gc.collect()\n\nprint(f\"\\n✅ Готово! Submission збережено у '{OUTPUT_FILE}'\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T21:46:15.573548Z","iopub.execute_input":"2025-10-06T21:46:15.574261Z","iopub.status.idle":"2025-10-06T21:48:07.896640Z","shell.execute_reply.started":"2025-10-06T21:46:15.574239Z","shell.execute_reply":"2025-10-06T21:48:07.895973Z"}},"outputs":[],"execution_count":null}]}