{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# for Google colab\nfor useing fast GPU, I used Google colab","metadata":{"id":"zT5dMiQK3qD3"}},{"cell_type":"code","source":"from google.colab import drive\ndrive.mount('/content/drive')","metadata":{"executionInfo":{"elapsed":15958,"status":"ok","timestamp":1658458012539,"user":{"displayName":"松田龍","userId":"10818678393291437164"},"user_tz":-540},"id":"_C1se0Es2vQA","outputId":"7adc02ab-7a5c-416a-d126-96a84da12a10"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd ./drive/MyDrive/Colab Notebooks/Kaggle/AMEX/notebook","metadata":{"executionInfo":{"elapsed":452,"status":"ok","timestamp":1658458012987,"user":{"displayName":"松田龍","userId":"10818678393291437164"},"user_tz":-540},"id":"AedahjB92C39","outputId":"d04f3c2c-4b79-4071-c59a-e6988a83ccfc"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## pip installs\n\n```\n# !pip install xxx\n```\n","metadata":{"id":"B73fWukU5DjM"}},{"cell_type":"code","source":"!pip install catboost","metadata":{"executionInfo":{"elapsed":10741,"status":"ok","timestamp":1658458023722,"user":{"displayName":"松田龍","userId":"10818678393291437164"},"user_tz":-540},"id":"pBy4U7Ea5HiK","outputId":"3c6bcaa5-43c2-4434-dcd4-ee0f96890904"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --upgrade wandb","metadata":{"executionInfo":{"elapsed":7855,"status":"ok","timestamp":1658458031572,"user":{"displayName":"松田龍","userId":"10818678393291437164"},"user_tz":-540},"id":"h3FZpmHv5r14","outputId":"05c65c5f-93eb-46c7-fda2-2df9e7d877c5"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# pre settings","metadata":{"id":"WjzaHeuHYOwx"}},{"cell_type":"markdown","source":"import","metadata":{"id":"x9zk5JVJVhuD"}},{"cell_type":"code","source":"import os\nimport gc\nimport random\nfrom tqdm.notebook import tqdm\nimport scipy as sp\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport itertools\nfrom typing import Tuple\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import StratifiedKFold\nfrom catboost import CatBoostClassifier\nimport wandb","metadata":{"id":"NTSA9KuY1rRD","executionInfo":{"status":"ok","timestamp":1658458032685,"user_tz":-540,"elapsed":1115,"user":{"displayName":"松田龍","userId":"10818678393291437164"}}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"methods","metadata":{"id":"e2Ybs9rpVni7"}},{"cell_type":"code","source":"def seed_everything(seed: int=42) -> None:\n    '''\n    update os seed\n    '''\n    random.seed(seed)\n    np.random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n\ndef pre_proccesed_data(path: str) -> pd.DataFrame:\n    \"\"\"\n    :path: str - path to data with format pickle\n        ../input/amex-agg-data-pickle/train_agg.pkl\n    \"\"\"\n    data = pd.read_pickle(path, compression=\"gzip\")\n    for col in data.columns:\n        if data[col].dtype=='float16':\n            data[col] = data[col].astype('float32').round(decimals=2).astype('float16')\n            \n    cat_features = [\n    \"B_30\",\n    \"B_38\",\n    \"D_114\",\n    \"D_116\",\n    \"D_117\",\n    \"D_120\",\n    \"D_126\",\n    \"D_63\",\n    \"D_64\",\n    \"D_66\",\n    \"D_68\"\n    ]\n    \n    cat_features = [f\"{cf}_last\" for cf in cat_features]\n    le_encoder = LabelEncoder()\n    \n    for categorical_feature in cat_features:\n        data[categorical_feature] = le_encoder.fit_transform(data[categorical_feature])\n        #data[categorical_feature] = le_encoder.transform(data[categorical_feature])\n        \n    return data\n\ndef amex_metric_mod(y_true, y_pred) -> Tuple[float, float, float]:\n\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n    \n    normalized_gini_coefficient = gini[1]/gini[0]\n    evaluation_metric = 0.5 * (normalized_gini_coefficient + top_four)\n\n    return normalized_gini_coefficient, top_four, evaluation_metric","metadata":{"id":"4leMAE1U1rRE","executionInfo":{"status":"ok","timestamp":1658458032686,"user_tz":-540,"elapsed":8,"user":{"displayName":"松田龍","userId":"10818678393291437164"}}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"setting","metadata":{"id":"K9mJOFPLVjc0"}},{"cell_type":"code","source":"# pd display option\npd.set_option('display.max_rows', 500)\npd.set_option('display.max_columns', 500)\npd.set_option('display.width', 1000)\n\n# set dataframe and feature columns\ntrain_df = pre_proccesed_data('../input/amex-agg-data-pickle/train_agg.pkl')\nFEATURES = [col for col in train_df.columns if col not in ['target']]","metadata":{"executionInfo":{"elapsed":19949,"status":"ok","timestamp":1658458052628,"user":{"displayName":"松田龍","userId":"10818678393291437164"},"user_tz":-540},"id":"ssqol1ksVwh3"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"based notebook: https://www.kaggle.com/code/kartushovdanil/baseline-amex-catboost-blending-wandb","metadata":{"id":"kSiag3SQLPzu"}},{"cell_type":"markdown","source":"# Oputuna\n\nref: https://github.com/optuna/optuna-examples/blob/main/catboost/catboost_simple.py","metadata":{"id":"M1WtHIQ-zp5p"}},{"cell_type":"markdown","source":"install oputuna","metadata":{"id":"pGogQxi4zvLy"}},{"cell_type":"code","source":"!pip install optuna","metadata":{"id":"eQC9RTzazub1","executionInfo":{"status":"ok","timestamp":1658458059667,"user_tz":-540,"elapsed":7057,"user":{"displayName":"松田龍","userId":"10818678393291437164"}},"outputId":"b2e6c0ac-726b-40ac-8bc9-480dc0cfb2aa"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import optuna\nimport logging\nimport sys","metadata":{"id":"YJSWmCjzz6SC","executionInfo":{"status":"ok","timestamp":1658458060539,"user_tz":-540,"elapsed":894,"user":{"displayName":"松田龍","userId":"10818678393291437164"}}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"define objective","metadata":{"id":"MuL3W5AW0AJi"}},{"cell_type":"code","source":"# parameter class\nclass CONFIG:\n    SEED = 42\n    FOLDS = 5\n    ITERATIONS = 400\n    START_FOLD = 1\n    FEATURETOP = 0\nseed_everything(CONFIG.SEED)","metadata":{"id":"bPsMJsrH1v5u","executionInfo":{"status":"ok","timestamp":1658458060540,"user_tz":-540,"elapsed":11,"user":{"displayName":"松田龍","userId":"10818678393291437164"}}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def objective(trial):\n\n  importance = []\n  y_test = pd.DataFrame()\n  gc.collect()\n  skf = StratifiedKFold(n_splits=CONFIG.FOLDS, shuffle=True, random_state=CONFIG.SEED)\n\n  FEATURES = train_df.drop(columns='target').columns\n  train_idx, valid_idx = next(skf.split(train_df[FEATURES], train_df.target))\n  train_x, train_y = train_df.iloc[train_idx].reset_index(drop=True)[FEATURES], train_df.iloc[train_idx].reset_index(drop=True)['target']\n  print(f'train X shape: {train_x.shape}, train Y shape: {train_y.shape}')\n  valid_x, valid_y = train_df.iloc[valid_idx].reset_index(drop=True)[FEATURES], train_df.iloc[valid_idx].reset_index(drop=True)['target']\n  print(f'valid X shape: {valid_x.shape}, valid Y shape: {valid_y.shape}')\n  print('#'*50)\n  print(' ')\n\n  param = {\n        \"objective\": trial.suggest_categorical(\"objective\", [\"Logloss\", \"CrossEntropy\"]),\n        # \"colsample_bylevel\": trial.suggest_float(\"colsample_bylevel\", 0.01, 0.1),\n        \"depth\": trial.suggest_int(\"depth\", 1, 12),\n        \"random_strength\": trial.suggest_float(\"random_strength\", 0, 1),\n        \"learning_rate\": trial.suggest_float(\"learning_rate\", 0.001, 0.01),\n        #\"boosting_type\": trial.suggest_categorical(\"boosting_type\", [\"Ordered\", \"Plain\"]),\n        \"bootstrap_type\": trial.suggest_categorical(\n            \"bootstrap_type\", [\"Bayesian\", \"Bernoulli\", \"MVS\"]\n        ),\n        #\"used_ram_limit\": \"3gb\",\n    }\n\n  if param[\"bootstrap_type\"] == \"Bayesian\":\n      param[\"bagging_temperature\"] = trial.suggest_float(\"bagging_temperature\", 0, 10)\n  elif param[\"bootstrap_type\"] == \"Bernoulli\":\n      param[\"subsample\"] = trial.suggest_float(\"subsample\", 0.1, 1)\n\n  gbm = CatBoostClassifier(**param, \n                           iterations=CONFIG.ITERATIONS, random_state=CONFIG.SEED, task_type='GPU',\n                           use_best_model=True,\n                           eval_metric='Logloss'\n                           )\n\n  gbm.fit(train_x, train_y, eval_set=[(valid_x, valid_y)], verbose=100, early_stopping_rounds=100)\n\n  preds = gbm.predict(valid_x)\n  pred_labels = np.rint(preds)\n  normalized_gini_coefficient, top_four, evaluation_metric = amex_metric_mod(valid_y, pred_labels)\n  return evaluation_metric","metadata":{"id":"dRKOftUyz-Gc","executionInfo":{"status":"ok","timestamp":1658458216924,"user_tz":-540,"elapsed":269,"user":{"displayName":"松田龍","userId":"10818678393291437164"}}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study = optuna.create_study(direction='maximize')\nstudy.optimize(objective, n_trials=50)\n\nprint(\"Number of finished trials: {}\".format(len(study.trials)))\n\nprint(\"Best trial:\")\ntrial = study.best_trial\n\nprint(\"  Value: {}\".format(trial.value))\n\nprint(\"  Params: \")\nfor key, value in trial.params.items():\n    print(\"    {}: {}\".format(key, value))","metadata":{"id":"PKQPt76T2rjH","executionInfo":{"status":"ok","timestamp":1658465798479,"user_tz":-540,"elapsed":6792673,"user":{"displayName":"松田龍","userId":"10818678393291437164"}},"outputId":"efe2295e-03a6-4834-91d9-88c884d2f267"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Result is below\n\nNumber of finished trials: 50\nBest trial:\n  Value: 0.5734734881504024\n  Params: \n    objective: Logloss\n    depth: 11\n    random_strength: 0.00803073910474994\n    learning_rate: 0.009992242536155601\n    bootstrap_type: Bayesian\n    bagging_temperature: 1.2390740088643806","metadata":{}},{"cell_type":"code","source":"","metadata":{"id":"AZsaTe5ukVkA","executionInfo":{"status":"aborted","timestamp":1658458126229,"user_tz":-540,"elapsed":7,"user":{"displayName":"松田龍","userId":"10818678393291437164"}}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"8R81XRQGqEOJ","executionInfo":{"status":"aborted","timestamp":1658458126230,"user_tz":-540,"elapsed":7,"user":{"displayName":"松田龍","userId":"10818678393291437164"}}},"execution_count":null,"outputs":[]}]}