{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%%time\n%reset -f\n! pip install pyfarmhash\nimport gc; gc.collect()\n\nimport cudf\nfrom sklearn.model_selection import StratifiedKFold\nimport numpy as np\nimport pandas as pd\n\nfrom sklearn.preprocessing import PowerTransformer\n# importing necessary libraries \nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport scipy.stats as stats\nimport statsmodels.api as sm\nimport farmhash\nfrom typing import Any, Callable\nimport pandas as pd\nfrom scipy.stats import skew\nsns.set_theme()\nsns.set_palette(palette = \"rainbow\")\n%matplotlib inline","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-15T20:06:57.348303Z","iopub.execute_input":"2022-08-15T20:06:57.348780Z","iopub.status.idle":"2022-08-15T20:07:12.865414Z","shell.execute_reply.started":"2022-08-15T20:06:57.348683Z","shell.execute_reply":"2022-08-15T20:07:12.864015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SEED=1024\nFOLDS = BUCKETS = 17\nMODEL_NAME = \"Xgboost\" #Xgboost,LightGBM\nTEST_RATIO = 1/FOLDS","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:07:12.867861Z","iopub.execute_input":"2022-08-15T20:07:12.868273Z","iopub.status.idle":"2022-08-15T20:07:12.875286Z","shell.execute_reply.started":"2022-08-15T20:07:12.868233Z","shell.execute_reply":"2022-08-15T20:07:12.874191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fixing_skewness(df, num_col):\n    \"\"\"\n    This function takes in a dataframe and return fixed skewed dataframe\n    \"\"\"\n    ## Import necessary modules \n    from scipy.stats import skew\n    from scipy.special import boxcox1p\n    from scipy.stats import boxcox_normmax\n    \n    ## Getting all the data that are not of \"object\" type. \n    numeric_feats = df.dtypes[num_col].index\n\n    # Check the skew of all numerical features\n    skewed_feats = df[numeric_feats].apply(lambda x: skew(x)).sort_values(ascending=False)\n    high_skew = skewed_feats[abs(skewed_feats) > 0.75]\n    skewed_features = high_skew.index\n    \n    print(\"skewed_features:\", skewed_features)\n\n    for feat in skewed_features:\n        df[feat] = boxcox1p(df[feat], boxcox_normmax(df[feat] + 1))\n        \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:07:12.877429Z","iopub.execute_input":"2022-08-15T20:07:12.878188Z","iopub.status.idle":"2022-08-15T20:07:12.887079Z","shell.execute_reply.started":"2022-08-15T20:07:12.878149Z","shell.execute_reply":"2022-08-15T20:07:12.885902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess(dataset):\n    dataset['customer_ID'] = dataset['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n    dataset['S_2'] = cudf.to_datetime(dataset['S_2'])\n    dataset['cid'],_ = dataset['customer_ID'].factorize()\n    dataset = dataset.sort_values('cid')\n    dataset.set_index(['customer_ID', 'S_2'], inplace=True)\n    dataset = dataset.to_pandas()\n    dataset = reduce_mem_usage(dataset)\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:07:12.891148Z","iopub.execute_input":"2022-08-15T20:07:12.891537Z","iopub.status.idle":"2022-08-15T20:07:12.900038Z","shell.execute_reply.started":"2022-08-15T20:07:12.891501Z","shell.execute_reply":"2022-08-15T20:07:12.899000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    gc.collect()\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:07:12.901405Z","iopub.execute_input":"2022-08-15T20:07:12.901955Z","iopub.status.idle":"2022-08-15T20:07:12.917479Z","shell.execute_reply.started":"2022-08-15T20:07:12.901918Z","shell.execute_reply":"2022-08-15T20:07:12.916313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def flatten_columns(df):\n    df.columns = [\"_\".join(column) for column in df.columns]\n    return df\n\ndef engineer(dataset, feature_set):\n    if feature_set == 0:\n        dataset = dataset.groupby(level='customer_ID').last()\n        return dataset\n    \n    if feature_set == 1:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['last']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std', 'sum']).pipe(flatten_columns)\n\n        dataset = cudf.concat([cat_feat, num_feat], axis=1)\n        return dataset\n    \n    if feature_set == 2:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['last']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n        \n        diff_cols_a = [f\"B_{i}\" for i in [11, 14, 17]] + [\"D_39\", \"D_131\"] + [f\"S_{i}\" for i in [16, 23]]\n        diff_cols_b = [\"P_2\", \"P_3\"]\n        diff_feat = dataset[diff_cols_a + diff_cols_b]\n        for a in diff_cols_a:\n            for b in diff_cols_b:\n                    diff_feat[f\"{a}-{b}\"] = diff_feat[a] - diff_feat[b]\n        diff_feat.drop(diff_cols_a + diff_cols_b, axis=1, inplace=True)\n        diff_feat = diff_feat.groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n\n        dataset = cudf.concat([cat_feat, num_feat, diff_feat], axis=1)\n        return dataset\n    \n    if feature_set == 3:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['count', 'last', 'nunique']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n        \n        diff_cols_a = [f\"B_{i}\" for i in [11, 14, 17]] + [\"D_39\", \"D_131\"] + [f\"S_{i}\" for i in [16, 23]]\n        diff_cols_b = [\"P_2\", \"P_3\"]\n        diff_feat = dataset[diff_cols_a + diff_cols_b]\n        for a in diff_cols_a:\n            for b in diff_cols_b:\n                    diff_feat[f\"{a}-{b}\"] = diff_feat[a] - diff_feat[b]\n        diff_feat.drop(diff_cols_a + diff_cols_b, axis=1, inplace=True)\n        diff_feat = diff_feat.groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n\n        dataset = cudf.concat([cat_feat, num_feat, diff_feat], axis=1)\n        return dataset\n    \n    if feature_set == 4:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['count', 'last', 'nunique']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std', 'min', 'max']).pipe(flatten_columns)\n        \n        diff_cols_a = [f\"B_{i}\" for i in [11, 14, 17]] + [\"D_39\", \"D_131\"] + [f\"S_{i}\" for i in [16, 23]]\n        diff_cols_b = [\"P_2\", \"P_3\"]\n        diff_feat = dataset[diff_cols_a + diff_cols_b]\n        for a in diff_cols_a:\n            for b in diff_cols_b:\n                    diff_feat[f\"{a}-{b}\"] = diff_feat[a] - diff_feat[b]\n        diff_feat.drop(diff_cols_a + diff_cols_b, axis=1, inplace=True)\n        diff_feat = diff_feat.groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std', 'min', 'max']).pipe(flatten_columns)\n\n        dataset = cudf.concat([cat_feat, num_feat, diff_feat], axis=1)\n        return dataset\n\n    if feature_set == 5:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['count', 'last', 'nunique']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n        \n        for col in num_feat.columns:\n            if 'last' in col and col.replace('last', 'first') in num_feat.columns:\n                num_feat[col + '_lag_sub'] = num_feat[col] - num_feat[col.replace('last', 'first')]\n                num_feat[col + '_lag_div'] = num_feat[col] / num_feat[col.replace('last', 'first')]\n        \n        diff_cols_a = [f\"B_{i}\" for i in [11, 14, 17]] + [\"D_39\", \"D_131\"] + [f\"S_{i}\" for i in [16, 23]]\n        diff_cols_b = [\"P_2\", \"P_3\"]\n        diff_feat = dataset[diff_cols_a + diff_cols_b]\n        for a in diff_cols_a:\n            for b in diff_cols_b:\n                    diff_feat[f\"{a}-{b}\"] = diff_feat[a] - diff_feat[b]\n        diff_feat.drop(diff_cols_a + diff_cols_b, axis=1, inplace=True)\n        diff_feat = diff_feat.groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n\n        dataset = cudf.concat([cat_feat, num_feat, diff_feat], axis=1)\n        return dataset\n\n    if feature_set == 6:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['count', 'last', 'nunique']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n        \n        diff_cols_a = [f\"B_{i}\" for i in [11, 14, 17]] + [\"D_39\", \"D_131\"] + [f\"S_{i}\" for i in [16, 23]]\n        diff_cols_b = [\"P_2\", \"P_3\"]\n        diff_feat = dataset[diff_cols_a + diff_cols_b]\n        for a in diff_cols_a:\n            for b in diff_cols_b:\n                    diff_feat[f\"{a}-{b}\"] = diff_feat[a] - diff_feat[b]\n        diff_feat.drop(diff_cols_a + diff_cols_b, axis=1, inplace=True)\n        diff_feat = diff_feat.groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n\n        for col in num_feat.columns:\n            if 'last' in col:\n                num_feat[col + '_round2'] = num_feat[col].round(2)\n        \n        dataset = cudf.concat([cat_feat, num_feat, diff_feat], axis=1)\n        return dataset\n\n    if feature_set == 7:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['count', 'last', 'nunique']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target','cid']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std','sum', 'max']).pipe(flatten_columns)\n        gc.collect()\n        for col in num_cols:\n            num_feat[col + \"_sub_mean\"] = num_feat[col + \"_last\"] - num_feat[col + \"_mean\"]\n        \n        diff_cols_a = [f\"B_{i}\" for i in [11, 14, 17]] + [\"D_39\", \"D_131\"] + [f\"S_{i}\" for i in [16, 23]]\n        diff_cols_b = [\"P_2\", \"P_3\"]\n        diff_feat = dataset[diff_cols_a + diff_cols_b]\n        for a in diff_cols_a:\n            for b in diff_cols_b:\n                    diff_feat[f\"{a}-{b}\"] = diff_feat[a] - diff_feat[b]\n        diff_feat.drop(diff_cols_a + diff_cols_b, axis=1, inplace=True)\n        gc.collect()\n        diff_feat = diff_feat.groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std','sum', 'max']).pipe(flatten_columns)\n\n        dataset = cudf.concat([cat_feat, num_feat, diff_feat], axis=1)\n        return dataset, cat_feat.columns\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:07:12.919339Z","iopub.execute_input":"2022-08-15T20:07:12.919771Z","iopub.status.idle":"2022-08-15T20:07:12.961241Z","shell.execute_reply.started":"2022-08-15T20:07:12.919729Z","shell.execute_reply":"2022-08-15T20:07:12.959831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from typing import Any, Callable\n\n# define some custom types (optional)\nHashFunc = Callable[[Any], int]\nTestSetCheckFunc = Callable[[int], bool]\n\n\ndef generate_farmhash_fingerprint(value: Any) -> int:\n    \"\"\"Convert a value into a hashed value using farmhash\"\"\"\n    return farmhash.fingerprint64(str(value))\n\n\ndef convert_hash_to_bucket(hashed_value: int, total_buckets: int) -> int:\n    \"\"\"Assign a bucket using modulo operator\"\"\"\n    return hashed_value % total_buckets\n\n\ndef test_set_check(bucket: int) -> bool:\n    \"\"\"Check if the bucket should be included in the test set\n\n    This is an arbirtary function, you could change this for your own\n    requirements\n\n    In this case, the datapoint is assigned to the test set if the bucket\n    number is less than the test ratio x total buckets.\n    \"\"\"\n    return bucket < TEST_RATIO * BUCKETS\n\n\ndef assign_hash_bucket(value: Any, hash_func: Callable) -> int:\n    \"\"\"Assign a bucket to an input value using hashing algorithm\"\"\"\n    hashed_value = hash_func(value)\n    bucket = convert_hash_to_bucket(hashed_value, total_buckets=BUCKETS)\n    return bucket\n\n\ndef hash_train_test_split(\n    df: pd.DataFrame,\n    split_col: str,\n    approx_test_ratio: float,\n    hash_func: HashFunc,\n    test_set_check_func: TestSetCheckFunc,\n    ):\n    \"\"\"Split the data into a training and test set based of a specific column\n\n    This function adds an additional column to the dataframe. This is for\n    demonstration purposes and is not required. The test set check could all\n    be completed in memory by adapting the test_set_check_func\n\n    Args:\n        df (pd.DataFrame): original dataset\n        split_col: name of the column to use for hashing which uniquely\n            identifies a datapoint\n        approx_test_ratio: float between 0-1. This is an approximate ratio as\n            the hashing algo will not necessarily provide a uniform bucket\n            distribution for small datasets\n        hash_func: hash function to use to encode the data\n        test_set_check_func: function used to check if the bucket should be\n            included in the test set\n    Returns:\n        tuple: Two dataframes, the first is the training set and the second\n            is the test set\n    \"\"\"\n\n    # assign bucket\n    df[\"bucket\"] = df[split_col].apply(assign_hash_bucket, hash_func=hash_func)\n\n    # generate 'mask' of boolean values which define the train/test split\n    in_test_set = df[\"bucket\"].apply(test_set_check_func)\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:07:12.962596Z","iopub.execute_input":"2022-08-15T20:07:12.963301Z","iopub.status.idle":"2022-08-15T20:07:12.977716Z","shell.execute_reply.started":"2022-08-15T20:07:12.963264Z","shell.execute_reply":"2022-08-15T20:07:12.976718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_test_split(train, method=\"hash\"):\n    if method == \"kfold\":\n        kfold = StratifiedKFold(n_splits=FOLDS, shuffle=True, random_state=SEED)\n        train['fold'] = -1\n        for fold_ix, (train_ixs, valid_ixs) in enumerate(kfold.split(train, train['target'].to_array())):\n            train['fold'].iloc[valid_ixs] = fold_ix\n            \n    elif method == \"hash\":\n        print(train.columns)\n        train = hash_train_test_split(\n                                    train,\n                                    split_col=\"cid\",\n                                    approx_test_ratio=TEST_RATIO,\n                                    hash_func=generate_farmhash_fingerprint,\n                                    test_set_check_func=test_set_check,\n                                )\n        \n        for i in range(BUCKETS):\n            train_df = train[train.bucket != i]\n            valid_df = train[train.bucket == i]\n\n            print(\"train shape:\", train_df.shape)\n            print(\"valid shape:\", valid_df.shape)\n        \n    return train","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:07:12.979259Z","iopub.execute_input":"2022-08-15T20:07:12.979900Z","iopub.status.idle":"2022-08-15T20:07:12.993280Z","shell.execute_reply.started":"2022-08-15T20:07:12.979861Z","shell.execute_reply":"2022-08-15T20:07:12.992378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_train_labels(train):\n    train_labels = cudf.read_csv('/kaggle/input/amex-default-prediction/train_labels.csv')\n    train_labels['customer_ID'] = train_labels['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n    train_labels['cid'],_ = train_labels['customer_ID'].factorize()\n    train_labels = train_labels.to_pandas()\n    train_labels = train_test_split(train_labels)\n    train_labels = cudf.DataFrame(train_labels)\n    train_labels.set_index('customer_ID', inplace=True)\n    \n    return cudf.merge(train, train_labels, how='inner', left_index=True, right_index=True).sort_index()","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:07:12.994601Z","iopub.execute_input":"2022-08-15T20:07:12.995943Z","iopub.status.idle":"2022-08-15T20:07:13.011251Z","shell.execute_reply.started":"2022-08-15T20:07:12.995905Z","shell.execute_reply":"2022-08-15T20:07:13.010228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def main(*, feature_set, num_rows):\n    train = cudf.read_parquet(\"../input/amex-data-integer-dtypes-parquet-format/train.parquet\", num_rows=num_rows)\n    train = preprocess(train)\n    train = cudf.DataFrame(train)\n    train, cat_feat = engineer(train, feature_set)\n    train = add_train_labels(train)\n    print(train.head())\n    print(feature_set, train.shape)\n    gc.collect()\n    train.to_feather('train.ftr')\n    \n    return cat_feat, train","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:07:13.014915Z","iopub.execute_input":"2022-08-15T20:07:13.015602Z","iopub.status.idle":"2022-08-15T20:07:13.022925Z","shell.execute_reply.started":"2022-08-15T20:07:13.015573Z","shell.execute_reply":"2022-08-15T20:07:13.021780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_labels = cudf.read_csv('/kaggle/input/amex-default-prediction/train_labels.csv')\n# train_labels['cid'],_ = train_labels['customer_ID'].factorize()\n# train_labels = train_labels.to_pandas()\n# train_labels = train_test_split(train_labels)\n# train_labels = cudf.DataFrame(train_labels)\n# train_labels['customer_ID'] = train_labels['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n# train_labels.set_index('customer_ID', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:07:13.026581Z","iopub.execute_input":"2022-08-15T20:07:13.027192Z","iopub.status.idle":"2022-08-15T20:07:13.037149Z","shell.execute_reply.started":"2022-08-15T20:07:13.027144Z","shell.execute_reply":"2022-08-15T20:07:13.035846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_labels","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:07:13.038737Z","iopub.execute_input":"2022-08-15T20:07:13.039510Z","iopub.status.idle":"2022-08-15T20:07:13.046899Z","shell.execute_reply.started":"2022-08-15T20:07:13.039404Z","shell.execute_reply":"2022-08-15T20:07:13.045932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_feat, train = main(\n    feature_set = 7,\n    num_rows = None\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:07:13.048162Z","iopub.execute_input":"2022-08-15T20:07:13.049006Z","iopub.status.idle":"2022-08-15T20:08:47.398739Z","shell.execute_reply.started":"2022-08-15T20:07:13.048969Z","shell.execute_reply":"2022-08-15T20:08:47.397183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:08:47.406177Z","iopub.execute_input":"2022-08-15T20:08:47.414935Z","iopub.status.idle":"2022-08-15T20:08:49.446018Z","shell.execute_reply.started":"2022-08-15T20:08:47.414884Z","shell.execute_reply":"2022-08-15T20:08:49.444979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:08:49.447771Z","iopub.execute_input":"2022-08-15T20:08:49.448187Z","iopub.status.idle":"2022-08-15T20:08:49.606176Z","shell.execute_reply.started":"2022-08-15T20:08:49.448147Z","shell.execute_reply":"2022-08-15T20:08:49.604963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Training","metadata":{}},{"cell_type":"code","source":"%%time\n%reset -f\nimport gc; gc.collect()\nimport cudf\nfrom colorama import Style, Fore\nimport pandas as pd\nimport xgboost as xgb\nimport cupy as cp\nfrom catboost import CatBoostClassifier\nimport numpy as np\nfrom catboost import CatBoostRegressor, Pool, CatBoostClassifier\n# import lightgbm as lgb\n\nSEED=1024\nFOLDS = BUCKETS = 17\nMODEL_NAME = \"LightGBM\" #Xgboost,LightGBM,Catboost\nTEST_RATIO = 1/FOLDS\n\ndef check_input(arr):\n    if type(arr) is pd.DataFrame:\n        arr = arr[arr.columns[0]]\n        \n    if type(arr) is pd.Series:\n        arr = arr.values\n        \n    if len(arr.shape) > 1:\n        arr = arr[:, 0]\n        \n    return arr\n\n\ndef gini(cs_0, cs_1, sum_0, sum_1):\n    auc_ = (cs_0 - sum_0 / 2) * sum_1\n    tot = cs_0[-1] * cs_1[-1]\n\n    return 2 * float(auc_.sum() / tot) - 1\n\n\ndef recall_at4(cs_0, cs_1, sum_1):\n    cs_tot = cs_0 + cs_1\n    th = cs_tot[-1] * 0.96\n    \n    return float(sum_1[cs_tot >= th].sum() / cs_1[-1])\n\n\ndef amex_metric_cupy(y_true, y_pred):\n    y_true = cp.asarray(check_input(y_true))\n    y_pred = cp.asarray(check_input(y_pred))\n    \n    unique = cp.unique(y_pred)\n    rank = cp.searchsorted(unique, y_pred)\n    \n    sum_1 = cp.zeros_like(unique, dtype=cp.float64)\n    sum_1.scatter_add(rank, y_true)\n    \n    sum_0 = cp.zeros_like(unique, dtype=cp.float64)\n    sum_0.scatter_add(rank, 1 - y_true)\n    sum_0 *= 20\n    \n    cs_0, cs_1 = sum_0.cumsum(), sum_1.cumsum()\n    \n    g = gini(cs_0, cs_1, sum_0, sum_1)\n    d = recall_at4(cs_0, cs_1, sum_1)\n    \n    return (g + d) / 2\n\n\ndef xgb_amex(y_pred, dmatrix):\n    return \"amex\", amex_metric_cupy(dmatrix.get_label(), y_pred)\n\n\ndef amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = pd.concat([y_true, y_pred], axis=\"columns\").sort_values(\n            \"prediction\", ascending=False\n        )\n        df[\"weight\"] = df[\"target\"].apply(lambda x: 20 if x == 0 else 1)\n        four_pct_cutoff = int(0.04 * df[\"weight\"].sum())\n        df[\"weight_cumsum\"] = df[\"weight\"].cumsum()\n        df_cutoff = df.loc[df[\"weight_cumsum\"] <= four_pct_cutoff]\n        return (df_cutoff[\"target\"] == 1).sum() / (df[\"target\"] == 1).sum()\n\n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = pd.concat([y_true, y_pred], axis=\"columns\").sort_values(\n            \"prediction\", ascending=False\n        )\n        df[\"weight\"] = df[\"target\"].apply(lambda x: 20 if x == 0 else 1)\n        df[\"random\"] = (df[\"weight\"] / df[\"weight\"].sum()).cumsum()\n        total_pos = (df[\"target\"] * df[\"weight\"]).sum()\n        df[\"cum_pos_found\"] = (df[\"target\"] * df[\"weight\"]).cumsum()\n        df[\"lorentz\"] = df[\"cum_pos_found\"] / total_pos\n        df[\"gini\"] = (df[\"lorentz\"] - df[\"random\"]) * df[\"weight\"]\n        return df[\"gini\"].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={\"target\": \"prediction\"})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)\n\n\ndef main_xgboost(*, xgb_parameters=None, num_rows=None):\n    folds = cudf.read_feather(\"train.ftr\")\n    \n    features = [col for col in folds.columns if col not in ['target', 'fold']]\n    print(len(features))\n    \n    predictions = []\n    \n    for fold_ix in range(FOLDS):\n        print(Fore.BLUE + \"#\" * 10, f\"Fold {fold_ix}\", \"#\" * 10 + Style.RESET_ALL)\n        \n        train = folds[folds.fold != fold_ix]\n        valid = folds[folds.fold == fold_ix]\n        \n        print(\"train shape:\", train.shape)\n        print(\"valid shape:\", train.valid)\n        \n        dtrain = xgb.DMatrix(data=train[features], label=train['target'])\n        dvalid = xgb.DMatrix(data=valid[features], label=valid['target'])\n\n        model = xgb.train(\n            xgb_parameters,\n            dtrain=dtrain,\n            num_boost_round=10000,\n\n            evals=[(dtrain, \"train\"), (dvalid, \"valid\")],\n            early_stopping_rounds=1500,\n            \n            custom_metric=xgb_amex,\n            maximize=True,\n\n            verbose_eval=500\n        )\n        \n        model.save_model(f\"xgb_fold{fold_ix}_seed_{xgb_parameters['random_state']}.xgb\")\n        \n        prediction = pd.DataFrame({\n            \"prediction\": model.predict(dvalid, iteration_range=(0, model.best_iteration + 1)),\n            \"target\": valid['target'].to_array()\n        })\n        \n        print(f\"Fold: {amex_metric(prediction[['target']], prediction[['prediction']]):.4f} CV\")\n        predictions.append(prediction)\n\n        del dtrain, dvalid, model, prediction, train, valid\n        gc.collect()\n    \n        print(Fore.BLUE + \"#\" * 28, \"\\n\" + Style.RESET_ALL)\n    \n    prediction = pd.concat(predictions)\n    print(Style.BRIGHT + f\"Results: {amex_metric(prediction[['target']], prediction[['prediction']]):.4f} CV\" + Style.RESET_ALL)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:08:49.610842Z","iopub.execute_input":"2022-08-15T20:08:49.611517Z","iopub.status.idle":"2022-08-15T20:08:50.632386Z","shell.execute_reply.started":"2022-08-15T20:08:49.611476Z","shell.execute_reply":"2022-08-15T20:08:50.631109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lgb_amex_metric(y_pred, y_true):\n    y_true = y_true.get_label()\n    return 'amex_metric', amex_metric(y_true, y_pred), True\n#     return 'amex_metric', amex_metric_cupy(y_true, y_pred), True","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:08:50.634051Z","iopub.execute_input":"2022-08-15T20:08:50.634494Z","iopub.status.idle":"2022-08-15T20:08:50.640046Z","shell.execute_reply.started":"2022-08-15T20:08:50.634445Z","shell.execute_reply":"2022-08-15T20:08:50.638958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip uninstall -y lightgbm\n!apt-get install -y libboost-all-dev\n!git clone --recursive https://github.com/Microsoft/LightGBM","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:08:50.641708Z","iopub.execute_input":"2022-08-15T20:08:50.642418Z","iopub.status.idle":"2022-08-15T20:09:30.298765Z","shell.execute_reply.started":"2022-08-15T20:08:50.642378Z","shell.execute_reply":"2022-08-15T20:09:30.297404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%bash\ncd LightGBM\nrm -r build\nmkdir build\ncd build\ncmake -DUSE_GPU=1 -DOpenCL_LIBRARY=/usr/local/cuda/lib64/libOpenCL.so -DOpenCL_INCLUDE_DIR=/usr/local/cuda/include/ ..\nmake -j$(nproc)","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:09:30.301451Z","iopub.execute_input":"2022-08-15T20:09:30.302293Z","iopub.status.idle":"2022-08-15T20:12:51.160510Z","shell.execute_reply.started":"2022-08-15T20:09:30.302240Z","shell.execute_reply":"2022-08-15T20:12:51.159183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cd LightGBM/python-package/;python setup.py install --precompile","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:12:51.162868Z","iopub.execute_input":"2022-08-15T20:12:51.163316Z","iopub.status.idle":"2022-08-15T20:12:53.002867Z","shell.execute_reply.started":"2022-08-15T20:12:51.163270Z","shell.execute_reply":"2022-08-15T20:12:53.001583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir -p /etc/OpenCL/vendors && echo \"libnvidia-opencl.so.1\" > /etc/OpenCL/vendors/nvidia.icd\n!rm -r LightGBM","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:12:53.005265Z","iopub.execute_input":"2022-08-15T20:12:53.005718Z","iopub.status.idle":"2022-08-15T20:12:55.295481Z","shell.execute_reply.started":"2022-08-15T20:12:53.005674Z","shell.execute_reply":"2022-08-15T20:12:55.294076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:12:55.297841Z","iopub.execute_input":"2022-08-15T20:12:55.298289Z","iopub.status.idle":"2022-08-15T20:12:55.901048Z","shell.execute_reply.started":"2022-08-15T20:12:55.298243Z","shell.execute_reply":"2022-08-15T20:12:55.900005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def xgb_train(train,valid,features):\n        \n    train = cudf.DataFrame(train)\n    gc.collect()\n    print(\"train xgboost\")\n    dtrain = xgb.DMatrix(data=train[features], label=train['target'])\n    dvalid = xgb.DMatrix(data=valid[features], label=valid['target'])\n\n    xgb_parameters={\n        'max_depth': 10,\n        'eta': 0.03,\n\n        'subsample': 0.9,\n        'colsample_bytree': 0.5,\n        \n        'objective': 'binary:logistic',\n        \n        'tree_method': 'gpu_hist',\n        'predictor': 'gpu_predictor',\n        \n        'random_state': 432031,\n        \n        'gamma': 1.8,\n        'min_child_weight': 64,\n        'lambda': 70,\n        'eval_metric':'logloss',\n    }\n    \n    model = xgb.train(\n        xgb_parameters,\n        dtrain=dtrain,\n        num_boost_round=20000,\n\n        evals=[(dtrain, \"train\"), (dvalid, \"valid\")],\n        early_stopping_rounds=3000,\n\n        custom_metric=xgb_amex,\n        maximize=True,\n\n        verbose_eval=500\n    )\n    \n    prediction = pd.DataFrame({\n            \"prediction\": model.predict(dvalid, iteration_range=(0, model.best_iteration + 1)),\n            \"target\": valid['target'].values\n        })\n    \n    del dtrain, dvalid\n    gc.collect()\n    \n    return model, prediction\n\n\ndef catboost_train(train,valid,features,cat_feat=None):\n    \n    dtrain = Pool(train[features], label=train['target'])\n    dvalid = Pool(valid[features], label=valid['target'])\n    \n    catboost_parameters={\n                       'iterations': 10000,\n                       'learning_rate': 0.05,\n                       'depth':6,\n                       'eval_metric': 'RMSE',\n                       'random_seed': SEED,\n                       'bagging_temperature': 0.2,\n                       'od_type': 'Iter',\n#                        'metric_period': 50,\n                        'early_stopping_rounds':1500,\n#                        'od_wait': 20,\n                       'use_best_model':True,\n                       'task_type':'GPU',\n                       'l2_leaf_reg':4,\n#                        'border_count':128,\n                        }\n        \n    model = CatBoostRegressor(**catboost_parameters)\n    model = CatBoostClassifier(iterations=5000, random_state=22)\n    \n    model.fit(dtrain, eval_set=[dtrain, dvalid],cat_features=cat_feat, verbose=500, early_stopping_rounds=1500)\n    \n    prediction = pd.DataFrame({\n#             \"prediction\": model.predict(dvalid),\n            \"prediction\": clf.predict_proba(dvalid)[:, 1],\n            \"target\": valid['target'].values\n        })\n    \n    del dtrain, dvalid\n    gc.collect()\n    \n    return model, prediction\n\n\ndef light_gbm_train(train, valid, features):\n    params = {\n        'objective': 'binary',\n        'metric': \"binary_logloss\",\n        'boosting': 'dart', #goss, gbdt\n        'seed': SEED,\n#         'max_depth':4,\n#         'num_leaves': 100,\n#         'learning_rate': 0.01,\n#         'feature_fraction': 0.20,\n#         'bagging_freq': 10,\n#         'bagging_fraction': 0.50,\n#         'n_jobs': -1,\n#         'lambda_l2': 2,\n#         'min_data_in_leaf': 40,\n        \"n_estimators\": 10000,\n#         \"early_stopping_round\": 300,\n        \"device\": \"gpu\",\n#         \"gpu_platform_id\": 0,\n#         \"gpu_device_id\": 0,\n        }\n    # Given its a regression case, I am using the RMSE as the metric.\n\n    dtrain = lgb.Dataset(train[features], label=train['target'])\n    dvalid = lgb.Dataset(valid[features], label=valid['target'])\n\n    model_light_gbm = lgb.train(params, dtrain, 15000,\n                               valid_sets=[dtrain, dvalid],\n                               feval = lgb_amex_metric\n                                )\n    \n    prediction = pd.DataFrame({\n            \"prediction\": model_light_gbm.predict(dvalid, num_iteration=model_light_gbm.best_iteration ),\n            \"target\": valid['target'].values\n        })\n    \n    del dtrain, dvalid\n    gc.collect()\n\n    return prediction","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:12:55.902730Z","iopub.execute_input":"2022-08-15T20:12:55.903117Z","iopub.status.idle":"2022-08-15T20:12:55.922009Z","shell.execute_reply.started":"2022-08-15T20:12:55.903074Z","shell.execute_reply":"2022-08-15T20:12:55.921084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex_model_run(*,df, model_name='Xgboost', num_rows=None):\n    gc.collect()\n    features = [col for col in df.columns if col not in ['target', 'fold', 'cid', 'bucket']]\n    cat_feat = ['B_30_count', 'B_30_last', 'B_30_nunique', 'B_38_count', 'B_38_last',\n       'B_38_nunique', 'D_114_count', 'D_114_last', 'D_114_nunique',\n       'D_116_count', 'D_116_last', 'D_116_nunique', 'D_117_count',\n       'D_117_last', 'D_117_nunique', 'D_120_count', 'D_120_last',\n       'D_120_nunique', 'D_126_count', 'D_126_last', 'D_126_nunique',\n       'D_63_count', 'D_63_last', 'D_63_nunique', 'D_64_count', 'D_64_last',\n       'D_64_nunique', 'D_66_count', 'D_66_last', 'D_66_nunique', 'D_68_count',\n       'D_68_last', 'D_68_nunique']\n\n\n    print(len(features))\n    \n    predictions = []\n    \n    for fold_ix in range(BUCKETS):\n        print(Fore.BLUE + \"#\" * 10, f\"Fold {fold_ix}\", \"#\" * 10 + Style.RESET_ALL)\n        \n        train = df[df.bucket != fold_ix].to_pandas()\n        valid = df[df.bucket == fold_ix].to_pandas()\n        \n        gc.collect()\n        \n        print(\"train shape:\", train.shape)\n        print(\"valid shape:\", valid.shape)\n        \n        if model_name == 'Xgboost':\n            model,prediction = xgb_train(train,valid,features)\n        if model_name == 'Catboost':\n            model, prediction = catboost_train(train,valid,features, cat_feat)\n        if model_name == \"LightGBM\":\n             model, prediction = light_gbm_train(train,valid,features)\n        \n        model.save_model(f\"{model_name}_fold{fold_ix}_seed_{SEED}.json\")\n        \n        print(f\"Fold: {amex_metric(prediction[['target']], prediction[['prediction']]):.4f} CV\")\n        predictions.append(prediction)\n\n        del prediction, train, valid, model\n        gc.collect()\n    \n        print(Fore.BLUE + \"#\" * 28, \"\\n\" + Style.RESET_ALL)\n    \n    prediction = pd.concat(predictions)\n    print(Style.BRIGHT + f\"Results: {amex_metric(prediction[['target']], prediction[['prediction']]):.4f} CV\" + Style.RESET_ALL)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:12:55.923625Z","iopub.execute_input":"2022-08-15T20:12:55.924172Z","iopub.status.idle":"2022-08-15T20:12:55.936298Z","shell.execute_reply.started":"2022-08-15T20:12:55.924125Z","shell.execute_reply":"2022-08-15T20:12:55.935198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = cudf.read_feather(\"train.ftr\")","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:12:55.938076Z","iopub.execute_input":"2022-08-15T20:12:55.938563Z","iopub.status.idle":"2022-08-15T20:13:00.701752Z","shell.execute_reply.started":"2022-08-15T20:12:55.938525Z","shell.execute_reply":"2022-08-15T20:13:00.700413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:13:00.703706Z","iopub.execute_input":"2022-08-15T20:13:00.704485Z","iopub.status.idle":"2022-08-15T20:13:02.624584Z","shell.execute_reply.started":"2022-08-15T20:13:00.704440Z","shell.execute_reply":"2022-08-15T20:13:02.623468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:13:02.626217Z","iopub.execute_input":"2022-08-15T20:13:02.627619Z","iopub.status.idle":"2022-08-15T20:13:02.839586Z","shell.execute_reply.started":"2022-08-15T20:13:02.627577Z","shell.execute_reply":"2022-08-15T20:13:02.837390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# amex_model_run(df=train_df, model_name='Catboost')","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:13:02.840943Z","iopub.execute_input":"2022-08-15T20:13:02.841522Z","iopub.status.idle":"2022-08-15T20:13:02.850249Z","shell.execute_reply.started":"2022-08-15T20:13:02.841480Z","shell.execute_reply":"2022-08-15T20:13:02.848991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# amex_model_run(df=train_df, model_name='Xgboost')","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:13:02.851660Z","iopub.execute_input":"2022-08-15T20:13:02.852562Z","iopub.status.idle":"2022-08-15T20:13:02.860120Z","shell.execute_reply.started":"2022-08-15T20:13:02.852524Z","shell.execute_reply":"2022-08-15T20:13:02.858975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:13:02.865171Z","iopub.execute_input":"2022-08-15T20:13:02.865767Z","iopub.status.idle":"2022-08-15T20:13:03.030531Z","shell.execute_reply.started":"2022-08-15T20:13:02.865734Z","shell.execute_reply":"2022-08-15T20:13:03.029351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amex_model_run(df=train_df, model_name='LightGBM')","metadata":{"execution":{"iopub.status.busy":"2022-08-15T20:13:03.032293Z","iopub.execute_input":"2022-08-15T20:13:03.032704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n%reset -f\nimport gc; gc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n%reset -f\nimport gc; gc.collect()\nimport cudf\nimport xgboost as xgb\nimport numpy as np\nfrom catboost import CatBoostClassifier\nimport numpy as np\nfrom catboost import CatBoostRegressor, Pool, CatBoostClassifier\n\nSEED=1024\nFOLDS = BUCKETS = 17\n\ndef reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df\n\ndef flatten_columns(df):\n    df.columns = [\"_\".join(column) for column in df.columns]\n    return df\n\n\n# def preprocess(dataset):\n#     dataset['customer_ID'] = dataset['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n#     dataset['S_2'] = cudf.to_datetime(dataset['S_2'])\n#     dataset['cid'],_ = dataset['customer_ID'].factorize()\n#     dataset = dataset.sort_values('cid')\n#     dataset.set_index(['customer_ID', 'S_2'], inplace=True)\n#     dataset = dataset.to_pandas()\n#     dataset = reduce_mem_usage(dataset)\n#     return dataset\n\ndef preprocess(dataset):\n    dataset['customer_ID'] = dataset['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n    dataset['S_2'] = cudf.to_datetime(dataset['S_2'])\n    dataset.set_index(['customer_ID', 'S_2'], inplace=True)\n    dataset = dataset.to_pandas()\n    dataset = reduce_mem_usage(dataset)\n    return dataset\n\n\ndef flatten_columns(df):\n    df.columns = [\"_\".join(column) for column in df.columns]\n    return df\n\ndef engineer(dataset, feature_set):\n    if feature_set == 0:\n        dataset = dataset.groupby(level='customer_ID').last()\n        return dataset\n    \n    if feature_set == 1:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['last']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std', 'sum']).pipe(flatten_columns)\n\n        dataset = cudf.concat([cat_feat, num_feat], axis=1)\n        return dataset\n    \n    if feature_set == 2:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['last']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n        \n        diff_cols_a = [f\"B_{i}\" for i in [11, 14, 17]] + [\"D_39\", \"D_131\"] + [f\"S_{i}\" for i in [16, 23]]\n        diff_cols_b = [\"P_2\", \"P_3\"]\n        diff_feat = dataset[diff_cols_a + diff_cols_b]\n        for a in diff_cols_a:\n            for b in diff_cols_b:\n                    diff_feat[f\"{a}-{b}\"] = diff_feat[a] - diff_feat[b]\n        diff_feat.drop(diff_cols_a + diff_cols_b, axis=1, inplace=True)\n        diff_feat = diff_feat.groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n\n        dataset = cudf.concat([cat_feat, num_feat, diff_feat], axis=1)\n        return dataset\n    \n    if feature_set == 3:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['count', 'last', 'nunique']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n        \n        diff_cols_a = [f\"B_{i}\" for i in [11, 14, 17]] + [\"D_39\", \"D_131\"] + [f\"S_{i}\" for i in [16, 23]]\n        diff_cols_b = [\"P_2\", \"P_3\"]\n        diff_feat = dataset[diff_cols_a + diff_cols_b]\n        for a in diff_cols_a:\n            for b in diff_cols_b:\n                    diff_feat[f\"{a}-{b}\"] = diff_feat[a] - diff_feat[b]\n        diff_feat.drop(diff_cols_a + diff_cols_b, axis=1, inplace=True)\n        diff_feat = diff_feat.groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n\n        dataset = cudf.concat([cat_feat, num_feat, diff_feat], axis=1)\n        return dataset\n    \n    if feature_set == 4:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['count', 'last', 'nunique']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std', 'min', 'max']).pipe(flatten_columns)\n        \n        diff_cols_a = [f\"B_{i}\" for i in [11, 14, 17]] + [\"D_39\", \"D_131\"] + [f\"S_{i}\" for i in [16, 23]]\n        diff_cols_b = [\"P_2\", \"P_3\"]\n        diff_feat = dataset[diff_cols_a + diff_cols_b]\n        for a in diff_cols_a:\n            for b in diff_cols_b:\n                    diff_feat[f\"{a}-{b}\"] = diff_feat[a] - diff_feat[b]\n        diff_feat.drop(diff_cols_a + diff_cols_b, axis=1, inplace=True)\n        diff_feat = diff_feat.groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std', 'min', 'max']).pipe(flatten_columns)\n\n        dataset = cudf.concat([cat_feat, num_feat, diff_feat], axis=1)\n        return dataset\n\n    if feature_set == 5:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['count', 'last', 'nunique']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n        \n        for col in num_feat.columns:\n            if 'last' in col and col.replace('last', 'first') in num_feat.columns:\n                num_feat[col + '_lag_sub'] = num_feat[col] - num_feat[col.replace('last', 'first')]\n                num_feat[col + '_lag_div'] = num_feat[col] / num_feat[col.replace('last', 'first')]\n        \n        diff_cols_a = [f\"B_{i}\" for i in [11, 14, 17]] + [\"D_39\", \"D_131\"] + [f\"S_{i}\" for i in [16, 23]]\n        diff_cols_b = [\"P_2\", \"P_3\"]\n        diff_feat = dataset[diff_cols_a + diff_cols_b]\n        for a in diff_cols_a:\n            for b in diff_cols_b:\n                    diff_feat[f\"{a}-{b}\"] = diff_feat[a] - diff_feat[b]\n        diff_feat.drop(diff_cols_a + diff_cols_b, axis=1, inplace=True)\n        diff_feat = diff_feat.groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n\n        dataset = cudf.concat([cat_feat, num_feat, diff_feat], axis=1)\n        return dataset\n\n    if feature_set == 6:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['count', 'last', 'nunique']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n        \n        diff_cols_a = [f\"B_{i}\" for i in [11, 14, 17]] + [\"D_39\", \"D_131\"] + [f\"S_{i}\" for i in [16, 23]]\n        diff_cols_b = [\"P_2\", \"P_3\"]\n        diff_feat = dataset[diff_cols_a + diff_cols_b]\n        for a in diff_cols_a:\n            for b in diff_cols_b:\n                    diff_feat[f\"{a}-{b}\"] = diff_feat[a] - diff_feat[b]\n        diff_feat.drop(diff_cols_a + diff_cols_b, axis=1, inplace=True)\n        diff_feat = diff_feat.groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n\n        for col in num_feat.columns:\n            if 'last' in col:\n                num_feat[col + '_round2'] = num_feat[col].round(2)\n        \n        dataset = cudf.concat([cat_feat, num_feat, diff_feat], axis=1)\n        return dataset\n\n    if feature_set == 7:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['count', 'last', 'nunique']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target','cid']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std','sum', 'max']).pipe(flatten_columns)\n        gc.collect()\n        for col in num_cols:\n            num_feat[col + \"_sub_mean\"] = num_feat[col + \"_last\"] - num_feat[col + \"_mean\"]\n        \n        diff_cols_a = [f\"B_{i}\" for i in [11, 14, 17]] + [\"D_39\", \"D_131\"] + [f\"S_{i}\" for i in [16, 23]]\n        diff_cols_b = [\"P_2\", \"P_3\"]\n        diff_feat = dataset[diff_cols_a + diff_cols_b]\n        for a in diff_cols_a:\n            for b in diff_cols_b:\n                    diff_feat[f\"{a}-{b}\"] = diff_feat[a] - diff_feat[b]\n        diff_feat.drop(diff_cols_a + diff_cols_b, axis=1, inplace=True)\n        gc.collect()\n        diff_feat = diff_feat.groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std','sum', 'max']).pipe(flatten_columns)\n\n        dataset = cudf.concat([cat_feat, num_feat, diff_feat], axis=1)\n        return dataset, cat_feat.columns\n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def catboost_prediction(test,features,cat_feat=None):\n    test_preds=[]\n    test=test.to_pandas()\n    dtest = Pool(test)\n    model = CatBoostRegressor()\n    model.load_model(f\"Catboost_fold0_seed_1024.json\")\n    for f in range(1, FOLDS):\n        model.load_model(f\"Catboost_fold{f}_seed_1024.xgb\")\n        preds += model.predict(dtest)\n    preds /= FOLDS\n    test_preds.append(preds)\n    \n    return model, prediction","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict(customers, rows, num_cust, model_name=\"catboost\"):\n    skip_rows = 0\n    skip_cust = 0\n    test_preds = []\n\n    for k in range(len(rows)):\n        print(\"#\" * 25)\n        print(f\"### {k}\")\n        print(\"#\" * 25)\n        \n        test = cudf.read_parquet(\n            \"../input/amex-data-integer-dtypes-parquet-format/test.parquet\",\n            skiprows=skip_rows, num_rows=rows[k]\n        )\n\n        test['customer_ID'] = test['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n        test['S_2'] = cudf.to_datetime(test['S_2'])\n        test.set_index(['customer_ID', 'S_2'], inplace=True)\n\n        gc.collect()\n        skip_rows += rows[k]\n\n        test = test.to_pandas()\n        test = reduce_mem_usage(test)\n        test = cudf.DataFrame(test)\n        test,_ = engineer(test, 7)\n#         print(\"test:\", test.head())\n        if k == len(rows) - 1:\n            test = test.loc[customers[skip_cust:]]\n        else:\n            test = test.loc[customers[skip_cust : skip_cust + num_cust]]\n        \n        skip_cust += num_cust\n        \n        if model_name==\"Xgboost\":\n            # Prepare data for inference\n            dtest = xgb.DMatrix(data=test)\n            gc.collect()\n            model = xgb.Booster()\n            model.load_model(f\"Xgboost_fold0_seed_1024.json\")\n            preds = model.predict(dtest)\n            for f in range(1, FOLDS):\n                model.load_model(f\"Xgboost_fold{f}_seed_1024.json\")\n                preds += model.predict(dtest, iteration_range=(0, model.best_iteration + 1))\n            preds /= FOLDS\n            test_preds.append(preds)\n        \n        elif model_name==\"catboost\":\n            test=test.to_pandas()\n            dtest = Pool(test)\n            gc.collect()\n            model = CatBoostRegressor()\n            model.load_model(f\"Catboost_fold0_seed_1024.json\")\n            preds = model.predict(dtest)\n            for f in range(1, FOLDS):\n                model.load_model(f\"Catboost_fold{f}_seed_1024.json\")\n                preds += model.predict(dtest)\n            preds /= FOLDS\n            test_preds.append(preds)\n            \n        elif model_name==\"lightgbm\":\n            test=test.to_pandas()\n            dtest = Pool(test)\n            gc.collect()\n            model = CatBoostRegressor()\n            model.load_model(f\"Catboost_fold0_seed_1024.json\")\n            preds = model.predict(dtest)\n            for f in range(1, FOLDS):\n                model.load_model(f\"Catboost_fold{f}_seed_1024.json\")\n                preds += model.predict(dtest)\n            preds /= FOLDS\n            test_preds.append(preds)\n            \n\n        # Cleanup\n        del dtest, model\n        _ = gc.collect()\n\n    return test_preds\n\n\ndef main():\n    test_customers = cudf.read_parquet(\n        \"../input/amex-data-integer-dtypes-parquet-format/test.parquet\",\n        columns=['customer_ID']\n    )\n    test_customers[\"customer_ID\"] = test_customers[\"customer_ID\"].str[-16:].str.hex_to_int().astype(\"int64\")\n    \n    def get_rows(customers, test, num_parts):\n        \"\"\"Divides the test dataset in `num_parts` parts.\n        Each part contains approximately `chunk` customers.\n        Returns the number of rows and then number of customers in\n        each part, except the last which has fewer.\n        \"\"\"\n        chunk = len(customers) // num_parts\n        rows = []\n\n        for k in range(num_parts):\n            if k == num_parts - 1:\n                cc = customers[k * chunk :]\n            else:\n                cc = customers[k * chunk : (k + 1) * chunk]\n\n            s = test.loc[test.customer_ID.isin(cc)].shape[0]\n            rows.append(s)\n\n        return rows, chunk\n    \n    customers = test_customers[[\"customer_ID\"]].drop_duplicates().sort_index().values.flatten()\n    rows, num_cust = get_rows(customers, test_customers[[\"customer_ID\"]], num_parts=10)\n    \n    test_preds = predict(customers, rows, num_cust, model_name=\"Xgboost\")\n    \n    test_preds = np.concatenate(test_preds)\n    test = cudf.DataFrame(index=customers, data={\"prediction\": test_preds})\n    sub = cudf.read_csv(\"../input/amex-default-prediction/sample_submission.csv\")[\n        [\"customer_ID\"]\n    ]\n    sub[\"customer_ID_hash\"] = sub[\"customer_ID\"].str[-16:].str.hex_to_int().astype(\"int64\")\n    sub = sub.set_index(\"customer_ID_hash\")\n    sub = sub.merge(test[[\"prediction\"]], left_index=True, right_index=True, how=\"left\")\n    sub = sub.reset_index(drop=True)\n\n    # Display predictions\n    sub.to_csv(f\"submission_xgb.csv\", index=False)\n    print(\"Submission file shape is\", sub.shape)\n\n\nmain()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Stacking models","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import ElasticNet, Lasso,  BayesianRidge, LassoLarsIC\nfrom sklearn.ensemble import RandomForestRegressor,  GradientBoostingRegressor\nfrom sklearn.kernel_ridge import KernelRidge\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import RobustScaler\nfrom sklearn.base import BaseEstimator, TransformerMixin, RegressorMixin, clone\nfrom sklearn.model_selection import KFold, cross_val_score, train_test_split\nfrom sklearn.metrics import mean_squared_error\nimport xgboost as xgb\nimport lightgbm as lgb","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lasso = make_pipeline(Lasso(alpha =0.0005, random_state=1))\nENet = make_pipeline(RobustScaler(), ElasticNet(alpha=0.0005, l1_ratio=.9, random_state=3))\nKRR = KernelRidge(alpha=0.6, kernel='polynomial', degree=2, coef0=2.5)\nGBoost = GradientBoostingRegressor(n_estimators=3000, learning_rate=0.05,\n                                   max_depth=4, max_features='sqrt',\n                                   min_samples_leaf=15, min_samples_split=10, \n                                   loss='huber', random_state =5)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class AveragingModels(BaseEstimator, RegressorMixin, TransformerMixin):\n    def __init__(self, models):\n        self.models = models\n        \n    # we define clones of the original models to fit the data in\n    def fit(self, X, y):\n        self.models_ = [clone(x) for x in self.models]\n        \n        # Train cloned base models\n        for model in self.models_:\n            model.fit(X, y)\n\n        return self\n    \n    #Now we do the predictions for cloned models and average them\n    def predict(self, X):\n        predictions = np.column_stack([\n            model.predict(X) for model in self.models_\n        ])\n        return np.mean(predictions, axis=1)   ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class StackingAveragedModels(BaseEstimator, RegressorMixin, TransformerMixin):\n    def __init__(self, base_models, meta_model, n_folds=5):\n        self.base_models = base_models\n        self.meta_model = meta_model\n        self.n_folds = n_folds\n   \n    # We again fit the data on clones of the original models\n    def fit(self, X, y):\n        self.base_models_ = [list() for x in self.base_models]\n        self.meta_model_ = clone(self.meta_model)\n        kfold = KFold(n_splits=self.n_folds, shuffle=True, random_state=156)\n        \n        # Train cloned base models then create out-of-fold predictions\n        # that are needed to train the cloned meta-model\n        out_of_fold_predictions = np.zeros((X.shape[0], len(self.base_models)))\n        for i, model in enumerate(self.base_models):\n            for train_index, holdout_index in kfold.split(X, y):\n                instance = clone(model)\n                self.base_models_[i].append(instance)\n                instance.fit(X[train_index], y[train_index])\n                y_pred = instance.predict(X[holdout_index])\n                out_of_fold_predictions[holdout_index, i] = y_pred\n                \n        # Now train the cloned  meta-model using the out-of-fold predictions as new feature\n        self.meta_model_.fit(out_of_fold_predictions, y)\n        return self\n   \n    #Do the predictions of all base models on the test data and use the averaged predictions as \n    #meta-features for the final prediction which is done by the meta-model\n    def predict(self, X):\n        meta_features = np.column_stack([\n            np.column_stack([model.predict(X) for model in base_models]).mean(axis=1)\n            for base_models in self.base_models_ ])\n        return self.meta_model_.predict(meta_features)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# stacked_averaged_models = StackingAveragedModels(base_models = (ENet, GBoost, KRR),\n#                                                  meta_model = lasso)\n\n# score = rmsle_cv(stacked_averaged_models)\n# print(\"Stacking Averaged models score: {:.4f} ({:.4f})\".format(score.mean(), score.std()))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"averaged_models = AveragingModels(models = (ENet, GBoost, KRR, lasso))\n\nscore = rmsle_cv(averaged_models)\nprint(\" Averaged base models score: {:.4f} ({:.4f})\\n\".format(score.mean(), score.std()))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stacked_averaged_models = StackingAveragedModels(base_models = (ENet, GBoost, KRR),\n                                                 meta_model = lasso)\n\nscore = rmsle_cv(stacked_averaged_models)\nprint(\"Stacking Averaged models score: {:.4f} ({:.4f})\".format(score.mean(), score.std()))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stacked_averaged_models.fit(train.values, y_train)\nstacked_train_pred = stacked_averaged_models.predict(train.values)\nstacked_pred = np.expm1(stacked_averaged_models.predict(test.values))\nprint(rmsle(y_train, stacked_train_pred))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Ensembling StackedRegressor, XGBoost and LightGBM","metadata":{}},{"cell_type":"code","source":"ensemble = xgb_pred*0.85  + lgb_pred*0.15","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !rm train.pq","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train catboost model","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n%reset -f\nimport gc; gc.collect()\nimport cudf\nfrom colorama import Style, Fore\nimport pandas as pd\nimport xgboost as xgb\nimport cupy as cp\nimport cudf","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from catboost import CatBoostClassifier\nimport numpy as np\nfrom catboost import CatBoostRegressor, Pool, CatBoostClassifier","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SEED=1024\nFOLDS = 11","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def check_input(arr):\n    if type(arr) is pd.DataFrame:\n        arr = arr[arr.columns[0]]\n        \n    if type(arr) is pd.Series:\n        arr = arr.values\n        \n    if len(arr.shape) > 1:\n        arr = arr[:, 0]\n        \n    return arr\n\n\ndef gini(cs_0, cs_1, sum_0, sum_1):\n    auc_ = (cs_0 - sum_0 / 2) * sum_1\n    tot = cs_0[-1] * cs_1[-1]\n\n    return 2 * float(auc_.sum() / tot) - 1\n\n\ndef recall_at4(cs_0, cs_1, sum_1):\n    cs_tot = cs_0 + cs_1\n    th = cs_tot[-1] * 0.96\n    \n    return float(sum_1[cs_tot >= th].sum() / cs_1[-1])\n\n\ndef amex_metric_cupy(y_true, y_pred):\n    y_true = cp.asarray(check_input(y_true))\n    y_pred = cp.asarray(check_input(y_pred))\n    \n    unique = cp.unique(y_pred)\n    rank = cp.searchsorted(unique, y_pred)\n    \n    sum_1 = cp.zeros_like(unique, dtype=cp.float64)\n    sum_1.scatter_add(rank, y_true)\n    \n    sum_0 = cp.zeros_like(unique, dtype=cp.float64)\n    sum_0.scatter_add(rank, 1 - y_true)\n    sum_0 *= 20\n    \n    cs_0, cs_1 = sum_0.cumsum(), sum_1.cumsum()\n    \n    g = gini(cs_0, cs_1, sum_0, sum_1)\n    d = recall_at4(cs_0, cs_1, sum_1)\n    \n    return (g + d) / 2\n\n\ndef xgb_amex(y_pred, dmatrix):\n    return \"amex\", amex_metric_cupy(dmatrix.get_label(), y_pred)\n\n\ndef amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = pd.concat([y_true, y_pred], axis=\"columns\").sort_values(\n            \"prediction\", ascending=False\n        )\n        df[\"weight\"] = df[\"target\"].apply(lambda x: 20 if x == 0 else 1)\n        four_pct_cutoff = int(0.04 * df[\"weight\"].sum())\n        df[\"weight_cumsum\"] = df[\"weight\"].cumsum()\n        df_cutoff = df.loc[df[\"weight_cumsum\"] <= four_pct_cutoff]\n        return (df_cutoff[\"target\"] == 1).sum() / (df[\"target\"] == 1).sum()\n\n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = pd.concat([y_true, y_pred], axis=\"columns\").sort_values(\n            \"prediction\", ascending=False\n        )\n        df[\"weight\"] = df[\"target\"].apply(lambda x: 20 if x == 0 else 1)\n        df[\"random\"] = (df[\"weight\"] / df[\"weight\"].sum()).cumsum()\n        total_pos = (df[\"target\"] * df[\"weight\"]).sum()\n        df[\"cum_pos_found\"] = (df[\"target\"] * df[\"weight\"]).cumsum()\n        df[\"lorentz\"] = df[\"cum_pos_found\"] / total_pos\n        df[\"gini\"] = (df[\"lorentz\"] - df[\"random\"]) * df[\"weight\"]\n        return df[\"gini\"].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={\"target\": \"prediction\"})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ProfitMetric:\n    @staticmethod\n    def amex_metric_cupy(y_true, y_pred):\n        y_true = cp.asarray(check_input(y_true))\n        y_pred = cp.asarray(check_input(y_pred))\n\n        unique = cp.unique(y_pred)\n        rank = cp.searchsorted(unique, y_pred)\n\n        sum_1 = cp.zeros_like(unique, dtype=cp.float64)\n        sum_1.scatter_add(rank, y_true)\n\n        sum_0 = cp.zeros_like(unique, dtype=cp.float64)\n        sum_0.scatter_add(rank, 1 - y_true)\n        sum_0 *= 20\n\n        cs_0, cs_1 = sum_0.cumsum(), sum_1.cumsum()\n\n        g = gini(cs_0, cs_1, sum_0, sum_1)\n        d = recall_at4(cs_0, cs_1, sum_1)\n\n        return (g + d) / 2\n    \n    def is_max_optimal(self):\n        return True # greater is better\n\n    def evaluate(self, approxes, target, weight):            \n        assert len(approxes) == 1\n        assert len(target) == len(approxes[0])\n        y_true = cp.array(target).astype(int)\n        approx = approxes[0]\n        score = self.amex_metric_cupy(y_true, approx)\n        return score, 1","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def main_catboost(*, catboost_parameters=None, num_rows=None):\n    folds = cudf.read_feather(\"train.ftr\")\n    \n    features = [col for col in folds.columns if col not in ['target', 'fold']]\n    print(len(features))\n    \n    predictions = []\n    \n    for fold_ix in range(FOLDS):\n        print(Fore.BLUE + \"#\" * 10, f\"Fold {fold_ix}\", \"#\" * 10 + Style.RESET_ALL)\n        \n        train = folds[folds.fold != fold_ix].to_pandas()\n        valid = folds[folds.fold == fold_ix].to_pandas()\n        \n        dtrain = Pool(train[features], label=train['target'])\n        dvalid = Pool(valid[features], label=valid['target'])\n\n        model = CatBoostRegressor(iterations=10000, \n                             objective=\"RMSE\",\n                             task_type='GPU',\n                             bagging_temperature = 0.2,\n                             od_type='Iter',\n                             metric_period = 50,\n#                              eval_metric=ProfitMetric(),\n                             od_wait=20) \n#         CatBoostRegressor(iterations=500,\n#                                    learning_rate=0.01,\n#                                    depth=10,\n#                                    eval_metric='RMSE',\n#                                    random_seed = 42,\n#                                    bagging_temperature=0.2,\n#                                    od_type='Iter',\n#                                    metric_period=50,\n#                                    od_wait=20\n#                                    )\n    \n        model.fit(dtrain, eval_set=dvalid, verbose=100, early_stopping_rounds=500)\n        \n        model.save_model(f\"cat_fold{fold_ix}_seed{catboost_parameters['random_state']}.json\")\n        \n        prediction = pd.DataFrame({\n            \"prediction\": model.predict(dvalid),\n            \"target\": valid['target'].values\n        })\n        \n        print(f\"Fold: {amex_metric(prediction[['target']], prediction[['prediction']]):.4f} CV\")\n        predictions.append(prediction)\n\n        del dtrain, dvalid, model, prediction, train, valid\n        gc.collect()\n    \n        print(Fore.BLUE + \"#\" * 28, \"\\n\" + Style.RESET_ALL)\n    \n    prediction = pd.concat(predictions)\n    print(Style.BRIGHT + f\"Results: {amex_metric(prediction[['target']], prediction[['prediction']]):.4f} CV\" + Style.RESET_ALL)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_feat = ['B_30_count', 'B_30_last', 'B_30_nunique', 'B_38_count', 'B_38_last',\n       'B_38_nunique', 'D_114_count', 'D_114_last', 'D_114_nunique',\n       'D_116_count', 'D_116_last', 'D_116_nunique', 'D_117_count',\n       'D_117_last', 'D_117_nunique', 'D_120_count', 'D_120_last',\n       'D_120_nunique', 'D_126_count', 'D_126_last', 'D_126_nunique',\n       'D_63_count', 'D_63_last', 'D_63_nunique', 'D_64_count', 'D_64_last',\n       'D_64_nunique', 'D_66_count', 'D_66_last', 'D_66_nunique', 'D_68_count',\n       'D_68_last', 'D_68_nunique']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"main_catboost(\n    catboost_parameters = {\n        \"objective\": \"Logloss\",\n        \"learning_rate\": 0.05,\n        \"n_estimators\": 15000,\n        \"max_depth\": 10,\n        \"od_type\": \"Iter\",\n        \"task_type\":\"GPU\",\n        \"use_best_model\": True,\n        'random_state': 1024,\n        'early_stopping_rounds': 1500,\n        'verbose': 500,\n#         'cat_features': cat_feat,\n        \"border_count\": 128,\n        \"eval_metric\": \"Logloss\",\n\n    },\n    num_rows = None\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n%reset -f\nimport gc; gc.collect()\nimport cudf\nimport xgboost as xgb\nimport numpy as np\nfrom catboost import CatBoostClassifier\nimport numpy as np\nfrom catboost import CatBoostRegressor, Pool, CatBoostClassifier\n\nSEED=1024\nFOLDS = 11\n\ndef reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df\n\ndef flatten_columns(df):\n    df.columns = [\"_\".join(column) for column in df.columns]\n    return df\n\n\n# def preprocess(dataset):\n#     dataset['customer_ID'] = dataset['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n#     dataset['S_2'] = cudf.to_datetime(dataset['S_2'])\n#     dataset.set_index(['customer_ID', 'S_2'], inplace=True)\n    \n#     return dataset\n\ndef preprocess(dataset):\n    dataset['customer_ID'] = dataset['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n    dataset['S_2'] = cudf.to_datetime(dataset['S_2'])\n    dataset['cid'],_ = dataset['customer_ID'].factorize()\n    dataset = dataset.sort_values('cid')\n    dataset.set_index(['customer_ID', 'S_2'], inplace=True)\n    dataset = dataset.to_pandas()\n    dataset = reduce_mem_usage(dataset)\n    return dataset\n\n\ndef engineer(dataset, feature_set):\n    if feature_set == 0:\n        dataset = dataset.groupby(level='customer_ID').last()\n        return dataset\n    \n    if feature_set == 1:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['last']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n\n        dataset = cudf.concat([cat_feat, num_feat], axis=1)\n        return dataset\n    \n    if feature_set == 2:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['last']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n        \n        diff_cols_a = [f\"B_{i}\" for i in [11, 14, 17]] + [\"D_39\", \"D_131\"] + [f\"S_{i}\" for i in [16, 23]]\n        diff_cols_b = [\"P_2\", \"P_3\"]\n        diff_feat = dataset[diff_cols_a + diff_cols_b]\n        for a in diff_cols_a:\n            for b in diff_cols_b:\n                    diff_feat[f\"{a}-{b}\"] = diff_feat[a] - diff_feat[b]\n        diff_feat.drop(diff_cols_a + diff_cols_b, axis=1, inplace=True)\n        diff_feat = diff_feat.groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n\n        dataset = cudf.concat([cat_feat, num_feat, diff_feat], axis=1)\n        return dataset\n    \n    if feature_set == 3:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['count', 'last', 'nunique']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n        \n        diff_cols_a = [f\"B_{i}\" for i in [11, 14, 17]] + [\"D_39\", \"D_131\"] + [f\"S_{i}\" for i in [16, 23]]\n        diff_cols_b = [\"P_2\", \"P_3\"]\n        diff_feat = dataset[diff_cols_a + diff_cols_b]\n        for a in diff_cols_a:\n            for b in diff_cols_b:\n                    diff_feat[f\"{a}-{b}\"] = diff_feat[a] - diff_feat[b]\n        diff_feat.drop(diff_cols_a + diff_cols_b, axis=1, inplace=True)\n        diff_feat = diff_feat.groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n\n        dataset = cudf.concat([cat_feat, num_feat, diff_feat], axis=1)\n        return dataset\n    \n    if feature_set == 4:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['count', 'last', 'nunique']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std', 'min', 'max']).pipe(flatten_columns)\n        \n        diff_cols_a = [f\"B_{i}\" for i in [11, 14, 17]] + [\"D_39\", \"D_131\"] + [f\"S_{i}\" for i in [16, 23]]\n        diff_cols_b = [\"P_2\", \"P_3\"]\n        diff_feat = dataset[diff_cols_a + diff_cols_b]\n        for a in diff_cols_a:\n            for b in diff_cols_b:\n                    diff_feat[f\"{a}-{b}\"] = diff_feat[a] - diff_feat[b]\n        diff_feat.drop(diff_cols_a + diff_cols_b, axis=1, inplace=True)\n        diff_feat = diff_feat.groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std', 'min', 'max']).pipe(flatten_columns)\n\n        dataset = cudf.concat([cat_feat, num_feat, diff_feat], axis=1)\n        return dataset\n\n    if feature_set == 5:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['count', 'last', 'nunique']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n        \n        for col in num_feat.columns:\n            if 'last' in col and col.replace('last', 'first') in num_feat.columns:\n                num_feat[col + '_lag_sub'] = num_feat[col] - num_feat[col.replace('last', 'first')]\n                num_feat[col + '_lag_div'] = num_feat[col] / num_feat[col.replace('last', 'first')]\n        \n        diff_cols_a = [f\"B_{i}\" for i in [11, 14, 17]] + [\"D_39\", \"D_131\"] + [f\"S_{i}\" for i in [16, 23]]\n        diff_cols_b = [\"P_2\", \"P_3\"]\n        diff_feat = dataset[diff_cols_a + diff_cols_b]\n        for a in diff_cols_a:\n            for b in diff_cols_b:\n                    diff_feat[f\"{a}-{b}\"] = diff_feat[a] - diff_feat[b]\n        diff_feat.drop(diff_cols_a + diff_cols_b, axis=1, inplace=True)\n        diff_feat = diff_feat.groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n\n        dataset = cudf.concat([cat_feat, num_feat, diff_feat], axis=1)\n        return dataset\n\n    if feature_set == 6:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['count', 'last', 'nunique']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n        \n        diff_cols_a = [f\"B_{i}\" for i in [11, 14, 17]] + [\"D_39\", \"D_131\"] + [f\"S_{i}\" for i in [16, 23]]\n        diff_cols_b = [\"P_2\", \"P_3\"]\n        diff_feat = dataset[diff_cols_a + diff_cols_b]\n        for a in diff_cols_a:\n            for b in diff_cols_b:\n                    diff_feat[f\"{a}-{b}\"] = diff_feat[a] - diff_feat[b]\n        diff_feat.drop(diff_cols_a + diff_cols_b, axis=1, inplace=True)\n        diff_feat = diff_feat.groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n\n        for col in num_feat.columns:\n            if 'last' in col:\n                num_feat[col + '_round2'] = num_feat[col].round(2)\n        \n        dataset = cudf.concat([cat_feat, num_feat, diff_feat], axis=1)\n        return dataset\n\n    if feature_set == 7:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['count', 'last', 'nunique']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std','sum', 'max']).pipe(flatten_columns)\n        \n        for col in num_cols:\n            num_feat[col + \"_sub_mean\"] = num_feat[col + \"_last\"] - num_feat[col + \"_mean\"]\n        \n        diff_cols_a = [f\"B_{i}\" for i in [11, 14, 17]] + [\"D_39\", \"D_131\"] + [f\"S_{i}\" for i in [16, 23]]\n        diff_cols_b = [\"P_2\", \"P_3\"]\n        diff_feat = dataset[diff_cols_a + diff_cols_b]\n        for a in diff_cols_a:\n            for b in diff_cols_b:\n                    diff_feat[f\"{a}-{b}\"] = diff_feat[a] - diff_feat[b]\n        diff_feat.drop(diff_cols_a + diff_cols_b, axis=1, inplace=True)\n        diff_feat = diff_feat.groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std','sum', 'max']).pipe(flatten_columns)\n\n        dataset = cudf.concat([cat_feat, num_feat, diff_feat], axis=1)\n        return dataset\n\n\ndef predict(customers, rows, num_cust):\n    skip_rows = 0\n    skip_cust = 0\n    test_preds = []\n\n    for k in range(len(rows)):\n        print(\"#\" * 25)\n        print(f\"### {k}\")\n        print(\"#\" * 25)\n        \n        test = cudf.read_parquet(\n            \"../input/amex-data-integer-dtypes-parquet-format/test.parquet\",\n            skiprows=skip_rows, num_rows=rows[k]\n        )\n\n        test['customer_ID'] = test['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n        test['S_2'] = cudf.to_datetime(test['S_2'])\n        test.set_index(['customer_ID', 'S_2'], inplace=True)\n\n        gc.collect()\n        skip_rows += rows[k]\n\n        test = test.to_pandas()\n        test = reduce_mem_usage(test)\n        test = cudf.DataFrame(test)\n        test = engineer(test, 7)\n        \n        if k == len(rows) - 1:\n            test = test.loc[customers[skip_cust:]]\n        else:\n            test = test.loc[customers[skip_cust : skip_cust + num_cust]]\n        \n        skip_cust += num_cust\n\n        # Prepare data for inference\n        dtest = Pool(test.to_pandas())\n        gc.collect()\n\n        # Compute predictions and average blend all fold models\n        model = CatBoostRegressor()\n        model.load_model(f\"cat_fold0_seed1024.json\")\n        preds = model.predict(dtest)\n        for f in range(1, FOLDS):\n            model.load_model(f\"cat_fold{f}_seed1024.json\")\n            preds += model.predict(dtest)\n        preds /= FOLDS\n        test_preds.append(preds)\n\n        # Cleanup\n        del dtest, model\n        _ = gc.collect()\n\n    return test_preds\n\n\ndef main():\n    test_customers = cudf.read_parquet(\n        \"../input/amex-data-integer-dtypes-parquet-format/test.parquet\",\n        columns=['customer_ID']\n    )\n    test_customers[\"customer_ID\"] = test_customers[\"customer_ID\"].str[-16:].str.hex_to_int().astype(\"int64\")\n    \n    def get_rows(customers, test, num_parts):\n        \"\"\"Divides the test dataset in `num_parts` parts.\n        Each part contains approximately `chunk` customers.\n        Returns the number of rows and then number of customers in\n        each part, except the last which has fewer.\n        \"\"\"\n        chunk = len(customers) // num_parts\n        rows = []\n\n        for k in range(num_parts):\n            if k == num_parts - 1:\n                cc = customers[k * chunk :]\n            else:\n                cc = customers[k * chunk : (k + 1) * chunk]\n\n            s = test.loc[test.customer_ID.isin(cc)].shape[0]\n            rows.append(s)\n\n        return rows, chunk\n    \n    customers = test_customers[[\"customer_ID\"]].drop_duplicates().sort_index().values.flatten()\n    rows, num_cust = get_rows(customers, test_customers[[\"customer_ID\"]], num_parts=10)\n    \n    test_preds = predict(customers, rows, num_cust)\n    \n    test_preds = np.concatenate(test_preds)\n    test = cudf.DataFrame(index=customers, data={\"prediction\": test_preds})\n    sub = cudf.read_csv(\"../input/amex-default-prediction/sample_submission.csv\")[\n        [\"customer_ID\"]\n    ]\n    sub[\"customer_ID_hash\"] = sub[\"customer_ID\"].str[-16:].str.hex_to_int().astype(\"int64\")\n    sub = sub.set_index(\"customer_ID_hash\")\n    sub = sub.merge(test[[\"prediction\"]], left_index=True, right_index=True, how=\"left\")\n    sub = sub.reset_index(drop=True)\n\n    # Display predictions\n    sub.to_csv(f\"catboost_submission_xgb.csv\", index=False)\n    print(\"Submission file shape is\", sub.shape)\n\n\nmain()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\nimport numpy as np","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from ambrosm notebook\ndef lgb_amex_metric(y_true, y_pred):\n    \"\"\"The competition metric with lightgbm's calling convention\"\"\"\n    return ('amex',\n            amex_metric(pd.DataFrame({'target': y_true}), pd.Series(y_pred, name='prediction')),\n            True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#see : https://www.kaggle.com/competitions/amex-default-prediction/discussion/327534\ndef amex_metric_mod_lgbm(y_pred: np.ndarray, data: lgb.Dataset):\n\n    y_true = data.get_label()\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n\n    return 'AMEX', 0.5 * (gini[1]/gini[0]+ top_four), True","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from xgboost import XGBClassifier\nfrom lightgbm import LGBMClassifier\nfrom catboost import CatBoostClassifier","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb=XGBClassifier(n_estimators = 100,max_features=None,\n            bootstrap=True,random_state=42,warm_start=True, \n                      oob_score=True,tree_method=\"gpu_hist\" ,verbosity = 0 )\nlgbm=LGBMClassifier(n_estimators = 100,max_features=None,silent=True,\n                    random_state=42, \n                              verbose =-100,device='gpu' )\ncatboost=CatBoostClassifier(n_estimators = 100,\n                    random_state=42, \n                              bootstrap_type=\"Bayesian\",task_type=\"GPU\",verbose=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def main_lgbm(*, lgbm_parameters=None, num_rows=None):\n    \n    gc.collect()\n    \n    folds = cudf.read_feather(\"train.ftr\")\n    \n    folds = folds.to_pandas()\n    \n    features = [col for col in folds.columns if col not in ['target', 'fold']]\n    print(len(features))\n    \n    predictions = []\n    \n    for fold_ix in range(5):\n        print(Fore.BLUE + \"#\" * 10, f\"Fold {fold_ix}\", \"#\" * 10 + Style.RESET_ALL)\n        \n        train = folds[folds.fold != fold_ix]\n        valid = folds[folds.fold == fold_ix]\n        \n        print(type(train))\n        \n        X_tr = train[features]\n        y_tr = train['target']\n        X_va = valid[features]\n        y_va = valid['target']\n        \n        lgb_train_data = lgb.Dataset(X_tr, label=y_tr)\n        lgb_val_data = lgb.Dataset(X_va, label=y_va)\n        \n        print(valid['target'].values)\n        \n#         model = lgb.train(lgbm_parameters,\n#                                lgb_train_data,\n#                                valid_sets = [lgb_train_data,lgb_val_data],\n# #                                feval=amex_metric_mod_lgbm,\n# #                                early_stopping_rounds=200\n#                              )\n        model = lgbm.fit(X_tr,y_tr,\n#                 eval_set=[X_va,y_va]\n                        )\n        \n        model.save_model(f\"catboost_fold{fold_ix}_seed{catboost_parameters['random_state']}.cat\")\n\n        prediction = pd.DataFrame({\n            \"prediction\": model.predict(X_va),\n            \"target\": valid['target'].values\n        })\n        \n        print(f\"Fold: {amex_metric(prediction[['target']], prediction[['prediction']]):.4f} CV\")\n        predictions.append(prediction)\n\n        del X_tr, y_tr, X_va, y_va, model, prediction, train, valid\n        gc.collect()\n    \n        print(Fore.BLUE + \"#\" * 28, \"\\n\" + Style.RESET_ALL)\n    \n    prediction = pd.concat(predictions)\n    print(Style.BRIGHT + f\"Results: {amex_metric(prediction[['target']], prediction[['prediction']]):.4f} CV\" + Style.RESET_ALL)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n%reset -f\nimport gc; gc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n%reset -f\nimport gc; gc.collect()\nimport cudf\nimport xgboost as xgb\nimport numpy as np\nfrom catboost import CatBoostClassifier\nimport numpy as np\nfrom catboost import CatBoostRegressor, Pool, CatBoostClassifier\n\nSEED=1024\nFOLDS = 11\n\ndef reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df\n\ndef flatten_columns(df):\n    df.columns = [\"_\".join(column) for column in df.columns]\n    return df\n\n\n# def preprocess(dataset):\n#     dataset['customer_ID'] = dataset['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n#     dataset['S_2'] = cudf.to_datetime(dataset['S_2'])\n#     dataset.set_index(['customer_ID', 'S_2'], inplace=True)\n    \n#     return dataset\n\ndef preprocess(dataset):\n    dataset['customer_ID'] = dataset['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n    dataset['S_2'] = cudf.to_datetime(dataset['S_2'])\n    dataset['cid'],_ = dataset['customer_ID'].factorize()\n    dataset = dataset.sort_values('cid')\n    dataset.set_index(['customer_ID', 'S_2'], inplace=True)\n    dataset = dataset.to_pandas()\n    dataset = reduce_mem_usage(dataset)\n    return dataset\n\n\ndef flatten_columns(df):\n    df.columns = [\"_\".join(column) for column in df.columns]\n    return df\n\ndef engineer(dataset, feature_set):\n    if feature_set == 0:\n        dataset = dataset.groupby(level='customer_ID').last()\n        return dataset\n    \n    if feature_set == 1:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['last']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std', 'sum']).pipe(flatten_columns)\n\n        dataset = cudf.concat([cat_feat, num_feat], axis=1)\n        return dataset\n    \n    if feature_set == 2:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['last']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n        \n        diff_cols_a = [f\"B_{i}\" for i in [11, 14, 17]] + [\"D_39\", \"D_131\"] + [f\"S_{i}\" for i in [16, 23]]\n        diff_cols_b = [\"P_2\", \"P_3\"]\n        diff_feat = dataset[diff_cols_a + diff_cols_b]\n        for a in diff_cols_a:\n            for b in diff_cols_b:\n                    diff_feat[f\"{a}-{b}\"] = diff_feat[a] - diff_feat[b]\n        diff_feat.drop(diff_cols_a + diff_cols_b, axis=1, inplace=True)\n        diff_feat = diff_feat.groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n\n        dataset = cudf.concat([cat_feat, num_feat, diff_feat], axis=1)\n        return dataset\n    \n    if feature_set == 3:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['count', 'last', 'nunique']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n        \n        diff_cols_a = [f\"B_{i}\" for i in [11, 14, 17]] + [\"D_39\", \"D_131\"] + [f\"S_{i}\" for i in [16, 23]]\n        diff_cols_b = [\"P_2\", \"P_3\"]\n        diff_feat = dataset[diff_cols_a + diff_cols_b]\n        for a in diff_cols_a:\n            for b in diff_cols_b:\n                    diff_feat[f\"{a}-{b}\"] = diff_feat[a] - diff_feat[b]\n        diff_feat.drop(diff_cols_a + diff_cols_b, axis=1, inplace=True)\n        diff_feat = diff_feat.groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n\n        dataset = cudf.concat([cat_feat, num_feat, diff_feat], axis=1)\n        return dataset\n    \n    if feature_set == 4:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['count', 'last', 'nunique']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std', 'min', 'max']).pipe(flatten_columns)\n        \n        diff_cols_a = [f\"B_{i}\" for i in [11, 14, 17]] + [\"D_39\", \"D_131\"] + [f\"S_{i}\" for i in [16, 23]]\n        diff_cols_b = [\"P_2\", \"P_3\"]\n        diff_feat = dataset[diff_cols_a + diff_cols_b]\n        for a in diff_cols_a:\n            for b in diff_cols_b:\n                    diff_feat[f\"{a}-{b}\"] = diff_feat[a] - diff_feat[b]\n        diff_feat.drop(diff_cols_a + diff_cols_b, axis=1, inplace=True)\n        diff_feat = diff_feat.groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std', 'min', 'max']).pipe(flatten_columns)\n\n        dataset = cudf.concat([cat_feat, num_feat, diff_feat], axis=1)\n        return dataset\n\n    if feature_set == 5:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['count', 'last', 'nunique']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n        \n        for col in num_feat.columns:\n            if 'last' in col and col.replace('last', 'first') in num_feat.columns:\n                num_feat[col + '_lag_sub'] = num_feat[col] - num_feat[col.replace('last', 'first')]\n                num_feat[col + '_lag_div'] = num_feat[col] / num_feat[col.replace('last', 'first')]\n        \n        diff_cols_a = [f\"B_{i}\" for i in [11, 14, 17]] + [\"D_39\", \"D_131\"] + [f\"S_{i}\" for i in [16, 23]]\n        diff_cols_b = [\"P_2\", \"P_3\"]\n        diff_feat = dataset[diff_cols_a + diff_cols_b]\n        for a in diff_cols_a:\n            for b in diff_cols_b:\n                    diff_feat[f\"{a}-{b}\"] = diff_feat[a] - diff_feat[b]\n        diff_feat.drop(diff_cols_a + diff_cols_b, axis=1, inplace=True)\n        diff_feat = diff_feat.groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n\n        dataset = cudf.concat([cat_feat, num_feat, diff_feat], axis=1)\n        return dataset\n\n    if feature_set == 6:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['count', 'last', 'nunique']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n        \n        diff_cols_a = [f\"B_{i}\" for i in [11, 14, 17]] + [\"D_39\", \"D_131\"] + [f\"S_{i}\" for i in [16, 23]]\n        diff_cols_b = [\"P_2\", \"P_3\"]\n        diff_feat = dataset[diff_cols_a + diff_cols_b]\n        for a in diff_cols_a:\n            for b in diff_cols_b:\n                    diff_feat[f\"{a}-{b}\"] = diff_feat[a] - diff_feat[b]\n        diff_feat.drop(diff_cols_a + diff_cols_b, axis=1, inplace=True)\n        diff_feat = diff_feat.groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std']).pipe(flatten_columns)\n\n        for col in num_feat.columns:\n            if 'last' in col:\n                num_feat[col + '_round2'] = num_feat[col].round(2)\n        \n        dataset = cudf.concat([cat_feat, num_feat, diff_feat], axis=1)\n        return dataset\n\n    if feature_set == 7:\n        cat_cols = [\n            \"B_30\",\n            \"B_38\",\n            \"D_114\",\n            \"D_116\",\n            \"D_117\",\n            \"D_120\",\n            \"D_126\",\n            \"D_63\",\n            \"D_64\",\n            \"D_66\",\n            \"D_68\",\n        ]\n        cat_feat = dataset[cat_cols].groupby(level='customer_ID').agg(['count', 'last', 'nunique']).pipe(flatten_columns)\n\n        num_cols = [col for col in dataset.columns if col not in cat_cols + ['target','cid']]\n        num_feat = dataset[num_cols].groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std','sum', 'max']).pipe(flatten_columns)\n        gc.collect()\n        for col in num_cols:\n            num_feat[col + \"_sub_mean\"] = num_feat[col + \"_last\"] - num_feat[col + \"_mean\"]\n        \n        diff_cols_a = [f\"B_{i}\" for i in [11, 14, 17]] + [\"D_39\", \"D_131\"] + [f\"S_{i}\" for i in [16, 23]]\n        diff_cols_b = [\"P_2\", \"P_3\"]\n        diff_feat = dataset[diff_cols_a + diff_cols_b]\n        for a in diff_cols_a:\n            for b in diff_cols_b:\n                    diff_feat[f\"{a}-{b}\"] = diff_feat[a] - diff_feat[b]\n        diff_feat.drop(diff_cols_a + diff_cols_b, axis=1, inplace=True)\n        gc.collect()\n        diff_feat = diff_feat.groupby(level='customer_ID').agg(['first', 'last', 'mean', 'std','sum', 'max']).pipe(flatten_columns)\n\n        dataset = cudf.concat([cat_feat, num_feat, diff_feat], axis=1)\n        return dataset, cat_feat.columns\n    \n\n\ndef predict(customers, rows, num_cust):\n    skip_rows = 0\n    skip_cust = 0\n    test_preds = []\n\n    for k in range(len(rows)):\n        print(\"#\" * 25)\n        print(f\"### {k}\")\n        print(\"#\" * 25)\n        \n        test = cudf.read_parquet(\n            \"../input/amex-data-integer-dtypes-parquet-format/test.parquet\",\n            skiprows=skip_rows, num_rows=rows[k]\n        )\n\n        test['customer_ID'] = test['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n        test['S_2'] = cudf.to_datetime(test['S_2'])\n        test.set_index(['customer_ID', 'S_2'], inplace=True)\n\n        gc.collect()\n        skip_rows += rows[k]\n\n        test = test.to_pandas()\n        test = reduce_mem_usage(test)\n        test = cudf.DataFrame(test)\n        test = engineer(test, 7)\n        \n        if k == len(rows) - 1:\n            test = test.loc[customers[skip_cust:]]\n        else:\n            test = test.loc[customers[skip_cust : skip_cust + num_cust]]\n        \n        skip_cust += num_cust\n\n        # Prepare data for inference\n        dtest = xgb.DMatrix(data=test)\n        gc.collect()\n\n        # Compute predictions and average blend all fold models\n        model = xgb.Booster()\n        model.load_model(f\"xgb_fold0_seed_1024.xgb\")\n        preds = model.predict(dtest)\n        for f in range(1, FOLDS):\n            model.load_model(f\"xgb_fold{f}_seed_1024.xgb\")\n            preds += model.predict(dtest, iteration_range=(0, model.best_iteration + 1))\n        preds /= FOLDS\n        test_preds.append(preds)\n\n        # Cleanup\n        del dtest, model\n        _ = gc.collect()\n\n    return test_preds\n\n\ndef main():\n    test_customers = cudf.read_parquet(\n        \"../input/amex-data-integer-dtypes-parquet-format/test.parquet\",\n        columns=['customer_ID']\n    )\n    test_customers[\"customer_ID\"] = test_customers[\"customer_ID\"].str[-16:].str.hex_to_int().astype(\"int64\")\n    \n    def get_rows(customers, test, num_parts):\n        \"\"\"Divides the test dataset in `num_parts` parts.\n        Each part contains approximately `chunk` customers.\n        Returns the number of rows and then number of customers in\n        each part, except the last which has fewer.\n        \"\"\"\n        chunk = len(customers) // num_parts\n        rows = []\n\n        for k in range(num_parts):\n            if k == num_parts - 1:\n                cc = customers[k * chunk :]\n            else:\n                cc = customers[k * chunk : (k + 1) * chunk]\n\n            s = test.loc[test.customer_ID.isin(cc)].shape[0]\n            rows.append(s)\n\n        return rows, chunk\n    \n    customers = test_customers[[\"customer_ID\"]].drop_duplicates().sort_index().values.flatten()\n    rows, num_cust = get_rows(customers, test_customers[[\"customer_ID\"]], num_parts=10)\n    \n    test_preds = predict(customers, rows, num_cust)\n    \n    test_preds = np.concatenate(test_preds)\n    test = cudf.DataFrame(index=customers, data={\"prediction\": test_preds})\n    sub = cudf.read_csv(\"../input/amex-default-prediction/sample_submission.csv\")[\n        [\"customer_ID\"]\n    ]\n    sub[\"customer_ID_hash\"] = sub[\"customer_ID\"].str[-16:].str.hex_to_int().astype(\"int64\")\n    sub = sub.set_index(\"customer_ID_hash\")\n    sub = sub.merge(test[[\"prediction\"]], left_index=True, right_index=True, how=\"left\")\n    sub = sub.reset_index(drop=True)\n\n    # Display predictions\n    sub.to_csv(f\"submission_xgb.csv\", index=False)\n    print(\"Submission file shape is\", sub.shape)\n\n\nmain()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls -al","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PseudoLabeler.py","metadata":{}},{"cell_type":"code","source":"from sklearn.utils import shuffle\nfrom sklearn.base import BaseEstimator, RegressorMixin\nimport pandas as pd\nimport numpy as np\nimport random\n\nclass PseudoLabeler(BaseEstimator, RegressorMixin):\n    '''\n    Sci-kit learn wrapper for creating pseudo-lebeled estimators.\n    '''\n    \n    def __init__(self, model, unlabled_data, sample_rate=0.2, seed=42):\n        '''\n        @sample_rate - percent of samples used as pseudo-labelled data\n                       from the unlabled dataset\n        '''\n        assert sample_rate <= 1.0, 'Sample_rate should be between 0.0 and 1.0.'\n        \n        self.sample_rate = sample_rate\n        self.seed = seed\n        self.model = model\n        self.model.seed = seed\n        \n        self.unlabled_data = unlabled_data\n        # self.features = features\n        # self.target = target\n        \n    def get_params(self, deep=True):\n        return {\n            \"sample_rate\": self.sample_rate,\n            \"seed\": self.seed,\n            \"model\": self.model,\n            \"unlabled_data\": self.unlabled_data,\n            # \"features\": self.features,\n            # \"target\": self.target\n        }\n\n    def set_params(self, **parameters):\n        for parameter, value in parameters.items():\n            setattr(self, parameter, value)\n        return self\n\n        \n    def fit(self, X, y):\n        '''\n        Fit the data using pseudo labeling.\n        '''\n\n        augemented_train = self.__create_augmented_train(X, y)\n        self.model.fit(\n            augemented_train[:,:-1],\n            augemented_train[:,-1]\n        )\n        \n        return self\n\n\n    def __create_augmented_train(self, X, y):\n        '''\n        Create and return the augmented_train set that consists\n        of pseudo-labeled and labeled data.\n        '''        \n        num_of_samples = int(len(self.unlabled_data) * self.sample_rate)\n        \n        # Train the model and creat the pseudo-labels\n        self.model.fit(X, y)\n        pseudo_labels = self.model.predict(self.unlabled_data)\n        \n        # Add the pseudo-labels to the test set\n        length = len(self.unlabled_data)\n        indexs = list(range(0,length))\n        random.seed(a=19)\n        sampleindex = random.sample(indexs,num_of_samples)\n\n        \n        # Take a subset of the test set with pseudo-labels and append in onto\n        # the training set\n        pseudo_labels = np.array([[p] for p in pseudo_labels])\n        sampled_pseudo_data = np.hstack((self.unlabled_data[sampleindex], pseudo_labels[sampleindex]))\n        y1 = np.array([[t] for t in y])\n        temp_train = np.hstack((X, y1))\n       \n        augemented_train = np.vstack((sampled_pseudo_data, temp_train))\n        augemented_train = np.array(augemented_train)\n        return shuffle(augemented_train)\n        \n    def predict(self, X):\n        '''\n        Returns the predicted values.\n        '''\n        return self.model.predict(X)\n    \n    def get_model_name(self):\n        return self.model.__class__.__name__","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fixing_skewness(df, num_col):\n    \"\"\"\n    This function takes in a dataframe and return fixed skewed dataframe\n    \"\"\"\n    ## Import necessary modules \n    from scipy.stats import skew\n    from scipy.special import boxcox1p\n    from scipy.stats import boxcox_normmax\n    \n    ## Getting all the data that are not of \"object\" type. \n    numeric_feats = df.dtypes[num_col].index\n\n    # Check the skew of all numerical features\n    skewed_feats = df[numeric_feats].apply(lambda x: skew(x)).sort_values(ascending=False)\n    high_skew = skewed_feats[abs(skewed_feats) > 0.5]\n    skewed_features = high_skew.index\n\n    for feat in skewed_features:\n        df[feat] = boxcox1p(df[feat], boxcox_normmax(df[feat] + 1))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fixing_skewness(alldf,num_columns)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}