{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"There are a ton of notebooks for the competition, but I didn't find any that ran top-to-bottom in a default 16GB RAM Kaggle kernel. This one does.\n\nThis is a very simple quick-and-dirty solution to the problem. We simply load the data, optionally impute missing values with median or mode (don't need to for LGBM, though), then fit an LGBM model and make predictions.\n\nFirst, you need to add the feather dataset from here: https://www.kaggle.com/datasets/ruchi798/parquet-files-amexdefault-prediction\nThere are a bunch of datasets that come up when searching for 'amex feather' in the 'add data' tool, others might work too.\n\nSince the data is massive, we need to reduce it using known techniques (https://www.kaggle.com/competitions/amex-default-prediction/discussion/328054) and chunk through the test data when making predictions. But since the test set needs to have the actual IDs, we aren't going through the trouble of reducing the IDs to int64, then converting back to actual IDs (although we could).","metadata":{}},{"cell_type":"code","source":"import gc\nimport os\n\nimport numpy as np\nimport pandas as pd\nfrom tqdm.notebook import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nfrom lightgbm import LGBMClassifier, early_stopping\n\ntrain_feather_path = \"../input/parquet-files-amexdefault-prediction/train_data.ftr\"\ntest_feather_path = \"../input/parquet-files-amexdefault-prediction/test_data.ftr\"\ntrain_pq_path = \"../input/parquet-files-amexdefault-prediction/train_data.parquet\"\ntest_pq_path = \"../input/parquet-files-amexdefault-prediction/test_data.parquet\"\ntrain_labels_path = \"../input/amex-default-prediction/train_labels.csv\"\n\nSAMPLE = False\n\ntrain_sample_path = '../train_sample.ftr'\ntest_sample_path = '../train_sample.ftr'","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-19T19:29:45.827726Z","iopub.execute_input":"2022-08-19T19:29:45.828390Z","iopub.status.idle":"2022-08-19T19:29:48.247634Z","shell.execute_reply.started":"2022-08-19T19:29:45.828255Z","shell.execute_reply":"2022-08-19T19:29:48.246272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Load data and fill missing values. Use SAMPLE = True to only use samples of data for development work.","metadata":{}},{"cell_type":"code","source":"# quickly load one row so we can get the column names for categorical and numeric columns\ntrain_df_for_cols = pd.read_csv('../input/amex-default-prediction/train_data.csv', nrows=1)\ncat_cols = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\nnum_cols = set(train_df_for_cols.columns).difference(cat_cols + ['customer_ID', 'S_2', 'B_31'])\n\nclass DataPrepper:\n    def __init__(self, fillna=False):\n        self.fillna = fillna  # if True,fills categorical with mode and numeric with median from train data\n        self.cat_les = {}  # label encoders for categorical columns\n        self.num_medians = {}  # median values for filling NA from train data\n        \n    def label_encode_categorical_cols(self, df, train):\n        print(\"converting categorical columns\")\n        for col in cat_cols:\n            # careful to only get the NA fill values from the train set so we don't have data leakage\n            if train:\n                le = LabelEncoder()\n                le_fit = le.fit(df[col].unique())\n                _ = self.cat_les.setdefault(col, le_fit)\n            df[col] = self.cat_les[col].transform(df[col])\n            df[col] = df[col].astype('int8')\n            gc.collect()\n        \n        return df\n\n    \n    def convert_numeric_and_fill_median(self, df, train):\n        print(\"converting numeric columns\")\n        if train:\n            _ = self.num_medians.setdefault('B_31', df['B_31'].median())\n\n        if self.fillna:\n            df['B_31'].fillna(self.num_medians['B_31'], inplace=True)\n        \n        df['B_31'] = df['B_31'].astype('int8')\n\n        for col in num_cols:\n            if train:\n                _ = self.num_medians.setdefault(col, df[col].median())\n\n            if self.fillna:\n                df[col].fillna(self.num_medians[col], inplace=True)\n            \n            df[col] = df[col].astype('float16')\n            \n            gc.collect()\n        \n        return df\n    \n        \n    def reduce_memory(self, df, is_label_df=False, train=True):\n        \"\"\"\n        Reduces the memomry size of pandas df with techniques. \n        \"\"\"\n        if train:\n            df['customer_ID'] =\\\n                df['customer_ID'].apply(lambda x: int(x[-16:],16) ).astype('int64')\n        \n        df.set_index(['customer_ID'], inplace=True)\n\n        if not is_label_df:\n            df['S_2'] = pd.to_datetime(df['S_2'])\n            df = self.label_encode_categorical_cols(df, train)\n            df = self.convert_numeric_and_fill_median(df, train)\n\n        gc.collect()\n\n        return df\n\n    \n    def get_df_max_date_per_customer(self, df, label_df=None, train=True):\n        \"\"\"\n        Each customer has a series of dates with data. This simply gets the latest date per customer ID.\n\n        If train=True, label_df must not be None and should contain labels for each customer ID.\n        \"\"\"\n        max_date_idx = df['S_2'].groupby(df.index).transform(max) == df['S_2']\n        df = df.loc[max_date_idx]\n        df_max_date_ids = df.index\n        if train:\n            label_df = train_label_df.loc[df_max_date_ids]\n            return df, label_df\n        else:\n            return df","metadata":{"execution":{"iopub.status.busy":"2022-08-19T19:29:48.250256Z","iopub.execute_input":"2022-08-19T19:29:48.250720Z","iopub.status.idle":"2022-08-19T19:29:48.304746Z","shell.execute_reply.started":"2022-08-19T19:29:48.250673Z","shell.execute_reply":"2022-08-19T19:29:48.303321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dp = DataPrepper()","metadata":{"execution":{"iopub.status.busy":"2022-08-19T19:29:48.309018Z","iopub.execute_input":"2022-08-19T19:29:48.311912Z","iopub.status.idle":"2022-08-19T19:29:48.316928Z","shell.execute_reply.started":"2022-08-19T19:29:48.311867Z","shell.execute_reply":"2022-08-19T19:29:48.315456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if SAMPLE:\n    print(\"loading train data sample\")\n    if os.path.exists(train_sample_path):\n        train_df = pd.read_feather(train_sample_path)\n    else:\n        train_df = pd.read_feather(train_feather_path).sample(10000, random_state=42).reset_index(drop=True)\n        train_df.to_feather(train_sample_path)\n        \n    train_df = dp.reduce_memory(train_df, train=True)\nelse:\n    print(\"loading train data\")\n    train_df = dp.reduce_memory(pd.read_feather(train_feather_path), train=True)\n    \ntrain_label_df = dp.reduce_memory(pd.read_csv(train_labels_path), is_label_df=True, train=True)\n\ntrain_df, train_label_df = dp.get_df_max_date_per_customer(df=train_df, label_df=train_label_df)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T19:29:48.320283Z","iopub.execute_input":"2022-08-19T19:29:48.320732Z","iopub.status.idle":"2022-08-19T19:31:20.335410Z","shell.execute_reply.started":"2022-08-19T19:29:48.320687Z","shell.execute_reply":"2022-08-19T19:31:20.332837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_split_df_X, val_split_df_X, train_split_df_y, val_split_df_y = train_test_split(train_df, train_label_df, stratify=train_label_df['target'], random_state=42, test_size=0.1)\n\ndel train_df, train_label_df\n\nlgbm_model = LGBMClassifier(n_estimators=1000)\nlgbm_model.fit(X=train_split_df_X.drop(columns=['S_2']),\n                                y=train_split_df_y,\n                                categorical_feature=cat_cols,\n                                eval_set=(val_split_df_X.drop(columns=['S_2']), val_split_df_y),\n                                callbacks=[early_stopping(20)],\n                                verbose=True)\n\ndel train_split_df_X, val_split_df_X, train_split_df_y, val_split_df_y\ngc.collect()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-08-19T19:31:20.337226Z","iopub.execute_input":"2022-08-19T19:31:20.337585Z","iopub.status.idle":"2022-08-19T19:32:21.046951Z","shell.execute_reply.started":"2022-08-19T19:31:20.337553Z","shell.execute_reply":"2022-08-19T19:32:21.045774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if SAMPLE:\n    priabsnt('loading test data sample')\n    if os.path.exists(test_sample_path):\n        test_df = dp.reduce_memory(pd.read_feather(test_sample_path), train=False)\n    else:\n        test_df = dp.reduce_memory(pd.read_feather(test_feather_path).sample(10000, random_state=42).reset_index(drop=True), train=False)\n        test_df.to_feather(test_sample_path)\n    \n    test_df = dp.reduce_memory(test_df, train=False)\nelse:\n    print('loading test data')\n    test_df = dp.reduce_memory(pd.read_feather(test_feather_path), train=False)\n\nunique_test_users = len(test_df.index.unique())","metadata":{"execution":{"iopub.status.busy":"2022-08-19T19:32:21.048778Z","iopub.execute_input":"2022-08-19T19:32:21.049171Z","iopub.status.idle":"2022-08-19T19:33:43.552795Z","shell.execute_reply.started":"2022-08-19T19:32:21.049140Z","shell.execute_reply":"2022-08-19T19:33:43.551445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since the test data is too big to fit in memory and do a group by, we instead chunk the data.","metadata":{}},{"cell_type":"code","source":"test_len = len(test_df)\n\ntest_idx = 0\nstep = 2000000  # you can decrease the step if you have more memory-intensive processing for each chunk. total test length 11363762\nall_preds = []\nuser_ids = []\nlast_chunk = False\n\nwith tqdm(total=int(test_len/step) + 1) as pbar:\n    while len(test_df) > 0:\n        # get chunk of \"step\" size test data, or whatever's left\n        test_chunk = test_df.iloc[:step]\n        test_df = test_df.iloc[len(test_chunk):]\n        \n        if len(test_chunk) == step:\n            # drop last ID since we're not sure we have all data for that user,\n            # only if we aren't at the end of the test set\n            last_id = test_chunk.index[-1]\n            test_chunk = test_chunk.drop(last_id)\n            \n        # advance index by the amount of data we've processed so far\n        test_idx += len(test_chunk)\n\n        test_df_for_preds = dp.get_df_max_date_per_customer(df=test_chunk, train=False)\n        all_preds.extend(lgbm_model.predict_proba(test_df_for_preds.drop(columns='S_2'))[:, 1])\n        user_ids.extend(test_df_for_preds.index)\n        gc.collect()\n        pbar.update()","metadata":{"execution":{"iopub.status.busy":"2022-08-19T19:33:43.554698Z","iopub.execute_input":"2022-08-19T19:33:43.556465Z","iopub.status.idle":"2022-08-19T19:34:22.390319Z","shell.execute_reply.started":"2022-08-19T19:33:43.556413Z","shell.execute_reply":"2022-08-19T19:34:22.388997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# make sure we have all test users IDs in our predictions\nassert unique_test_users == len(user_ids)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T19:34:22.391973Z","iopub.execute_input":"2022-08-19T19:34:22.392452Z","iopub.status.idle":"2022-08-19T19:34:22.398747Z","shell.execute_reply.started":"2022-08-19T19:34:22.392408Z","shell.execute_reply":"2022-08-19T19:34:22.397514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({'prediction': all_preds, 'customer_ID': user_ids}).set_index('customer_ID')\n# uncomment to be able to submit this to the competition\n# submission.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-19T19:34:22.400258Z","iopub.execute_input":"2022-08-19T19:34:22.400640Z","iopub.status.idle":"2022-08-19T19:34:25.618786Z","shell.execute_reply.started":"2022-08-19T19:34:22.400607Z","shell.execute_reply":"2022-08-19T19:34:25.617889Z"},"trusted":true},"execution_count":null,"outputs":[]}]}