{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Loading unique customer data \nI use the fact that data is sorted by `customer_ID` to load data related to each customer in the training process. \n","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-05-27T21:47:00.384584Z","iopub.execute_input":"2022-05-27T21:47:00.385446Z","iopub.status.idle":"2022-05-27T21:47:00.4008Z","shell.execute_reply.started":"2022-05-27T21:47:00.385398Z","shell.execute_reply":"2022-05-27T21:47:00.399955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = \"/kaggle/input/amex-default-prediction/\"","metadata":{"execution":{"iopub.status.busy":"2022-05-27T21:47:00.826504Z","iopub.execute_input":"2022-05-27T21:47:00.827156Z","iopub.status.idle":"2022-05-27T21:47:00.832267Z","shell.execute_reply.started":"2022-05-27T21:47:00.827107Z","shell.execute_reply":"2022-05-27T21:47:00.831223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\ndata_temp = pd.read_csv(data_dir+\"train_data.csv\", nrows=5)\nnum_cols = [col for col in data_temp.columns.to_list() if col not in cat_cols]","metadata":{"execution":{"iopub.status.busy":"2022-05-27T21:47:01.449824Z","iopub.execute_input":"2022-05-27T21:47:01.45074Z","iopub.status.idle":"2022-05-27T21:47:01.491608Z","shell.execute_reply.started":"2022-05-27T21:47:01.450691Z","shell.execute_reply":"2022-05-27T21:47:01.490648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Read the customer IDs from train and test data","metadata":{}},{"cell_type":"code","source":"train_customers = pd.read_csv(data_dir+\"train_data.csv\", usecols=[\"customer_ID\"])\ntest_customers = pd.read_csv(data_dir+\"test_data.csv\", usecols=[\"customer_ID\"])","metadata":{"execution":{"iopub.status.busy":"2022-05-27T21:47:02.849452Z","iopub.execute_input":"2022-05-27T21:47:02.850063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get the indices related to each customer\nSince the data is sorted based on the `customer_ID`, one may use the `skiprows` and `nrows` arguments of the `pd.read_csv` to read the customer data without the need to read the whole dataset. ","metadata":{}},{"cell_type":"code","source":"train_customer_indices = train_customers.reset_index().set_index(\"customer_ID\").groupby('customer_ID').apply(lambda x : x.to_numpy().reshape(-1, )).to_dict()\ntest_customer_indices = test_customers.reset_index().set_index(\"customer_ID\").groupby('customer_ID').apply(lambda x : x.to_numpy().reshape(-1, )).to_dict()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I use the `Dataset` class from `pytorch` to form batches of customers by loading each customers data while training","metadata":{}},{"cell_type":"code","source":"import torch \nfrom torch.utils.data import Dataset ","metadata":{"execution":{"iopub.status.busy":"2022-05-27T21:46:53.173173Z","iopub.execute_input":"2022-05-27T21:46:53.173659Z","iopub.status.idle":"2022-05-27T21:46:55.009174Z","shell.execute_reply.started":"2022-05-27T21:46:53.173623Z","shell.execute_reply":"2022-05-27T21:46:55.008124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TrainCustomerData(Dataset):\n    def __init__(self, customer_indices, data_dir=None, cat_cols=['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']):\n        #self.customer_indices = customer_indices\n        self.customer_ids = tuple(customer_indices)\n        self.customer_indices = tuple(customer_indices.values())\n        self.train_data_dir = data_dir + \"train_data.csv\"\n        self.train_labels = pd.read_csv(data_dir+\"train_labels.csv\").set_index(\"customer_ID\")\n        self.data_columns = pd.read_csv(data_dir+\"train_data.csv\", nrows=5).columns.to_list()\n        self.cont_cols = [col for col in self.data_columns if col not in cat_cols + [\"customer_ID\", \"S_2\"]]\n        self.data_dir = data_dir\n\n    def __len__(self):\n        return len(self.customer_indices)\n\n    def __getitem__(self, index):\n        customer_data_indices = self.customer_indices[index]\n        skiprows = range(1, customer_data_indices[0]+1)\n        nrows = customer_data_indices[-1] - customer_data_indices[0] + 1\n        customer_data = pd.read_csv(self.train_data_dir, skiprows=skiprows, nrows=nrows, header=0)\n        customer_id = customer_data.customer_ID.iloc[0]\n        \n        customer_data.drop([\"customer_ID\", \"S_2\"], axis=1, inplace=True)\n        \n        customer_cont_data = customer_data[self.cont_cols]\n        customer_cont_tensor_data = torch.as_tensor(customer_cont_data.values, dtype=torch.float32)\n        \n        customer_cat_data = customer_data[cat_cols].values\n        \n        customer_label = torch.as_tensor(self.train_labels.loc[customer_id].values, dtype=torch.int32)\n        \n        return customer_cont_tensor_data, customer_cat_data, customer_label, customer_id\n\n","metadata":{"execution":{"iopub.status.busy":"2022-05-27T21:46:19.506729Z","iopub.execute_input":"2022-05-27T21:46:19.507316Z","iopub.status.idle":"2022-05-27T21:46:19.615105Z","shell.execute_reply.started":"2022-05-27T21:46:19.507197Z","shell.execute_reply":"2022-05-27T21:46:19.613852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label = pd.read_csv(\"/kaggle/input/amex-default-prediction/train_labels.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-05-26T15:52:02.464594Z","iopub.execute_input":"2022-05-26T15:52:02.466985Z","iopub.status.idle":"2022-05-26T15:52:03.723612Z","shell.execute_reply.started":"2022-05-26T15:52:02.46694Z","shell.execute_reply":"2022-05-26T15:52:03.722485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-05-26T15:52:03.725765Z","iopub.execute_input":"2022-05-26T15:52:03.726107Z","iopub.status.idle":"2022-05-26T15:52:03.737533Z","shell.execute_reply.started":"2022-05-26T15:52:03.726077Z","shell.execute_reply":"2022-05-26T15:52:03.736202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-05-26T15:52:03.739723Z","iopub.execute_input":"2022-05-26T15:52:03.740222Z","iopub.status.idle":"2022-05-26T15:52:03.825197Z","shell.execute_reply.started":"2022-05-26T15:52:03.740178Z","shell.execute_reply":"2022-05-26T15:52:03.824132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-05-26T15:53:05.855708Z","iopub.execute_input":"2022-05-26T15:53:05.856205Z","iopub.status.idle":"2022-05-26T15:53:05.891042Z","shell.execute_reply.started":"2022-05-26T15:53:05.856166Z","shell.execute_reply":"2022-05-26T15:53:05.889734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:35:36.98019Z","iopub.execute_input":"2022-05-26T14:35:36.980618Z","iopub.status.idle":"2022-05-26T14:35:36.988859Z","shell.execute_reply.started":"2022-05-26T14:35:36.980585Z","shell.execute_reply":"2022-05-26T14:35:36.987668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:35:43.947003Z","iopub.execute_input":"2022-05-26T14:35:43.948085Z","iopub.status.idle":"2022-05-26T14:35:43.95604Z","shell.execute_reply.started":"2022-05-26T14:35:43.948028Z","shell.execute_reply":"2022-05-26T14:35:43.954888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:35:48.946224Z","iopub.execute_input":"2022-05-26T14:35:48.946928Z","iopub.status.idle":"2022-05-26T14:35:49.006075Z","shell.execute_reply.started":"2022-05-26T14:35:48.946891Z","shell.execute_reply":"2022-05-26T14:35:49.004634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-05-26T13:28:51.348888Z","iopub.execute_input":"2022-05-26T13:28:51.349496Z","iopub.status.idle":"2022-05-26T13:28:51.356372Z","shell.execute_reply.started":"2022-05-26T13:28:51.349457Z","shell.execute_reply":"2022-05-26T13:28:51.355223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-05-26T14:40:34.897198Z","iopub.execute_input":"2022-05-26T14:40:34.897593Z","iopub.status.idle":"2022-05-26T14:40:35.124212Z","shell.execute_reply.started":"2022-05-26T14:40:34.897562Z","shell.execute_reply":"2022-05-26T14:40:35.1232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}