{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Problem\nFor each customer in dataset there are typically 13 values for each feature. In many solutions authors use only last value for each feature. But I did not find why last values is enough. So this script is for providing some kind of proof that last values contains enough information. But it is not a proof in mathematical sense, and also some authors use mean and std of 13 values (and may be some other combinations) as another features.\n\nThis notebook is for providing some insights that last values are the most important ones. ","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nfrom scipy.special import expit, logit\n\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.model_selection import train_test_split, StratifiedKFold\nfrom sklearn.metrics import roc_auc_score, average_precision_score, log_loss\n\nimport torch\nfrom torch import nn\n\nfrom matplotlib import rcParams\nrcParams['figure.figsize'] = (16, 12)\nimport seaborn as sns\nsns.set_theme()\n\nfrom tqdm.notebook import tqdm\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-22T15:02:32.151165Z","iopub.execute_input":"2022-08-22T15:02:32.151759Z","iopub.status.idle":"2022-08-22T15:02:32.165608Z","shell.execute_reply.started":"2022-08-22T15:02:32.151695Z","shell.execute_reply":"2022-08-22T15:02:32.164302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Calculating competition specific metrics","metadata":{}},{"cell_type":"code","source":"ZERO_WEIGHT = 20.0\n\n\ndef top_n_recall_metric(y, prob, percent=4.0):\n    indexes = np.argsort(prob)[::-1]\n    y = y[indexes]\n    weights = np.ones_like(y)\n    weights[y == 0] = ZERO_WEIGHT\n    top_percent_weight = (percent / 100.0 * weights.sum())\n    top_n_position = np.where(weights.cumsum() > top_percent_weight)[0][0]\n    y_top = y[:top_n_position]\n    weights = weights[:top_n_position]\n    return y_top.sum() / y.sum()\n\n\ndef gini_metric(y, prob):\n    ranks = pd.DataFrame(dict(prob=prob)).rank(method='first', ascending=False).values.ravel()\n    y_sorted = y[np.argsort(ranks)]\n    weights = np.ones_like(y_sorted)\n    weights[y_sorted == 0] = ZERO_WEIGHT\n    gini_curve = y_sorted.cumsum() / y.sum()\n    return ((gini_curve * weights).sum() / weights.sum()) - 0.5\n\n\ndef normalized_gini_metric(y, prob):\n    return gini_metric(y, prob) / gini_metric(y, y)\n\n\ndef amex_metric(y, prob):\n    D = top_n_recall_metric(y, prob)\n    G = normalized_gini_metric(y, prob)\n    return 0.5 * (G + D)","metadata":{"execution":{"iopub.status.busy":"2022-08-22T14:40:12.275392Z","iopub.execute_input":"2022-08-22T14:40:12.276527Z","iopub.status.idle":"2022-08-22T14:40:12.295012Z","shell.execute_reply.started":"2022-08-22T14:40:12.276465Z","shell.execute_reply":"2022-08-22T14:40:12.293591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading data\nLoading only 2000_000 rows because of kaggle-notebook limitations.","metadata":{}},{"cell_type":"code","source":"%%time\ndf = pd.read_csv('/kaggle/input/amex-default-prediction/train_data.csv', nrows=2000_000)\ndf_labels =  pd.read_csv('/kaggle/input/amex-default-prediction/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-22T14:32:11.444588Z","iopub.execute_input":"2022-08-22T14:32:11.445338Z","iopub.status.idle":"2022-08-22T14:34:23.450581Z","shell.execute_reply.started":"2022-08-22T14:32:11.445286Z","shell.execute_reply":"2022-08-22T14:34:23.449180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let's leave only numerical features to simplify input for neural net","metadata":{}},{"cell_type":"code","source":"features = sorted(df.columns[2:], key=lambda item: (item.split('_')[0], int(item.split('_')[1])))\ncategorical_features = [\n    'B_30', 'B_38',\n    'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\nnumerical_features = list(filter(lambda f: f not in categorical_features, features))\nprint(numerical_features)","metadata":{"execution":{"iopub.status.busy":"2022-08-22T14:34:31.537723Z","iopub.execute_input":"2022-08-22T14:34:31.539203Z","iopub.status.idle":"2022-08-22T14:34:31.551064Z","shell.execute_reply.started":"2022-08-22T14:34:31.539138Z","shell.execute_reply":"2022-08-22T14:34:31.548969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Calculate Mean values for all numerical features\nWe need to replace Nans with numbers to send data to neural net, so let's calculate mean values for each feature first and then use these values to fill Nans","metadata":{}},{"cell_type":"code","source":"def get_mean_features(df, numerical_features):\n    X_values_mean = np.zeros(shape=(13, len(numerical_features)))\n    X_values_sum = np.zeros_like(X_values_mean)\n    X_n_values = np.zeros_like(X_values_sum)\n    for customer_id, g in tqdm(df.groupby('customer_ID')):\n        x = g[numerical_features].values\n        if x.shape[0] < 13:\n            # simply ignore if customer has less than 13 rows\n            continue\n        mask = ~np.isnan(x)\n        X_n_values[mask] += 1.0\n        X_values_sum[mask] += x[mask]\n    mask = (X_n_values > 0)\n    X_values_mean[mask] = X_values_sum[mask] / X_n_values[mask]\n    return X_values_mean\n\nX_values_mean = get_mean_features(df, numerical_features)\nprint(X_values_mean.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-22T14:34:37.010394Z","iopub.execute_input":"2022-08-22T14:34:37.011967Z","iopub.status.idle":"2022-08-22T14:37:28.677392Z","shell.execute_reply.started":"2022-08-22T14:34:37.011900Z","shell.execute_reply":"2022-08-22T14:37:28.675670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Transform numerical columns","metadata":{}},{"cell_type":"code","source":"def get_X(df, X_values_mean, numerical_features):\n    X = []\n    customer_ids = []\n    for customer_id, g in tqdm(df.groupby('customer_ID')):\n        customer_ids.append(customer_id)\n        x = g[numerical_features].values\n        assert(x.shape[0] <= 13)\n        if x.shape[0] < 13:\n            nan_values = np.full(shape=(13 - x.shape[0], x.shape[1]), fill_value=np.nan)\n            x = np.vstack((nan_values, x))\n        if np.isnan(x).sum() > 0:\n            mask = np.isnan(x)\n            x[mask] = X_values_mean[mask]\n        X.append(x.T)\n    X = np.array(X)\n    df_customers = pd.DataFrame({'customer_ID': customer_ids})\n    return X, df_customers","metadata":{"execution":{"iopub.status.busy":"2022-08-22T14:37:28.679717Z","iopub.execute_input":"2022-08-22T14:37:28.680538Z","iopub.status.idle":"2022-08-22T14:37:28.691737Z","shell.execute_reply.started":"2022-08-22T14:37:28.680487Z","shell.execute_reply":"2022-08-22T14:37:28.689837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X, df_customers = get_X(df, X_values_mean, numerical_features)\ny = pd.merge(df_customers, df_labels, how='left', on='customer_ID')['target'].values\nassert(np.isnan(X).sum() == 0)\nprint(X.shape, y.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-22T14:37:28.693579Z","iopub.execute_input":"2022-08-22T14:37:28.694620Z","iopub.status.idle":"2022-08-22T14:40:12.256818Z","shell.execute_reply.started":"2022-08-22T14:37:28.694553Z","shell.execute_reply":"2022-08-22T14:40:12.255277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Define neural network","metadata":{}},{"cell_type":"code","source":"class Net(nn.Module):\n    def __init__(self, n_features, sequence_len=13, sequence_output_dim=2, hidden_dim=64):\n        super().__init__()\n        # The most important layer is linear_0. It combines values from 13 rows for each customer\n        # with the same weights. After training nn we would analyze these weights.\n        self.linear_0 = nn.Linear(sequence_len, sequence_output_dim, bias=False)\n        self.relu_0 = nn.ReLU()\n        n_linear_0_output_features = n_features * sequence_output_dim\n        # -------------------------------------------------------------------------\n        # The rest is two dense layers with RELU nonlinearity\n        # You could add batchnorm and/or dropout layers here.\n        # But it does not change results significantly so we dont use them here for simplicity.\n        self.linear_1 = nn.Linear(n_linear_0_output_features, hidden_dim)\n        self.relu_1 = nn.ReLU()\n        self.linear_2 = nn.Linear(hidden_dim, 2)\n\n    def forward(self, x):\n        x = self.linear_0(x)\n        x = x.view(x.shape[0], -1)\n        x = self.relu_0(x)\n        x = self.linear_1(x)\n        x = self.relu_1(x)\n        x = self.linear_2(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2022-08-22T14:48:21.790623Z","iopub.execute_input":"2022-08-22T14:48:21.791233Z","iopub.status.idle":"2022-08-22T14:48:21.804349Z","shell.execute_reply.started":"2022-08-22T14:48:21.791183Z","shell.execute_reply":"2022-08-22T14:48:21.802609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_batch(X, y, batch_size=1024):\n    kf = StratifiedKFold(n_splits=(X.shape[0] // batch_size), shuffle=True)\n    for train_index, test_index in kf.split(X, y[:, 1]):\n        X_batch = torch.from_numpy(X[test_index, :].astype(np.float32))\n        y_batch = torch.from_numpy(y[test_index, :].astype(np.float32))\n        yield test_index, (X_batch, y_batch)","metadata":{"execution":{"iopub.status.busy":"2022-08-22T14:48:22.451063Z","iopub.execute_input":"2022-08-22T14:48:22.452003Z","iopub.status.idle":"2022-08-22T14:48:22.460774Z","shell.execute_reply.started":"2022-08-22T14:48:22.451938Z","shell.execute_reply":"2022-08-22T14:48:22.459523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_ohe = np.asarray(OneHotEncoder().fit_transform(y.reshape((-1, 1))).todense())\nX_tr, X_v, y_tr, y_v = train_test_split(X, y_ohe, test_size=0.4, random_state=20220822)","metadata":{"execution":{"iopub.status.busy":"2022-08-22T14:48:24.187922Z","iopub.execute_input":"2022-08-22T14:48:24.188647Z","iopub.status.idle":"2022-08-22T14:48:25.891199Z","shell.execute_reply.started":"2022-08-22T14:48:24.188607Z","shell.execute_reply":"2022-08-22T14:48:25.889361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_epoch(net, X, y, loss_fn, optimizer, batch_size=1024):\n    net.train()\n    n_batches = len(X) // batch_size\n    l1_lambda = 0.3\n    for indexes, (X_batch, y_batch) in tqdm(get_batch(X, y, batch_size=batch_size), total=n_batches):\n        y_pred = net(X_batch)\n        \n        # adding L1 regularization on linear_0 weights to get sparse weights\n        l1_norm = net.linear_0.weight.abs().sum()\n\n        loss = loss_fn(y_pred, y_batch) + l1_lambda * l1_norm\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()","metadata":{"execution":{"iopub.status.busy":"2022-08-22T14:48:25.893593Z","iopub.execute_input":"2022-08-22T14:48:25.894023Z","iopub.status.idle":"2022-08-22T14:48:25.901702Z","shell.execute_reply.started":"2022-08-22T14:48:25.893987Z","shell.execute_reply":"2022-08-22T14:48:25.900625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"net = Net(n_features=X.shape[1], sequence_len=X.shape[2])","metadata":{"execution":{"iopub.status.busy":"2022-08-22T14:48:27.348706Z","iopub.execute_input":"2022-08-22T14:48:27.349719Z","iopub.status.idle":"2022-08-22T14:48:27.359911Z","shell.execute_reply.started":"2022-08-22T14:48:27.349677Z","shell.execute_reply":"2022-08-22T14:48:27.358695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calc_metrics(y, prob):\n    metrics = []\n    metrics.append(('Cross_Entropy', log_loss(y, prob)))\n    metrics.append(('ROC-AUC', roc_auc_score(y, prob)))\n    metrics.append(('Average_Precision', average_precision_score(y, prob)))\n    \n    metrics.append(('D', top_n_recall_metric(y, prob)))\n    metrics.append(('G', normalized_gini_metric(y, prob)))\n    metrics.append(('AMEX', amex_metric(y, prob)))\n    return metrics","metadata":{"execution":{"iopub.status.busy":"2022-08-22T14:48:28.204075Z","iopub.execute_input":"2022-08-22T14:48:28.205013Z","iopub.status.idle":"2022-08-22T14:48:28.212714Z","shell.execute_reply.started":"2022-08-22T14:48:28.204957Z","shell.execute_reply":"2022-08-22T14:48:28.211497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train network","metadata":{}},{"cell_type":"code","source":"loss_fn = torch.nn.CrossEntropyLoss(reduction='mean')\nlearning_rate = 0.004_000\nfor epoch in range(5):\n    optimizer = torch.optim.RMSprop(net.parameters(), lr=learning_rate)\n    learning_rate /= 2.0\n    train_epoch(net, X_tr, y_tr, loss_fn, optimizer)\n    \n    predictions_raw_v = (net(torch.from_numpy(X_v.astype(np.float32))).detach().numpy())[:,1]\n    prob_v = expit(predictions_raw_v)\n    metrics = calc_metrics(y_v[:, 1], prob_v)\n    print('epoch:', epoch)\n    print('validation metrics:')\n    for name, value in metrics:\n        print(f'{name}: {value:0.6f}')\n    print('--------------------------------------------')","metadata":{"execution":{"iopub.status.busy":"2022-08-22T14:48:29.875940Z","iopub.execute_input":"2022-08-22T14:48:29.876464Z","iopub.status.idle":"2022-08-22T14:48:41.353609Z","shell.execute_reply.started":"2022-08-22T14:48:29.876408Z","shell.execute_reply":"2022-08-22T14:48:41.352560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"neural net has AMEX=0.76, ROC-AUC=0.9528. It shows that a network catches main dependencies in data. So we can look at weights it uses for customer rows.","metadata":{}},{"cell_type":"markdown","source":"## Finally let's analyze weights of 13 input rows for first and second output of layer linear_0\nOn graph we can see that only 13-th weight (it is weight of the last row) for second output has significant value. All other weights are approximately zero.\n\nOf course it does not mean that all other rows except the last one useless.\nIt simply shows that the last row contains the most important information for target prediction from neural network point of view. ","metadata":{}},{"cell_type":"code","source":"weights = net.linear_0.weight.detach().numpy()\nprint(weights.shape)\nfirst_output_weights = weights[0, :]\nsecond_output_weights = weights[1, :]\nsns.lineplot(x=np.arange(len(first_output_weights)), y=first_output_weights, label='first_output_weights')\nsns.lineplot(x=np.arange(len(second_output_weights)), y=second_output_weights, label='second_output_weights')","metadata":{"execution":{"iopub.status.busy":"2022-08-22T15:20:36.980172Z","iopub.execute_input":"2022-08-22T15:20:36.980872Z","iopub.status.idle":"2022-08-22T15:20:37.370542Z","shell.execute_reply.started":"2022-08-22T15:20:36.980822Z","shell.execute_reply":"2022-08-22T15:20:37.369210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}