{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"sourceType":"competition"},{"sourceId":10238503,"sourceType":"datasetVersion","datasetId":6331440}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\n\n# Path to the dataset\nordered_file_path = '/kaggle/input/intermediate-ordered-train-data/ordered_train_data_cleaned.parquet'\n\n# Load the dataset\ndata = pd.read_parquet(ordered_file_path)\n\n# Print all columns one by one\nprint(\"Columns in the dataset:\")\nfor col in data.columns:\n    print(col)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:56:16.692390Z","iopub.execute_input":"2024-12-19T09:56:16.692809Z","iopub.status.idle":"2024-12-19T09:56:28.218154Z","shell.execute_reply.started":"2024-12-19T09:56:16.692766Z","shell.execute_reply":"2024-12-19T09:56:28.217017Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Print the top 5 rows for each column\nprint(\"Top 5 rows of each column:\")\nfor col in data.columns:\n    print(f\"Column: {col}\")\n    print(data[col].head(5))\n    print(\"-\" * 50)  # Separator for better readability","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:58:34.925648Z","iopub.execute_input":"2024-12-19T09:58:34.926172Z","iopub.status.idle":"2024-12-19T09:58:35.166546Z","shell.execute_reply.started":"2024-12-19T09:58:34.926133Z","shell.execute_reply":"2024-12-19T09:58:35.165303Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# List of encoded features\nencoded_features = [\n    \"D_63_encoded\", \"D_64_encoded\", \"D_66_encoded\", \"D_68_encoded\", \n    \"B_30_encoded\", \"B_38_encoded\", \"D_114_encoded\", \"D_116_encoded\", \n    \"D_117_encoded\", \"D_120_encoded\", \"D_126_encoded\", \n    \"year_encoded\", \"month_encoded\", \"day_of_week_encoded\", \n    \"day_of_year_encoded\"\n]\n\n# List of original features corresponding to the encoded ones\noriginal_features = [feature.replace(\"_encoded\", \"\") for feature in encoded_features]\n\n# Check if both original and encoded features exist in the dataset\nboth_present = []\nfor orig, encoded in zip(original_features, encoded_features):\n    if orig in data.columns and encoded in data.columns:\n        both_present.append((orig, encoded))\n\nif both_present:\n    print(\"Both original and encoded features are present for the following pairs:\")\n    for orig, encoded in both_present:\n        print(f\"- Original: {orig}, Encoded: {encoded}\")\nelse:\n    print(\"No original and encoded features found together in the dataset.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T10:00:35.260308Z","iopub.execute_input":"2024-12-19T10:00:35.260817Z","iopub.status.idle":"2024-12-19T10:00:35.269197Z","shell.execute_reply.started":"2024-12-19T10:00:35.260779Z","shell.execute_reply":"2024-12-19T10:00:35.267936Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom torch_geometric.data import Data\nfrom torch_geometric.nn import GCNConv\nimport torch\nimport torch.nn.functional as F\nfrom sklearn.impute import SimpleImputer\nfrom annoy import AnnoyIndex\nimport psutil\n\n# Function to check memory usage\ndef check_memory_usage(stage=\"\"):\n    print(f\"{stage} - Memory usage: {psutil.virtual_memory().percent}%\")\n\n# Load the data\ntrain_data = pd.read_parquet('/kaggle/input/intermediate-ordered-train-data/ordered_train_data_cleaned.parquet')\ntrain_labels = pd.read_csv('/kaggle/input/amex-default-prediction/train_labels.csv', index_col='customer_ID')\n\n# Aggregate data for each customer_ID\naggregated_data = train_data.groupby('customer_ID').agg(['mean', 'max', 'min']).reset_index()\n\n# Fix column names\naggregated_data.columns = ['customer_ID' if col[0] == 'customer_ID' else '_'.join(col).strip('_') for col in aggregated_data.columns]\n\n# Merge with labels\naggregated_data = aggregated_data.merge(train_labels, on='customer_ID')\n\n# Ensure only numeric features are normalized\nnumeric_features = aggregated_data.select_dtypes(include=[np.number]).drop(columns=['target']).values\n\n# Handle NaN values using imputation\nimputer = SimpleImputer(strategy='mean')\nnumeric_features = imputer.fit_transform(numeric_features)\n\n# Normalize features, avoiding division by zero\nmean = np.mean(numeric_features, axis=0)\nstd = np.std(numeric_features, axis=0)\nstd[std == 0] = 1  # Avoid division by zero\nfeatures = (numeric_features - mean) / std\n\n# Verify no NaNs remain\nassert not np.isnan(features).any(), \"Features still contain NaN values!\"\ncheck_memory_usage(\"After Feature Normalization\")\n\n# Create edge connections using Annoy for approximate nearest neighbors\ndef create_edges(features, threshold=0.9):\n    num_features = features.shape[1]\n    annoy_index = AnnoyIndex(num_features, metric='angular')  # 'angular' approximates cosine similarity\n\n    # Build the index\n    for i, vec in enumerate(features):\n        annoy_index.add_item(i, vec)\n    annoy_index.build(10)  # Number of trees\n\n    # Find similar items\n    edges = []\n    for i in range(features.shape[0]):\n        neighbors, distances = annoy_index.get_nns_by_item(i, 10, include_distances=True)\n        for neighbor, distance in zip(neighbors, distances):\n            if neighbor != i and 1 - distance > threshold:  # Cosine similarity is 1 - distance\n                edges.append((i, neighbor))\n\n    print(f\"Number of edges created: {len(edges)}\")\n    if len(edges) == 0:\n        print(\"No edges created. Consider lowering the threshold.\")\n        edges = [(i, i) for i in range(features.shape[0])]  # Add self-loops\n\n    return np.array(edges)\n\n\nedges = create_edges(features, threshold=0.9)\ncheck_memory_usage(\"After Edge Creation\")\n\n# Create a PyTorch Geometric Data object\nedge_index = torch.tensor(edges.T, dtype=torch.long)\nx = torch.tensor(features, dtype=torch.float)\ny = torch.tensor(aggregated_data['target'].values, dtype=torch.long)\n\ndata = Data(x=x, edge_index=edge_index, y=y)\ncheck_memory_usage(\"After Data Object Creation\")\n\n# Define GCN Model\nclass GCN(torch.nn.Module):\n    def __init__(self, num_features, num_classes):\n        super(GCN, self).__init__()\n        self.conv1 = GCNConv(num_features, 16)\n        self.conv2 = GCNConv(16, num_classes)\n\n    def forward(self, data):\n        x, edge_index = data.x, data.edge_index\n        x = self.conv1(x, edge_index)\n        x = F.relu(x)\n        x = self.conv2(x, edge_index)\n        return F.log_softmax(x, dim=1)\n\n# Train the GCN\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nmodel = GCN(num_features=x.size(1), num_classes=2).to(device)\ndata = data.to(device)\noptimizer = torch.optim.Adam(model.parameters(), lr=0.01, weight_decay=5e-4)\n\nmodel.train()\nfor epoch in range(200):\n    optimizer.zero_grad()\n    out = model(data)\n    loss = F.nll_loss(out, data.y)\n    loss.backward()\n    optimizer.step()\n    print(f'Epoch {epoch+1}, Loss: {loss.item():.4f}')\n    if epoch % 10 == 0:\n        check_memory_usage(f\"Epoch {epoch+1}\")\n\n# Evaluate the model\nmodel.eval()\npred = model(data).argmax(dim=1)\naccuracy = (pred == data.y).sum().item() / data.num_nodes\nprint(f'Accuracy: {accuracy:.4f}')\n\n# Compute the Amex Metric\nfrom sklearn.metrics import roc_auc_score\n\ny_true = data.y.cpu().numpy()\ny_pred = pred.cpu().numpy()\namex_score = roc_auc_score(y_true, y_pred)\nprint(f'Amex Metric: {amex_score:.4f}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T16:23:09.397288Z","iopub.execute_input":"2024-12-25T16:23:09.397699Z","iopub.status.idle":"2024-12-25T16:24:22.458919Z","shell.execute_reply.started":"2024-12-25T16:23:09.397668Z","shell.execute_reply":"2024-12-25T16:24:22.457552Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls /kaggle/input/amex-default-prediction/","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T09:50:40.810939Z","iopub.execute_input":"2024-12-25T09:50:40.811374Z","iopub.status.idle":"2024-12-25T09:50:41.017924Z","shell.execute_reply.started":"2024-12-25T09:50:40.811340Z","shell.execute_reply":"2024-12-25T09:50:41.016677Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install torch torchvision torchaudio --index-url https://download.pytorch.org/whl/cpu\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T16:22:56.585129Z","iopub.execute_input":"2024-12-25T16:22:56.585542Z","iopub.status.idle":"2024-12-25T16:23:02.165002Z","shell.execute_reply.started":"2024-12-25T16:22:56.585512Z","shell.execute_reply":"2024-12-25T16:23:02.163718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install torch-scatter torch-sparse torch-cluster torch-spline-conv torch-geometric -f https://data.pyg.org/whl/torch-2.0.0+cpu.html\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T16:23:02.167183Z","iopub.execute_input":"2024-12-25T16:23:02.167618Z","iopub.status.idle":"2024-12-25T16:23:08.742828Z","shell.execute_reply.started":"2024-12-25T16:23:02.167574Z","shell.execute_reply":"2024-12-25T16:23:08.741005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch_geometric\nprint(\"PyTorch Geometric installed successfully!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T09:48:52.725982Z","iopub.execute_input":"2024-12-25T09:48:52.726393Z","iopub.status.idle":"2024-12-25T09:48:59.570193Z","shell.execute_reply.started":"2024-12-25T09:48:52.726359Z","shell.execute_reply":"2024-12-25T09:48:59.569089Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install annoy psutil","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T10:39:10.394391Z","iopub.execute_input":"2024-12-25T10:39:10.396471Z","iopub.status.idle":"2024-12-25T10:39:17.178479Z","shell.execute_reply.started":"2024-12-25T10:39:10.396366Z","shell.execute_reply":"2024-12-25T10:39:17.176591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom torch_geometric.data import Data\nfrom torch_geometric.nn import GCNConv\nimport torch\nimport torch.nn.functional as F\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import train_test_split\nfrom annoy import AnnoyIndex\nimport psutil\n\n# Function to check memory usage\ndef check_memory_usage(stage=\"\"):\n    print(f\"{stage} - Memory usage: {psutil.virtual_memory().percent}%\")\n\n# Load the data\ntrain_data = pd.read_parquet('/kaggle/input/intermediate-ordered-train-data/ordered_train_data_cleaned.parquet')\ntrain_labels = pd.read_csv('/kaggle/input/amex-default-prediction/train_labels.csv', index_col='customer_ID')\n\n# Aggregate data for each customer_ID\naggregated_data = train_data.groupby('customer_ID').agg(['mean', 'max', 'min']).reset_index()\n\n# Fix column names\naggregated_data.columns = ['customer_ID' if col[0] == 'customer_ID' else '_'.join(col).strip('_') for col in aggregated_data.columns]\n\n# Merge with labels\naggregated_data = aggregated_data.merge(train_labels, on='customer_ID')\n\n# Ensure only numeric features are normalized\nnumeric_features = aggregated_data.select_dtypes(include=[np.number]).drop(columns=['target']).values\n\n# Handle NaN values using imputation\nimputer = SimpleImputer(strategy='mean')\nnumeric_features = imputer.fit_transform(numeric_features)\n\n# Normalize features, avoiding division by zero\nmean = np.mean(numeric_features, axis=0)\nstd = np.std(numeric_features, axis=0)\nstd[std == 0] = 1  # Avoid division by zero\nfeatures = (numeric_features - mean) / std\n\n# Verify no NaNs remain\nassert not np.isnan(features).any(), \"Features still contain NaN values!\"\ncheck_memory_usage(\"After Feature Normalization\")\n\n# Create edge connections using Annoy for approximate nearest neighbors\ndef create_edges(features, threshold=0.9):\n    num_features = features.shape[1]\n    annoy_index = AnnoyIndex(num_features, metric='angular')  # 'angular' approximates cosine similarity\n\n    # Build the index\n    for i, vec in enumerate(features):\n        annoy_index.add_item(i, vec)\n    annoy_index.build(10)  # Number of trees\n\n    # Find similar items\n    edges = []\n    for i in range(features.shape[0]):\n        neighbors, distances = annoy_index.get_nns_by_item(i, 10, include_distances=True)\n        for neighbor, distance in zip(neighbors, distances):\n            if neighbor != i and 1 - distance > threshold:  # Cosine similarity is 1 - distance\n                edges.append((i, neighbor))\n\n    print(f\"Number of edges created: {len(edges)}\")\n    if len(edges) == 0:\n        print(\"No edges created. Consider lowering the threshold.\")\n        edges = [(i, i) for i in range(features.shape[0])]  # Add self-loops\n\n    return np.array(edges)\n\nedges = create_edges(features, threshold=0.9)\ncheck_memory_usage(\"After Edge Creation\")\n\n# Create a PyTorch Geometric Data object\nedge_index = torch.tensor(edges.T, dtype=torch.long)\nx = torch.tensor(features, dtype=torch.float)\ny = torch.tensor(aggregated_data['target'].values, dtype=torch.long)\n\ndata = Data(x=x, edge_index=edge_index, y=y)\ncheck_memory_usage(\"After Data Object Creation\")\n\n# Split the nodes into training and validation sets\ntrain_indices, val_indices = train_test_split(\n    np.arange(data.num_nodes),  # Node indices\n    test_size=0.1,             # 10% for validation\n    random_state=42,           # Reproducibility\n    stratify=data.y.numpy()    # Stratify by class labels\n)\n\n# Create masks for training and validation\ntrain_mask = torch.zeros(data.num_nodes, dtype=torch.bool)\nval_mask = torch.zeros(data.num_nodes, dtype=torch.bool)\n\ntrain_mask[train_indices] = True\nval_mask[val_indices] = True\n\ndata.train_mask = train_mask\ndata.val_mask = val_mask\n\n# Define GCN Model\nclass GCN(torch.nn.Module):\n    def __init__(self, num_features, num_classes):\n        super(GCN, self).__init__()\n        self.conv1 = GCNConv(num_features, 16)\n        self.conv2 = GCNConv(16, num_classes)\n\n    def forward(self, data):\n        x, edge_index = data.x, data.edge_index\n        x = self.conv1(x, edge_index)\n        x = F.relu(x)\n        x = self.conv2(x, edge_index)\n        return F.log_softmax(x, dim=1)\n\n# Train the GCN\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nmodel = GCN(num_features=x.size(1), num_classes=2).to(device)\ndata = data.to(device)\noptimizer = torch.optim.Adam(model.parameters(), lr=0.01, weight_decay=5e-4)\n\nmodel.train()\nfor epoch in range(200):\n    optimizer.zero_grad()\n    out = model(data)\n    loss = F.nll_loss(out[data.train_mask], data.y[data.train_mask])  # Use only training nodes\n    loss.backward()\n    optimizer.step()\n    print(f'Epoch {epoch+1}, Loss: {loss.item():.4f}')\n    if epoch % 10 == 0:\n        check_memory_usage(f\"Epoch {epoch+1}\")\n\n# Evaluate the model\nmodel.eval()\nout = model(data)  # Forward pass\npred = out.argmax(dim=1)\naccuracy = (pred[data.val_mask] == data.y[data.val_mask]).sum().item() / data.val_mask.sum().item()\nprint(f'Validation Accuracy: {accuracy:.4f}')\n\n# Compute Amex Metric for validation nodes\nval_true = data.y[data.val_mask].cpu().numpy()\nval_probs = out[data.val_mask].softmax(dim=1)[:, 1].detach().cpu().numpy()  # Corrected line\namex_score = roc_auc_score(val_true, val_probs)\nprint(f'Validation Amex Metric: {amex_score:.4f}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T16:35:51.940506Z","iopub.execute_input":"2024-12-25T16:35:51.941015Z","iopub.status.idle":"2024-12-25T16:36:56.108981Z","shell.execute_reply.started":"2024-12-25T16:35:51.940977Z","shell.execute_reply":"2024-12-25T16:36:56.107576Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom torch_geometric.data import Data\nfrom torch_geometric.nn import GCNConv\nimport torch\nimport torch.nn.functional as F\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import train_test_split\nfrom annoy import AnnoyIndex\nimport psutil\n\n# Function to check memory usage\ndef check_memory_usage(stage=\"\"):\n    print(f\"{stage} - Memory usage: {psutil.virtual_memory().percent}%\")\n\n# Load the data\ndata = pd.read_parquet('/kaggle/input/intermediate-ordered-train-data/ordered_train_data_cleaned.parquet')\nlabels = pd.read_csv('/kaggle/input/amex-default-prediction/train_labels.csv', index_col='customer_ID')\n\n# Split customer_IDs into train and test\ntrain_customer_ids, test_customer_ids = train_test_split(\n    labels.index, test_size=0.2, random_state=42, stratify=labels['target']\n)\n\ntrain_data = data[data['customer_ID'].isin(train_customer_ids)]\ntest_data = data[data['customer_ID'].isin(test_customer_ids)]\n\ntrain_labels = labels.loc[train_customer_ids]\ntest_labels = labels.loc[test_customer_ids]\n\n# Aggregate training data for each customer_ID\naggregated_train_data = train_data.groupby('customer_ID').agg(['mean', 'max', 'min']).reset_index()\naggregated_train_data.columns = ['customer_ID' if col[0] == 'customer_ID' else '_'.join(col).strip('_') for col in aggregated_train_data.columns]\naggregated_train_data = aggregated_train_data.merge(train_labels, on='customer_ID')\n\n# Extract features and labels for training\ntrain_features = aggregated_train_data.drop(columns=['customer_ID', 'target'])\ntrain_features = train_features.select_dtypes(include=[np.number]).values  # Select only numeric columns\ntrain_targets = aggregated_train_data['target'].values\n\n# Handle NaN values using imputation\nimputer = SimpleImputer(strategy='mean')\ntrain_features = imputer.fit_transform(train_features)\n\n# Normalize features (store mean and std for use on test set)\nmean = np.mean(train_features, axis=0)\nstd = np.std(train_features, axis=0)\nstd[std == 0] = 1  # Avoid division by zero\ntrain_features = (train_features - mean) / std\n\n# Verify no NaNs remain\nassert not np.isnan(train_features).any(), \"Train features still contain NaN values!\"\ncheck_memory_usage(\"After Train Feature Normalization\")\n\n# Create edges using only training data\n# Create edges using only training data\ndef create_edges(features, threshold=0.9):\n    num_features = features.shape[1]\n    annoy_index = AnnoyIndex(num_features, metric='angular')\n\n    # Build the index\n    for i, vec in enumerate(features):\n        annoy_index.add_item(i, vec)\n    annoy_index.build(10)\n\n    # Find similar items\n    edges = []\n    for i in range(features.shape[0]):\n        neighbors, distances = annoy_index.get_nns_by_item(i, 10, include_distances=True)\n        for neighbor, distance in zip(neighbors, distances):\n            if neighbor != i and 1 - distance > threshold:\n                edges.append((i, neighbor))\n\n    # If no edges were created, add self-loops\n    if len(edges) == 0:\n        print(\"No edges created. Adding self-loops for all nodes.\")\n        edges = [(i, i) for i in range(features.shape[0])]\n\n    # Debugging: Print edge information\n    print(f\"Number of edges created: {len(edges)}\")\n    if len(edges) < 10:\n        print(\"Sample edges:\", edges[:10])\n\n    return np.array(edges)\n\n# Create edges for the training set\ntrain_edges = create_edges(train_features, threshold=0.9)\ncheck_memory_usage(\"After Edge Creation for Train Set\")\n\n# Ensure edges exist\nif train_edges.size == 0:\n    print(\"Warning: No edges created in training data. Check feature normalization or threshold.\")\n\n# Create PyTorch Geometric Data object for training\ntrain_edge_index = torch.tensor(train_edges.T, dtype=torch.long)\ntrain_x = torch.tensor(train_features, dtype=torch.float)\ntrain_y = torch.tensor(train_targets, dtype=torch.long)\n\ntrain_data = Data(x=train_x, edge_index=train_edge_index, y=train_y)\n\n# Verify that edge_index is not empty\nassert train_data.edge_index.size(1) > 0, \"Error: train_data.edge_index is empty!\"\n\n\n# Split training nodes into train and validation sets\ntrain_indices, val_indices = train_test_split(\n    np.arange(train_data.num_nodes),\n    test_size=0.2,\n    random_state=42,\n    stratify=train_data.y.numpy()\n)\n\n# Create masks for training and validation\ntrain_mask = torch.zeros(train_data.num_nodes, dtype=torch.bool)\nval_mask = torch.zeros(train_data.num_nodes, dtype=torch.bool)\ntrain_mask[train_indices] = True\nval_mask[val_indices] = True\n\ntrain_data.train_mask = train_mask\ntrain_data.val_mask = val_mask\n\n# Define GCN Model\nclass GCN(torch.nn.Module):\n    def __init__(self, num_features, num_classes):\n        super(GCN, self).__init__()\n        self.conv1 = GCNConv(num_features, 16)\n        self.conv2 = GCNConv(16, num_classes)\n\n    def forward(self, data):\n        x, edge_index = data.x, data.edge_index\n        x = self.conv1(x, edge_index)\n        x = F.relu(x)\n        x = self.conv2(x, edge_index)\n        return F.log_softmax(x, dim=1)\n\n# Train the GCN\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nmodel = GCN(num_features=train_x.size(1), num_classes=2).to(device)\ntrain_data = train_data.to(device)\noptimizer = torch.optim.Adam(model.parameters(), lr=0.01, weight_decay=5e-4)\n\nmodel.train()\nfor epoch in range(200):\n    optimizer.zero_grad()\n    out = model(train_data)\n    loss = F.nll_loss(out[train_data.train_mask], train_data.y[train_data.train_mask])\n    loss.backward()\n    optimizer.step()\n    print(f'Epoch {epoch+1}, Loss: {loss.item():.4f}')\n\n# Preprocess and predict on the test set\naggregated_test_data = test_data.groupby('customer_ID').agg(['mean', 'max', 'min']).reset_index()\naggregated_test_data.columns = ['customer_ID' if col[0] == 'customer_ID' else '_'.join(col).strip('_') for col in aggregated_test_data.columns]\naggregated_test_data = aggregated_test_data.merge(test_labels, on='customer_ID')\n\ntest_features = aggregated_test_data.drop(columns=['customer_ID', 'target'])\ntest_features = test_features.select_dtypes(include=[np.number]).values  # Select only numeric columns\ntest_features = imputer.transform(test_features)  # Use train imputer\ntest_features = (test_features - mean) / std  # Normalize with train mean and std\n\ntest_x = torch.tensor(test_features, dtype=torch.float)\ntest_edge_index = torch.tensor(create_edges(test_features, threshold=0.9).T, dtype=torch.long)\n\ntest_data = Data(x=test_x, edge_index=test_edge_index)\n\n# Predict on test data\nmodel.eval()\ntest_data = test_data.to(device)\ntest_out = model(test_data)\ntest_probs = test_out.softmax(dim=1)[:, 1].cpu().detach().numpy()\n\nprint(\"Test Predictions:\", test_probs)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T17:22:02.947372Z","iopub.execute_input":"2024-12-25T17:22:02.947881Z","iopub.status.idle":"2024-12-25T17:22:59.550728Z","shell.execute_reply.started":"2024-12-25T17:22:02.947849Z","shell.execute_reply":"2024-12-25T17:22:59.549266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, roc_auc_score\n\n# Evaluate the model on the validation data\nmodel.eval()\nwith torch.no_grad():\n    val_out = model(train_data)  # Forward pass on the validation data\n    val_preds = val_out[train_data.val_mask].argmax(dim=1).cpu().numpy()  # Predicted labels\n    val_probs = val_out[train_data.val_mask].softmax(dim=1)[:, 1].cpu().numpy()  # Predicted probabilities\n    val_true = train_data.y[train_data.val_mask].cpu().numpy()  # Ground truth labels\n\n# Calculate Conventional Metrics\nval_accuracy = accuracy_score(val_true, val_preds)\nval_precision = precision_score(val_true, val_preds)\nval_recall = recall_score(val_true, val_preds)\nval_f1 = f1_score(val_true, val_preds)\n\n# Calculate Amex Metric\namex_metric = roc_auc_score(val_true, val_probs)\n\n# Print Results\nprint(f\"Validation Accuracy: {val_accuracy:.4f}\")\nprint(f\"Validation Precision: {val_precision:.4f}\")\nprint(f\"Validation Recall: {val_recall:.4f}\")\nprint(f\"Validation F1 Score: {val_f1:.4f}\")\nprint(f\"Validation Amex Metric (AUC): {amex_metric:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T17:24:58.363827Z","iopub.execute_input":"2024-12-25T17:24:58.364327Z","iopub.status.idle":"2024-12-25T17:24:58.464783Z","shell.execute_reply.started":"2024-12-25T17:24:58.364290Z","shell.execute_reply":"2024-12-25T17:24:58.463668Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the Amex metric\ndef amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x == 0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n    \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x == 0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n    \n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n    \n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n    return 0.5 * (g + d)\n\n# Evaluate the model on the validation data\nmodel.eval()\nwith torch.no_grad():\n    val_out = model(train_data)  # Forward pass on the validation data\n    val_preds = val_out[train_data.val_mask].argmax(dim=1).cpu().numpy()  # Predicted labels\n    val_probs = val_out[train_data.val_mask].softmax(dim=1)[:, 1].cpu().numpy()  # Predicted probabilities\n    val_true = train_data.y[train_data.val_mask].cpu().numpy()  # Ground truth labels\n\n# Calculate Conventional Metrics\nval_accuracy = accuracy_score(val_true, val_preds)\nval_precision = precision_score(val_true, val_preds)\nval_recall = recall_score(val_true, val_preds)\nval_f1 = f1_score(val_true, val_preds)\n\n# Prepare data for Amex metric\nval_true_df = pd.DataFrame({'target': val_true})\nval_probs_df = pd.DataFrame({'prediction': val_probs})\n\n# Calculate Amex Metric\namex_metric_value = amex_metric(val_true_df, val_probs_df)\n\n# Print Results\nprint(f\"Validation Accuracy: {val_accuracy:.4f}\")\nprint(f\"Validation Precision: {val_precision:.4f}\")\nprint(f\"Validation Recall: {val_recall:.4f}\")\nprint(f\"Validation F1 Score: {val_f1:.4f}\")\nprint(f\"Validation Amex Metric: {amex_metric_value:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T17:26:13.481201Z","iopub.execute_input":"2024-12-25T17:26:13.481582Z","iopub.status.idle":"2024-12-25T17:26:13.605364Z","shell.execute_reply.started":"2024-12-25T17:26:13.481551Z","shell.execute_reply":"2024-12-25T17:26:13.604220Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#best till now\nimport pandas as pd\nimport numpy as np\nfrom torch_geometric.data import Data\nfrom torch_geometric.nn import GCNConv\nimport torch\nimport torch.nn.functional as F\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, roc_auc_score\nfrom sklearn.model_selection import train_test_split\nfrom annoy import AnnoyIndex\nimport psutil\n\n# Function to check memory usage\ndef check_memory_usage(stage=\"\"):\n    print(f\"{stage} - Memory usage: {psutil.virtual_memory().percent}%\")\n\n# Define the Amex metric\ndef amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x == 0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n\n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x == 0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n    return 0.5 * (g + d)\n\n# Load the data\ndata = pd.read_parquet('/kaggle/input/intermediate-ordered-train-data/ordered_train_data_cleaned.parquet')\nlabels = pd.read_csv('/kaggle/input/amex-default-prediction/train_labels.csv', index_col='customer_ID')\n\n# Split customer_IDs into train and test\ntrain_customer_ids, test_customer_ids = train_test_split(\n    labels.index, test_size=0.2, random_state=42, stratify=labels['target']\n)\n\ntrain_data = data[data['customer_ID'].isin(train_customer_ids)]\ntest_data = data[data['customer_ID'].isin(test_customer_ids)]\n\ntrain_labels = labels.loc[train_customer_ids]\ntest_labels = labels.loc[test_customer_ids]\n\n# Aggregate training data for each customer_ID\naggregated_train_data = train_data.groupby('customer_ID').agg(['mean', 'max', 'min']).reset_index()\naggregated_train_data.columns = ['customer_ID' if col[0] == 'customer_ID' else '_'.join(col).strip('_') for col in aggregated_train_data.columns]\naggregated_train_data = aggregated_train_data.merge(train_labels, on='customer_ID')\n\n# Extract features and labels for training\ntrain_features = aggregated_train_data.drop(columns=['customer_ID', 'target'])\ntrain_features = train_features.select_dtypes(include=[np.number]).values  # Select only numeric columns\ntrain_targets = aggregated_train_data['target'].values\n\n# Handle NaN values using imputation\nimputer = SimpleImputer(strategy='mean')\ntrain_features = imputer.fit_transform(train_features)\n\n# Normalize features (store mean and std for use on test set)\nmean = np.mean(train_features, axis=0)\nstd = np.std(train_features, axis=0)\nstd[std == 0] = 1  # Avoid division by zero\ntrain_features = (train_features - mean) / std\n\n# Verify no NaNs remain\nassert not np.isnan(train_features).any(), \"Train features still contain NaN values!\"\ncheck_memory_usage(\"After Train Feature Normalization\")\n\n# Create edges using only training data\ndef create_edges(features, threshold=0.9):\n    num_features = features.shape[1]\n    annoy_index = AnnoyIndex(num_features, metric='angular')\n\n    # Build the index\n    for i, vec in enumerate(features):\n        annoy_index.add_item(i, vec)\n    annoy_index.build(10)\n\n    # Find similar items\n    edges = []\n    for i in range(features.shape[0]):\n        neighbors, distances = annoy_index.get_nns_by_item(i, 10, include_distances=True)\n        for neighbor, distance in zip(neighbors, distances):\n            if neighbor != i and 1 - distance > threshold:\n                edges.append((i, neighbor))\n\n    # If no edges were created, add self-loops\n    if len(edges) == 0:\n        print(\"No edges created. Adding self-loops for all nodes.\")\n        edges = [(i, i) for i in range(features.shape[0])]\n\n    return np.array(edges)\n\n# Create edges for the training set\ntrain_edges = create_edges(train_features, threshold=0.9)\ncheck_memory_usage(\"After Edge Creation for Train Set\")\n\n# Create PyTorch Geometric Data object for training\ntrain_edge_index = torch.tensor(train_edges.T, dtype=torch.long)\ntrain_x = torch.tensor(train_features, dtype=torch.float)\ntrain_y = torch.tensor(train_targets, dtype=torch.long)\n\ntrain_data = Data(x=train_x, edge_index=train_edge_index, y=train_y)\n\n# Split training nodes into train and validation sets\ntrain_indices, val_indices = train_test_split(\n    np.arange(train_data.num_nodes),\n    test_size=0.2,\n    random_state=42,\n    stratify=train_data.y.numpy()\n)\n\n# Create masks for training and validation\ntrain_mask = torch.zeros(train_data.num_nodes, dtype=torch.bool)\nval_mask = torch.zeros(train_data.num_nodes, dtype=torch.bool)\ntrain_mask[train_indices] = True\nval_mask[val_indices] = True\n\ntrain_data.train_mask = train_mask\ntrain_data.val_mask = val_mask\n\n# Define GCN Model\nclass GCN(torch.nn.Module):\n    def __init__(self, num_features, num_classes):\n        super(GCN, self).__init__()\n        self.conv1 = GCNConv(num_features, 16)\n        self.conv2 = GCNConv(16, num_classes)\n\n    def forward(self, data):\n        x, edge_index = data.x, data.edge_index\n        x = self.conv1(x, edge_index)\n        x = F.relu(x)\n        x = self.conv2(x, edge_index)\n        return F.log_softmax(x, dim=1)\n\n# Train the GCN\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nmodel = GCN(num_features=train_x.size(1), num_classes=2).to(device)\ntrain_data = train_data.to(device)\noptimizer = torch.optim.Adam(model.parameters(), lr=0.01, weight_decay=5e-4)\n\nmodel.train()\nfor epoch in range(200):\n    optimizer.zero_grad()\n    out = model(train_data)\n    loss = F.nll_loss(out[train_data.train_mask], train_data.y[train_data.train_mask])\n    loss.backward()\n    optimizer.step()\n    print(f'Epoch {epoch+1}, Loss: {loss.item():.4f}')\n\n# Preprocess and predict on the test set\naggregated_test_data = test_data.groupby('customer_ID').agg(['mean', 'max', 'min']).reset_index()\naggregated_test_data.columns = ['customer_ID' if col[0] == 'customer_ID' else '_'.join(col).strip('_') for col in aggregated_test_data.columns]\naggregated_test_data = aggregated_test_data.merge(test_labels, on='customer_ID')\n\ntest_features = aggregated_test_data.drop(columns=['customer_ID', 'target'])\ntest_features = test_features.select_dtypes(include=[np.number]).values  # Select only numeric columns\ntest_features = imputer.transform(test_features)  # Use train imputer\ntest_features = (test_features - mean) / std  # Normalize with train mean and std\n\n# Make predictions for each row independently\ntest_probs = []\ntest_predictions = []\ntest_true = aggregated_test_data['target'].values\n\nmodel.eval()  # Set model to evaluation mode\nwith torch.no_grad():\n    for row in test_features:\n        test_x = torch.tensor(row, dtype=torch.float).unsqueeze(0)  # Single-node feature\n        test_edge_index = torch.tensor([[0], [0]], dtype=torch.long)  # Self-loop\n        \n        test_data = Data(x=test_x, edge_index=test_edge_index).to(device)\n        test_out = model(test_data)\n        test_prob = test_out.softmax(dim=1)[:, 1].item()  # Probability of positive class\n        test_pred = test_out.argmax(dim=1).item()  # Predicted class label\n        \n        test_probs.append(test_prob)\n        test_predictions.append(test_pred)\n\n# Evaluate predictions\ntest_accuracy = accuracy_score(test_true, test_predictions)\ntest_precision = precision_score(test_true, test_predictions)\ntest_recall = recall_score(test_true, test_predictions)\ntest_f1 = f1_score(test_true, test_predictions)\ntest_auc = roc_auc_score(test_true, test_probs)\n\n# Calculate Amex Metric\ntest_true_df = pd.DataFrame({'target': test_true})\ntest_probs_df = pd.DataFrame({'prediction': test_probs})\ntest_amex_metric = amex_metric(test_true_df, test_probs_df)\n\n# Print results\nprint(f\"Test Accuracy: {test_accuracy:.4f}\")\nprint(f\"Test Precision: {test_precision:.4f}\")\nprint(f\"Test Recall: {test_recall:.4f}\")\nprint(f\"Test F1 Score: {test_f1:.4f}\")\nprint(f\"Test AUC: {test_auc:.4f}\")\nprint(f\"Test Amex Metric: {test_amex_metric:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T17:34:43.053570Z","iopub.execute_input":"2024-12-25T17:34:43.054128Z","iopub.status.idle":"2024-12-25T17:35:56.222581Z","shell.execute_reply.started":"2024-12-25T17:34:43.054095Z","shell.execute_reply":"2024-12-25T17:35:56.221254Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}