{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-13T12:37:16.842238Z","iopub.execute_input":"2025-11-13T12:37:16.842499Z","iopub.status.idle":"2025-11-13T12:37:19.134917Z","shell.execute_reply.started":"2025-11-13T12:37:16.842455Z","shell.execute_reply":"2025-11-13T12:37:19.133745Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install protobuf==3.20.3","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-13T12:39:29.940276Z","iopub.execute_input":"2025-11-13T12:39:29.940677Z","iopub.status.idle":"2025-11-13T12:39:36.782442Z","shell.execute_reply.started":"2025-11-13T12:39:29.940642Z","shell.execute_reply":"2025-11-13T12:39:36.780596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport gc\nfrom sklearn.preprocessing import StandardScaler\nimport tensorflow as tf\nfrom tensorflow.keras import models, layers, callbacks\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-13T12:42:58.49562Z","iopub.execute_input":"2025-11-13T12:42:58.495988Z","iopub.status.idle":"2025-11-13T12:42:58.508749Z","shell.execute_reply.started":"2025-11-13T12:42:58.495962Z","shell.execute_reply":"2025-11-13T12:42:58.507584Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Paths\ntrain_path = \"/kaggle/input/amex-default-prediction/train_data.csv\"\nlabels_path = \"/kaggle/input/amex-default-prediction/train_labels.csv\"\n\n# Load labels\nlabels = pd.read_csv(labels_path)\nlabels_dict = labels.set_index(\"customer_ID\")[\"target\"].to_dict()\ndel labels\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-13T12:43:13.603021Z","iopub.execute_input":"2025-11-13T12:43:13.603444Z","iopub.status.idle":"2025-11-13T12:43:16.145271Z","shell.execute_reply.started":"2025-11-13T12:43:13.603414Z","shell.execute_reply":"2025-11-13T12:43:16.143887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CHUNKED EDA (Light)\n\nmissing_summary = {}\ncorr_summary = {}\n\ntrain_chunks = pd.read_csv(TRAIN_FILE, chunksize=CHUNK_SIZE)\nfor i, chunk in enumerate(train_chunks):\n    print(f\"[EDA] Processing chunk {i+1}\")\n\n    # Merge target\n    chunk['target'] = chunk['customer_ID'].map(labels_dict)\n\n    # Compute missing %\n    missing_pct = chunk.isnull().mean() * 100\n    for col, pct in missing_pct.items():\n        if col not in missing_summary:\n            missing_summary[col] = []\n        missing_summary[col].append(pct)\n\n    # Numeric columns\n    num_cols = chunk.select_dtypes(include=[np.number]).columns.tolist()\n    num_cols = [c for c in num_cols if c != 'target']\n\n    # Fill missing with median\n    chunk[num_cols] = chunk[num_cols].fillna(chunk[num_cols].median())\n\n    # Compute correlation with target\n    for col in num_cols:\n        corr_val = chunk[[col, 'target']].corr().iloc[0, 1]\n        if not np.isnan(corr_val):\n            if col not in corr_summary:\n                corr_summary[col] = []\n            corr_summary[col].append(corr_val)\n\n    del chunk\n    gc.collect()\n\n# Aggregate missing %\nmissing_mean = {col: np.mean(pcts) for col, pcts in missing_summary.items()}\nmissing_df = pd.DataFrame.from_dict(missing_mean, orient='index', columns=['missing_pct']).sort_values(by='missing_pct', ascending=False)\nprint(\"\\nTop 10 missing features:\")\nprint(missing_df.head(10))\n\n# Identify columns to drop\ndrop_cols = missing_df[missing_df['missing_pct'] > 95].index.tolist()\nprint(f\"\\nDropping {len(drop_cols)} columns with >95% missing values.\")\n\n# Aggregate correlations\ncorr_mean = {col: np.mean(vals) for col, vals in corr_summary.items()}\ncorr_df = pd.DataFrame.from_dict(corr_mean, orient='index', columns=['corr']).sort_values(by='corr', ascending=False)\n\n# Plot correlations\nplt.figure(figsize=(6, 4))\nsns.barplot(x='corr', y=corr_df.index[:10], data=corr_df.head(10), palette='viridis')\nplt.title(\"Top 10 Positive Correlations with Target\")\nplt.show()\n\nplt.figure(figsize=(6, 4))\nsns.barplot(x='corr', y=corr_df.index[-10:], data=corr_df.tail(10), palette='magma')\nplt.title(\"Top 10 Negative Correlations with Target\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-13T12:44:37.927664Z","iopub.execute_input":"2025-11-13T12:44:37.92813Z","iopub.status.idle":"2025-11-13T12:55:50.053771Z","shell.execute_reply.started":"2025-11-13T12:44:37.92809Z","shell.execute_reply":"2025-11-13T12:55:50.052558Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CHUNKED PREPROCESSING\n\nos.makedirs(\"/kaggle/working/processed_chunks\", exist_ok=True)\ntrain_chunks = pd.read_csv(train_path, chunksize=CHUNK_SIZE)\nprocessed_files = []\n\nfor i, chunk in enumerate(train_chunks):\n    print(f\"[Preprocessing] Chunk {i+1}\")\n\n    # Merge labels\n    chunk['target'] = chunk['customer_ID'].map(labels_dict)\n\n    # Drop high-missing columns\n    chunk.drop(columns=[c for c in drop_cols if c in chunk.columns], inplace=True, errors='ignore')\n\n    # Keep only numeric columns (ignore strings, dates, IDs)\n    num_cols = chunk.select_dtypes(include=[np.number]).columns.tolist()\n    if 'target' in num_cols:\n        num_cols.remove('target')\n\n    # Fill missing + scale\n    chunk[num_cols] = chunk[num_cols].fillna(chunk[num_cols].median())\n    scaler = StandardScaler()\n    chunk[num_cols] = scaler.fit_transform(chunk[num_cols])\n\n    # Reconstruct final chunk\n    final_chunk = chunk[['customer_ID', 'target'] + num_cols].copy()\n\n    # Save chunk\n    file_name = f\"/kaggle/working/processed_chunks/processed_chunk_{i+1}.parquet\"\n    final_chunk.to_parquet(file_name, index=False)\n    processed_files.append(file_name)\n\n    del chunk, final_chunk\n    gc.collect()\n\nprint(\"\\nAll processed chunks saved!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-13T13:37:52.66737Z","iopub.execute_input":"2025-11-13T13:37:52.66782Z","iopub.status.idle":"2025-11-13T13:47:02.558654Z","shell.execute_reply.started":"2025-11-13T13:37:52.66776Z","shell.execute_reply":"2025-11-13T13:47:02.557508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# TRAIN/VALIDATION SPLIT (by chunks)\n\nnp.random.seed(42)\nnp.random.shuffle(processed_files)\nsplit_idx = int(len(processed_files) * (1 - VAL_RATIO))\ntrain_files = processed_files[:split_idx]\nval_files = processed_files[split_idx:]\n\nprint(f\"\\nTrain chunks: {len(train_files)}, Validation chunks: {len(val_files)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-13T13:47:38.136632Z","iopub.execute_input":"2025-11-13T13:47:38.137071Z","iopub.status.idle":"2025-11-13T13:47:38.145252Z","shell.execute_reply.started":"2025-11-13T13:47:38.137046Z","shell.execute_reply":"2025-11-13T13:47:38.143624Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# DATA GENERATOR\n\ndef data_generator(file_list, batch_size=1024):\n    while True:\n        for file in file_list:\n            chunk = pd.read_parquet(file)\n\n            # Convert numeric features to float32\n            X = chunk.drop(['customer_ID', 'target'], axis=1).astype('float32').values\n            y = chunk['target'].astype('float32').values\n\n            for start in range(0, len(X), batch_size):\n                end = start + batch_size\n                yield X[start:end], y[start:end]\n\n            del chunk\n            gc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-13T13:48:32.091564Z","iopub.execute_input":"2025-11-13T13:48:32.091915Z","iopub.status.idle":"2025-11-13T13:48:32.099226Z","shell.execute_reply.started":"2025-11-13T13:48:32.091892Z","shell.execute_reply":"2025-11-13T13:48:32.097985Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# NEURAL NETWORK MODEL\n\nos.environ[\"CUDA_VISIBLE_DEVICES\"] = \"-1\"\n\ndef build_nn(input_dim):\n    model = models.Sequential([\n        layers.Input(shape=(input_dim,)),\n        layers.Dense(128, activation='relu'),\n        layers.Dropout(0.3),\n        layers.Dense(64, activation='relu'),\n        layers.Dropout(0.2),\n        layers.Dense(32, activation='relu'),\n        layers.Dense(1, activation='sigmoid')\n    ])\n    model.compile(\n        optimizer='adam',\n        loss='binary_crossentropy',\n        metrics=[tf.keras.metrics.AUC(name='auc')]\n    )\n    return model\n\n# Determine input dimension\nsample = pd.read_parquet(processed_files[0])\ninput_dim = sample.drop(['customer_ID', 'target'], axis=1).shape[1]\ndel sample\ngc.collect()\n\nmodel = build_nn(input_dim)\nearly_stop = callbacks.EarlyStopping(monitor='val_auc', mode='max', patience=5, restore_best_weights=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-13T13:48:37.297111Z","iopub.execute_input":"2025-11-13T13:48:37.297713Z","iopub.status.idle":"2025-11-13T13:48:40.462024Z","shell.execute_reply.started":"2025-11-13T13:48:37.297683Z","shell.execute_reply":"2025-11-13T13:48:40.460524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# TRAIN MODEL\n\ntrain_gen = data_generator(train_files, batch_size=1024)\nval_gen = data_generator(val_files, batch_size=1024)\n\nsteps_per_epoch = 100\nval_steps = 20\n\nhistory = model.fit(\n    train_gen,\n    validation_data=val_gen,\n    steps_per_epoch=steps_per_epoch,\n    validation_steps=val_steps,\n    epochs=10,\n    callbacks=[early_stop],\n    verbose=2\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-13T13:48:53.289128Z","iopub.execute_input":"2025-11-13T13:48:53.289523Z","iopub.status.idle":"2025-11-13T13:49:16.860137Z","shell.execute_reply.started":"2025-11-13T13:48:53.289466Z","shell.execute_reply":"2025-11-13T13:49:16.859234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# EVALUATION\n\nplt.plot(history.history['auc'], label='Train AUC')\nplt.plot(history.history['val_auc'], label='Val AUC')\nplt.legend()\nplt.title(\"Neural Network AUC\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-13T13:50:14.0528Z","iopub.execute_input":"2025-11-13T13:50:14.053214Z","iopub.status.idle":"2025-11-13T13:50:14.269696Z","shell.execute_reply.started":"2025-11-13T13:50:14.053178Z","shell.execute_reply":"2025-11-13T13:50:14.268435Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create Submission CSV\n\nfrom sklearn.preprocessing import StandardScaler\n\n# Create Submission CSV\ntest_path = \"/kaggle/input/amex-default-prediction/test_data.csv\"\nsubmission_file = \"/kaggle/working/submission.csv\"\nchunk_size = 500_000  # adjust based on RAM\n\nsubmission_rows = []\n\n# Read test data in chunks using python engine\ntest_chunks = pd.read_csv(test_path, chunksize=chunk_size, engine='python', dtype={'customer_ID': str})\nfor i, chunk in enumerate(test_chunks):\n    print(f\"[Submission] Processing test chunk {i+1}\")\n\n    # Drop columns that were dropped in training\n    chunk.drop(columns=[c for c in drop_cols if c in chunk.columns], inplace=True, errors='ignore')\n\n    # Keep only numeric columns\n    num_cols = chunk.select_dtypes(include=[np.number]).columns.tolist()\n\n    # Fill missing values with training median\n    for idx, col in enumerate(num_cols):\n        median_value = train_medians[idx] if idx < len(train_medians) else 0\n        chunk[col] = chunk[col].fillna(median_value)\n\n    # Scale using training scaler\n    X_test = scaler.transform(chunk[num_cols].values).astype('float32')\n\n    # Predict\n    preds = model.predict(X_test, batch_size=1024, verbose=0).flatten()\n\n    # Store customer_ID and predictions\n    submission_rows.extend(zip(chunk['customer_ID'].values, preds))\n\n    del chunk, X_test, preds\n    gc.collect()\n\n# Create submission DataFrame\nsubmission_df = pd.DataFrame(submission_rows, columns=['customer_ID', 'prediction'])\n\nsubmission_df['customer_ID'] = submission_df['customer_ID'].astype(str)\nsubmission_df['prediction'].replace([np.inf, -np.inf], np.nan, inplace=True)\nsubmission_df['prediction'].fillna(0.0, inplace=True)\nsubmission_df['prediction'] = submission_df['prediction'].astype(float)\n\nsubmission_df.to_csv(submission_file, index=False)\nprint(\"Submission CSV created at:\", submission_file)\nsubmission_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-13T15:21:25.347386Z","iopub.execute_input":"2025-11-13T15:21:25.347805Z","iopub.status.idle":"2025-11-13T16:10:39.271743Z","shell.execute_reply.started":"2025-11-13T15:21:25.347776Z","shell.execute_reply":"2025-11-13T16:10:39.269836Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Histogram of predictions\nplt.figure(figsize=(8,5))\nsns.histplot(submission_df['prediction'], bins=50, kde=True, color='skyblue')\nplt.title(\"Distribution of Test Predictions\")\nplt.xlabel(\"Predicted Probability of Default\")\nplt.ylabel(\"Count\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-13T16:11:28.499551Z","iopub.execute_input":"2025-11-13T16:11:28.500338Z","iopub.status.idle":"2025-11-13T16:12:21.339522Z","shell.execute_reply.started":"2025-11-13T16:11:28.500305Z","shell.execute_reply":"2025-11-13T16:12:21.338433Z"}},"outputs":[],"execution_count":null}]}